Compare commits
14
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
3e6e21bf0a | ||
|
|
0fbd2c8f12 | ||
|
|
1379b699f3 | ||
|
|
1ff0991452 | ||
|
|
ac01f627d1 | ||
|
|
f3bc23cb81 | ||
|
|
8d4a7cf240 | ||
|
|
c4f16a5448 | ||
|
|
e4da4e0610 | ||
|
|
9a3238025d | ||
|
|
c96ef87714 | ||
|
|
de9870175f | ||
|
|
be04ebe954 | ||
|
|
62d6a7a150 |
+19
-61
@@ -31,60 +31,20 @@ pub fn freestandingTarget(b: *std.Build) std.Build.ResolvedTarget {
|
|||||||
});
|
});
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Which library domain package exports each importable module — the one
|
/// Resolve one imported module by searching the packages this binary DECLARED
|
||||||
/// name -> home table. When a domain grows a module, it gets a row here; a
|
/// in its own build.zig.zon — the C include path made literal: an import can
|
||||||
/// binary naming a module whose home is missing from its own build.zig.zon
|
/// only be satisfied by a domain the binary claims, and each domain's own
|
||||||
/// fails loudly at dependency resolution.
|
/// build.zig (its addModule exports) is the single statement of who owns
|
||||||
const ModuleHome = struct { name: []const u8, home: []const u8 };
|
/// what. There is no name table here to drift.
|
||||||
const module_homes = [_]ModuleHome{
|
fn moduleFromDeclaredDependencies(b: *std.Build, name: []const u8) *std.Build.Module {
|
||||||
// library/kernel — the userspace private-ABI library, split by concern.
|
for (b.available_deps) |declared| {
|
||||||
.{ .name = "abi", .home = "kernel" },
|
const dependency = b.dependency(declared[0], .{});
|
||||||
.{ .name = "system-call", .home = "kernel" },
|
if (dependency.builder.modules.get(name)) |module| return module;
|
||||||
.{ .name = "ipc", .home = "kernel" },
|
|
||||||
.{ .name = "time", .home = "kernel" },
|
|
||||||
.{ .name = "thread", .home = "kernel" },
|
|
||||||
.{ .name = "logging", .home = "kernel" },
|
|
||||||
.{ .name = "process", .home = "kernel" },
|
|
||||||
.{ .name = "file-system", .home = "kernel" },
|
|
||||||
.{ .name = "memory", .home = "kernel" },
|
|
||||||
.{ .name = "service", .home = "kernel" },
|
|
||||||
.{ .name = "start", .home = "kernel" },
|
|
||||||
// library/device — driver-side libraries + the flat reference data.
|
|
||||||
.{ .name = "mmio", .home = "device" },
|
|
||||||
.{ .name = "acpi-ids", .home = "device" },
|
|
||||||
.{ .name = "device-abi", .home = "device" },
|
|
||||||
.{ .name = "aml", .home = "device" },
|
|
||||||
.{ .name = "usb-abi", .home = "device" },
|
|
||||||
.{ .name = "usb-ids", .home = "device" },
|
|
||||||
.{ .name = "usb", .home = "device" },
|
|
||||||
.{ .name = "driver", .home = "device" },
|
|
||||||
.{ .name = "block", .home = "device" },
|
|
||||||
.{ .name = "pci", .home = "device" },
|
|
||||||
.{ .name = "pci-class", .home = "device" },
|
|
||||||
.{ .name = "device-registry", .home = "device" },
|
|
||||||
// library/client — userspace service clients.
|
|
||||||
.{ .name = "display-client", .home = "client" },
|
|
||||||
.{ .name = "input-client", .home = "client" },
|
|
||||||
// library/protocol — the wire protocols.
|
|
||||||
.{ .name = "vfs-protocol", .home = "protocol" },
|
|
||||||
.{ .name = "input-protocol", .home = "protocol" },
|
|
||||||
.{ .name = "block-protocol", .home = "protocol" },
|
|
||||||
.{ .name = "usb-transfer-protocol", .home = "protocol" },
|
|
||||||
.{ .name = "device-manager-protocol", .home = "protocol" },
|
|
||||||
.{ .name = "display-protocol", .home = "protocol" },
|
|
||||||
.{ .name = "scanout-protocol", .home = "protocol" },
|
|
||||||
.{ .name = "power-protocol", .home = "protocol" },
|
|
||||||
// library/csv — the /etc/*.csv helpers.
|
|
||||||
.{ .name = "csv", .home = "csv" },
|
|
||||||
// library/xkeyboard-config — keycode -> keysym/character tables.
|
|
||||||
.{ .name = "xkeyboard-config", .home = "xkeyboard-config" },
|
|
||||||
};
|
|
||||||
|
|
||||||
fn moduleHome(name: []const u8) ?[]const u8 {
|
|
||||||
for (module_homes) |entry| {
|
|
||||||
if (std.mem.eql(u8, entry.name, name)) return entry.home;
|
|
||||||
}
|
}
|
||||||
return null;
|
@panic(b.fmt(
|
||||||
|
"no declared dependency exports a module named '{s}' — declare the domain that owns it in this package's build.zig.zon",
|
||||||
|
.{name},
|
||||||
|
));
|
||||||
}
|
}
|
||||||
|
|
||||||
/// What `userBinary` needs to know about one user binary.
|
/// What `userBinary` needs to know about one user binary.
|
||||||
@@ -95,8 +55,8 @@ pub const UserBinaryOptions = struct {
|
|||||||
root_source_file: std.Build.LazyPath,
|
root_source_file: std.Build.LazyPath,
|
||||||
/// Exactly the modules the program's source @imports (directly or through
|
/// Exactly the modules the program's source @imports (directly or through
|
||||||
/// its same-directory files) — no more, no less. Order is free; sorted
|
/// its same-directory files) — no more, no less. Order is free; sorted
|
||||||
/// reads best. An undeclared @import fails the compile; a declared name no
|
/// reads best. An undeclared @import fails the compile; a name no
|
||||||
/// domain exports fails the build graph with a pointer to module_homes.
|
/// declared domain exports fails the build graph, naming the miss.
|
||||||
imports: []const []const u8,
|
imports: []const []const u8,
|
||||||
/// Built multi-threaded (`single_threaded = false`) so real atomics/TLS
|
/// Built multi-threaded (`single_threaded = false`) so real atomics/TLS
|
||||||
/// work — required before a binary may call `Thread.spawn`
|
/// work — required before a binary may call `Thread.spawn`
|
||||||
@@ -121,12 +81,10 @@ pub fn userBinary(b: *std.Build, options: UserBinaryOptions) *std.Build.Step.Com
|
|||||||
const kernel = b.dependency("kernel", .{});
|
const kernel = b.dependency("kernel", .{});
|
||||||
var imports: std.ArrayListUnmanaged(std.Build.Module.Import) = .empty;
|
var imports: std.ArrayListUnmanaged(std.Build.Module.Import) = .empty;
|
||||||
for (options.imports) |name| {
|
for (options.imports) |name| {
|
||||||
const home = moduleHome(name) orelse @panic(b.fmt(
|
imports.append(b.allocator, .{
|
||||||
"no library domain exports a module named '{s}' — if a domain grew it, add its row to module_homes in build-support/build.zig",
|
.name = name,
|
||||||
.{name},
|
.module = moduleFromDeclaredDependencies(b, name),
|
||||||
));
|
}) catch @panic("OOM");
|
||||||
const dependency = if (std.mem.eql(u8, home, "kernel")) kernel else b.dependency(home, .{});
|
|
||||||
imports.append(b.allocator, .{ .name = name, .module = dependency.module(name) }) catch @panic("OOM");
|
|
||||||
}
|
}
|
||||||
// Settings (target, optimize, code model, ...) live on the root module
|
// Settings (target, optimize, code model, ...) live on the root module
|
||||||
// only; the program module inherits them.
|
// only; the program module inherits them.
|
||||||
|
|||||||
@@ -96,6 +96,48 @@ fn addKernel(
|
|||||||
return exe;
|
return exe;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// One row of the production ship table: which package, which of its
|
||||||
|
/// artifacts, and the FHS boot path. For most binaries all three share one
|
||||||
|
/// name; the helpers below make a row from just that name.
|
||||||
|
const ShipRow = struct { path: []const u8, package: []const u8, artifact: []const u8 };
|
||||||
|
|
||||||
|
fn service(comptime name: []const u8) ShipRow {
|
||||||
|
return .{ .path = "system/services/" ++ name, .package = name, .artifact = name };
|
||||||
|
}
|
||||||
|
fn driver(comptime name: []const u8) ShipRow {
|
||||||
|
return .{ .path = "system/drivers/" ++ name, .package = name, .artifact = name };
|
||||||
|
}
|
||||||
|
/// An extra artifact of a multi-binary driver package (ps2-bus, usb-hid),
|
||||||
|
/// bundled at its own flattened /system/drivers path.
|
||||||
|
fn driverArtifact(comptime package: []const u8, comptime artifact: []const u8) ShipRow {
|
||||||
|
return .{ .path = "system/drivers/" ++ artifact, .package = package, .artifact = artifact };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The production ship table — what a plain `zig build` image contains,
|
||||||
|
/// beyond the specials the build fn adds around it (init, discovery, the
|
||||||
|
/// /system/configuration data files; the /test fixtures join only under
|
||||||
|
/// -Dtest-case).
|
||||||
|
/// Selecting what goes into a build = selecting rows: a package in no row is
|
||||||
|
/// not just unshipped, its build file is never even loaded
|
||||||
|
/// (docs/build-packages-plan.md).
|
||||||
|
const production_ship = [_]ShipRow{
|
||||||
|
service("fat"),
|
||||||
|
service("display"),
|
||||||
|
service("display-demo"),
|
||||||
|
service("device-manager"),
|
||||||
|
service("input"),
|
||||||
|
service("logger"),
|
||||||
|
driver("pci-bus"),
|
||||||
|
driver("ps2-bus"),
|
||||||
|
driverArtifact("ps2-bus", "ps2-keyboard"),
|
||||||
|
driverArtifact("ps2-bus", "ps2-mouse"),
|
||||||
|
driver("usb-xhci-bus"),
|
||||||
|
driverArtifact("usb-hid", "usb-hid-keyboard"),
|
||||||
|
driverArtifact("usb-hid", "usb-hid-mouse"),
|
||||||
|
driver("usb-storage"),
|
||||||
|
driver("virtio-gpu"),
|
||||||
|
};
|
||||||
|
|
||||||
pub fn build(b: *std.Build) void {
|
pub fn build(b: *std.Build) void {
|
||||||
ensureZigVersion();
|
ensureZigVersion();
|
||||||
|
|
||||||
@@ -211,48 +253,27 @@ pub fn build(b: *std.Build) void {
|
|||||||
const exe = addKernel(b, kernel_target, optimize, kernel_modules, test_case, serial);
|
const exe = addKernel(b, kernel_target, optimize, kernel_modules, test_case, serial);
|
||||||
const exe_serial = addKernel(b, kernel_target, optimize, kernel_modules, test_case, true);
|
const exe_serial = addKernel(b, kernel_target, optimize, kernel_modules, test_case, true);
|
||||||
|
|
||||||
// --- the user-space binaries, every one of them a package ---
|
// --- what ships: the boot tree ---
|
||||||
// Binary packages (docs/build-packages-plan.md, phase 2): each binary
|
// Every user binary and its FHS home on the boot volume. There is no packed
|
||||||
// builds itself against the domain packages via build-support's shared
|
// ramdisk artifact any more: make-fat-image.py lays each binary out at its
|
||||||
// recipe, started in ring 3 by the kernel's user-ELF loader like always;
|
// path on the image, and the EFI loader walks /system and /test at boot and
|
||||||
// the root build just takes artifacts for the boot image. init receives
|
// builds the in-RAM initial_ramdisk table from the trees — the volume's file
|
||||||
// the root's -Dserial as a dependency option (its liveness heartbeat is a
|
// structure is the single source of truth. Entry names (and hence argv[0] and
|
||||||
// serial/test-build diagnostic the QEMU harness asserts on; a flashable
|
// task names) are these paths with a leading slash.
|
||||||
// image leaves it out).
|
//
|
||||||
const init_exe = b.dependency("init", .{ .serial = serial }).artifact("init");
|
// The uniform rows live in `production_ship` (the table above `build`);
|
||||||
|
// spelled out here are only the genuinely non-uniform entries: init
|
||||||
// --- the rest of the boot tree: /system services and drivers ---
|
// (receives the root's -Dserial as a dependency option — its liveness
|
||||||
// Each is built by the same user-binary recipe and laid out at its FHS path on
|
// heartbeat is a serial/test-build diagnostic the QEMU harness asserts
|
||||||
// the boot volume (see `bundled` below). The EFI loader walks the tree at boot
|
// on; a flashable image leaves it out), discovery (the -Ddiscovery pick),
|
||||||
// and hands the kernel an in-RAM initial_ramdisk of it (system/initial-ramdisk.zig).
|
// and the /system/configuration data files. Each binary builds itself
|
||||||
// (The /test fixtures are lazy dependencies, resolved further down only
|
// against the domain packages via build-support's shared recipe; the root
|
||||||
// for a -Dtest-case build.)
|
// just takes artifacts (docs/build-packages-plan.md).
|
||||||
// The drivers, each directory its own package: the PS/2 bus family (bus +
|
var bundled_list: std.ArrayListUnmanaged(images.BundledBinary) = .empty;
|
||||||
// keyboard + mouse from one package), the xHCI bus driver, the USB HID
|
bundled_list.append(b.allocator, .{
|
||||||
// class drivers, and USB mass storage. Their unit tests ride along.
|
.path = "system/services/init",
|
||||||
const ps2_bus_package = b.dependency("ps2-bus", .{});
|
.binary = b.dependency("init", .{ .serial = serial }).artifact("init").getEmittedBin(),
|
||||||
const ps2_bus_exe = ps2_bus_package.artifact("ps2-bus");
|
}) catch @panic("OOM");
|
||||||
const ps2_keyboard_exe = ps2_bus_package.artifact("ps2-keyboard");
|
|
||||||
const ps2_mouse_exe = ps2_bus_package.artifact("ps2-mouse");
|
|
||||||
const usb_xhci_bus_exe = b.dependency("usb-xhci-bus", .{}).artifact("usb-xhci-bus");
|
|
||||||
const usb_hid_package = b.dependency("usb-hid", .{});
|
|
||||||
const usb_hid_keyboard_exe = usb_hid_package.artifact("usb-hid-keyboard");
|
|
||||||
const usb_hid_mouse_exe = usb_hid_package.artifact("usb-hid-mouse");
|
|
||||||
const usb_storage_package = b.dependency("usb-storage", .{});
|
|
||||||
const usb_storage_exe = usb_storage_package.artifact("usb-storage");
|
|
||||||
// The FAT filesystem server and the display stack, each its own package
|
|
||||||
// (fat's and display's unit tests ride along in their packages).
|
|
||||||
const fat_package = b.dependency("fat", .{});
|
|
||||||
const fat_exe = fat_package.artifact("fat");
|
|
||||||
const display_package = b.dependency("display", .{});
|
|
||||||
const display_exe = display_package.artifact("display");
|
|
||||||
const display_demo_exe = b.dependency("display-demo", .{}).artifact("display-demo");
|
|
||||||
const virtio_gpu_package = b.dependency("virtio-gpu", .{});
|
|
||||||
const virtio_gpu_exe = virtio_gpu_package.artifact("virtio-gpu");
|
|
||||||
// The first binary package (docs/build-packages-plan.md, phase 2): pci-bus
|
|
||||||
// builds itself against the domain packages; the root build just takes the
|
|
||||||
// artifact for the boot image.
|
|
||||||
const pci_bus_exe = b.dependency("pci-bus", .{}).artifact("pci-bus");
|
|
||||||
// The discovery service: one swappable process per firmware
|
// The discovery service: one swappable process per firmware
|
||||||
// (docs/discovery.md), bundled under the neutral ramdisk name
|
// (docs/discovery.md), bundled under the neutral ramdisk name
|
||||||
// "discovery" so the device manager never learns which firmware it is on.
|
// "discovery" so the device manager never learns which firmware it is on.
|
||||||
@@ -269,59 +290,39 @@ pub fn build(b: *std.Build) void {
|
|||||||
.acpi => (b.lazyDependency("acpi", .{}) orelse @panic("system/services/acpi is missing")).artifact("discovery"),
|
.acpi => (b.lazyDependency("acpi", .{}) orelse @panic("system/services/acpi is missing")).artifact("discovery"),
|
||||||
.fdt => (b.lazyDependency("fdt", .{}) orelse @panic("system/services/fdt is missing")).artifact("discovery"),
|
.fdt => (b.lazyDependency("fdt", .{}) orelse @panic("system/services/fdt is missing")).artifact("discovery"),
|
||||||
};
|
};
|
||||||
const device_manager_exe = b.dependency("device-manager", .{}).artifact("device-manager");
|
bundled_list.append(b.allocator, .{
|
||||||
// The input service and its exercisers: the fan-out server, a hardware-free synthetic
|
.path = "system/services/discovery",
|
||||||
// source, and a subscriber that doubles as the `input` test's oracle. See docs/input.md.
|
.binary = discovery_exe.getEmittedBin(),
|
||||||
const input_exe = b.dependency("input", .{}).artifact("input");
|
}) catch @panic("OOM");
|
||||||
const logger_exe = b.dependency("logger", .{}).artifact("logger");
|
// The ship table: every uniform row, one line each.
|
||||||
|
for (production_ship) |row| {
|
||||||
// Every user binary and its FHS home on the boot volume. There is no packed
|
bundled_list.append(b.allocator, .{
|
||||||
// ramdisk artifact any more: make-fat-image.py lays each binary out at this
|
.path = row.path,
|
||||||
// path on the image, and the EFI loader walks /system and /test at boot and
|
.binary = b.dependency(row.package, .{}).artifact(row.artifact).getEmittedBin(),
|
||||||
// builds the in-RAM initial_ramdisk table from the trees — the volume's file
|
}) catch @panic("OOM");
|
||||||
// structure is the single source of truth. Entry names (and hence argv[0] and
|
}
|
||||||
// task names) are these paths with a leading slash. Test fixtures mirror their
|
// Data files, not binaries: packing them under /system/configuration rides
|
||||||
// repo home: test/system/services/<name> in the source tree IS the boot path.
|
// the kernel's read-only initrd mount of /system (system/kernel/vfs.zig
|
||||||
// init's boot service list is data (/etc/init.csv). -Ddiagnose selects the
|
// setInitialRamdisk) — the device manager reads its registry and init its
|
||||||
// variant that omits the display stack (so the kernel's boot transcript stays
|
// service list with no filesystem service running. -Ddiagnose selects the
|
||||||
// on screen); both are bundled at the same /etc/init.csv path.
|
// init.csv variant that omits the display stack (so the kernel's boot
|
||||||
const init_csv_source = if (diagnose) "etc/init-diagnose.csv" else "etc/init.csv";
|
// transcript stays on screen); both bundle at the same
|
||||||
const production_bundled = [_]images.BundledBinary{
|
// /system/configuration/init.csv path.
|
||||||
.{ .path = "system/services/init", .binary = init_exe.getEmittedBin() },
|
const init_csv_source = if (diagnose) "system/configuration/init-diagnose.csv" else "system/configuration/init.csv";
|
||||||
.{ .path = "system/services/fat", .binary = fat_exe.getEmittedBin() },
|
bundled_list.append(b.allocator, .{ .path = "system/configuration/devices.csv", .binary = b.path("system/configuration/devices.csv") }) catch @panic("OOM");
|
||||||
.{ .path = "system/services/display", .binary = display_exe.getEmittedBin() },
|
bundled_list.append(b.allocator, .{ .path = "system/configuration/init.csv", .binary = b.path(init_csv_source) }) catch @panic("OOM");
|
||||||
.{ .path = "system/services/display-demo", .binary = display_demo_exe.getEmittedBin() },
|
// The protocol grants: who may claim which name under /protocol
|
||||||
.{ .path = "system/services/device-manager", .binary = device_manager_exe.getEmittedBin() },
|
// (docs/os-development/protocol-namespace.md). init reads it beside init.csv,
|
||||||
.{ .path = "system/services/input", .binary = input_exe.getEmittedBin() },
|
// out of the same read-only initrd, before it spawns anything — the registrar
|
||||||
.{ .path = "system/services/discovery", .binary = discovery_exe.getEmittedBin() },
|
// has to know its policy before the first provider asks.
|
||||||
.{ .path = "system/services/logger", .binary = logger_exe.getEmittedBin() },
|
bundled_list.append(b.allocator, .{ .path = "system/configuration/protocol.csv", .binary = b.path("system/configuration/protocol.csv") }) catch @panic("OOM");
|
||||||
// A data file, not a binary: the device registry the manager reads at boot.
|
|
||||||
// Packing it under /etc makes the kernel auto-mount /etc as a read-only
|
|
||||||
// initrd tree (system/kernel/vfs.zig setInitialRamdisk), so the manager can
|
|
||||||
// fs.open("/etc/devices.csv") with no filesystem service running.
|
|
||||||
.{ .path = "etc/devices.csv", .binary = b.path("etc/devices.csv") },
|
|
||||||
// init's service list, likewise read from the kernel-served initrd /etc.
|
|
||||||
.{ .path = "etc/init.csv", .binary = b.path(init_csv_source) },
|
|
||||||
.{ .path = "system/drivers/ps2-bus", .binary = ps2_bus_exe.getEmittedBin() },
|
|
||||||
.{ .path = "system/drivers/ps2-keyboard", .binary = ps2_keyboard_exe.getEmittedBin() },
|
|
||||||
.{ .path = "system/drivers/ps2-mouse", .binary = ps2_mouse_exe.getEmittedBin() },
|
|
||||||
.{ .path = "system/drivers/usb-xhci-bus", .binary = usb_xhci_bus_exe.getEmittedBin() },
|
|
||||||
.{ .path = "system/drivers/usb-hid-keyboard", .binary = usb_hid_keyboard_exe.getEmittedBin() },
|
|
||||||
.{ .path = "system/drivers/usb-hid-mouse", .binary = usb_hid_mouse_exe.getEmittedBin() },
|
|
||||||
.{ .path = "system/drivers/usb-storage", .binary = usb_storage_exe.getEmittedBin() },
|
|
||||||
.{ .path = "system/drivers/virtio-gpu", .binary = virtio_gpu_exe.getEmittedBin() },
|
|
||||||
.{ .path = "system/drivers/pci-bus", .binary = pci_bus_exe.getEmittedBin() },
|
|
||||||
};
|
|
||||||
// A no-option build assumes neither -Dtest-case nor -Ddiagnose: it ships the
|
// A no-option build assumes neither -Dtest-case nor -Ddiagnose: it ships the
|
||||||
// production set only. The userspace test fixtures under /test join in only
|
// production set only. The userspace test fixtures under /test join in only
|
||||||
// for a test build — which the QEMU harness signals by passing
|
// for a test build — which the QEMU harness signals by passing
|
||||||
// -Dtest-case=<name> for every scenario, exactly when they must be on the
|
// -Dtest-case=<name> for every scenario, exactly when they must be on the
|
||||||
// boot volume. They are LAZY dependencies: a plain build neither compiles
|
// boot volume. They are LAZY dependencies too. Fixture packages are
|
||||||
// them nor loads their build files (docs/build-packages-plan.md). Fixture
|
// uniform — the dependency name, the artifact name, and the boot path's
|
||||||
// packages are uniform — the dependency name, the artifact name, and the
|
// leaf all match the directory — so a name is a whole entry.
|
||||||
// boot path's leaf all match the directory — so a name is a whole entry.
|
|
||||||
var bundled_list: std.ArrayListUnmanaged(images.BundledBinary) = .empty;
|
|
||||||
bundled_list.appendSlice(b.allocator, &production_bundled) catch @panic("OOM");
|
|
||||||
if (test_case != null) for ([_][]const u8{
|
if (test_case != null) for ([_][]const u8{
|
||||||
"vfs-test", // the user-space VFS round-trip client
|
"vfs-test", // the user-space VFS round-trip client
|
||||||
"fat-test",
|
"fat-test",
|
||||||
@@ -336,6 +337,9 @@ pub fn build(b: *std.Build) void {
|
|||||||
"args-echo",
|
"args-echo",
|
||||||
"process-test",
|
"process-test",
|
||||||
"thread-test", // the multi-threaded fixture (its package sets .threaded)
|
"thread-test", // the multi-threaded fixture (its package sets .threaded)
|
||||||
|
"user-memory-test", // aims deliberately bad user pointers at the checked copy layer
|
||||||
|
"protocol-registry-test", // drives the registrar: ungranted bind, collision, restart
|
||||||
|
"protocol-denied-test", // restriction stage one: an ungranted open answers as absence
|
||||||
}) |fixture| {
|
}) |fixture| {
|
||||||
const package = b.lazyDependency(fixture, .{}) orelse
|
const package = b.lazyDependency(fixture, .{}) orelse
|
||||||
@panic("a test fixture package is missing under test/system/services");
|
@panic("a test fixture package is missing under test/system/services");
|
||||||
@@ -419,12 +423,12 @@ pub fn build(b: *std.Build) void {
|
|||||||
protocol_library,
|
protocol_library,
|
||||||
csv_library,
|
csv_library,
|
||||||
xkeyboard_config_library,
|
xkeyboard_config_library,
|
||||||
fat_package,
|
b.dependency("fat", .{}),
|
||||||
display_package,
|
b.dependency("display", .{}),
|
||||||
ps2_bus_package,
|
b.dependency("ps2-bus", .{}),
|
||||||
usb_hid_package,
|
b.dependency("usb-hid", .{}),
|
||||||
usb_storage_package,
|
b.dependency("usb-storage", .{}),
|
||||||
virtio_gpu_package,
|
b.dependency("virtio-gpu", .{}),
|
||||||
}) |package| {
|
}) |package| {
|
||||||
test_step.dependOn(&package.builder.top_level_steps.get("test").?.step);
|
test_step.dependOn(&package.builder.top_level_steps.get("test").?.step);
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -74,6 +74,9 @@
|
|||||||
.@"args-echo" = .{ .path = "test/system/services/args-echo", .lazy = true },
|
.@"args-echo" = .{ .path = "test/system/services/args-echo", .lazy = true },
|
||||||
.@"process-test" = .{ .path = "test/system/services/process-test", .lazy = true },
|
.@"process-test" = .{ .path = "test/system/services/process-test", .lazy = true },
|
||||||
.@"thread-test" = .{ .path = "test/system/services/thread-test", .lazy = true },
|
.@"thread-test" = .{ .path = "test/system/services/thread-test", .lazy = true },
|
||||||
|
.@"user-memory-test" = .{ .path = "test/system/services/user-memory-test", .lazy = true },
|
||||||
|
.@"protocol-registry-test" = .{ .path = "test/system/services/protocol-registry-test", .lazy = true },
|
||||||
|
.@"protocol-denied-test" = .{ .path = "test/system/services/protocol-denied-test", .lazy = true },
|
||||||
// See `zig fetch --save <url>` for a command-line interface for adding dependencies.
|
// See `zig fetch --save <url>` for a command-line interface for adding dependencies.
|
||||||
//.example = .{
|
//.example = .{
|
||||||
// // When updating this field to a new URL, be sure to delete the corresponding
|
// // When updating this field to a new URL, be sure to delete the corresponding
|
||||||
|
|||||||
+1
-1
@@ -83,7 +83,7 @@ pub fn addImageSteps(b: *std.Build, options: Options) std.Build.LazyPath {
|
|||||||
// holding the EFI stub, the kernel, and the whole /system tree of user
|
// holding the EFI stub, the kernel, and the whole /system tree of user
|
||||||
// binaries at their FHS paths. QEMU presents this image as a USB mass-storage
|
// binaries at their FHS paths. QEMU presents this image as a USB mass-storage
|
||||||
// device the guest boots from (see run-x86-64 and the test harness), and the
|
// device the guest boots from (see run-x86-64 and the test harness), and the
|
||||||
// danos fat driver mounts the same image at /mnt/usb.
|
// danos fat driver mounts the same image at /volumes/usb.
|
||||||
const fat_image = addBootImage(b, options.kernel.getEmittedBin(), options.efi.getEmittedBin(), manifest_file, capsule_img, options.bundled);
|
const fat_image = addBootImage(b, options.kernel.getEmittedBin(), options.efi.getEmittedBin(), manifest_file, capsule_img, options.bundled);
|
||||||
const fat_image_install = b.addInstallFile(fat_image, "danos-usb.img");
|
const fat_image_install = b.addInstallFile(fat_image, "danos-usb.img");
|
||||||
b.getInstallStep().dependOn(&fat_image_install.step);
|
b.getInstallStep().dependOn(&fat_image_install.step);
|
||||||
|
|||||||
+2
-2
@@ -42,8 +42,8 @@ pub fn addRunSteps(b: *std.Build, fat_image_serial: std.Build.LazyPath) void {
|
|||||||
const vars_out = vars_copy.addOutputFileArg("OVMF_VARS.4m.fd");
|
const vars_out = vars_copy.addOutputFileArg("OVMF_VARS.4m.fd");
|
||||||
|
|
||||||
// Capture the guest's serial0 (danos's machine-readable log) to the qemu-test
|
// Capture the guest's serial0 (danos's machine-readable log) to the qemu-test
|
||||||
// scratch area — a dev/host artifact, kept out of the FHS boot volume we mount.
|
// scratch area — a dev/host artifact, kept out of the boot volume we mount.
|
||||||
// (/var/log/system is reserved for the kernel's own logging system later.) One
|
// (/system/logs on the volume belongs to the guest's own logger.) One
|
||||||
// timestamped file per run.
|
// timestamped file per run.
|
||||||
const log_dir = b.fmt("{s}/qemu-test", .{b.install_path});
|
const log_dir = b.fmt("{s}/qemu-test", .{b.install_path});
|
||||||
const make_log_dir = b.addSystemCommand(&.{ "mkdir", "-p", log_dir });
|
const make_log_dir = b.addSystemCommand(&.{ "mkdir", "-p", log_dir });
|
||||||
|
|||||||
+2
-2
@@ -208,7 +208,7 @@ the whole reason for the arrangement ([vision.md](vision.md)).
|
|||||||
danos is a **monorepo of sub-projects**. Each service or driver is a directory that is
|
danos is a **monorepo of sub-projects**. Each service or driver is a directory that is
|
||||||
its own Zig module — it can hold as many files as it needs, and other sub-projects
|
its own Zig module — it can hold as many files as it needs, and other sub-projects
|
||||||
reach it *by module name*, never by a path into its files. The source tree deliberately
|
reach it *by module name*, never by a path into its files. The source tree deliberately
|
||||||
**mirrors the runtime FHS** ([danos-file-system-hierarchy-FSH.md](file-system-development/danos-file-system-hierarchy-FSH.md)):
|
**mirrors the runtime file-system hierarchy** ([file-system-hierarchy.md](file-system-development/file-system-hierarchy.md)):
|
||||||
what you see under `system/` in the source is what a running danos represents under
|
what you see under `system/` in the source is what a running danos represents under
|
||||||
`/system`.
|
`/system`.
|
||||||
|
|
||||||
@@ -216,7 +216,7 @@ what you see under `system/` in the source is what a running danos represents un
|
|||||||
name.** `system/services/init/` contains `init.zig` (its root), and produces a binary
|
name.** `system/services/init/` contains `init.zig` (its root), and produces a binary
|
||||||
addressed as **`system/services/init`** — the repeated leaf resolves away:
|
addressed as **`system/services/init`** — the repeated leaf resolves away:
|
||||||
|
|
||||||
| Source (root file) | Addressed as (module / binary / FHS path) |
|
| Source (root file) | Addressed as (module / binary / hierarchy path) |
|
||||||
|----------------------------------------|--------------------------------------------|
|
|----------------------------------------|--------------------------------------------|
|
||||||
| `system/services/init/init.zig` | `system/services/init` → `/system/services/init` |
|
| `system/services/init/init.zig` | `system/services/init` → `/system/services/init` |
|
||||||
| `system/drivers/ps2-bus/ps2-bus.zig` | `system/drivers/ps2-bus` → `/system/drivers/ps2-bus` |
|
| `system/drivers/ps2-bus/ps2-bus.zig` | `system/drivers/ps2-bus` → `/system/drivers/ps2-bus` |
|
||||||
|
|||||||
@@ -62,8 +62,10 @@ Rules:
|
|||||||
include list — and its zon names only the domains those modules come from
|
include list — and its zon names only the domains those modules come from
|
||||||
(plus `build-support` and `kernel`, which is implicit in every binary: the
|
(plus `build-support` and `kernel`, which is implicit in every binary: the
|
||||||
root shim and user link script live there). Nothing is pre-wired: an
|
root shim and user link script live there). Nothing is pre-wired: an
|
||||||
undeclared `@import` is a compile error, and build-support's one
|
undeclared `@import` is a compile error, and build-support resolves each
|
||||||
module-to-domain table (`module_homes`) resolves each name. Availability
|
name by searching the packages the zon declares — the domains' own
|
||||||
|
addModule exports are the single statement of who owns what, with no name
|
||||||
|
table anywhere to drift. Availability
|
||||||
never meant bloat — Zig only compiles what a program actually imports — but
|
never meant bloat — Zig only compiles what a program actually imports — but
|
||||||
exactness makes the declared interface honest and machine-checked.
|
exactness makes the declared interface honest and machine-checked.
|
||||||
- **Modules export source, not artifacts** — each consumer compiles libraries
|
- **Modules export source, not artifacts** — each consumer compiles libraries
|
||||||
@@ -132,9 +134,9 @@ rewritten against the package template.
|
|||||||
## Execution notes (the finished shape)
|
## Execution notes (the finished shape)
|
||||||
|
|
||||||
- The shared recipe lives in `build-support/build.zig`: `userBinary` (what
|
- The shared recipe lives in `build-support/build.zig`: `userBinary` (what
|
||||||
every binary package calls, resolving each named import through the
|
every binary package calls; each named import resolves by searching the
|
||||||
`module_homes` table) and `programModule` (for per-binary addOptions
|
packages the binary's zon declares) and `programModule` (for per-binary
|
||||||
modules). The `start` root shim and `user.ld` are named through the kernel
|
addOptions modules). The `start` root shim and `user.ld` are named through the kernel
|
||||||
package (Dependency.path).
|
package (Dependency.path).
|
||||||
- Adding a binary = adding a directory with source + a ~15-line build.zig +
|
- Adding a binary = adding a directory with source + a ~15-line build.zig +
|
||||||
zon (copy any existing binary package, e.g.
|
zon (copy any existing binary package, e.g.
|
||||||
|
|||||||
@@ -0,0 +1,205 @@
|
|||||||
|
# The C library compatibility layer
|
||||||
|
|
||||||
|
A design note and milestone plan for **libdanos-c** — the mini C library that lets
|
||||||
|
`zig cc` cross-compile C programs for danos. It is milestone **P0** of
|
||||||
|
[python-on-danos-milestones.md](python-on-danos-milestones.md), expanded here the
|
||||||
|
way [character-devices-and-tty.md](character-devices-and-tty.md) expands P1.
|
||||||
|
CPython is the driving consumer, but the layer is general: any portable C program
|
||||||
|
within its surface should build.
|
||||||
|
|
||||||
|
## What it is — and the three things it is not
|
||||||
|
|
||||||
|
The deliverable is a **sysroot**: a set of C headers plus a static `libdanos-c.a`,
|
||||||
|
handed to `zig cc -target x86_64-freestanding-none` via `-isystem` and linked into
|
||||||
|
every C binary. Three explicit non-goals keep it small:
|
||||||
|
|
||||||
|
- **Not a musl port.** Whole-musl assumes Linux syscall semantics at its bottom
|
||||||
|
(the door the Zig roadmap deferred, twice now). We *lift* musl's pure-computation
|
||||||
|
source files and *write* a danos-native bottom — see the layer split below.
|
||||||
|
- **Not full POSIX — *yet*.** Stage 1's surface is "what CPython's minimal
|
||||||
|
configuration and ordinary portable C need" — roughly 100–150 functions — and
|
||||||
|
at that stage absence is a *feature*: configure scripts probe and adapt, and a
|
||||||
|
linker error is honest. But the end state is a **full C compatibility layer**
|
||||||
|
(see "The road to full coverage" below); the absence table is a schedule of
|
||||||
|
arrivals, not a wall.
|
||||||
|
- **Not a second runtime.** The library is a thin C-ABI re-spelling of the same
|
||||||
|
danos-native surface `runtime` already provides. It contains no policy of its
|
||||||
|
own; when the Zig track's `runtime.os` seam is authored, the libc bottom
|
||||||
|
re-targets it near-mechanically — the fourth appearance of the roadmap's "same
|
||||||
|
surface" symmetry.
|
||||||
|
|
||||||
|
One scoping rule sits above all three — the **size doctrine**: this layer serves
|
||||||
|
**applications only**. The kernel and the system services never link libdanos-c;
|
||||||
|
they stay danos-native Zig over `runtime`, small and static, because leanness is
|
||||||
|
an operating-system property. Applications have their own budget and may be as
|
||||||
|
big as they need to be. The libc is how big software *lands on* danos, never how
|
||||||
|
danos itself is built.
|
||||||
|
|
||||||
|
## The layer split: lift the mathematics, write the plumbing
|
||||||
|
|
||||||
|
The realization that makes 100–150 functions tractable: a libc is two very
|
||||||
|
different kinds of code, and the hard kind is portable.
|
||||||
|
|
||||||
|
| Layer | Contents | Source |
|
||||||
|
|-------|----------|--------|
|
||||||
|
| **Pure computation** | `string.h`/`memcpy` family, all of libm, `strtod`/`dtoa`, `strtol`, `qsort`, `ctype` tables, `gmtime` calendar math, the `printf`/`scanf` engines, `setjmp` (a dozen instructions of x86-64 asm) | **Lift from musl**, vendored under `library/c/third-party/musl/` (MIT; files compile standalone) |
|
||||||
|
| **OS plumbing** | fds (`open`/`read`/`write`/`close`/`lseek`/`stat`/`getcwd`/`chdir`/`isatty`), `mmap`/`munmap`, clocks, `exit`, `getenv`, `getentropy` | **Write in Zig**, exporting C ABI over the `runtime` syscall + VFS client surface |
|
||||||
|
| **The middle** | `malloc` over danos `mmap` (simple free-list; CPython's arenas sit above), `FILE*` buffering, `errno` | **Write in Zig** (small, danos-shaped) |
|
||||||
|
| **Entry** | `crt0`: the existing danos entry shim ([sysv.md](os-development/sysv.md)) bridged to C `main(argc, argv, envp)`, `environ` initialised, `exit` flushing stdio | **Write** |
|
||||||
|
|
||||||
|
Two liftings deserve their own line because getting them wrong is silent
|
||||||
|
corruption rather than a linker error:
|
||||||
|
|
||||||
|
- **`strtod`/float formatting.** Python's float `repr` guarantees shortest
|
||||||
|
round-trip; that property lives entirely in these routines. musl's are correct;
|
||||||
|
an improvised one would be subtly wrong for years. Lift, never write.
|
||||||
|
- **The stdio engines.** musl's `vfprintf`/`vfscanf` are self-contained around
|
||||||
|
its `FILE` abstraction (function-pointer read/write slots), so the whole
|
||||||
|
formatted-I/O engine lifts too — we implement only the fd-backed slots
|
||||||
|
(`__stdio_write`-shaped) and the buffering glue.
|
||||||
|
|
||||||
|
## Header policy
|
||||||
|
|
||||||
|
Hand-write the headers as danos's own minimal set rather than importing musl's
|
||||||
|
(musl's are entangled with Linux ABI details), borrowing declarations freely.
|
||||||
|
Freestanding compiler headers (`stdint.h`, `stddef.h`, `stdarg.h`, `stdbool.h`,
|
||||||
|
`float.h`, `limits.h`) come from clang via `zig cc` — do not duplicate them.
|
||||||
|
`errno.h` values are the danos errno enum re-spelled with POSIX names; there is no
|
||||||
|
Linux numbering to be compatible with, so the enum is the truth.
|
||||||
|
|
||||||
|
Deliberate absences, and their planned arrivals — this table is the
|
||||||
|
compatibility matrix, and "the road to full coverage" below is the schedule
|
||||||
|
that empties it:
|
||||||
|
|
||||||
|
| Absent | Arrives with |
|
||||||
|
|--------|--------------|
|
||||||
|
| `pthread.h` | the post-P5 pthread subset over `thread_spawn`/futex — but see the risk below |
|
||||||
|
| real `signal.h` (beyond no-op `signal()`/`raise` stubs) | M17 signals-over-IPC in the libc |
|
||||||
|
| `dlfcn.h` | [dynamic-libraries.md](dynamic-libraries.md) D1 |
|
||||||
|
| `fork`/`exec*`/`wait*` | P5 exposes danos spawn as `posix_spawn`; `fork` itself never (see below) |
|
||||||
|
| `socket.h` | a future networking track |
|
||||||
|
| locale beyond `"C"` | stage 3 evaluation (CPython is UTF-8-mode happy without it) |
|
||||||
|
| pipes (`pipe()`) | P5 process-control cluster |
|
||||||
|
|
||||||
|
## The road to full coverage
|
||||||
|
|
||||||
|
The layer grows in three stages; only stage 1 is a current milestone (P0), but
|
||||||
|
the stages exist so stage-1 decisions never have to be unmade:
|
||||||
|
|
||||||
|
- **Stage 1 — CPython-minimal** (P0, the slicing below): ~100–150 functions,
|
||||||
|
static-only, absences honest.
|
||||||
|
- **Stage 2 — the danos-complete layer**: the full hosted C11 standard library,
|
||||||
|
plus every POSIX facility danos semantics support, landing as its enabling
|
||||||
|
milestone lands — pipes and `posix_spawn` at P5, real signals at M17, the
|
||||||
|
pthread subset after P5, `dlfcn.h` at
|
||||||
|
[dynamic-libraries](dynamic-libraries.md) D1, sockets with networking. Stage 2
|
||||||
|
is not one milestone but the standing rule that **every system capability
|
||||||
|
gets its C spelling when it ships**, so the matrix above drains as the OS
|
||||||
|
grows.
|
||||||
|
- **Stage 3 — ecosystem grade**: the point where "portable C program" generally
|
||||||
|
means "builds on danos" (autotools-style probing included). Reaching it is
|
||||||
|
mostly stage 2 compounding, plus the long tail (locale, wide-char,
|
||||||
|
`fnmatch`/`glob`/`regex` — the last three lift from musl like the rest). At
|
||||||
|
this stage, re-evaluate hand-grown-vs-musl-port once with real data; the
|
||||||
|
standing recommendation remains danos-native — musl's bottom assumes Linux
|
||||||
|
syscall semantics, and by stage 3 the danos bottom exists and is tested —
|
||||||
|
with musl continuing as the quarry for computation code.
|
||||||
|
|
||||||
|
Two boundaries are permanent and worth stating at every stage: **`fork` never
|
||||||
|
comes** — danos is a spawn-shaped OS, and `fork`'s address-space-duplication
|
||||||
|
semantics are hostile to everything from capabilities to threads; software that
|
||||||
|
hard-requires `fork` (not `posix_spawn`) stays off the platform. And the
|
||||||
|
**public ABI stays the vDSO + IPC protocols** — a full libc is a compatibility
|
||||||
|
*layer*, not a second stable system ABI.
|
||||||
|
|
||||||
|
## Milestone slicing
|
||||||
|
|
||||||
|
1. **sysroot-skeleton** — layout under `library/c/` (a build package:
|
||||||
|
`include/`, Zig sources, vendored musl subtree); `crt0`; string/mem +
|
||||||
|
`ctype` lifted; a `build.zig` step making C binaries first-class targets.
|
||||||
|
*Test:* a C program using only computation links and runs in QEMU
|
||||||
|
(`c-hello` printing via a raw `write` extern to `debug_write`).
|
||||||
|
2. **fd-plumbing** — `errno`; open/read/write/close/lseek/stat/unlink/mkdir/
|
||||||
|
rename over the `runtime` VFS client; `getcwd`/`chdir`/`getenv`/
|
||||||
|
`getentropy` arriving as P1 lands them (stubbed truthfully until then:
|
||||||
|
`getenv` empty, `getentropy` `ENOSYS`). *Test:* QEMU `c-file-io` — create,
|
||||||
|
write, reopen, read back, stat size + mtime through FAT.
|
||||||
|
3. **malloc** — free-list allocator over danos `mmap`; `calloc`/`realloc`/
|
||||||
|
`free`; alignment guarantees documented. *Test:* host + QEMU allocator
|
||||||
|
torture (interleaved sizes, realloc growth, alignment asserts).
|
||||||
|
4. **stdio** — `FILE*`, buffering modes, the lifted printf/scanf engines wired
|
||||||
|
to the fd slots; `snprintf` family; stdin/stdout/stderr over fd 0/1/2.
|
||||||
|
*Test:* host round-trip suite for format engines (especially `%.17g`
|
||||||
|
float round-trip); QEMU `c-stdio` cooked-line echo once P1's console exists.
|
||||||
|
5. **mathematics-and-time** — libm lifted wholesale; `strtod`/`strtol`;
|
||||||
|
`clock_gettime` (monotonic + realtime over `clock`/`wall_clock`);
|
||||||
|
`gmtime`/`mktime`/`strftime` (UTC only — no timezone database);
|
||||||
|
`setjmp`/`longjmp`; `qsort`/`bsearch`; `abort`/`assert`. *Test:* host
|
||||||
|
`strtod`/`dtoa` vectors against known-hard cases; QEMU `c-time` sanity
|
||||||
|
against the wall clock.
|
||||||
|
|
||||||
|
Slices 1, 3, 4-host, and 5-host have **no dependency on P1** and can start
|
||||||
|
immediately; slice 2 and the QEMU halves interleave with P1 as it lands.
|
||||||
|
|
||||||
|
**Exit for the layer as a whole** (= P0's exit): `c-hello` and `c-file-io` green
|
||||||
|
in the QEMU suite, and the host-side computation tests green — at which point P2
|
||||||
|
(CPython configure) becomes the layer's real integration test.
|
||||||
|
|
||||||
|
## Testing strategy: two targets, on purpose
|
||||||
|
|
||||||
|
The computation layer is target-independent, so it is unit-tested **on the host**
|
||||||
|
(built for the host triple, compared against the host libc's answers —
|
||||||
|
thousands of cheap oracle checks for `strtod`, `printf`, libm edge cases). The
|
||||||
|
plumbing layer only means anything **on danos**, so it is tested in the QEMU
|
||||||
|
suite like every other subsystem. Keeping the split explicit stops the slow-QEMU
|
||||||
|
suite from absorbing tests that a host `zig test` runs in milliseconds.
|
||||||
|
|
||||||
|
## Risks and gotchas
|
||||||
|
|
||||||
|
- **CPython's configure may insist on pthreads.** WASI-class targets build
|
||||||
|
threadless, but verify this *first* in P2 bring-up; the fallback is a
|
||||||
|
truthfully-single-threaded `pthread.h` stub set (create returns `EAGAIN`,
|
||||||
|
mutexes are no-ops — valid when only one thread can exist). Decide from
|
||||||
|
evidence, not assumption.
|
||||||
|
- **`long double` is x87 80-bit on x86-64.** musl's libm handles it, but keep
|
||||||
|
CPython away from it (`configure` uses `double` throughout by default);
|
||||||
|
don't hand-write anything touching x87.
|
||||||
|
- **errno is a contract, not a convention.** The Zig plumbing must map every
|
||||||
|
`runtime` error to a POSIX name consistently — CPython turns errno into
|
||||||
|
exception types (`FileNotFoundError` is `ENOENT`). One table, tested.
|
||||||
|
- **`malloc` alignment**: 16-byte minimum on x86-64 (SSE spills in
|
||||||
|
compiled C). The free-list must guarantee it from day one; retrofitting
|
||||||
|
alignment bugs out of an allocator is misery.
|
||||||
|
- **Vendoring discipline.** The musl subtree is lift-only — never edited in
|
||||||
|
place (patches live beside it if ever needed), pinned to one musl release,
|
||||||
|
with the file list documented so a version bump is a re-copy, not an
|
||||||
|
archaeology dig.
|
||||||
|
- **stdio buffering vs. crashes.** Buffered stdout + a crashing program eats
|
||||||
|
output — the classic debugging trap. `stderr` stays unbuffered (per C
|
||||||
|
standard) and `exit`/`abort` flush; document that `_exit` does not.
|
||||||
|
|
||||||
|
## Decisions needing sign-off
|
||||||
|
|
||||||
|
- **Lift-from-musl for all pure computation** (vendored, pinned, unedited) rather
|
||||||
|
than writing or porting whole-musl.
|
||||||
|
- **Hand-written danos-native headers**; danos errno values are the numbering.
|
||||||
|
- **`library/c/` as a build package** producing both the sysroot and the
|
||||||
|
first-class C-binary build step.
|
||||||
|
- The **deliberate-absence table** as the living compatibility matrix, drained
|
||||||
|
by the three-stage road above — with exactly one permanent "never": `fork`.
|
||||||
|
- **Full coverage as the end state** (stage 3), reached by the standing rule
|
||||||
|
that every system capability ships with its C spelling — not by a musl port.
|
||||||
|
|
||||||
|
## Related
|
||||||
|
|
||||||
|
- [python-on-danos-milestones.md](python-on-danos-milestones.md) — this is P0.
|
||||||
|
- [dynamic-libraries.md](dynamic-libraries.md) — ships in this sysroot
|
||||||
|
(`dlfcn.h` + the loader) once its D1 lands.
|
||||||
|
- [python-on-danos.md](python-on-danos.md) — the design note that scoped the
|
||||||
|
layer.
|
||||||
|
- [character-devices-and-tty.md](character-devices-and-tty.md) — P1; supplies
|
||||||
|
the console that makes stdio interactive.
|
||||||
|
- [zig-self-hosting.md](zig-self-hosting.md) — the `runtime.os` seam the
|
||||||
|
plumbing layer will re-target when it exists.
|
||||||
|
- [os-development/sysv.md](os-development/sysv.md) — the entry stack `crt0`
|
||||||
|
bridges.
|
||||||
@@ -0,0 +1,174 @@
|
|||||||
|
# Character devices, the console, and the tty question
|
||||||
|
|
||||||
|
A design note for the **stream** half of the device world. danos has block devices
|
||||||
|
(the USB storage service behind the FAT mount) but no character devices — and three
|
||||||
|
tracks now need them at once: the terminal application, Zig self-hosting Phase 1
|
||||||
|
("wire fd 0/1/2 to a console byte stream"), and [Python on danos](python-on-danos.md)
|
||||||
|
Phase 1. This note settles what a character device *is* on danos before any of those
|
||||||
|
tracks build one.
|
||||||
|
|
||||||
|
## The Unix picture, briefly
|
||||||
|
|
||||||
|
Unix splits devices in two: **block devices** are seekable arrays of fixed-size
|
||||||
|
sectors (disks); **character devices** are unseekable byte streams (keyboards,
|
||||||
|
serial ports, terminals, `/dev/null`, entropy). A **tty** is the canonical
|
||||||
|
character device — a byte stream plus a *line discipline* (echo, line buffering,
|
||||||
|
erase handling, Ctrl-C-to-signal) that lives in the kernel. A **pty** is a pair of
|
||||||
|
character devices (master/slave) that exists so a *userspace* program — a terminal
|
||||||
|
emulator — can impersonate terminal hardware to the kernel's in-kernel line
|
||||||
|
discipline.
|
||||||
|
|
||||||
|
The identification asked for and confirmed: yes, tty and pty are character
|
||||||
|
devices in this taxonomy.
|
||||||
|
|
||||||
|
## The realization that shapes everything: danos already has the mechanism
|
||||||
|
|
||||||
|
A Unix character device is an in-kernel dispatch table: major/minor numbers route
|
||||||
|
`read()`/`write()` to a driver. danos already has exactly that dispatch — the VFS:
|
||||||
|
`fs_resolve` routes a path to a mounted backend service, and `Operation.mount`
|
||||||
|
attaches a backend *endpoint* at a prefix. What is missing is not a device model;
|
||||||
|
it is **one node kind with stream semantics**. And the protocol already reserved
|
||||||
|
it: `NodeKind.character_device = 2` sits unimplemented in
|
||||||
|
[vfs-protocol.zig](../library/protocol/vfs/vfs-protocol.zig), exactly like
|
||||||
|
`symbolic_link`.
|
||||||
|
|
||||||
|
So the design is small:
|
||||||
|
|
||||||
|
**A character device on danos is a VFS node, served by an ordinary service over
|
||||||
|
the existing VFS wire protocol, whose read/write have stream semantics.**
|
||||||
|
|
||||||
|
No device numbers, no `/dev` special casing, no new syscalls, no new protocol —
|
||||||
|
a service is reachable at a path, clients open it with `runtime.fs` like any
|
||||||
|
file, and the node kind says what it is. (Since the protocol namespace landed
|
||||||
|
in design, that path is `/protocol/console` — a protocol node, see
|
||||||
|
[os-development/protocol-namespace.md](os-development/protocol-namespace.md) —
|
||||||
|
rather than a mounted device file; the stream semantics below are unchanged.)
|
||||||
|
|
||||||
|
### Stream semantics (the actual contract change)
|
||||||
|
|
||||||
|
For a node whose kind is `character_device`:
|
||||||
|
|
||||||
|
- **`offset` is ignored** on read and write; there is no seek position. (`lseek`,
|
||||||
|
when the C layer exists, returns `ESPIPE`.)
|
||||||
|
- **Reads block** until at least one byte is available, then return what is there —
|
||||||
|
**short reads are normal**, not EOF. A zero-length read reply means the stream
|
||||||
|
is closed (hangup), not end-of-file-at-size.
|
||||||
|
- **`FileStatus.size` is 0** and means nothing; `mtime` may be 0.
|
||||||
|
- Writes may be short if the service's buffer is full; the client loops as it
|
||||||
|
already must for the 256-byte message cap.
|
||||||
|
|
||||||
|
This is a semantics note on existing operations, not a wire change — the `Request`
|
||||||
|
and `Reply` structs are untouched. The one true protocol addition is a **`control`
|
||||||
|
operation** (appended to `Operation`, values stable): a typed request the stream's
|
||||||
|
service interprets. Deliberately *not* an `ioctl` grab-bag — the control payloads
|
||||||
|
are enumerated per protocol, starting with the terminal set below.
|
||||||
|
|
||||||
|
## The first character device is a pseudo-device
|
||||||
|
|
||||||
|
The first device is deliberately **not hardware**: an in-memory **loopback** — a
|
||||||
|
byte queue served over the stream contract, where bytes written to one end are
|
||||||
|
read from the other. It is the reference implementation of the semantics above
|
||||||
|
(blocking reads, short reads, hangup on close, the `control` round-trip), it
|
||||||
|
tests deterministically with no QEMU serial scripting, and it keeps hardware off
|
||||||
|
the critical path entirely. `null` and `zero` come along nearly for free as
|
||||||
|
degenerate cases. This is a decision, not a convenience: the dead-COM1 boot bug
|
||||||
|
on real hardware already proved serial cannot be assumed present or alive, so
|
||||||
|
**nothing in this milestone writes to COM1**. (A serial-backed stream node can
|
||||||
|
exist *later* as one more optional backend for headless debugging; it is on
|
||||||
|
nobody's critical path.)
|
||||||
|
|
||||||
|
The loopback is also not throwaway — it is the seed of P5's `pipe()`, which is
|
||||||
|
the same object with two fds.
|
||||||
|
|
||||||
|
## The console service
|
||||||
|
|
||||||
|
A `console` service owns the line discipline — **in userspace**, where a
|
||||||
|
microkernel wants it, not in the kernel as Unix has it:
|
||||||
|
|
||||||
|
- **The discipline is a pure library first**: bytes and key events in, bytes
|
||||||
|
out, no I/O of its own — developed and host-tested against in-memory buffers,
|
||||||
|
then shared verbatim between the console and the future terminal application.
|
||||||
|
- **Input**: subscribes to keyboard `InputEvent` IPC (the structured events that
|
||||||
|
exist today) and cooks them into bytes. Cooked mode is the default: echo, line
|
||||||
|
buffering, backspace/erase, so a line is delivered on Enter. Raw mode delivers
|
||||||
|
bytes as they come (the REPL's line editor and any full-screen program need it).
|
||||||
|
- **Output is a pluggable sink**, and the stream contract is independent of it:
|
||||||
|
the bring-up sink is in-memory (readable back by tests, mirrored to the boot
|
||||||
|
log), and the real one is the framebuffer text renderer when the display
|
||||||
|
track's font work lands.
|
||||||
|
- **Control set** (the `control` payloads): mode raw/cooked, echo on/off, and
|
||||||
|
window-size query — the minimal termios. Ctrl-C-to-signal joins when M17
|
||||||
|
signals-over-IPC lands; until then Ctrl-C is just a byte.
|
||||||
|
- Mounts itself at `/device/console` as a `character_device` node.
|
||||||
|
|
||||||
|
**fd 0/1/2** then stop being special: spawn hands the child three open handles
|
||||||
|
(console by default; anything else if the parent chooses), and `runtime`'s fd
|
||||||
|
table maps 0/1/2 to them. `isatty` is simply "does `status` say
|
||||||
|
`character_device`" — no side channel needed.
|
||||||
|
|
||||||
|
## The pty answer: there is no pty
|
||||||
|
|
||||||
|
The pty exists in Unix *because the line discipline is in the kernel* — userspace
|
||||||
|
terminal emulators need a kernel gadget to impersonate hardware. On danos the
|
||||||
|
terminal emulator is already a userspace server, so the pair collapses:
|
||||||
|
|
||||||
|
**The graphical terminal application serves the VFS stream protocol itself and
|
||||||
|
hands its own endpoints to the children it spawns as their fd 0/1/2.**
|
||||||
|
|
||||||
|
The terminal *is* the console service for its children — same protocol, same
|
||||||
|
control set, same line discipline code (shared as a library with the boot
|
||||||
|
console). No master/slave device pair, no `/dev/pts`, no new kernel object. When
|
||||||
|
CPython arrives, the libc's `isatty`/read/write see a character device and are
|
||||||
|
none the wiser; when xonsh eventually wants job control, that lands as control
|
||||||
|
messages + M17 signals, still with no pty object.
|
||||||
|
|
||||||
|
What this costs: programs that *specifically* manipulate Unix ptys
|
||||||
|
(`os.openpty()`, `pexpect`-style tools) have no direct equivalent — the danos
|
||||||
|
answer is "spawn the child yourself with your own stream endpoints," which is the
|
||||||
|
same capability with less machinery. Accepted.
|
||||||
|
|
||||||
|
## Milestone slicing
|
||||||
|
|
||||||
|
1. **pseudo-devices** — VFS honors `character_device` semantics end to end;
|
||||||
|
`Operation.control` added; the in-memory **loopback** (plus `null`/`zero`)
|
||||||
|
as the first device. QEMU test: one client writes, another reads — open,
|
||||||
|
offsetless read/write, blocking read, short read, hangup on close, control
|
||||||
|
round-trip. No hardware anywhere.
|
||||||
|
2. **console-service** — the line-discipline library (host-tested, pure) plus
|
||||||
|
the console composing keyboard `InputEvent`s with an in-memory output sink;
|
||||||
|
mounted at `/device/console`. QEMU test injects key events and reads cooked
|
||||||
|
lines and raw bytes back through the sink.
|
||||||
|
3. **fd-inheritance** — spawn passes 0/1/2 handles; `runtime` fd table; `isatty`
|
||||||
|
via `status`; existing binaries' stdout migrates from `debug_write` to fd 1
|
||||||
|
(the logger keeps its own path).
|
||||||
|
4. **terminal-as-server** — deferred to the terminal application milestone
|
||||||
|
(Python track P3): the terminal reuses the discipline library and serves its
|
||||||
|
children directly.
|
||||||
|
|
||||||
|
Steps 1–3 are exactly the shared seam that Zig self-hosting Phase 1 and Python
|
||||||
|
Phase 1 both list; neither track repeats them.
|
||||||
|
|
||||||
|
## Decisions needing sign-off
|
||||||
|
|
||||||
|
- **No pty object; the terminal serves its children directly** (the section
|
||||||
|
above) — the load-bearing simplification.
|
||||||
|
- **`control` as an enumerated, typed operation** rather than an ioctl-style
|
||||||
|
opaque pass-through.
|
||||||
|
- **Line discipline in userspace services** (console + terminal, shared library),
|
||||||
|
never in the kernel.
|
||||||
|
|
||||||
|
## Related
|
||||||
|
|
||||||
|
- [python-on-danos.md](python-on-danos.md) — consumes this as its Phase 1.
|
||||||
|
- [zig-self-hosting.md](zig-self-hosting.md) — ditto ("stdio as fds").
|
||||||
|
- [file-system-development/vfs-protocol.md](file-system-development/vfs-protocol.md) —
|
||||||
|
the wire protocol this note extends.
|
||||||
|
- [file-system-development/file-system-hierarchy.md](file-system-development/file-system-hierarchy.md)
|
||||||
|
— the tree the console surfaces in.
|
||||||
|
- [os-development/protocol-namespace.md](os-development/protocol-namespace.md) —
|
||||||
|
supersedes this note's device-node naming: the console lands as a protocol
|
||||||
|
(`/protocol/console`, a protocol node), not a `/dev`-style device file. The
|
||||||
|
stream semantics designed here (line discipline, cooked/raw modes) carry over
|
||||||
|
unchanged.
|
||||||
|
- [device-driver-development/input.md](device-driver-development/input.md) — the
|
||||||
|
`InputEvent` stream the console cooks.
|
||||||
@@ -87,13 +87,13 @@ rest of the system hasn't had to face:
|
|||||||
│ (ResourceKind.memory = [base, height*pitch], write-combining hint,
|
│ (ResourceKind.memory = [base, height*pitch], write-combining hint,
|
||||||
│ plus DisplayInfo{width, height, pitch, format, refresh_hz})
|
│ plus DisplayInfo{width, height, pitch, format, refresh_hz})
|
||||||
▼
|
▼
|
||||||
display service (system/services/display/, ServiceId.display) ← the compositor
|
display service (system/services/display/, /protocol/display) ← the compositor
|
||||||
│ device.claim(display node) → mmio_map(WRITE-COMBINING) = FRONT buffer (the LFB)
|
│ device.claim(display node) → mmio_map(WRITE-COMBINING) = FRONT buffer (the LFB)
|
||||||
│ mmap(cacheable) a BACK buffer of the same geometry
|
│ mmap(cacheable) a BACK buffer of the same geometry
|
||||||
│ owns: an ordered LAYER STACK + a per-frame DAMAGE tracker (rect list or tile grid)
|
│ owns: an ordered LAYER STACK + a per-frame DAMAGE tracker (rect list or tile grid)
|
||||||
│ loop: composite dirty layers → back buffer → present dirty rects → front
|
│ loop: composite dirty layers → back buffer → present dirty rects → front
|
||||||
│ backend is an INTERNAL interface: {gop-fb} at boot; {virtio-gpu} on hot-attach (v2)
|
│ backend is an INTERNAL interface: {gop-fb} at boot; {virtio-gpu} on hot-attach (v2)
|
||||||
▼ reached by name (ipc_lookup); clients drive it over the display protocol
|
▼ reached by name (open /protocol/display); clients drive it over the display protocol
|
||||||
┌────────────────────────────────────┬──────────────────────────────────────┐
|
┌────────────────────────────────────┬──────────────────────────────────────┐
|
||||||
drawing clients (v1) surface clients (deferred)
|
drawing clients (v1) surface clients (deferred)
|
||||||
display commands: display surfaces:
|
display commands: display surfaces:
|
||||||
|
|||||||
@@ -1,34 +1,70 @@
|
|||||||
# IPC: message-passing channels
|
# IPC: the kernel-ipc transport
|
||||||
|
|
||||||
Inter-process communication is the **backbone of a microkernel**. Once drivers and
|
Inter-process communication is the **backbone of a microkernel**. Once drivers and
|
||||||
services run isolated in their own address spaces ([vision](../vision.md)), they can't
|
services run isolated in their own address spaces ([vision](../vision.md)), they can't
|
||||||
just call each other — a request becomes a **message**. In a microkernel, whatever
|
just call each other — a request becomes bytes on a wire. In a microkernel, whatever
|
||||||
was a function call across a monolithic kernel is IPC, so it's a first-class
|
was a function call across a monolithic kernel is IPC, so it's a first-class
|
||||||
concern, not an afterthought.
|
concern, not an afterthought.
|
||||||
|
|
||||||
There are two layers, built a milestone apart:
|
This document describes **one transport** — the bottom layer (L0) of the
|
||||||
|
communication stack defined in
|
||||||
|
[communication.md](../os-development/communication.md), which owns the model
|
||||||
|
and the vocabulary (*protocol*, *channel*, *packet*, *signal*, *endpoint*).
|
||||||
|
kernel-ipc is the **first** transport, not the only possible one: in
|
||||||
|
buffer-plus-doorbell terms it is a kernel-owned mailbox with the scheduler as
|
||||||
|
the doorbell. Its distinguishing properties, which the layers above may rely
|
||||||
|
on where they say so:
|
||||||
|
|
||||||
- **`system/kernel/ipc.zig`** — a bounded blocking channel between *kernel threads*,
|
- **Rendezvous.** A call is a synchronous meeting, copied sender-page to
|
||||||
described below. The primitive, and where the blocking discipline was worked out.
|
receiver-page — natural backpressure, no queue to size.
|
||||||
- **`system/kernel/ipc-synchronous.zig`** — synchronous call/reply between *processes*, across
|
- **Capability carriage.** The *only* transport that can move a handle
|
||||||
address spaces. What user-space servers and drivers actually talk over. It's the
|
between processes. Channels are therefore always established over
|
||||||
second half of this document.
|
kernel-ipc, and it remains every channel's control path even when bulk
|
||||||
|
data is negotiated onto a fatter transport (a shared-memory ring).
|
||||||
|
- **Verified source.** Every delivery carries the kernel-stamped badge — the
|
||||||
|
identity the channel layer attaches to received packets.
|
||||||
|
- **Bounded packets.** 256 bytes call/reply, 64 pushed — the floor every
|
||||||
|
protocol may assume on any transport.
|
||||||
|
|
||||||
## The channel
|
Three properties keep the networking analogy honest — kernel-ipc is
|
||||||
|
networking-*shaped*, not TCP:
|
||||||
|
|
||||||
The first form is a **bounded blocking channel** (`system/kernel/ipc.zig`): a fixed-size
|
- **Channels over it are RPC-shaped, not streams.** Packets, call/reply,
|
||||||
ring buffer of messages with a producer/consumer rendezvous, built on the
|
datagram pushes — closer to UDP plus RPC than to a byte stream. Ordering
|
||||||
scheduler's [wait queues](../os-development/scheduling.md).
|
exists per exchange (a reply answers its call), not across a channel.
|
||||||
|
- **Possession is the connection.** There is no handshake state in the
|
||||||
|
kernel: holding the capability *is* having the channel. A provider's one
|
||||||
|
endpoint terminates every client's channel at once, demultiplexed by badge
|
||||||
|
— like every client sharing the server's listening socket, with
|
||||||
|
per-connection state living in the provider, keyed by badge. A *private*
|
||||||
|
channel (a dedicated endpoint pair) is built when wanted: that is exactly
|
||||||
|
what `subscribe` does.
|
||||||
|
- **Packets never fragment.** If it doesn't fit in a packet, it isn't a
|
||||||
|
packet: bulk data lives in shared memory and a packet (or signal) is the
|
||||||
|
doorbell. The display path already works this way.
|
||||||
|
|
||||||
|
The rest of this document is the implementation, bottom-up: the kernel-thread
|
||||||
|
queue the blocking discipline was worked out on, then endpoints — this
|
||||||
|
transport's termination points.
|
||||||
|
|
||||||
|
## The kernel-thread queue
|
||||||
|
|
||||||
|
The first form is a **bounded blocking queue** (`system/kernel/ipc.zig`): a
|
||||||
|
fixed-size ring buffer of messages with a producer/consumer rendezvous, built
|
||||||
|
on the scheduler's [wait queues](../os-development/scheduling.md). (Its type
|
||||||
|
is still named `Channel(T, capacity)` — it predates the vocabulary above, and
|
||||||
|
is a *queue between kernel threads in one address space*, not a channel in
|
||||||
|
the model's sense; a rename can ride a later flag-day.)
|
||||||
|
|
||||||
`Channel(T, capacity)` is generic over the message type and buffer size. It holds a
|
`Channel(T, capacity)` is generic over the message type and buffer size. It holds a
|
||||||
ring buffer, a count, and two wait queues:
|
ring buffer, a count, and two wait queues:
|
||||||
|
|
||||||
- **`send(msg)`** — if the channel is full, block on the *not-full* queue; otherwise
|
- **`send(msg)`** — if the queue is full, block on the *not-full* queue; otherwise
|
||||||
write the message, bump the count, and wake a waiting receiver.
|
write the message, bump the count, and wake a waiting receiver.
|
||||||
- **`receive()`** — if the channel is empty, block on the *not-empty* queue; otherwise
|
- **`receive()`** — if the queue is empty, block on the *not-empty* queue; otherwise
|
||||||
take a message, drop the count, and wake a waiting sender.
|
take a message, drop the count, and wake a waiting sender.
|
||||||
|
|
||||||
Neither side busy-waits: a full channel parks the sender, an empty one parks the
|
Neither side busy-waits: a full queue parks the sender, an empty one parks the
|
||||||
receiver, and each operation wakes the other side when it makes progress possible.
|
receiver, and each operation wakes the other side when it makes progress possible.
|
||||||
|
|
||||||
Two details make it correct:
|
Two details make it correct:
|
||||||
@@ -45,47 +81,53 @@ Two details make it correct:
|
|||||||
CPU. `waitLocked` / `wakeLocked` are the variants that assume the caller already
|
CPU. `waitLocked` / `wakeLocked` are the variants that assume the caller already
|
||||||
holds that critical section.
|
holds that critical section.
|
||||||
|
|
||||||
## Verifying it
|
### Verifying it
|
||||||
|
|
||||||
The `ipc` test (see [testing.md](../testing.md)) runs a producer and a consumer passing
|
The `ipc` test (see [testing.md](../testing.md)) runs a producer and a consumer passing
|
||||||
**100 messages through a 4-slot channel**. The small buffer means the channel goes
|
**100 messages through a 4-slot queue**. The small buffer means the queue goes
|
||||||
full and empty over and over, so both the blocking-send and blocking-receive paths are
|
full and empty over and over, so both the blocking-send and blocking-receive paths are
|
||||||
exercised heavily. The messages arrive intact and in order (their sum is the
|
exercised heavily. The messages arrive intact and in order (their sum is the
|
||||||
expected `5050`), and neither task busy-waits — they block and wake each other.
|
expected `5050`), and neither task busy-waits — they block and wake each other.
|
||||||
|
|
||||||
## Endpoints: call/reply across address spaces
|
## Endpoints: the termination points
|
||||||
|
|
||||||
A channel connects two kernel threads sharing one address space. Real servers are
|
A queue connects two kernel threads sharing one address space. Real providers are
|
||||||
*processes*, so the payload has to cross an address-space boundary. That's
|
*processes*, so a packet has to cross an address-space boundary. That's
|
||||||
`system/kernel/ipc-synchronous.zig`, and its shape is L4's: a synchronous **rendezvous** at an
|
`system/kernel/ipc-synchronous.zig`, and its shape is L4's: a synchronous **rendezvous** at an
|
||||||
`Endpoint`, with the message copied directly from the sender's pages to the receiver's
|
`Endpoint`, with the packet copied directly from the sender's pages to the receiver's
|
||||||
(`copyAcross` walks both sets of page tables through the physmap — no CR3 switch, no
|
(`copyAcross` walks both sets of page tables through the physmap — no CR3 switch, no
|
||||||
bounce buffer).
|
bounce buffer).
|
||||||
|
|
||||||
Two syscalls carry it:
|
Two syscalls carry the request/reply exchange:
|
||||||
|
|
||||||
- **`ipc_call(h, msg, reply)`** — copy `msg` to the server, block until it replies.
|
- **`ipc_call(h, msg, reply)`** — copy the request packet to the provider, block
|
||||||
|
until the reply packet comes back.
|
||||||
- **`ipc_reply_wait(h, reply, recv)`** — reply to the client you're still holding (if
|
- **`ipc_reply_wait(h, reply, recv)`** — reply to the client you're still holding (if
|
||||||
any), then block for the next request. One syscall, because a server's steady state
|
any), then block for the next request. One syscall, because a provider's steady state
|
||||||
is *always* "finish the last one, wait for the next".
|
is *always* "finish the last one, wait for the next".
|
||||||
|
|
||||||
An endpoint is reached by **handle** — a small integer index into the process's handle
|
An endpoint is reached by **handle** — a small integer index into the process's handle
|
||||||
table (`Task.handles`), exactly like a file descriptor, and just as unforgeable. The
|
table (`Task.handles`), exactly like a file descriptor, and just as unforgeable.
|
||||||
bootstrap problem (how do you get the first handle?) is solved by a tiny name registry:
|
The provider never learns the client's identity beyond the **badge** delivered
|
||||||
a server calls `ipc_register(service_id, h)` under a well-known small integer, and a
|
alongside each packet: the caller's task id, stamped by the kernel —
|
||||||
client calls `ipc_lookup(service_id)`.
|
unforgeable source addressing, a property a network's source field lacks.
|
||||||
|
|
||||||
The server never learns the client's identity beyond a **badge**, delivered alongside
|
The bootstrap problem — how a channel is first established — is the subject of
|
||||||
the message: the caller's task id.
|
[protocol-namespace.md](../os-development/protocol-namespace.md): a protocol is
|
||||||
|
resolved by name and the channel arrives as a capability. (The mechanism it
|
||||||
|
replaced — `ipc_register`/`ipc_lookup` under compile-time `ServiceId` integers —
|
||||||
|
is gone: both syscalls and the enum were deleted when the registry landed, and
|
||||||
|
their syscall numbers are left vacant.)
|
||||||
|
|
||||||
### Interrupts are messages too
|
### Interrupts are signals
|
||||||
|
|
||||||
`notifyFromIsr` posts an *asynchronous* notification to an endpoint — no payload, no
|
`notifyFromIsr` posts an *asynchronous* signal to an endpoint — no payload, no
|
||||||
reply owed — and wakes whoever is blocked in `reply_wait`. Its badge has the top bit
|
reply owed — and wakes whoever is blocked in `reply_wait`. Its badge has the top bit
|
||||||
set (`notify_badge_bit`), which is how a driver's single event loop distinguishes "a
|
set (`notify_badge_bit`), which is how a driver's single event loop distinguishes "a
|
||||||
client wants something" from "the hardware wants something". Notifications sit in a
|
client wants something" from "the hardware wants something". Signals sit in a
|
||||||
small coalescing ring on the endpoint, so an interrupt taken while the driver was busy
|
small coalescing ring on the endpoint, so an interrupt taken while the driver was busy
|
||||||
elsewhere is not lost.
|
elsewhere is not lost — coalesced, never dropped, which is exactly a signal's
|
||||||
|
contract (the *count* may collapse; the *fact* may not).
|
||||||
|
|
||||||
This is what makes a user-space driver possible at all, and it's the subject of
|
This is what makes a user-space driver possible at all, and it's the subject of
|
||||||
[drivers.md](drivers.md).
|
[drivers.md](drivers.md).
|
||||||
@@ -93,40 +135,65 @@ This is what makes a user-space driver possible at all, and it's the subject of
|
|||||||
## What's next (partly done since)
|
## What's next (partly done since)
|
||||||
|
|
||||||
- **Priority inheritance** through IPC — still open: a high-priority client
|
- **Priority inheritance** through IPC — still open: a high-priority client
|
||||||
blocked on a low-priority server suffers unbounded priority inversion.
|
blocked on a low-priority provider suffers unbounded priority inversion.
|
||||||
- **Handle transfer.** *Landed as cap-passing (M13)*: `ipc_call` and
|
- **Handle transfer.** *Landed as cap-passing (M13)*: `ipc_call` and
|
||||||
`ipc_reply_wait` carry an optional capability alongside the bytes (`send_cap`),
|
`ipc_reply_wait` carry an optional capability alongside the bytes (`send_cap`),
|
||||||
copying an endpoint or shared-memory handle into the peer's table. First user:
|
copying an endpoint or shared-memory handle into the peer's table — the
|
||||||
[input](input.md) subscribers register by handing over their own endpoint, and
|
mechanism by which channels are established and private channels built. First
|
||||||
class drivers get a private channel to one device.
|
user: [input](input.md) subscribers register by handing over their own
|
||||||
|
endpoint, and class drivers get a private channel to one device.
|
||||||
- **Asynchronous / buffered send** for the cases where a rendezvous is the wrong
|
- **Asynchronous / buffered send** for the cases where a rendezvous is the wrong
|
||||||
shape (logging, notifications between servers). *Landed as `ipc_send`* — a
|
shape (logging, event fan-out). *Landed as `ipc_send`* — a
|
||||||
non-blocking post to an endpoint's bounded payload queue, delivered through
|
non-blocking post of an event packet (≤ 64 bytes) to an endpoint's bounded
|
||||||
`reply_wait` as a buffered message (badge bit `notify_message_bit`). Built for, and
|
queue, delivered through `reply_wait` (badge bit `notify_message_bit`). Built
|
||||||
first used by, the [input service](input.md)'s keyboard-event broadcast, where a
|
for, and first used by, the [input service](input.md)'s keyboard-event
|
||||||
synchronous push would let one dead subscriber hang the fan-out. A full queue drops
|
broadcast, where a synchronous push would let one dead subscriber hang the
|
||||||
the oldest (discrete messages, not a coalescing level like the notification ring).
|
fan-out. A full queue drops the oldest — event packets are droppable by
|
||||||
- **A bounded reply** — half landed. The copy is still 256 bytes
|
design ([protocol-namespace.md](../os-development/protocol-namespace.md)'s
|
||||||
(`MESSAGE_MAXIMUM`) under the big kernel lock, but bulk transfer got its shared
|
wiring section states the rule).
|
||||||
|
- **A bounded reply** — half landed. The copy is still one packet
|
||||||
|
(256 bytes) under the big kernel lock, but bulk transfer got its shared
|
||||||
pages: `shared_memory_create`/`map`/`physical`, the region handle delegated as
|
pages: `shared_memory_create`/`map`/`physical`, the region handle delegated as
|
||||||
a capability (above). virtio-gpu's scanout surface is the first user
|
a capability (above) — the packets-never-fragment rule in practice.
|
||||||
|
virtio-gpu's scanout surface is the first user
|
||||||
([display-v2.md](display-v2.md)).
|
([display-v2.md](display-v2.md)).
|
||||||
|
|
||||||
## Lifecycle conventions over IPC (M17)
|
## Lifecycle conventions over IPC (M17)
|
||||||
|
|
||||||
Three conventions from [process-lifecycle.md](../os-development/process-lifecycle.md) ride the
|
Three conventions from [process-lifecycle.md](../os-development/process-lifecycle.md) ride the
|
||||||
notification mechanism:
|
signal mechanism:
|
||||||
|
|
||||||
- **Signals** arrive as notifications on the endpoint a process nominated with
|
- **Process signals** arrive as endpoint signals on the endpoint a process
|
||||||
`signal_bind` (`process.bindSignals`): badge = the signal bit plus the
|
nominated with `signal_bind` (`process.bindSignals`): badge = the signal bit
|
||||||
coalesced pending mask (`process.signalsFrom` decodes). Statements,
|
plus the coalesced pending mask (`process.signalsFrom` decodes). Statements,
|
||||||
never questions; no payload, no reply.
|
never questions; no payload, no reply.
|
||||||
- **One-shot timers** (`timer_bind`, `time.timerOnce`) land as a
|
- **One-shot timers** (`timer_bind`, `time.timerOnce`) land as a
|
||||||
timer-bit notification — the timed wait: a service arms a deadline and keeps
|
timer-bit signal — the timed wait: a service arms a deadline and keeps
|
||||||
serving, instead of blocking in sleep.
|
serving, instead of blocking in sleep.
|
||||||
|
- **Kernel notifications go only to your own endpoint.** `signal_bind`,
|
||||||
|
`timer_bind`, `process_subscribe`, `irq_bind`, `msi_bind`, and spawn's exit
|
||||||
|
endpoint all *nominate where the kernel will speak*, and all of them refuse an
|
||||||
|
endpoint the caller did not create (`-EPERM`; the check is `ipc.ownedBy`,
|
||||||
|
normalized to the process, so any thread may nominate an endpoint a sibling
|
||||||
|
created). Holding a handle is not enough, because holding a handle is cheap:
|
||||||
|
`fs_resolve` installs a mounted backend's capability in *any* caller's table,
|
||||||
|
so every process holds a handle to PID 1's mailbox. Without the rule, "bind
|
||||||
|
init's endpoint, then signal yourself" is a genuine, kernel-stamped `terminate`
|
||||||
|
badge in PID 1's queue — a shutdown a receiver has no way to disbelieve — and
|
||||||
|
timers, which carry no identity at all, multiply any loop that re-arms on its
|
||||||
|
own landing.
|
||||||
|
- **A capability that arrives belongs to the turn.** The kernel installs a sent
|
||||||
|
capability in the receiver's table whatever the message's length or kind, so a
|
||||||
|
receive loop must dispose of one on *every* path — the ping, the notification,
|
||||||
|
the malformed request. The service harness (`service.run`) and PID 1 both hold
|
||||||
|
it in an `ipc.Arrival`, released by a `defer`, and a handler that means to keep
|
||||||
|
it says `take()`: forgetting closes, keeping is explicit. The reverse
|
||||||
|
arrangement leaks a handle-table slot per request, and thirty-two unauthorized
|
||||||
|
zero-length pings then end a service's ability to accept any capability —
|
||||||
|
no subscribe, no shared-memory handover — for the rest of the boot.
|
||||||
- **The universal ping**: a **zero-length request is the liveness probe**,
|
- **The universal ping**: a **zero-length request is the liveness probe**,
|
||||||
answered with a zero-length reply by the service harness itself
|
answered with a zero-length reply by the service harness itself
|
||||||
(`service.run`). No protocol's requests start at length zero, so the
|
(`service.run`). No protocol's requests start at length zero, so the
|
||||||
encoding cannot collide, and a wedged service simply fails to answer — which
|
encoding cannot collide, and a wedged service simply fails to answer — which
|
||||||
is the diagnosis. Deep health ("can I reach my hardware?") stays a per-service
|
is the diagnosis. Deep health ("can I reach my hardware?") stays a per-service
|
||||||
protocol message.
|
protocol packet.
|
||||||
|
|||||||
@@ -0,0 +1,124 @@
|
|||||||
|
# Dynamic libraries on danos
|
||||||
|
|
||||||
|
A design note and milestone plan for shared objects: building them, loading them
|
||||||
|
with `dlopen`, and — the part that needs kernel work — actually *sharing* them
|
||||||
|
between processes. Directional, post-P5 of
|
||||||
|
[python-on-danos-milestones.md](python-on-danos-milestones.md); nothing on the
|
||||||
|
CPython bring-up path depends on it.
|
||||||
|
|
||||||
|
## Reconciling the earlier "rejected"
|
||||||
|
|
||||||
|
Dynamic libraries were evaluated once before and rejected — but as an answer to a
|
||||||
|
*different question*: whether they could claw back ReleaseSafe's measured ~2×
|
||||||
|
code size. They cannot (the safety checks inline at every call site; no library
|
||||||
|
scheme dedups them), and that verdict stands for that question. The reasons to
|
||||||
|
build them now are the ones that investigation never weighed:
|
||||||
|
|
||||||
|
- **`ctypes` and runtime FFI** — Python calling into a danos library without
|
||||||
|
rebuilding the interpreter. This is the piece that makes Python prototyping
|
||||||
|
self-serve: drop a `.so` on the image, `ctypes.CDLL` it, iterate.
|
||||||
|
- **Loadable CPython extension modules** — today every C extension means
|
||||||
|
relinking the interpreter (`Modules/Setup`); with `dlopen`, an extension is a
|
||||||
|
file.
|
||||||
|
- **One interpreter image, many Python services** — a statically-linked CPython
|
||||||
|
is tens of megabytes *per process*. A shared `libpython` mapped read-only once
|
||||||
|
(milestone D3 below) makes Python services cheap enough to be the default way
|
||||||
|
to prototype one.
|
||||||
|
- **Plugin-shaped applications** — the UI toolkit and the terminal will want
|
||||||
|
them eventually.
|
||||||
|
|
||||||
|
The scoping that dissolves the apparent contradiction is the **size doctrine**:
|
||||||
|
leanness is an *operating-system* property — the kernel and system services stay
|
||||||
|
small and statically linked, and none of them ever link the loader — while
|
||||||
|
*applications* have their own budget and may be big. Dynamic libraries are an
|
||||||
|
**application-layer facility**, full stop.
|
||||||
|
|
||||||
|
What also does **not** change: the public ABI stays the vDSO + the IPC
|
||||||
|
protocols. Shared objects are artifacts *within* one system image, versioned by
|
||||||
|
the build — not a new stable ABI surface for the OS.
|
||||||
|
|
||||||
|
## Design
|
||||||
|
|
||||||
|
- **Format and codegen are free.** ELF shared objects with position-independent
|
||||||
|
code; `zig cc -fPIC -shared` against the [libdanos-c](c-library-compatibility.md)
|
||||||
|
sysroot already emits them. The work is entirely on the loading side.
|
||||||
|
- **The loader lives in userspace, inside the libc.** `dlopen` reads the `.so`
|
||||||
|
through the VFS, maps its segments, applies relocations, resolves symbols
|
||||||
|
against the process and the `DT_NEEDED` dependency graph, runs constructors,
|
||||||
|
returns a handle. No kernel loader changes in v1 — segments land in anonymous
|
||||||
|
`mmap` as private copies.
|
||||||
|
- **Bind-now, always.** All relocations resolved at `dlopen` time
|
||||||
|
(`RTLD_NOW` semantics only). Lazy PLT binding buys startup latency danos does
|
||||||
|
not care about, at the price of a writable GOT dance and a much subtler
|
||||||
|
loader. Not worth it; keep it out permanently.
|
||||||
|
- **W^X from day one.** Map, relocate, then flip text pages read-execute —
|
||||||
|
which requires memory-protection change (`mprotect`-shaped) in the danos
|
||||||
|
`mmap` surface if it is not already there. No page is ever writable and
|
||||||
|
executable at once.
|
||||||
|
- **TLS in shared objects is deferred.** Thread-local storage models
|
||||||
|
(initial-exec vs. general-dynamic) are the deep end of every dynamic linker.
|
||||||
|
v1 refuses a `.so` with a TLS segment; revisit alongside the post-P5 pthread
|
||||||
|
subset, which is when it could matter.
|
||||||
|
- **Executables stay static until D4.** v1 is "a static binary that can
|
||||||
|
`dlopen`" — no `PT_INTERP`, no program interpreter, no dynamically-linked
|
||||||
|
`main` binaries. That keeps process startup untouched.
|
||||||
|
|
||||||
|
## Milestones
|
||||||
|
|
||||||
|
1. **D1 — dlopen in-process.** The `.so` build target; the loader in libdanos-c:
|
||||||
|
map, relocate (`RELATIVE`/`GLOB_DAT`/`JUMP_SLOT`), resolve, constructors;
|
||||||
|
`dlopen`/`dlsym`/`dlerror`/`dlclose`; private anonymous mappings; no TLS.
|
||||||
|
*Test:* QEMU `dlopen-hello` — load a `.so`, call a symbol, unload, reload.
|
||||||
|
2. **D2 — the FFI payoff.** `DT_NEEDED` dependency graphs; a **libffi port**
|
||||||
|
(x86-64 SysV assembly is upstream; the port is its closure-allocation paths,
|
||||||
|
which must respect W^X); CPython's `ctypes` enabled; extension modules
|
||||||
|
loadable from file. *Test:* QEMU — a Python script `ctypes.CDLL`s a danos
|
||||||
|
`.so` and round-trips a call; a `.so` extension module imports.
|
||||||
|
3. **D3 — actual sharing (the kernel milestone).** Shared read-only file-backed
|
||||||
|
mappings — a page-cache-shaped facility so N processes mapping `libpython`
|
||||||
|
hold one physical copy. This is the memory-win milestone and the only one
|
||||||
|
touching the kernel; design it with the existing shm machinery in view
|
||||||
|
(the shared-fate walks already locked the relevant paths). *Test:* N Python
|
||||||
|
services up; measure physical pages against N× the static baseline.
|
||||||
|
4. **D4 — dynamically-linked executables** (optional, evaluate after D3):
|
||||||
|
`PT_INTERP`, a danos program interpreter, and the spawn path teaching the
|
||||||
|
loader about it. Only worth it if the image-size or update story demands it.
|
||||||
|
|
||||||
|
## Risks and gotchas
|
||||||
|
|
||||||
|
- **Scope creep is the failure mode.** Every dynamic linker grows toward glibc.
|
||||||
|
The fences: bind-now only, no lazy binding ever, no TLS until pthreads demand
|
||||||
|
it, no dlopen-from-memory, no versioned symbols. Each fence removed is a
|
||||||
|
design discussion, not a patch.
|
||||||
|
- **Code loading is a security event.** `dlopen` turns file bytes into executable
|
||||||
|
code, so W^X discipline is table stakes and *what may be dlopened* is a
|
||||||
|
capability question — the natural danos answer is that loadability follows VFS
|
||||||
|
readability of the `.so`, and services' images are supervised like any other
|
||||||
|
artifact. Revisit explicitly at D3 when mappings become shared.
|
||||||
|
- **`dlclose` is where loaders go to die.** Constructors/destructors,
|
||||||
|
dangling function pointers, re-open identity. Keep v1 semantics honest and
|
||||||
|
simple: `dlclose` runs destructors and unmaps; holding pointers past it is
|
||||||
|
undefined; no reference-counted deferral cleverness.
|
||||||
|
- **The ReleaseSafe fact still applies to `.so`s** — a ReleaseSafe shared object
|
||||||
|
carries its inlined checks like any static code; D3's sharing saves *copies*,
|
||||||
|
not check overhead. Size expectations should be set accordingly.
|
||||||
|
|
||||||
|
## Decisions needing sign-off
|
||||||
|
|
||||||
|
- Dynamic libraries join the roadmap at all (this note exists because the
|
||||||
|
earlier size-motivated rejection was re-opened for ABI/sharing reasons).
|
||||||
|
- **Bind-now only; no lazy binding, permanently.**
|
||||||
|
- **Loader in userspace libc; kernel involvement only at D3** (shared read-only
|
||||||
|
mappings).
|
||||||
|
- **Static executables until D4**, and D4 only on demonstrated need.
|
||||||
|
|
||||||
|
## Related
|
||||||
|
|
||||||
|
- [c-library-compatibility.md](c-library-compatibility.md) — the sysroot the
|
||||||
|
loader ships in; its absence table gains `dlfcn.h` at D1.
|
||||||
|
- [python-on-danos.md](python-on-danos.md) — the `ctypes` story this unlocks.
|
||||||
|
- [python-on-danos-milestones.md](python-on-danos-milestones.md) — sequencing;
|
||||||
|
this work is post-P5.
|
||||||
|
- [os-development/memory-map.md](os-development/memory-map.md) /
|
||||||
|
[os-development/paging.md](os-development/paging.md) — where W^X and shared
|
||||||
|
mappings land.
|
||||||
@@ -1,128 +0,0 @@
|
|||||||
# DanOS Filesystem Hierarchy Standard (DFHS)
|
|
||||||
|
|
||||||
Most modern Unix and Unix-like operating systems follow the FHS. DanOS has its own FHS structure which extends the unix FHS. Root path resolution is provided by the kernel-resident VFS root (`fs_resolve`, `system/kernel/vfs.zig`); mounted filesystem servers serve the subtrees they own.
|
|
||||||
|
|
||||||
## Directory structure
|
|
||||||
|
|
||||||
| Path | Description |
|
|
||||||
|------------------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------|
|
|
||||||
| / | Primary hierarchy root and root directory of the entire file system hierarchy. |
|
|
||||||
| /bin | Essential command binaries that need to be available in single-user mode, including to bring up the system or repair it, for all users (e.g., cat, ls, cp). |
|
|
||||||
| /boot | Boot loader files (e.g., EFI, initial-ramdisk.img ). |
|
|
||||||
| /dev | POSIX Device files (e.g., /dev/null, /dev/disk0, /dev/tty, /dev/random). |
|
|
||||||
| /etc | Host-specific system-wide configuration files. |
|
|
||||||
| /home | Users' home directories, containing saved files, personal settings, etc. |
|
|
||||||
| /lib | Libraries essential for the binaries in /bin and /sbin. eg realtime, system, ipc etc. |
|
|
||||||
| /sbin | Essential system binaries (e.g init) |
|
|
||||||
| /srv | Site-specific data served by this system, such as data and scripts for web servers, data offered by FTP servers, and repositories for version control systems |
|
|
||||||
| /system | DanOS operating system files (similar idea to C:\Windows). A true representation of danos — its layout mirrors the source tree, so `/system` is what danos *is*. |
|
|
||||||
| /system/devices | danos virtual device tree e.g. similar to /sys on linux but with danos device tree conventions (the structures in the devices module) |
|
|
||||||
| /system/drivers | driver binaries, one sub-project each (e.g. /system/drivers/pci-bus, /system/drivers/ps2-bus) |
|
|
||||||
| /system/services | system-service binaries — init, the FAT server, and other user-mode servers (e.g. /system/services/init, /system/services/fat) |
|
|
||||||
| /system/kernel | the kernel image |
|
|
||||||
| /test | Test fixtures for the QEMU integration suite. Read-only and initrd-backed like /system, and its layout likewise mirrors the source tree (the repo's test/ directory). Present on development and test images; a volume without it still boots. |
|
|
||||||
| /test/system/services | test-fixture binaries (e.g. /test/system/services/vfs-test, /test/system/services/thread-test) — the same path in the repo source tree and on the boot volume |
|
|
||||||
| /tmp | Directory for temporary files (see also /var/tmp). Often not preserved between system reboots and may be severely size-restricted. |
|
|
||||||
| /usr | Secondary hierarchy for read-only user data; contains the majority of (multi-)user utilities and applications. Should be shareable and read-only. |
|
|
||||||
| /var | Variable files: files whose content is expected to continually change during normal operation of the system, such as logs, spool files, and temporary e-mail files. |
|
|
||||||
|
|
||||||
## File types
|
|
||||||
|
|
||||||
POSIX specifies the long format of the ls command to represent the Unix file type as the first letter for an entry.
|
|
||||||
|
|
||||||
| type | symbol | Description |
|
|
||||||
|-------------------|--------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------|
|
|
||||||
| regular | - | An ordinary file holding an uninterpreted byte stream. Reads and writes are positional, and the file grows on demand (e.g., a binary in /bin, a config file in /etc). |
|
|
||||||
| directory | d | A container mapping names to other files. It may only be modified through directory operations, never written to directly. |
|
|
||||||
| symbolic link | l | A file whose contents are a path that is resolved in its place. The target need not exist, and may cross mount points. |
|
|
||||||
| FIFO special | p | A named pipe: an in-order byte stream between processes, where writers block until a reader opens the other end. |
|
|
||||||
| block special | b | A device node addressed in fixed-size blocks with the kernel free to buffer and reorder access (e.g., /dev/disk0). |
|
|
||||||
| character special | c | A device node addressed as an unbuffered byte stream, delivered to the driver in order (e.g., /dev/tty, /dev/null). |
|
|
||||||
| socket | s | A named endpoint for bidirectional message-passing between processes, bound to a path rather than an address. |
|
|
||||||
|
|
||||||
## /dev
|
|
||||||
|
|
||||||
`/dev` holds the names through which processes reach devices. It is deliberately not
|
|
||||||
the device tree: the tree — every node discovered by ACPI or PCI enumeration, with its
|
|
||||||
resources and its parent — lives under [/system/devices](#directory-structure) and is
|
|
||||||
addressed by device id. `/dev` is the much smaller set of devices that have a driver
|
|
||||||
willing to serve them, addressed by name.
|
|
||||||
|
|
||||||
A device node is not a file the VFS can read. The bytes live in a driver process
|
|
||||||
([drivers.md](../device-driver-development/drivers.md)), so opening a `/dev` name has to resolve to that driver's
|
|
||||||
IPC endpoint, and subsequent reads and writes are calls against it. Resolve-to-endpoint
|
|
||||||
is exactly what the kernel's `fs_resolve` already does for any mounted backend, and
|
|
||||||
`FileStatus.kind` is the field that marks a device node; **what is not implemented today
|
|
||||||
is `/dev` itself** — no service mounts it. (The flat eight-node ramfs this section once
|
|
||||||
described is retired: the kernel-resident VFS root in `system/kernel/vfs.zig` serves a
|
|
||||||
read-only initrd mount per top-level tree — `/system`, and `/test` on images that carry
|
|
||||||
the fixtures — with real directories and node kinds, and filesystem
|
|
||||||
backends such as the FAT server mount the rest.) The three sections below describe the
|
|
||||||
intended shape, and are honest about which parts the kernel can already support.
|
|
||||||
|
|
||||||
### Character devices
|
|
||||||
|
|
||||||
A character device is a byte stream with no addressable position: bytes are delivered
|
|
||||||
to the driver in the order written, and a read consumes what is there. Terminals,
|
|
||||||
serial lines, keyboards and mice are all of this shape. These are the natural first
|
|
||||||
device nodes in danos, because a character driver needs nothing the kernel doesn't
|
|
||||||
already provide — it claims its device, maps its registers with `mmio_map`, and blocks
|
|
||||||
on `replyWait` for either an interrupt or a client request. `system/drivers/ps2-bus/ps2-bus.zig`
|
|
||||||
is already that program, minus the file-node client half.
|
|
||||||
|
|
||||||
The obstacle was never the file type; it is which hardware a ring-3 driver can reach.
|
|
||||||
Direct `in`/`out` from user space is still a #GP (no TSS I/O bitmap, IOPL never raised),
|
|
||||||
but a driver no longer needs it: **`io_read`/`io_write`** grant port access the same way
|
|
||||||
`mmio_map` grants memory — gated by `device_claim` and the device's discovered `io_port`
|
|
||||||
resource. So the 16550 UART at `0x3F8` and the PS/2 controller at `0x60`/`0x64` (and thus
|
|
||||||
`/dev/ttyS0` and a keyboard node) are now writable as ordinary ring-3 drivers; the
|
|
||||||
low-rate legacy hardware that needs port I/O is fine with a syscall per access. A
|
|
||||||
memory-mapped device such as the framebuffer, needing no port I/O at all, remains the
|
|
||||||
easiest first entry.
|
|
||||||
|
|
||||||
### Block devices
|
|
||||||
|
|
||||||
A block device is addressed in fixed-size blocks and, unlike a character device, the
|
|
||||||
layer above is free to buffer, reorder, coalesce and retry requests against it. Disks
|
|
||||||
and other persistent storage are the whole population of this class.
|
|
||||||
|
|
||||||
A block driver is now **writable, but not yet memory-safe.** Every storage controller
|
|
||||||
worth naming is a bus master: it is programmed by handing it the physical address of a
|
|
||||||
descriptor ring and left to read and write memory on its own. That ring is exactly what
|
|
||||||
**`dma_alloc`** now provides — physically contiguous, pinned, uncacheable, with its
|
|
||||||
physical address disclosed — and **`/lib/device/mmio`**'s barriers order the descriptor writes
|
|
||||||
against the doorbell, and **`msi_bind`** delivers completions. So an AHCI or NVMe driver
|
|
||||||
can be written today (the M14/M15 work in [driver-model.md](../device-driver-development/driver-model.md); the earlier
|
|
||||||
"cannot host a block driver at all" is no longer true).
|
|
||||||
|
|
||||||
What is *not* yet true is that it is safe. A device programmed with an arbitrary physical
|
|
||||||
address writes to arbitrary physical memory, and page tables do not sit between a device
|
|
||||||
and RAM — an IOMMU does. The IOMMU is now *detected* (M16), but no translation domains
|
|
||||||
are programmed, so granting a DMA-capable device to a driver process is still equivalent
|
|
||||||
to granting ring 0. Until per-device domains confine a driver's DMA to the buffers it
|
|
||||||
`dma_alloc`'d, a block driver works but forfeits the isolation that motivates user-space
|
|
||||||
drivers — enforcement is the next step, and lands with that first driver. A ramdisk over
|
|
||||||
the initial ramdisk remains the one block-shaped thing that needs no driver process at all.
|
|
||||||
|
|
||||||
### Pseudo-devices
|
|
||||||
|
|
||||||
A pseudo-device has the interface of a device and no hardware behind it: `/dev/null`
|
|
||||||
discarding writes and reading as end-of-file, `/dev/zero` reading as an endless run of
|
|
||||||
zero bytes, `/dev/full` failing writes with `ENOSPC`, `/dev/random` and `/dev/urandom`
|
|
||||||
yielding unpredictable bytes.
|
|
||||||
|
|
||||||
These are the only `/dev` entries danos can implement immediately, and they are the
|
|
||||||
sensible place to start, because they are exactly the entries that need no driver
|
|
||||||
process, no `device_claim`, no MMIO grant and no interrupt. A future pseudo-device
|
|
||||||
service would answer them out of its own address space — `null` and `zero` are a few
|
|
||||||
lines each in its `read` and `write` handlers — and mount itself at `/dev` the way the
|
|
||||||
FAT server mounts `/mnt/usb`. The two pieces of structure every later device node
|
|
||||||
depends on (and that the flat ramfs of the time lacked) exist now: directories, so that
|
|
||||||
`/dev/null` is a path rather than a name; and a populated `FileStatus.kind`, so that a
|
|
||||||
caller can tell a character device from a regular file.
|
|
||||||
|
|
||||||
`/dev/random` is the one that is not free. It needs an entropy source, and the honest
|
|
||||||
options on this kernel are `RDRAND`/`RDSEED` where CPUID advertises them, and the HPET
|
|
||||||
counter's low bits as a poor fallback. Neither is a seeded CSPRNG, and a `/dev/random`
|
|
||||||
that is merely unpredictable-looking is worse than none — nothing should be keyed from
|
|
||||||
it until it is a real one.
|
|
||||||
@@ -0,0 +1,90 @@
|
|||||||
|
# The danos file-system hierarchy
|
||||||
|
|
||||||
|
danos is not unix, and its tree does not follow the unix FHS. Paths are the
|
||||||
|
system's universal namespace — files, the device inventory, and protocol
|
||||||
|
endpoints all live in one tree — but what a path *yields* differs by subtree:
|
||||||
|
bytes, facts, or a connection. Root path resolution is provided by the
|
||||||
|
kernel-resident VFS root (`fs_resolve`, `system/kernel/vfs.zig`); mounted
|
||||||
|
backends serve the subtrees they own.
|
||||||
|
|
||||||
|
Naming follows the codebase conventions: kebab-case, full words, no
|
||||||
|
abbreviations. Every top-level name says what its subtree *is*.
|
||||||
|
|
||||||
|
## The tree
|
||||||
|
|
||||||
|
| Path | What it is |
|
||||||
|
|-------------------------|---------------------------------------------------------------------------------------------------------------------------------------------------------|
|
||||||
|
| `/` | The root of the one namespace. |
|
||||||
|
| `/applications` | Installed applications, one directory per application — the directory is the identity, the same rule as source sub-projects. *(Planned; empty today.)* |
|
||||||
|
| `/protocol` | The contract namespace: one protocol node per contract, grouped into directories by domain (`/protocol/display`, `/protocol/networking/ip`). Synthetic — no bytes; opening a name yields a connection to the current provider. See [protocol-namespace.md](../os-development/protocol-namespace.md). |
|
||||||
|
| `/system` | The operating system — what danos *is*. Its program subtrees mirror the source tree exactly. |
|
||||||
|
| `/system/kernel` | The kernel image. |
|
||||||
|
| `/system/drivers` | Driver binaries, one per sub-project (`/system/drivers/pci-bus`, `/system/drivers/ps2-bus`). |
|
||||||
|
| `/system/services` | System-service binaries (`/system/services/init`, `/system/services/fat`). |
|
||||||
|
| `/system/devices` | The device inventory: every node hardware discovery found, with its resources and parent — the structures of the devices module, as a browsable virtual tree. Informational only; you *read about* hardware here and *talk to* it through `/protocol`. *(Planned; served by device-manager.)* |
|
||||||
|
| `/system/configuration` | Machine configuration (`init.csv`, `devices.csv`). Writable, served from the boot volume. |
|
||||||
|
| `/system/logs` | Per-boot logs: `/system/logs/<boot-stamp>/<binary-path>.log`. Writable, served from the boot volume. |
|
||||||
|
| `/test` | Test fixtures for the QEMU integration suite. Read-only and initrd-backed like the program subtrees of `/system`, mirroring the repo's `test/` directory. Present on development and test images; a volume without it still boots. |
|
||||||
|
| `/volumes` | Attached storage volumes, one directory per volume (`/volumes/usb`). A volume's own tree appears beneath its name. |
|
||||||
|
|
||||||
|
Read-only and writable halves of `/system`: the program subtrees (`kernel`,
|
||||||
|
`drivers`, `services`) and the future `devices` are immutable at runtime —
|
||||||
|
initrd-backed or synthetic — while `configuration` and `logs` are mutable
|
||||||
|
machine state served by the boot-volume FAT backend. The kernel's
|
||||||
|
reserved-prefix rule (no mount may shadow `/system`, `/test`, or `/protocol`)
|
||||||
|
needs a carve-out for exactly these two writable subtrees; that lands with the
|
||||||
|
path migration below.
|
||||||
|
|
||||||
|
Deliberately not defined yet: a temporary-files location and per-application
|
||||||
|
mutable storage. Both belong to the `/applications` design and will be
|
||||||
|
specified there, not guessed at here.
|
||||||
|
|
||||||
|
## Node kinds
|
||||||
|
|
||||||
|
What a path resolves to. These fill `FileStatus.kind` and
|
||||||
|
`DirectoryEntry.kind` in the [vfs protocol](vfs-protocol.md)
|
||||||
|
(`library/protocol/vfs/vfs-protocol.zig`); enum values are append-only.
|
||||||
|
|
||||||
|
| Kind | Meaning |
|
||||||
|
|--------------------|-------------------------------------------------------------------------------------------------------------------------------------------------|
|
||||||
|
| `regular` | An ordinary file: an uninterpreted byte stream, positional reads and writes, grows on demand. |
|
||||||
|
| `directory` | A container mapping names to nodes; modified only through directory operations. |
|
||||||
|
| `character_device` | A node whose read/write have **stream semantics**: unseekable, reads block until bytes exist, size is meaningless. The console and every tty-shaped node ([character-devices-and-tty.md](../character-devices-and-tty.md)); what a POSIX layer's `isatty` detects. |
|
||||||
|
| `block_device` | A node addressed in fixed-size sectors — a raw volume. Reserved: recognized, nothing serves one yet. |
|
||||||
|
| `symbolic_link` | Reserved: a recognized value, not implemented by any backend. |
|
||||||
|
| `fifo` | Reserved for the future pipe object (wanted by the POSIX compatibility layer); not implemented. |
|
||||||
|
| `protocol` | A node naming a contract: `open` yields an IPC connection (an endpoint capability) instead of a file id — the kind of every leaf under `/protocol`. *(Being added; see protocol-namespace.md.)* |
|
||||||
|
|
||||||
|
Note the layering: `protocol` says what *opening the name* does (you get a
|
||||||
|
conversation); `character_device`/`block_device` say what *read and write
|
||||||
|
mean* on a node a provider serves you. The two compose — `/protocol/console`
|
||||||
|
is a protocol node in the registry, and the node opened over that connection
|
||||||
|
reports `character_device`, which is what gives it stream semantics. Only
|
||||||
|
`socket` is retired (its value stays reserved for wire stability): a named
|
||||||
|
rendezvous point is exactly what a protocol node is.
|
||||||
|
|
||||||
|
## What is deliberately absent
|
||||||
|
|
||||||
|
There is no `/bin`, `/boot`, `/dev`, `/etc`, `/home`, `/lib`, `/mnt`, `/sbin`,
|
||||||
|
`/srv`, `/tmp`, `/usr`, or `/var`. These encode unix history — the
|
||||||
|
binary/library split of small disks, configuration-as-scattered-text, devices
|
||||||
|
as magic files — that danos does not carry. A POSIX compatibility layer (the
|
||||||
|
Python track's mini-libc) may *present* whichever of these its programs
|
||||||
|
expect, mapped onto the real tree; the tree itself stays danos-native.
|
||||||
|
|
||||||
|
## Migration
|
||||||
|
|
||||||
|
The tree above is the specification; some code still writes the unix paths it
|
||||||
|
replaced. The flag-day converting them:
|
||||||
|
|
||||||
|
| Today (in code) | Becomes | Where |
|
||||||
|
|------------------------------------------|-------------------------------------------|-----------------------------------------------------------------|
|
||||||
|
| `/etc/init.csv` | `/system/configuration/init.csv` | `system/services/init/init.zig` |
|
||||||
|
| `/etc/devices.csv` | `/system/configuration/devices.csv` | `system/services/device-manager/device-manager.zig` |
|
||||||
|
| `/var/log/...` | `/system/logs/...` | `system/services/logger/logger.zig`, the FAT server's `/var` mount |
|
||||||
|
| `/mnt/usb` | `/volumes/usb` | `system/services/fat/fat.zig`, the fat/vfs tests |
|
||||||
|
| `ServiceId` lookup | resolve + open under `/protocol` | every service and client; [protocol-namespace.md](../os-development/protocol-namespace.md) |
|
||||||
|
|
||||||
|
The boot-image builder and the on-volume directory layout move in the same
|
||||||
|
change, so a freshly written image and the paths the services expect never
|
||||||
|
disagree.
|
||||||
@@ -149,8 +149,8 @@ Bitwise OR in `Request.flags`, meaningful for `open` only:
|
|||||||
|
|
||||||
## NodeKind
|
## NodeKind
|
||||||
|
|
||||||
Aligned to the FSH file-type table
|
Aligned to the node-kind table in the file-system hierarchy
|
||||||
(docs/danos-file-system-hierarchy-FSH.md):
|
(docs/file-system-development/file-system-hierarchy.md):
|
||||||
|
|
||||||
| value | kind |
|
| value | kind |
|
||||||
|------:|------|
|
|------:|------|
|
||||||
@@ -161,9 +161,25 @@ Aligned to the FSH file-type table
|
|||||||
| 4 | symbolic link |
|
| 4 | symbolic link |
|
||||||
| 5 | fifo |
|
| 5 | fifo |
|
||||||
| 6 | socket |
|
| 6 | socket |
|
||||||
|
| 7 | protocol |
|
||||||
|
|
||||||
Clients should map unknown values to *regular* rather than reject — the
|
Clients should map unknown values to *regular* rather than reject — the
|
||||||
table can grow.
|
table can grow. Kind 6 (`socket`) keeps its wire value but is retired from
|
||||||
|
the design — a named rendezvous point is exactly what a `protocol` node is,
|
||||||
|
landed as value 7 with the protocol namespace
|
||||||
|
(docs/os-development/protocol-namespace.md). `character_device` (stream
|
||||||
|
semantics — the tty/console shape) and `block_device` (raw sector-addressed
|
||||||
|
volumes, reserved) remain part of the design.
|
||||||
|
|
||||||
|
## An open reply may carry a capability
|
||||||
|
|
||||||
|
`open` rides `ipc_call`, whose reply direction can hand back an endpoint
|
||||||
|
capability alongside the `Reply` header. A file backend never uses it — FAT
|
||||||
|
answers with a node id and nothing else — but a **synthetic** backend does:
|
||||||
|
opening a `protocol` node returns the provider's endpoint, and possession of
|
||||||
|
that endpoint *is* the channel. The convention is per-backend, not
|
||||||
|
per-operation, so a client that opens an ordinary file simply receives no
|
||||||
|
capability, exactly as before.
|
||||||
|
|
||||||
## Lifetimes and trust
|
## Lifetimes and trust
|
||||||
|
|
||||||
@@ -183,8 +199,8 @@ What a non-Zig implementation may rely on, and what it must not:
|
|||||||
- Operation values, flag bits, `NodeKind` values, and struct layouts are
|
- Operation values, flag bits, `NodeKind` values, and struct layouts are
|
||||||
**append-only and frozen once shipped**. The unit test in
|
**append-only and frozen once shipped**. The unit test in
|
||||||
`library/protocol/vfs/vfs-protocol.zig` pins a sample of them (the `DirectoryEntry`
|
`library/protocol/vfs/vfs-protocol.zig` pins a sample of them (the `DirectoryEntry`
|
||||||
size, `NodeKind` 0–1, `Operation` values 0, 4 and 5); this page is the
|
size, `NodeKind` 0–1 and 6–7, `Operation` values 0, 4 and 5); this page is
|
||||||
full record of the frozen values.
|
the full record of the frozen values.
|
||||||
- The 256-byte message ceiling is a property of the current IPC transport,
|
- The 256-byte message ceiling is a property of the current IPC transport,
|
||||||
not a promise; clients should read `maximum_payload`-shaped limits from the
|
not a promise; clients should read `maximum_payload`-shaped limits from the
|
||||||
reply lengths they actually get (loop-until-done), not hard-code 224.
|
reply lengths they actually get (loop-until-done), not hard-code 224.
|
||||||
|
|||||||
@@ -0,0 +1,120 @@
|
|||||||
|
# Communication: the four layers
|
||||||
|
|
||||||
|
*Design, agreed 2026-07-31. The model document — the vocabulary and layering
|
||||||
|
every other communication document speaks.*
|
||||||
|
|
||||||
|
danos separates **what is said** from **how the bytes move**, so that the
|
||||||
|
mechanism is replaceable. The shape is a network stack's, cut into four
|
||||||
|
layers; a program only ever touches the top two.
|
||||||
|
|
||||||
|
```
|
||||||
|
L3 namespace /protocol/... names establishment points protocol-namespace.md
|
||||||
|
L2 protocol the language: packet schemas, verbs, targets the envelope, library/protocol/*
|
||||||
|
L1 channel two ends exchanging packets and signals the client library's Channel
|
||||||
|
L0 transport a buffer + a doorbell: moves the bytes ipc.md (kernel-ipc), later shm-ring, …
|
||||||
|
```
|
||||||
|
|
||||||
|
## Vocabulary
|
||||||
|
|
||||||
|
| Term | Meaning |
|
||||||
|
|---|---|
|
||||||
|
| **protocol** | The language: which packets exist, what their fields mean, which verbs a provider answers. Defined transport-independently in a `library/protocol/*` module. |
|
||||||
|
| **channel** | An open conversation between two processes, speaking one protocol. Established by opening a `/protocol/...` name; both ends can send and receive. |
|
||||||
|
| **packet** | The unit a protocol transmits: a bounded, atomic header+payload. Never fragmented — if it doesn't fit, it isn't a packet; bulk data rides shared memory with a packet as the doorbell. |
|
||||||
|
| **signal** | A payload-less poke below the packet layer: "something happened, come look." Coalescing — the count may collapse, the fact may not. |
|
||||||
|
| **transport** | What moves the bytes of one channel: a buffer plus a doorbell. Chosen (and upgradable) at establishment, invisible above L1. |
|
||||||
|
| **endpoint** | A termination point where a transport delivers. The kernel-ipc transport's endpoint is its kernel mailbox object. |
|
||||||
|
|
||||||
|
## Addressing: parties by channel, objects by target
|
||||||
|
|
||||||
|
There are no network-style addresses in a packet. The two questions addresses
|
||||||
|
answer are answered at different layers:
|
||||||
|
|
||||||
|
- **Who am I talking to?** The **channel**, decided once at establishment.
|
||||||
|
Opening `/protocol/input` yields a channel; every packet sent on it goes to
|
||||||
|
the peer. Nothing to route per-packet — like TCP, where no HTTP request
|
||||||
|
carries the server's IP.
|
||||||
|
- **Who sent this?** Attached to every received packet **by the channel
|
||||||
|
layer**, from identity the transport can verify — under kernel-ipc, the
|
||||||
|
kernel-stamped badge. The sender never writes a source field, which is what
|
||||||
|
makes source unforgeable (the property a network's spoofable source header
|
||||||
|
lacks).
|
||||||
|
- **Which of your things?** The packet's **`target`** field: *object*
|
||||||
|
addressing within the already-chosen peer — the vfs protocol's node id, the
|
||||||
|
display protocol's layer id, a block volume. `target = 0` addresses the
|
||||||
|
provider itself; a protocol without objects never uses it.
|
||||||
|
|
||||||
|
`target` is how instance multiplicity stays out of the namespace. Ten USB
|
||||||
|
sticks and the namespace still holds exactly one name, `/protocol/block`: a
|
||||||
|
channel to the provider, `enumerate` lists the current volumes as targets, a
|
||||||
|
`targets_changed` signal announces hotplug, and a read names its volume in
|
||||||
|
`target`. The unix `/dev/sda`,`/dev/sdb` problem is dissolved, not renamed.
|
||||||
|
|
||||||
|
If a future transport genuinely routes between machines, *it* carries real
|
||||||
|
source/destination addressing internally at L0 — the way IP runs under TCP —
|
||||||
|
and none of it surfaces into the packet header. Protocols stay ignorant of
|
||||||
|
distance.
|
||||||
|
|
||||||
|
## The transport (L0): a buffer and a doorbell
|
||||||
|
|
||||||
|
Strip any transport to its skeleton and the same two parts remain:
|
||||||
|
|
||||||
|
| Transport | Buffer | Doorbell | Status |
|
||||||
|
|---|---|---|---|
|
||||||
|
| **kernel-ipc** | kernel-owned mailbox (the `Endpoint`) | the scheduler (rendezvous wake) | the first transport — [ipc.md](../device-driver-development/ipc.md) |
|
||||||
|
| **shm-ring** | user-owned shared-memory ring | a signal | exists ad hoc (display bulk); to be formalized — the unlock for the 256-byte ceiling |
|
||||||
|
| network | NIC queue | an interrupt | someday, when danos networks |
|
||||||
|
|
||||||
|
Transports differ in their **properties**, which the channel layer exposes and
|
||||||
|
the protocol layer may depend on:
|
||||||
|
|
||||||
|
- **packet ceiling** — kernel-ipc: 256 bytes request/reply, 64 pushed. An
|
||||||
|
shm-ring's ceiling is its slot size. Kernel-ipc's 256 is the *floor* every
|
||||||
|
protocol may assume everywhere.
|
||||||
|
- **synchrony** — kernel-ipc's call is a rendezvous: natural backpressure, no
|
||||||
|
queue to size. An asynchronous transport buffers, so a channel over one
|
||||||
|
needs explicit flow control. Backpressure is a *transport property*, not a
|
||||||
|
channel guarantee — protocols that rely on it say so.
|
||||||
|
- **droppability** — pushed event packets may drop when a ring fills;
|
||||||
|
request/reply may not.
|
||||||
|
- **capability carriage** — **only kernel-ipc can move a capability.**
|
||||||
|
Handles are kernel objects; a user-space ring cannot transfer one. So
|
||||||
|
kernel-ipc is always the *establishment and control* transport — channels
|
||||||
|
are born on it, capabilities ride it — even when a channel's data is
|
||||||
|
negotiated onto something fatter.
|
||||||
|
|
||||||
|
That negotiation is the upgrade path: a channel starts on kernel-ipc; the
|
||||||
|
protocol's handshake may then delegate a shared-memory region (as a
|
||||||
|
capability, over kernel-ipc) and move its bulk traffic there. The display
|
||||||
|
path already does exactly this by hand; formalizing it in the channel layer
|
||||||
|
makes it every protocol's option.
|
||||||
|
|
||||||
|
## The channel (L1)
|
||||||
|
|
||||||
|
A channel has two ends, and **the ends are peers**: each may send packets,
|
||||||
|
each may receive, each may signal. Request/reply is a *pattern* over the
|
||||||
|
channel — a send with a correlated receive, which the kernel-ipc transport
|
||||||
|
happens to accelerate as a single rendezvous — not the definition of it. The
|
||||||
|
event stream (subscribe, then pushes) and the change signal (poke, then
|
||||||
|
re-read) are the other two patterns; all three are catalogued in
|
||||||
|
[protocol-namespace.md](protocol-namespace.md)'s wiring section.
|
||||||
|
|
||||||
|
The channel layer's obligations: deliver packets whole, attach the verified
|
||||||
|
source to every receive, expose the transport's properties, and hide the
|
||||||
|
transport's mechanics. The client library's `Channel` type is this layer made
|
||||||
|
concrete — a program holds channels that speak protocols and never touches a
|
||||||
|
raw handle.
|
||||||
|
|
||||||
|
## The protocol (L2) and the namespace (L3)
|
||||||
|
|
||||||
|
A protocol defines its packets through the envelope — every packet begins
|
||||||
|
`{operation, target}`, reserved verbs (`describe`, `enumerate`, `subscribe`,
|
||||||
|
`unsubscribe`) mean the same thing in every protocol, and `Define` checks
|
||||||
|
every packet against the transport floor at compile time. The full treatment,
|
||||||
|
including how names are granted, resolved, and restricted per process, is
|
||||||
|
[protocol-namespace.md](protocol-namespace.md).
|
||||||
|
|
||||||
|
Establishment points are named by contract — `/protocol/display`, never
|
||||||
|
`/protocol/ipc-1` — because the name must outlive the mechanism: a
|
||||||
|
transport named in the namespace could never be swapped, which would defeat
|
||||||
|
this document's premise.
|
||||||
@@ -231,8 +231,8 @@ Two consequences of neutrality bind on later work:
|
|||||||
|
|
||||||
- **Cross-firmware surfaces are named by domain, not firmware.** System power is
|
- **Cross-firmware surfaces are named by domain, not firmware.** System power is
|
||||||
a [`power`](power.md) protocol, not an "ACPI events" protocol: on x86 the acpi
|
a [`power`](power.md) protocol, not an "ACPI events" protocol: on x86 the acpi
|
||||||
service registers it, on ARM a PSCI/mailbox service registers the same
|
service binds it, on ARM a PSCI/mailbox service binds the same
|
||||||
`ServiceId.power`, and subscribers never learn the difference.
|
`/protocol/power`, and subscribers never learn the difference.
|
||||||
- **Identity must widen before the fdt service exists.** `DeviceDescriptor`'s
|
- **Identity must widen before the fdt service exists.** `DeviceDescriptor`'s
|
||||||
8-byte `hid` holds an EISA id but cannot hold an FDT `compatible` string
|
8-byte `hid` holds an EISA id but cannot hold an FDT `compatible` string
|
||||||
(`"brcm,bcm2835-aux-uart"`); the identity field grows before the ARM path can
|
(`"brcm,bcm2835-aux-uart"`); the identity field grows before the ARM path can
|
||||||
|
|||||||
@@ -15,9 +15,9 @@ Where the events come from is firmware-specific — on x86 they ride the ACPI SC
|
|||||||
([acpi.md](acpi.md)); on a Raspberry Pi they would come from PSCI or a mailbox.
|
([acpi.md](acpi.md)); on a Raspberry Pi they would come from PSCI or a mailbox.
|
||||||
What subscribers want is not: *the lid closed* means the same thing regardless of
|
What subscribers want is not: *the lid closed* means the same thing regardless of
|
||||||
who noticed. So the surface is **domain-named**. There is a `power-protocol`
|
who noticed. So the surface is **domain-named**. There is a `power-protocol`
|
||||||
module and a well-known `ServiceId.power = 5`; on x86 the **acpi service**
|
module and a contract named `/protocol/power`; on x86 the **acpi service**
|
||||||
registers it, and on ARM a PSCI/mailbox service will register the *same* id.
|
binds it, and on ARM a PSCI/mailbox service will bind the *same* name.
|
||||||
Subscribers call `ipc.lookup(.power)` and never learn which firmware they
|
Subscribers open `/protocol/power` and never learn which firmware they
|
||||||
are on — the neutrality the whole [discovery](discovery.md) migration exists to
|
are on — the neutrality the whole [discovery](discovery.md) migration exists to
|
||||||
preserve, carried one layer up into a running-system surface.
|
preserve, carried one layer up into a running-system surface.
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,474 @@
|
|||||||
|
# The protocol namespace
|
||||||
|
|
||||||
|
*Design, agreed 2026-07-31. Supersedes the `ServiceId` registry. P1–P3 of the
|
||||||
|
migration plan at the end have landed (the envelope, the registry and the
|
||||||
|
`ServiceId` flag-day, and restriction stage one); P4 and P5 are the remaining
|
||||||
|
work list.*
|
||||||
|
|
||||||
|
How a program finds, connects to, and is restricted from the things it talks to.
|
||||||
|
Three ideas, kept deliberately separate:
|
||||||
|
|
||||||
|
1. **Naming** — a path under `/protocol` names a *contract*, not a service.
|
||||||
|
2. **Access** — resolving that path yields an endpoint *capability*; what a process
|
||||||
|
cannot resolve, it cannot reach.
|
||||||
|
3. **Transport** — unchanged: packets over channels, moved by whichever
|
||||||
|
transport the channel rides (kernel-ipc first).
|
||||||
|
This document is layers **L3** (the namespace) and **L2** (the protocol
|
||||||
|
and its envelope) of the communication stack;
|
||||||
|
[communication.md](communication.md) owns the model and the vocabulary
|
||||||
|
(*protocol* the language, *channel* the conversation, *packet* the
|
||||||
|
transmitted unit, *signal* the payload-less poke, *transport* the
|
||||||
|
replaceable mechanism), and
|
||||||
|
[ipc.md](../device-driver-development/ipc.md) is the first transport.
|
||||||
|
|
||||||
|
## Why ServiceId has to go
|
||||||
|
|
||||||
|
Today a service calls `ipc_register(service_id, endpoint)` and a client calls
|
||||||
|
`ipc_lookup(service_id)`, where `ServiceId` is a compile-time enum in `abi.zig`
|
||||||
|
backed by a flat 16-slot table in the kernel. Three defects, in rising order:
|
||||||
|
|
||||||
|
- **Static.** The id space is baked into the ABI at compile time. A third-party
|
||||||
|
program can never introduce a service; the one place danos is *less* dynamic
|
||||||
|
than its own design.
|
||||||
|
- **Ungated.** `ipc_register` is callable by any process and *replaces* an
|
||||||
|
existing registration. Any process can hijack `.fat` or `.display` and
|
||||||
|
impersonate it. `ipc_lookup` is equally ambient.
|
||||||
|
- **Unrestrictable.** Because lookup is a syscall available to everyone, there is
|
||||||
|
no point at which "this process may not talk to the display" can be enforced.
|
||||||
|
Any future file-access restriction would be bypassable by speaking to the FAT
|
||||||
|
server directly.
|
||||||
|
|
||||||
|
## Naming: contracts, not services
|
||||||
|
|
||||||
|
`/protocol/<name>` names a protocol — the contract a conversation follows — and
|
||||||
|
resolving it connects you to whatever process currently provides that contract.
|
||||||
|
The client never cared *which* binary answers; it cares that its messages are
|
||||||
|
understood. Naming the contract makes that explicit, and buys:
|
||||||
|
|
||||||
|
- **Swappable providers.** Replace the display server; `/protocol/display`
|
||||||
|
routes to the new one; clients notice nothing.
|
||||||
|
- **Test fakes.** Spawn a program whose namespace wires `/protocol/display` to a
|
||||||
|
mock. The name promises the protocol; the mock speaks it.
|
||||||
|
- **One vocabulary.** The names mirror `library/protocol/`: a program imports
|
||||||
|
the `display-protocol` module, then opens `/protocol/display`. What you
|
||||||
|
compiled against and what you ask the namespace for are the same word.
|
||||||
|
|
||||||
|
A leaf names one contract — kebab-case, full words, matching the
|
||||||
|
`library/protocol/` module that defines its wire format — and related
|
||||||
|
contracts group into directories: `/protocol/networking/ip`,
|
||||||
|
`/protocol/networking/bluetooth`. Directories organize *contracts only*;
|
||||||
|
they never encode addressing (see below), so a directory appears because a
|
||||||
|
domain has several contracts, never because hardware multiplied. The module
|
||||||
|
tree mirrors the namespace (`library/protocol/networking/ip` ↔
|
||||||
|
`/protocol/networking/ip`), and registrar grants scope naturally to subtrees
|
||||||
|
— an application installed at `/applications/foo` can be granted
|
||||||
|
`/protocol/applications/foo/...` and nothing above it. `/protocol` is
|
||||||
|
top level, beside `/system` and `/applications`, because the boundary it names
|
||||||
|
is spoken on both sides: applications talk to protocols as much as the OS does
|
||||||
|
(see [file-system-hierarchy.md](../file-system-development/file-system-hierarchy.md)).
|
||||||
|
|
||||||
|
**Addressing lives inside the protocol, never in the path.** Which volume, which
|
||||||
|
layer, which input device — that is a destination field in the messages, the way
|
||||||
|
TCP carries a destination address, and the way danos protocols already work (the
|
||||||
|
display protocol multiplexes layer ids; the vfs protocol addresses node ids).
|
||||||
|
The namespace answers exactly one question — *may this process speak this
|
||||||
|
protocol at all* — so `/protocol/block` is one name no matter how many disks are
|
||||||
|
attached. The source address is never in the message either: it is the IPC
|
||||||
|
badge, stamped by the kernel per message, unforgeable — a property TCP's source
|
||||||
|
address does not have.
|
||||||
|
|
||||||
|
`/system/devices` (the device inventory) stays purely informational: facts for
|
||||||
|
diagnosis, never a routing mechanism. Unix conflated the two in `/dev`; danos
|
||||||
|
does not. You *read about* hardware in `/system/devices`; you *talk to* it
|
||||||
|
through `/protocol`.
|
||||||
|
|
||||||
|
## Resolution: a protocol node in the VFS
|
||||||
|
|
||||||
|
The kernel VFS router already does the hard part: `fs_resolve` matches a mount
|
||||||
|
prefix and installs the backend's endpoint capability in the caller's handle
|
||||||
|
table. The registry is just a backend mounted at `/protocol` — ring 3, like FAT.
|
||||||
|
Connecting is a normal vfs-protocol `open` with one twist in the reply:
|
||||||
|
|
||||||
|
```
|
||||||
|
client kernel router registry backend
|
||||||
|
│ fs_resolve("/protocol/display") │
|
||||||
|
│──────────────────────────▶│ prefix match: /protocol │
|
||||||
|
│◀── registry endpoint ─────│ (capability installed) │
|
||||||
|
│ vfs open("display") ──────────────────────────────────────▶│
|
||||||
|
│◀───────────────── Reply + capability = provider endpoint ──│
|
||||||
|
│ ipc_call(provider, display-protocol messages...) │
|
||||||
|
```
|
||||||
|
|
||||||
|
Both capability moves use machinery the kernel already has: request-direction
|
||||||
|
and reply-direction `send_cap` on `call`/`replyWait`. The vfs protocol needs two
|
||||||
|
additions, both append-only:
|
||||||
|
|
||||||
|
- `NodeKind.protocol` — a node that names a contract; its `open` establishes
|
||||||
|
a **channel** (delivered as an endpoint capability) instead of returning a
|
||||||
|
file id. The node is the protocol, the channel is the conversation, and the
|
||||||
|
addressing inside the packets decides where within the provider each one
|
||||||
|
lands. `readdir` over `/protocol` lists protocol nodes like any others, so
|
||||||
|
the tree stays browsable for diagnosis.
|
||||||
|
- The convention that an `open` reply may carry a capability. File backends
|
||||||
|
(FAT) never use it; synthetic backends (the registry, later the device
|
||||||
|
inventory) do.
|
||||||
|
|
||||||
|
The path lookup happens once, at connect time. The hot path — `ipc_call` on the
|
||||||
|
cached endpoint — is untouched. A provider crash turns the cached endpoint dead
|
||||||
|
(`-EPEER`), and the client's recovery is to re-resolve: the restart story falls
|
||||||
|
out of the naming layer for free.
|
||||||
|
|
||||||
|
## Registration: the registrar, held by init
|
||||||
|
|
||||||
|
The registry backend is **init**. It is already PID 1, already spawns every
|
||||||
|
service from its manifest, and already holds the supervision link to each — it
|
||||||
|
is the process that *knows* which binary is which. (If init grows
|
||||||
|
uncomfortable, the same design lifts into a dedicated registry service that
|
||||||
|
init spawns first and delegates to; nothing below changes.)
|
||||||
|
|
||||||
|
- **Binding.** A service creates its endpoint and sends the registry a `bind`
|
||||||
|
request with the protocol name as payload and the endpoint attached as the
|
||||||
|
call's capability.
|
||||||
|
- **Authorization.** Init's manifest gains a column: the protocols each spawned
|
||||||
|
binary may bind. A `bind` from any process not granted that name is refused
|
||||||
|
(`-EPERM`) — the badge identifies the caller, the supervision records map
|
||||||
|
badge to binary. This is the registrar authority; it never leaves init.
|
||||||
|
- **Collision is an error.** A name already bound refuses a second bind — never
|
||||||
|
last-writer-wins. When a provider dies, init (its supervisor) unbinds its
|
||||||
|
names; the restarted instance binds again.
|
||||||
|
- **Provenance.** The registry records name → task id → binary path, so a
|
||||||
|
diagnostic listing answers "who serves this?" at a glance:
|
||||||
|
|
||||||
|
```
|
||||||
|
/protocol/display pid 12 /system/services/display
|
||||||
|
/protocol/input pid 7 /system/services/input
|
||||||
|
```
|
||||||
|
|
||||||
|
`ipc_register` and `ipc_lookup` retire; the `ServiceId` enum leaves `abi.zig`.
|
||||||
|
The kernel keeps one residual rule: `/protocol` becomes a reserved prefix like
|
||||||
|
`/system` — `fs_mount` refuses to shadow it, and init's boot-time mount is the
|
||||||
|
only one it will ever hold. (Full gating of `fs_mount` is a separate item on
|
||||||
|
the security track; the reserved prefix closes the hole for this namespace
|
||||||
|
without waiting for it.)
|
||||||
|
|
||||||
|
## Restriction: per-process namespaces, not ACLs
|
||||||
|
|
||||||
|
danos has no users and no principals, deliberately. Restriction is therefore
|
||||||
|
**delegation**: what a process may open is decided by whoever spawned it, and
|
||||||
|
enforcement is absence — a protocol you cannot resolve does not exist for you.
|
||||||
|
"Permission denied" and "not found" are the same answer, which is the same
|
||||||
|
discipline the device layer already follows: the claim is the capability; here,
|
||||||
|
the resolvable name is the capability.
|
||||||
|
|
||||||
|
Two stages, deliberately ordered so the useful half lands first:
|
||||||
|
|
||||||
|
**Stage one — the registry filters by badge.** Init is both the spawner and the
|
||||||
|
registry, so its manifest already knows which binary may *open* which protocols
|
||||||
|
(a second manifest column, beside the bind grants). An `open` from a process
|
||||||
|
whose binary is not granted that protocol is refused. No new kernel mechanism
|
||||||
|
at all; the display driver's view can be narrowed to nothing, a future
|
||||||
|
downloaded application's to `display` and `input`, today.
|
||||||
|
|
||||||
|
**Stage two — spawn passes the namespace.** `spawn` gains an initial
|
||||||
|
capability: the child's connection to *its* registry view, chosen by the
|
||||||
|
spawner. A newly spawned process starts with an empty handle table and this one
|
||||||
|
handle — its world is whatever its parent wired in. This removes the last
|
||||||
|
ambient reach (`fs_resolve` finding `/protocol` globally), lets any supervisor
|
||||||
|
— not just init — narrow or fake a child's view (an application launcher
|
||||||
|
granting an app only what its manifest declares; a test harness substituting
|
||||||
|
every provider), and composes down the supervision tree. Stage one's manifest
|
||||||
|
column becomes the *content* of the view init builds, so nothing is thrown
|
||||||
|
away.
|
||||||
|
|
||||||
|
### A worked example: the microphone prompt
|
||||||
|
|
||||||
|
The scenario stage two exists for: an application opens
|
||||||
|
`/protocol/audio-input`, and the user should be asked. The supervisor is an
|
||||||
|
ordinary user process — an application launcher — and the flow needs no new
|
||||||
|
security concepts:
|
||||||
|
|
||||||
|
1. The launcher spawned the app with a namespace channel that terminates at
|
||||||
|
**the launcher itself**. The app's whole world is a conversation with its
|
||||||
|
supervisor.
|
||||||
|
2. The app's `open("audio-input")` packet lands in the launcher,
|
||||||
|
badge-stamped. The launcher spawned the app, so badge → binary path
|
||||||
|
(`/applications/foo`) is its own supervision record — "remember my choice"
|
||||||
|
needs no identity system.
|
||||||
|
3. Grant unknown → the launcher parks the request and shows a prompt (it is a
|
||||||
|
user process with display access; init never does UI). Blocking an open on
|
||||||
|
a human is architecturally fine: opens are connect-time, never hot-path.
|
||||||
|
4. **Yes** → the launcher opens `/protocol/audio-input` in *its own*
|
||||||
|
namespace and attaches the resulting channel to the parked reply. The app
|
||||||
|
cannot tell a prompt happened — a consented open is indistinguishable from
|
||||||
|
a direct one, merely slower.
|
||||||
|
5. **No** → refuse the open, indistinguishable from "no such protocol" — or
|
||||||
|
hand the app a **fake**: a silence-generating provider. The test-fake
|
||||||
|
mechanism doubles as a privacy feature.
|
||||||
|
|
||||||
|
The capability discipline holds throughout: the launcher can only grant what
|
||||||
|
it holds — if init never gave the launcher `audio-input`, no prompt can
|
||||||
|
conjure it. Consent is delegation flowing down the supervision tree, never a
|
||||||
|
global ACL edit. And the provider still sees the app's badge on every packet,
|
||||||
|
so a coarser second check at the audio service remains possible.
|
||||||
|
|
||||||
|
Two mechanical requirements this scenario pins on stage two:
|
||||||
|
|
||||||
|
- **Parked replies.** A prompt takes seconds, and the service loop holds one
|
||||||
|
outstanding reply today — the launcher must park request A, keep serving B
|
||||||
|
and C, and reply to A later (by badge). The kernel already tracks owed
|
||||||
|
replies (that is how death delivers `-EPEER`); multiple parked replies is
|
||||||
|
the extension, in the harness and, if needed, the kernel.
|
||||||
|
- **Granted channels are dedicated, hence revocable.** Once the app holds a
|
||||||
|
channel capability, nobody reaches into its handle table — so a
|
||||||
|
prompt-granted channel must be one that can be *killed*: a dedicated
|
||||||
|
endpoint pair (or per-client session at the provider) whose death turns
|
||||||
|
the app's capability into `-EPEER`. Revoking microphone access is then
|
||||||
|
killing that channel, using machinery that already exists.
|
||||||
|
|
||||||
|
One adjacent problem, named and deferred: **trusted UI**. The prompt is only
|
||||||
|
meaningful if the app cannot draw a convincing fake or overlay the real one —
|
||||||
|
a display-layer question (a reserved surface for the supervisor chain), owned
|
||||||
|
by the display track, not this one.
|
||||||
|
|
||||||
|
Fine-grained restriction *within* a protocol (this process may use volume A but
|
||||||
|
not volume B) is not the namespace's job. The capability-shaped answer, when it
|
||||||
|
is needed: the supervisor pre-opens a connection scoped to one target and passes
|
||||||
|
that connection to the child, which never opens `/protocol/block` at all.
|
||||||
|
Delegation again, not ACLs.
|
||||||
|
|
||||||
|
## The envelope: one addressing scheme for every protocol
|
||||||
|
|
||||||
|
Every protocol module today hand-rolls its `Request`/`Reply` with an
|
||||||
|
`operation` first field. That convention becomes a library, so addressing is
|
||||||
|
uniform and the rules are enforced by construction rather than by review. New
|
||||||
|
module: **`library/protocol/envelope`** (the one protocol-layer module that is
|
||||||
|
not itself a protocol).
|
||||||
|
|
||||||
|
```zig
|
||||||
|
/// Every packet a danos protocol transmits begins with this header.
|
||||||
|
pub const Header = extern struct {
|
||||||
|
operation: u32, // the verb; values 0..15 are reserved universal verbs
|
||||||
|
_padding: u32 = 0,
|
||||||
|
/// Object addressing, never party addressing: which of the peer's
|
||||||
|
/// objects this packet operates on — a volume, layer, node, device.
|
||||||
|
/// 0 addresses the provider itself. Parties are addressed by the
|
||||||
|
/// channel; the protocol defines target's meaning; the field's place
|
||||||
|
/// and width are universal.
|
||||||
|
target: u64 = 0,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Reserved verbs, answered by every provider.
|
||||||
|
pub const operation_describe: u32 = 0; // -> protocol name, version, target kinds
|
||||||
|
pub const operation_enumerate: u32 = 1; // -> the current targets, one per reply page
|
||||||
|
pub const operation_subscribe: u32 = 2; // capability = the subscriber's endpoint
|
||||||
|
pub const operation_unsubscribe: u32 = 3;
|
||||||
|
pub const first_protocol_operation: u32 = 16;
|
||||||
|
|
||||||
|
/// Every reply begins with this.
|
||||||
|
pub const Status = extern struct {
|
||||||
|
status: i32, // 0 or a negative errno
|
||||||
|
_padding: u32 = 0,
|
||||||
|
len: u32 = 0, // payload bytes following the header
|
||||||
|
_padding2: u32 = 0,
|
||||||
|
};
|
||||||
|
```
|
||||||
|
|
||||||
|
A protocol is then *defined through* the envelope, not beside it:
|
||||||
|
|
||||||
|
```zig
|
||||||
|
pub const Protocol = envelope.Define(.{
|
||||||
|
.name = "display",
|
||||||
|
.version = 1,
|
||||||
|
.operations = &.{
|
||||||
|
.{ .name = "configure_layer", .request = ConfigureLayer, .reply = void },
|
||||||
|
.{ .name = "blit", .request = Blit, .reply = void },
|
||||||
|
...
|
||||||
|
},
|
||||||
|
});
|
||||||
|
```
|
||||||
|
|
||||||
|
`Define` is comptime and is where the enforcement lives:
|
||||||
|
|
||||||
|
- verbs are numbered automatically from `first_protocol_operation`, so no
|
||||||
|
protocol can collide with the reserved range;
|
||||||
|
- every packet is size-checked at compile time against the kernel-ipc floor
|
||||||
|
— `packet_maximum` (256) for request/reply, `post_maximum` (64) for event
|
||||||
|
packets. Ceilings are transport properties
|
||||||
|
([communication.md](communication.md)); the floor is what every protocol
|
||||||
|
may assume on any transport. The errors that today surface as runtime
|
||||||
|
truncation become compile errors, and packets-never-fragment is enforced
|
||||||
|
at the source;
|
||||||
|
- the generated type carries encode/decode helpers and a provider-side dispatch
|
||||||
|
table, so a provider answers `describe` automatically and unknown operations
|
||||||
|
with `-ENOSYS` uniformly;
|
||||||
|
- the service harness (`library/kernel/service.zig`) accepts the generated
|
||||||
|
dispatch type, which is what makes the envelope *enforced*: a protocol that
|
||||||
|
bypasses `Define` does not plug into the harness.
|
||||||
|
|
||||||
|
Universal conventions that ride on the reserved verbs:
|
||||||
|
|
||||||
|
- **`describe`** is the version handshake. Version lives in the handshake, not
|
||||||
|
in every message — the 256-byte budget is too small to spend per call.
|
||||||
|
- **`enumerate`** is how multi-target protocols expose their targets, and the
|
||||||
|
standard `targets_changed` notification (a notify bit) tells subscribers to
|
||||||
|
re-enumerate — arrival and removal of volumes, layers, devices all take the
|
||||||
|
same shape. Hotplug fits the notification ring far better than a filesystem
|
||||||
|
tree ever did.
|
||||||
|
- **Source is the badge.** No protocol defines a "sender" field; the kernel's
|
||||||
|
per-message badge is the only source identity, and providers key per-client
|
||||||
|
state on it.
|
||||||
|
|
||||||
|
### Paths resolve once; integers do the work
|
||||||
|
|
||||||
|
A rule the envelope makes official: **a path appears in a conversation at most
|
||||||
|
once — at resolve or open — and everything after it addresses integers.** The
|
||||||
|
namespace resolves `/protocol/display` to an endpoint; a backend's `open`
|
||||||
|
resolves a path payload to a node id; from then on every packet carries the
|
||||||
|
integer in `target`. Integers compare in one instruction and fit the fixed
|
||||||
|
header, and the 256-byte message budget never re-carries path strings on the
|
||||||
|
hot path. This is already the system's shape — vfs node ids, display layer ids
|
||||||
|
— and the envelope pins it as the required shape for every protocol.
|
||||||
|
|
||||||
|
Two integer identities, not to be confused:
|
||||||
|
|
||||||
|
- **An open handle** — what vfs `open` returns today: transient, meaningful
|
||||||
|
only within one client's session with one provider, swept when the client
|
||||||
|
exits. Cheap, and all a protocol usually needs. Handles must be **scoped per
|
||||||
|
client** — validated against the badge, or drawn from a per-client id
|
||||||
|
namespace. (Today the FAT server's node ids are guessable small integers
|
||||||
|
honoured across clients; that hole closes with this rule.)
|
||||||
|
- **A persistent node identity** — a unix inode number, stable across opens
|
||||||
|
and renames. danos deliberately does not promise this, because FAT cannot
|
||||||
|
deliver it: a FAT file's identity is its directory entry, and rename or
|
||||||
|
truncation moves every candidate anchor. If a future filesystem or a cache
|
||||||
|
layer needs stable identity, that is the backend's promise to make, never
|
||||||
|
the protocol's assumption.
|
||||||
|
|
||||||
|
The five existing protocol modules (`vfs`, `display`, `input`, `power`,
|
||||||
|
`block`, plus `scanout`, `usb-transfer`, `device-manager`) rebase onto the
|
||||||
|
envelope during the migration flag-day. `input-protocol`'s subscribe/publish
|
||||||
|
split and `vfs-protocol`'s node addressing both map cleanly (`node` and layer
|
||||||
|
ids become `target`).
|
||||||
|
|
||||||
|
## Wiring: how conversations flow
|
||||||
|
|
||||||
|
The patterns below are channel-layer (L1) shapes; the delivery mechanics are
|
||||||
|
the kernel-ipc transport's, described here because it is the transport every
|
||||||
|
channel starts on. Kernel-ipc provides exactly three delivery shapes, and
|
||||||
|
every one is unicast. An endpoint is a mailbox owned by one process — its
|
||||||
|
creator receives; anyone holding its capability sends into it. That direction
|
||||||
|
never reverses:
|
||||||
|
|
||||||
|
1. **Synchronous call** — request/reply. The kernel parks the caller and
|
||||||
|
`replyWait` delivers the reply straight back, so the provider answers
|
||||||
|
without holding any capability to the client. Badge-stamped, blocking, and
|
||||||
|
the *only* shape that carries capabilities (in the request, and in the
|
||||||
|
reply — which is how a reverse path is bootstrapped).
|
||||||
|
2. **Asynchronous send** — an event packet pushed into the receiver's post
|
||||||
|
ring, at most `post_maximum` (64) bytes, no reply owed, never blocks the
|
||||||
|
sender. Strictly one-way: to be pushed to, you must first hand the pusher
|
||||||
|
your endpoint.
|
||||||
|
3. **Signals** — payload-less notification bits, below the packet layer,
|
||||||
|
coalescing: "something changed, come look."
|
||||||
|
|
||||||
|
A bidirectional link is therefore always **a pair of endpoints**, one per
|
||||||
|
direction, each delivered by cap-passing. Three conversation patterns are
|
||||||
|
built from these, and the envelope names all three:
|
||||||
|
|
||||||
|
- **Request/response** — the synchronous call. The default, and the only
|
||||||
|
place capabilities move.
|
||||||
|
- **Event stream** — `subscribe` (a synchronous call whose attached
|
||||||
|
capability is the subscriber's own endpoint), after which the provider
|
||||||
|
pushes events asynchronously; `unsubscribe` or subscriber exit ends it.
|
||||||
|
Listened-to, not blocked-on.
|
||||||
|
- **Change signal** — a signal plus re-read: `targets_changed` →
|
||||||
|
`enumerate`. For state whose truth lives with the provider.
|
||||||
|
|
||||||
|
**Broadcast is a provider pattern, never a kernel primitive.** The kernel
|
||||||
|
does not know subscriber sets — a service does. The input service is the
|
||||||
|
model: sources *publish* (a unicast call to the service), the service
|
||||||
|
*broadcasts* (a fan-out loop of asynchronous sends over its subscriber list,
|
||||||
|
so one dead subscriber can never stall the rest). One fan-out point per event
|
||||||
|
domain, owned by the service that defines the event.
|
||||||
|
|
||||||
|
The harness owns the machinery: the subscriber table, the dead-subscriber
|
||||||
|
sweep (via process-exit notifications), and the fan-out loop — all written by
|
||||||
|
hand in `input.zig` today, lifted into the service harness so every protocol
|
||||||
|
gets identical semantics. `Define` declares a protocol's events (`.events`),
|
||||||
|
and each event type is checked against `post_maximum` at compile time,
|
||||||
|
generalizing the assert `input-protocol` already carries.
|
||||||
|
|
||||||
|
**Event packets are droppable.** A slow subscriber's ring fills, and the
|
||||||
|
provider must not block on it — so an event stream is a hint or a coalescing
|
||||||
|
signal, never a ledger. Anything that must not be lost is either re-readable
|
||||||
|
state (the change-signal pattern) or bulk data in shared memory with a
|
||||||
|
packet as the doorbell, which is how the display path already works — the
|
||||||
|
packets-never-fragment rule and this one are the same rule seen from two
|
||||||
|
sides.
|
||||||
|
|
||||||
|
**Source direction (open point).** Today event sources are *clients*: an
|
||||||
|
input driver resolves `/protocol/input` and delivers each event as a
|
||||||
|
synchronous `publish` call — one capability, obtained by resolution, covers
|
||||||
|
everything, and the badge tells the service exactly who each event came from.
|
||||||
|
The inversion — the service subscribing to each driver — would require every
|
||||||
|
driver to be individually discoverable and its endpoint ferried to the
|
||||||
|
service, machinery whose payoff (the service choosing its sources) the
|
||||||
|
namespace already provides more cheaply: only a process granted open on
|
||||||
|
`/protocol/input` can publish into it. Sources stay clients for now;
|
||||||
|
revisited at restriction stage two, when a supervisor can wire capabilities
|
||||||
|
at spawn time.
|
||||||
|
|
||||||
|
## What this deliberately does not solve
|
||||||
|
|
||||||
|
The wider security track, for which this namespace is the foundation, not the
|
||||||
|
whole:
|
||||||
|
|
||||||
|
- **File access restriction** — the point of the exercise. The same stage-two
|
||||||
|
namespace mechanism extends from protocol names to file paths: the spawner
|
||||||
|
decides which subtrees resolve. Designed separately once this lands.
|
||||||
|
- `fs_mount` gating beyond the reserved prefixes; `system_spawn` gating;
|
||||||
|
`klog_read` being world-readable; backends checking the badge on per-node
|
||||||
|
operations (the FAT server honours node ids across clients today).
|
||||||
|
- Kernel hardening items already noted in-tree: SMEP/SMAP and SYSRET
|
||||||
|
canonical-RIP, now designed in [smep-smap.md](smep-smap.md).
|
||||||
|
- Pipes/FIFOs for the POSIX layer — a byte-stream object *beside* message IPC,
|
||||||
|
wanted by the Python track, unrelated to naming.
|
||||||
|
- **Trusted UI** — a permission prompt an application cannot fake or overlay
|
||||||
|
(see the microphone example). A display-track concern: the supervisor chain
|
||||||
|
needs a reserved surface.
|
||||||
|
|
||||||
|
## Migration plan
|
||||||
|
|
||||||
|
Flag-day per phase, in the style of the DMA-capability conversion — no
|
||||||
|
dual-stack periods, the QEMU suite green at each phase boundary.
|
||||||
|
|
||||||
|
**P1 — mechanics, no behavior change.** The `envelope` module with its comptime
|
||||||
|
`Define`, unit tests; `NodeKind.protocol` and the open-reply-capability
|
||||||
|
convention in `vfs-protocol`; existing protocols untouched.
|
||||||
|
|
||||||
|
**P2 — the registry.** Init serves `/protocol` (bind with manifest
|
||||||
|
authorization, collision refusal, unbind on provider death, provenance);
|
||||||
|
kernel reserves the `/protocol` prefix; every service converts from
|
||||||
|
`ipc_register` to `bind`, every client from `ipc_lookup` to resolve-and-open;
|
||||||
|
`ServiceId`, `ipc_register`, `ipc_lookup` deleted. Tests: unauthorized bind
|
||||||
|
refused, collision refused, provider restart re-binds and a client re-resolves.
|
||||||
|
|
||||||
|
**P3 — restriction, stage one.** The open-grant column in init's manifest;
|
||||||
|
registry refuses ungranted opens. Test: a fixture process denied a protocol its
|
||||||
|
neighbour is granted.
|
||||||
|
|
||||||
|
**P4 — protocol rebase.** Existing protocol modules re-expressed through
|
||||||
|
`Define`; providers move onto the generated dispatch; `describe`/`enumerate`
|
||||||
|
answered everywhere; the conformance test fixture exercises the reserved verbs
|
||||||
|
against every registered provider.
|
||||||
|
|
||||||
|
**P5 — restriction, stage two.** Spawn's initial capability; namespace views
|
||||||
|
built by the spawner; ambient resolution of `/protocol` retired. Includes the
|
||||||
|
two requirements the microphone example pins: **parked replies** (a
|
||||||
|
supervisor parks an open, keeps serving, replies later by badge) and
|
||||||
|
**dedicated, killable granted channels** (revocation = channel death →
|
||||||
|
`-EPEER`). Scoped separately — it touches `spawn`, the loader contract, and
|
||||||
|
every supervisor — and lands together with the file-path half of namespacing.
|
||||||
|
|
||||||
|
The unix-path migration ([file-system-hierarchy.md](../file-system-development/file-system-hierarchy.md#migration))
|
||||||
|
is independent of P1–P5 and can land before or after.
|
||||||
@@ -0,0 +1,143 @@
|
|||||||
|
# SMEP and SMAP — supervisor-mode hardening
|
||||||
|
|
||||||
|
*Design, 2026-07-31. Not yet implemented. Companion to
|
||||||
|
[protocol-namespace.md](protocol-namespace.md) on the security track — this is
|
||||||
|
the hardware half; that is the namespace half.*
|
||||||
|
|
||||||
|
Two CR4 bits that make the CPU refuse the two things a kernel should never do
|
||||||
|
with user memory:
|
||||||
|
|
||||||
|
- **SMEP** (Supervisor Mode Execution Prevention, CR4 bit 20): instruction
|
||||||
|
fetch in ring 0 from a page whose U/S bit says *user* → #PF. Kills the
|
||||||
|
classic ret2usr exploit shape — a kernel bug that redirects control flow
|
||||||
|
can no longer land in attacker-prepared user code.
|
||||||
|
- **SMAP** (Supervisor Mode Access Prevention, CR4 bit 21): data read/write
|
||||||
|
in ring 0 to a user page → #PF, unless `EFLAGS.AC` is set. `stac`/`clac`
|
||||||
|
open and close deliberate access windows; danos's design needs no windows
|
||||||
|
at all (below).
|
||||||
|
|
||||||
|
Detection is CPUID leaf 7, subleaf 0, EBX bit 7 (SMEP) and bit 20 (SMAP).
|
||||||
|
Both bits are per-core state: the BSP and every AP must set them.
|
||||||
|
|
||||||
|
## Why, in danos terms
|
||||||
|
|
||||||
|
Every syscall argument is an attacker-controlled integer, and several take
|
||||||
|
pointers. A kernel bug that dereferences a crafted pointer reads, writes, or
|
||||||
|
executes memory of the attacker's choosing — the exact bug class the
|
||||||
|
isolation tracks exist to prevent. SMEP/SMAP turn that class from "silent
|
||||||
|
compromise" into "immediate, attributable #PF with a kernel RIP in the log."
|
||||||
|
|
||||||
|
The second benefit matters as much as the first: **SMAP is a permanent
|
||||||
|
tripwire.** Once it is on, any *future* syscall that touches user memory
|
||||||
|
directly — instead of going through the checked copy layer — faults the
|
||||||
|
first time the QEMU suite runs it. The discipline stops depending on review.
|
||||||
|
|
||||||
|
## Where danos already stands
|
||||||
|
|
||||||
|
The design is closer than it looks, because the IPC layer was built right:
|
||||||
|
|
||||||
|
- **The copy layer is already SMAP-proof.** `copyAcross` and `copyFromUser`
|
||||||
|
(`system/kernel/ipc-synchronous.zig:305,333`) never dereference a user
|
||||||
|
virtual address: they walk the page tables and move bytes through the
|
||||||
|
physmap — kernel mappings throughout. SMAP cannot object.
|
||||||
|
- **Syscall entry already clears AC.** `SFMASK = 0x4_0700` clears IF, TF,
|
||||||
|
DF, **AC** on every `syscall`
|
||||||
|
(`system/kernel/architecture/x86_64/per-cpu.zig:76`). The syscall path is
|
||||||
|
SMAP-clean from day one.
|
||||||
|
- **The interrupt path is not.** Hardware does *not* clear AC on IDT
|
||||||
|
delivery, and ring 3 can set AC with `popfq` — so a hostile process could
|
||||||
|
take an interrupt with AC=1 and have the handler run with SMAP suspended.
|
||||||
|
`isr_common` (`system/kernel/architecture/x86_64/isr.s:366`) needs a
|
||||||
|
`clac` beside its `swapgs`.
|
||||||
|
- **CR4 today:** the BSP inherits firmware CR4 (no kernel write anywhere);
|
||||||
|
APs set PAE/OSFXSR/OSXMMEXCPT in `trampoline.s:62-68`. Neither path sets
|
||||||
|
SMEP/SMAP yet, and both must.
|
||||||
|
- **The stragglers.** Nine syscalls still dereference user pointers raw
|
||||||
|
after a bounds check — every one is a SMAP #PF waiting to happen, and
|
||||||
|
every one is *already* a latent kernel fault today (an unmapped-but-in-
|
||||||
|
range user page oopses the kernel instead of failing the call). The
|
||||||
|
verified sweep of `system/kernel/process.zig` (2026-07-31; a
|
||||||
|
whole-kernel `@ptrFromInt` audit found no user-address dereference
|
||||||
|
outside this file):
|
||||||
|
|
||||||
|
| Syscall | Raw access | Direction |
|
||||||
|
|---|---|---|
|
||||||
|
| `system_spawn` | name + argument blob (`:972`, `:980`) | read |
|
||||||
|
| `fs_resolve` | path in (`:1780`), result out (`:1797`) | read + write |
|
||||||
|
| `fs_mount` | prefix + rewrite strings (`:1864`, `:1865`) | read |
|
||||||
|
| `fs_unmount` | prefix string (`:1883`) | read |
|
||||||
|
| `fs_node` | read buffer out (`:1820`) | write |
|
||||||
|
| `debug_write` | message bytes (`:1700`; read twice — memcpy `:1710` and `log.append` `:1717`) | read |
|
||||||
|
| `klog_read` | log bytes out (`:1741`) | write |
|
||||||
|
| `klog_status` | status struct out (`:1758`) | write |
|
||||||
|
| `process_enumerate` | descriptor array out (`:1132`) | write |
|
||||||
|
| `device_enumerate` | descriptor array out (`:388`) | write |
|
||||||
|
|
||||||
|
For the write-direction rows the `@ptrFromInt` is in process.zig but the
|
||||||
|
stores happen in callees (`scheduler.enumerate`
|
||||||
|
`system/kernel/scheduler.zig:1209`, `devices_broker.enumerate`
|
||||||
|
`devices-broker.zig:136`, `log.readAt` `log.zig:209`, the vfs node calls
|
||||||
|
`vfs.zig:257/269/289`) — converting them means bounce buffers plus
|
||||||
|
`copyToUser` around those calls, not just editing the process.zig lines.
|
||||||
|
(Some paths already do it right — the futex word and the device-register
|
||||||
|
descriptor go through `copyFromUser` (`:1087`, `:924`). The write
|
||||||
|
direction has no public helper yet, but the mechanism exists:
|
||||||
|
`copyAcross` with a kernel source is exactly how IPC replies reach user
|
||||||
|
buffers, so `copyToUser` is a mechanical mirror.)
|
||||||
|
|
||||||
|
- **One known gap inside the copy layer itself:** the walk checks presence,
|
||||||
|
not the leaf U/S and writable bits (`ipc-synchronous.zig:20-22` flags
|
||||||
|
this). Today that is nearly moot — the user half contains only mappings
|
||||||
|
the kernel itself created for that process — but it must close before
|
||||||
|
shared or copy-on-write mappings exist, and closing it is part of making
|
||||||
|
the copy layer the single trusted door.
|
||||||
|
|
||||||
|
## The plan
|
||||||
|
|
||||||
|
**H1 — copy discipline (the real work).** A `user-memory` kernel module:
|
||||||
|
`copyFromUser` / `copyToUser` (the missing write direction) via the physmap
|
||||||
|
walk, with U/S and writable leaf checks closing the in-tree TODO. Convert
|
||||||
|
the nine stragglers. This fixes the latent unmapped-page kernel fault on
|
||||||
|
its own — it is worth doing even if SMEP/SMAP never shipped. QEMU suite
|
||||||
|
green; no behavior change visible to correct programs.
|
||||||
|
|
||||||
|
**H2 — SMEP.** A leaf-7 feature probe (the kernel has per-leaf `cpuid`
|
||||||
|
helpers in `apic.zig` to generalize); set CR4.SMEP during per-CPU bring-up
|
||||||
|
on BSP and APs — prefer the Zig-side per-CPU init over the trampoline
|
||||||
|
assembly, so one code path covers every core and the trampoline stays
|
||||||
|
minimal. Audit first that ring 0 never executes user-mapped pages: kernel
|
||||||
|
text lives in the kernel half, `jump_to_user` is kernel code, and the AP
|
||||||
|
trampoline page is kernel-mapped — expected clean, verify before flipping.
|
||||||
|
|
||||||
|
**H3 — SMAP.** Add `clac` at `isr_common` entry. `clac` is #UD on CPUs
|
||||||
|
without SMAP, so the instruction is a 3-byte NOP in the image, patched to
|
||||||
|
`clac` at boot when CPUID advertises SMAP (one-time patch beats a
|
||||||
|
conditional branch in the hottest path in the kernel). Then set CR4.SMAP in
|
||||||
|
the same per-CPU init. From this point the whole QEMU suite doubles as the
|
||||||
|
enforcement test: any missed raw dereference is a vector-14 with a kernel
|
||||||
|
RIP and a user CR2 — loud and attributable.
|
||||||
|
|
||||||
|
**H4 — keep it honest.** A line in the coding standards: kernel code
|
||||||
|
touches user memory only through `user-memory`; there is no `stac` anywhere
|
||||||
|
in the tree, and a PR that adds one is wrong by definition. SMAP enforces
|
||||||
|
the rule mechanically at test time.
|
||||||
|
|
||||||
|
Feature-gating follows the timekeeping rule (work on any VM, real Intel,
|
||||||
|
real AMD): both bits are probed, absence is logged and tolerated — like the
|
||||||
|
IOMMU's fail-open, the machine still boots, just unhardened. QEMU: TCG
|
||||||
|
implements both; KVM inherits the host (Intel Ivy Bridge+ for SMEP,
|
||||||
|
Broadwell+ for SMAP; AMD Zen+ for both). The test images should run with
|
||||||
|
`-cpu max` so the suite always exercises the enabled paths.
|
||||||
|
|
||||||
|
## Adjacent, deliberately separate
|
||||||
|
|
||||||
|
- **SYSRET canonical-RIP hardening** (`isr.s:192-194` documents it): a
|
||||||
|
non-canonical return RIP makes `sysretq` #GP *in ring 0* on Intel. Same
|
||||||
|
hardening bucket, independent fix (validate RCX before `sysretq`, fall
|
||||||
|
back to `iretq`), should ride the same branch as H2/H3 but is not
|
||||||
|
SMEP/SMAP.
|
||||||
|
- **KPTI / Meltdown-class leaks are out of scope.** SMEP/SMAP police
|
||||||
|
architectural accesses, not speculative ones. danos runs one kernel
|
||||||
|
mapping in every address space and accepts that on affected hardware;
|
||||||
|
revisit only if the threat model ever includes hostile native code on
|
||||||
|
shared machines.
|
||||||
@@ -11,7 +11,7 @@ sequential pass and hands the bytes to the kernel unmodified.
|
|||||||
|
|
||||||
The capsule is a *performance artifact*, not a source of truth. The boot
|
The capsule is a *performance artifact*, not a source of truth. The boot
|
||||||
volume's `/system` and `/test` file trees remain the canonical layout (see
|
volume's `/system` and `/test` file trees remain the canonical layout (see
|
||||||
[danos-file-system-hierarchy-FSH.md](../file-system-development/danos-file-system-hierarchy-FSH.md));
|
[file-system-hierarchy.md](../file-system-development/file-system-hierarchy.md));
|
||||||
the capsule is a pre-baked snapshot of the same binaries, derived from the same
|
the capsule is a pre-baked snapshot of the same binaries, derived from the same
|
||||||
build graph, so the running system is identical whether the loader read the
|
build graph, so the running system is identical whether the loader read the
|
||||||
capsule or walked the tree.
|
capsule or walked the tree.
|
||||||
@@ -36,11 +36,11 @@ so it need be no fancier. Little-endian throughout:
|
|||||||
|
|
||||||
```
|
```
|
||||||
Header magic: u32 = "DNR2" (0x32524E44), count: u32
|
Header magic: u32 = "DNR2" (0x32524E44), count: u32
|
||||||
Entry × count name: [64]u8 (NUL-padded FHS path), offset: u64, len: u64
|
Entry × count name: [64]u8 (NUL-padded hierarchy path), offset: u64, len: u64
|
||||||
blobs... each entry's file bytes, at its offset within the image
|
blobs... each entry's file bytes, at its offset within the image
|
||||||
```
|
```
|
||||||
|
|
||||||
- **Names are full FHS paths** (`/system/services/init`), not basenames — that
|
- **Names are full hierarchy paths** (`/system/services/init`), not basenames — that
|
||||||
is what "v2" means. The 64-byte capacity matches `abi.maximum_process_name`,
|
is what "v2" means. The 64-byte capacity matches `abi.maximum_process_name`,
|
||||||
so a task named after its binary path is never truncated. Paths longer than
|
so a task named after its binary path is never truncated. Paths longer than
|
||||||
63 bytes are a build error (`pack-system-image.py` rejects them).
|
63 bytes are a build error (`pack-system-image.py` rejects them).
|
||||||
@@ -54,14 +54,14 @@ blobs... each entry's file bytes, at its offset within the image
|
|||||||
|
|
||||||
## How it is built
|
## How it is built
|
||||||
|
|
||||||
`build.zig` maintains one `bundled` list — every user binary and its FHS home.
|
`build.zig` maintains one `bundled` list — every user binary and its hierarchy home.
|
||||||
Three artifacts are derived from that same list, in the same build graph, so
|
Three artifacts are derived from that same list, in the same build graph, so
|
||||||
they cannot drift apart:
|
they cannot drift apart:
|
||||||
|
|
||||||
1. **The tree**: each binary installed at its FHS path (`zig-out/system/...`
|
1. **The tree**: each binary installed at its hierarchy path (`zig-out/system/...`
|
||||||
and `zig-out/test/...`, mirrored onto the FAT boot volume by
|
and `zig-out/test/...`, mirrored onto the FAT boot volume by
|
||||||
`tools/make-fat-image.py`).
|
`tools/make-fat-image.py`).
|
||||||
2. **The manifest** (`system/manifest`): the FHS path of every bundled binary,
|
2. **The manifest** (`system/manifest`): the hierarchy path of every bundled binary,
|
||||||
one per line — the loader's per-file fallback input.
|
one per line — the loader's per-file fallback input.
|
||||||
3. **The capsule**: `tools/pack-system-image.py` packs the same binaries into
|
3. **The capsule**: `tools/pack-system-image.py` packs the same binaries into
|
||||||
the v2 container, installed at `zig-out/boot/system.img` and placed on the
|
the v2 container, installed at `zig-out/boot/system.img` and placed on the
|
||||||
@@ -103,7 +103,7 @@ the kernel (`kernel.zig`) then publishes the same bytes twice, to two
|
|||||||
consumers:
|
consumers:
|
||||||
|
|
||||||
- **The process layer** (`process.zig`): `system_spawn` looks binaries up in
|
- **The process layer** (`process.zig`): `system_spawn` looks binaries up in
|
||||||
the ramdisk via `Reader.find` — exact FHS path, or unique basename for
|
the ramdisk via `Reader.find` — exact hierarchy path, or unique basename for
|
||||||
pre-path callers — and loads them as fresh ring-3 processes. The stored path
|
pre-path callers — and loads them as fresh ring-3 processes. The stored path
|
||||||
becomes the task's name.
|
becomes the task's name.
|
||||||
- **The VFS root** (`vfs.zig`, `setInitialRamdisk`): the image is mounted as
|
- **The VFS root** (`vfs.zig`, `setInitialRamdisk`): the image is mounted as
|
||||||
@@ -111,7 +111,7 @@ consumers:
|
|||||||
paths, so `/system` and, when the fixtures are bundled, `/test`. Directory
|
paths, so `/system` and, when the fixtures are bundled, `/test`. Directory
|
||||||
nodes are derived from the entry paths (the unique parents), so the trees
|
nodes are derived from the entry paths (the unique parents), so the trees
|
||||||
are listable and their files readable over the normal VFS protocol — the
|
are listable and their files readable over the normal VFS protocol — the
|
||||||
FHS boot tree every process sees comes straight out of the capsule bytes.
|
boot tree every process sees comes straight out of the capsule bytes.
|
||||||
|
|
||||||
The image is never copied after the handoff and never mutated: the initrd is
|
The image is never copied after the handoff and never mutated: the initrd is
|
||||||
immutable, which is what makes the VFS's node serving lock-free.
|
immutable, which is what makes the VFS's node serving lock-free.
|
||||||
|
|||||||
@@ -27,7 +27,7 @@ packages whose build.zig calls `build_support.userBinary` (with `.threaded =
|
|||||||
true` where a binary spawns threads) and get packed into the initial-ramdisk;
|
true` where a binary spawns threads) and get packed into the initial-ramdisk;
|
||||||
new syscalls extend [abi.zig](../../system/abi.zig) `SystemCall` + a
|
new syscalls extend [abi.zig](../../system/abi.zig) `SystemCall` + a
|
||||||
`library/kernel` wrapper; test services live beside the code they exercise and
|
`library/kernel` wrapper; test services live beside the code they exercise and
|
||||||
register a `ServiceId` if they must be looked up.
|
bind a `/protocol/test/...` name if they must be reachable.
|
||||||
|
|
||||||
## How to verify along the way
|
## How to verify along the way
|
||||||
|
|
||||||
|
|||||||
@@ -270,13 +270,15 @@ stays single-threaded and lean.
|
|||||||
- *Handles do not cross threads.* The handle table lives on the `Task`
|
- *Handles do not cross threads.* The handle table lives on the `Task`
|
||||||
([scheduler.zig](../../system/kernel/scheduler.zig)), so a handle number is meaningful
|
([scheduler.zig](../../system/kernel/scheduler.zig)), so a handle number is meaningful
|
||||||
only to the thread that created it — thread A's endpoint handle `3` is not thread B's.
|
only to the thread that created it — thread A's endpoint handle `3` is not thread B's.
|
||||||
A thread that needs to reach an endpoint another thread owns looks it up
|
A thread that needs to reach an endpoint another thread owns opens the name
|
||||||
(`ipc.lookup(service)`) to install its **own** handle to the same underlying endpoint.
|
(`channel.openEndpoint("display")`) to install its **own** handle to the same
|
||||||
This is how the display's mouse-listener thread reaches the compositor loop's endpoint
|
underlying endpoint — an ordinary client open, with no special mechanism for the
|
||||||
to poke it awake (docs/display.md).
|
fact that the provider happens to be this process. This is how the display's
|
||||||
|
mouse-listener thread reaches the compositor loop's endpoint to poke it awake
|
||||||
|
(docs/display.md).
|
||||||
- *IPC syscalls that touch shared kernel state now serialize under the big kernel lock.*
|
- *IPC syscalls that touch shared kernel state now serialize under the big kernel lock.*
|
||||||
`create_ipc_endpoint`/`ipc_register`/`ipc_lookup` allocate from the kernel heap and
|
`create_ipc_endpoint` allocates from the kernel heap and
|
||||||
mutate the global service registry, endpoint refcounts, and handle tables. Those paths
|
mutates endpoint refcounts and handle tables. Those paths
|
||||||
were unlocked because a single-threaded process could not race itself; a multi-threaded
|
were unlocked because a single-threaded process could not race itself; a multi-threaded
|
||||||
one can, from two cores at once. They now take `sync.enter()` like `call`/`reply_wait`/
|
one can, from two cores at once. They now take `sync.enter()` like `call`/`reply_wait`/
|
||||||
`send` already did — the kernel heap has no lock of its own (heap.zig: "every kernel
|
`send` already did — the kernel heap has no lock of its own (heap.zig: "every kernel
|
||||||
|
|||||||
@@ -137,15 +137,15 @@ Grouped as `abi.zig` groups them:
|
|||||||
| process | `danos_exit`, `danos_yield`, `danos_sleep`, `danos_spawn`, `danos_process_enumerate`, `danos_process_kill`, `danos_process_exit_reason`, `danos_process_subscribe`, `danos_process_signal`, `danos_signal_bind` |
|
| process | `danos_exit`, `danos_yield`, `danos_sleep`, `danos_spawn`, `danos_process_enumerate`, `danos_process_kill`, `danos_process_exit_reason`, `danos_process_subscribe`, `danos_process_signal`, `danos_signal_bind` |
|
||||||
| threads | `danos_thread_spawn`, `danos_thread_exit`, `danos_current_core`, `danos_futex_wait`, `danos_futex_wake`, `danos_thread_self`, `danos_thread_join`, `danos_set_thread_pointer` |
|
| threads | `danos_thread_spawn`, `danos_thread_exit`, `danos_current_core`, `danos_futex_wait`, `danos_futex_wake`, `danos_thread_self`, `danos_thread_join`, `danos_set_thread_pointer` |
|
||||||
| memory | `danos_mmap`, `danos_munmap`, `danos_dma_alloc`, `danos_dma_free`, `danos_shared_memory_create`, `danos_shared_memory_map`, `danos_shared_memory_physical` |
|
| memory | `danos_mmap`, `danos_munmap`, `danos_dma_alloc`, `danos_dma_free`, `danos_shared_memory_create`, `danos_shared_memory_map`, `danos_shared_memory_physical` |
|
||||||
| ipc | `danos_endpoint_create`, `danos_ipc_register`, `danos_ipc_lookup`, `danos_ipc_call`, `danos_ipc_reply_wait`, `danos_ipc_send` |
|
| ipc | `danos_endpoint_create`, `danos_ipc_call`, `danos_ipc_reply_wait`, `danos_ipc_send` (naming is not a syscall: a provider binds its contract at the registry and a client resolves `/protocol/<name>` — see [protocol-namespace.md](protocol-namespace.md)) |
|
||||||
| devices | `danos_device_enumerate`, `danos_device_claim`, `danos_device_register`, `danos_mmio_map`, `danos_irq_bind`, `danos_irq_ack`, `danos_msi_bind`, `danos_io_read`, `danos_io_write` |
|
| devices | `danos_device_enumerate`, `danos_device_claim`, `danos_device_register`, `danos_mmio_map`, `danos_irq_bind`, `danos_irq_ack`, `danos_msi_bind`, `danos_io_read`, `danos_io_write` |
|
||||||
| time | `danos_clock`, `danos_wall_clock`, `danos_timer_bind` |
|
| time | `danos_clock`, `danos_wall_clock`, `danos_timer_bind` |
|
||||||
| diagnostics | `danos_debug_write` (leveled, kernel-stamped records), `danos_klog_read`, `danos_klog_status` |
|
| diagnostics | `danos_debug_write` (leveled, kernel-stamped records), `danos_klog_read`, `danos_klog_status` |
|
||||||
| filesystem naming | `danos_fs_resolve`, `danos_fs_node`, `danos_fs_mount`, `danos_fs_unmount` (naming only — file DATA still crosses the vfs-protocol IPC, see below) |
|
| filesystem naming | `danos_fs_resolve`, `danos_fs_node`, `danos_fs_mount`, `danos_fs_unmount` (naming only — file DATA still crosses the vfs-protocol IPC, see below) |
|
||||||
|
|
||||||
The constants that ride alongside the calls — mmap protection bits, DMA
|
The constants that ride alongside the calls — mmap protection bits, DMA
|
||||||
flags, notification badge bits, `ExitReason`, `Signal`, well-known service
|
flags, notification badge bits, `ExitReason`, `Signal`, `page_size`, the IPC
|
||||||
ids, `page_size`, the IPC message maximum — move to the public header too:
|
message maximum — move to the public header too:
|
||||||
they are wire values a Rust program needs verbatim. What stays private in
|
they are wire values a Rust program needs verbatim. What stays private in
|
||||||
`abi.zig` is exactly the thing the vDSO exists to hide: the `SystemCall`
|
`abi.zig` is exactly the thing the vDSO exists to hide: the `SystemCall`
|
||||||
numbers and the trap convention.
|
numbers and the trap convention.
|
||||||
|
|||||||
@@ -0,0 +1,192 @@
|
|||||||
|
# Python on danos: the milestone plan
|
||||||
|
|
||||||
|
The execution plan for [python-on-danos.md](python-on-danos.md). That note holds
|
||||||
|
the *why* and the design decisions; this one slices the work into milestones with
|
||||||
|
concrete deliverables, tests, and exit criteria. Milestones are numbered **P0–P5**
|
||||||
|
(track-local — the global M-series stays with the driver/lifecycle tracks).
|
||||||
|
|
||||||
|
Dependencies at a glance:
|
||||||
|
|
||||||
|
```
|
||||||
|
P0 toolchain + mini-libc ──┐
|
||||||
|
P1 streams + console + seam ─┴─→ P2 CPython minimal ─→ P3 terminal + REPL
|
||||||
|
│ │
|
||||||
|
└─→ P4 danos module │
|
||||||
|
+ Python service│
|
||||||
|
P5 process control + shell ←─────────────────────────────────┘
|
||||||
|
```
|
||||||
|
|
||||||
|
P0 and P1 are independent of each other and can proceed in parallel. P1 is shared
|
||||||
|
work — it is also Zig self-hosting Phase 1 and the first three slices of
|
||||||
|
[character-devices-and-tty.md](character-devices-and-tty.md).
|
||||||
|
|
||||||
|
## P0 — Toolchain + the C library compatibility layer
|
||||||
|
|
||||||
|
**Goal:** a C hello-world, cross-compiled on the host with `zig cc`, runs on danos.
|
||||||
|
|
||||||
|
Design and slicing live in
|
||||||
|
[c-library-compatibility.md](c-library-compatibility.md): the **libdanos-c**
|
||||||
|
sysroot (hand-written danos-native headers + `libc.a`) as a `library/c/` build
|
||||||
|
package — pure computation (string, libm, `strtod`, the printf/scanf engines)
|
||||||
|
lifted from a vendored, pinned musl subtree; the OS plumbing written in Zig over
|
||||||
|
the `runtime` surface (re-targeting `runtime.os` when the Zig track authors it);
|
||||||
|
`malloc` over danos `mmap`; a `crt0` bridging the danos entry shim to C `main`.
|
||||||
|
Driven by `zig cc -target x86_64-freestanding-none -isystem` (the triple becomes
|
||||||
|
`x86_64-danos` if the Zig fork lands first; nothing else changes).
|
||||||
|
|
||||||
|
Its five slices (sysroot-skeleton, fd-plumbing, malloc, stdio,
|
||||||
|
mathematics-and-time) carry their own tests — host-side oracle suites for the
|
||||||
|
computation layer, QEMU cases (`c-hello`, `c-file-io`, `c-stdio`, `c-time`) for
|
||||||
|
the plumbing.
|
||||||
|
|
||||||
|
**Exit:** `c-hello` and `c-file-io` green in the QEMU suite; host computation
|
||||||
|
tests green.
|
||||||
|
|
||||||
|
## P1 — Stream nodes, console, and the seam pieces
|
||||||
|
|
||||||
|
**Goal:** the shared Phase-1 surface exists: byte-stream stdio, cwd, environment,
|
||||||
|
entropy. Design and slicing live in
|
||||||
|
[character-devices-and-tty.md](character-devices-and-tty.md); this milestone is
|
||||||
|
its slices 1–3 plus three small seam pieces:
|
||||||
|
|
||||||
|
- **cwd/chdir** — per-process current directory used by path resolution (the
|
||||||
|
kernel already anchors a VFS root per `fs_resolve`; the cwd is the same idea,
|
||||||
|
process-scoped, with `getcwd`/`chdir` exposed through `runtime` and the libc).
|
||||||
|
- **Environment** — spawn carries an environment block; the SysV entry stack's
|
||||||
|
`envp` slot ([sysv.md](os-development/sysv.md)) stops being empty; `getenv`
|
||||||
|
reads it. An empty block stays valid.
|
||||||
|
- **Entropy** — a kernel `entropy` syscall (RDSEED/RDRAND with a jitter fallback,
|
||||||
|
mirroring the TSC-reliability posture of not trusting one CPU feature blindly);
|
||||||
|
the libc exposes `getentropy`.
|
||||||
|
|
||||||
|
- **Tests.** QEMU: the character-device tests from the tty note (offsetless
|
||||||
|
read/write, blocking read, cooked/raw control round-trip), plus `cwd-basics`
|
||||||
|
(chdir + relative open), `env-roundtrip` (spawn with env, child reads it),
|
||||||
|
`entropy-sane` (nonzero, changing, correct length).
|
||||||
|
|
||||||
|
**Exit:** a C program reads a cooked line from fd 0 and echoes it to fd 1 —
|
||||||
|
injected key events in, bytes read back through the console's in-memory sink,
|
||||||
|
all under QEMU with no hardware involved — and `getcwd`/`getenv`/`getentropy`
|
||||||
|
return real answers.
|
||||||
|
|
||||||
|
## P2 — CPython, minimal configuration
|
||||||
|
|
||||||
|
**Goal:** `python -c 'print(2**100)'` runs on danos under QEMU.
|
||||||
|
|
||||||
|
- Pin **CPython 3.13.x**; vendor as `third-party/cpython/` or fetch via the build
|
||||||
|
(decide with the build-packages conventions).
|
||||||
|
- Host build-Python of the same version (`--with-build-python`).
|
||||||
|
- `config.site` cache for the cross answers; `config.sub` patch so
|
||||||
|
`x86_64-unknown-danos` parses; a small `configure`/`pyconfig` patch set kept as
|
||||||
|
rebasable diffs, WASI-style.
|
||||||
|
- `--disable-shared`; static `Modules/Setup`: `posix errno _io _codecs _weakref
|
||||||
|
time math _stat _collections itertools _functools _locale _sre` plus what the
|
||||||
|
interpreter core insists on; threadless build (WASI precedent).
|
||||||
|
- `Lib/` on the FAT image under the hierarchy (e.g. `/system/python/lib`);
|
||||||
|
`PYTHONHOME` set accordingly; `.pyc` written with **checked-hash
|
||||||
|
invalidation** (FAT's 2-second mtime granularity makes mtime-based validation
|
||||||
|
lie during fast edit-run cycles).
|
||||||
|
- `PYTHONHASHSEED` pinned only if P1's entropy slipped — otherwise real
|
||||||
|
hash randomization from day one.
|
||||||
|
- **Tests.** QEMU: `python-expr` (the exit criterion), `python-file` (run a
|
||||||
|
script from FAT, write a file, read it back), then a curated slice of CPython's
|
||||||
|
own suite (`test_int`, `test_float`, `test_io`, `test_dict`) as a
|
||||||
|
longer-running target — the suite is the porting harness.
|
||||||
|
|
||||||
|
**Exit:** the four QEMU cases green; the CPython test slice green or with a
|
||||||
|
short, documented skip list.
|
||||||
|
|
||||||
|
## P3 — Terminal + REPL: the first real application
|
||||||
|
|
||||||
|
**Goal:** an interactive `python` REPL in a graphical danos terminal — the
|
||||||
|
milestone demo for the OS.
|
||||||
|
|
||||||
|
- Depends on the display track's font rendering (its stated next step) — until
|
||||||
|
that lands, the REPL is exercised end-to-end through the pseudo-device
|
||||||
|
harness from P1+P2 (scripted input in, output read back), so P2's exit is
|
||||||
|
never blocked on graphics; the graphical terminal is the *interactive* debut.
|
||||||
|
- The terminal application: draws with the UI toolkit / display client, consumes
|
||||||
|
keyboard `InputEvent`s, and — per the tty note's load-bearing decision —
|
||||||
|
**serves the VFS stream protocol itself** to its children, reusing the console's
|
||||||
|
line-discipline library. Spawns `python` with its endpoints as fd 0/1/2.
|
||||||
|
- Raw mode + the control set give the REPL line editing; window-size control
|
||||||
|
gives it wrapping.
|
||||||
|
- **Tests.** QEMU: scripted terminal session (inject key events, assert rendered
|
||||||
|
or captured output). Real-hardware smoke on the Intel box joins the existing
|
||||||
|
checklist.
|
||||||
|
|
||||||
|
**Exit:** typing `2+2` into the terminal on the QEMU GPU target prints `4`.
|
||||||
|
|
||||||
|
## P4 — The `danos` extension module + a Python service
|
||||||
|
|
||||||
|
**Goal:** Python can speak danos: IPC, capabilities, spawn.
|
||||||
|
|
||||||
|
- The `danos` module, **written in Zig against `Python.h`**, statically linked
|
||||||
|
via `Modules/Setup`: endpoints (create/send/receive), capability passing,
|
||||||
|
spawn + exit-notification, and the service bootstrap (announce, supervision
|
||||||
|
handshake) — the same surface Zig services use, re-exposed.
|
||||||
|
- UI-toolkit bindings as a second module once the toolkit's API settles.
|
||||||
|
- Prototype **one real service in Python** — policy-shaped, not data-plane
|
||||||
|
(candidates: hot-plug policy, a settings service) — speaking an existing wire
|
||||||
|
protocol, supervised by the device manager like any service.
|
||||||
|
- **Tests.** QEMU: `python-ipc-echo` (Python service echoes over an endpoint, a
|
||||||
|
Zig client asserts), plus the prototype service's own protocol test.
|
||||||
|
|
||||||
|
**Exit:** a Python process runs as a supervised danos service exchanging IPC
|
||||||
|
with Zig peers.
|
||||||
|
|
||||||
|
## P5 — Process control, then the shell
|
||||||
|
|
||||||
|
**Goal:** danos can spawn arbitrary programs with arguments and pipes; a small
|
||||||
|
Python shell uses it.
|
||||||
|
|
||||||
|
The kernel/VFS cluster a shell forces (any shell, any language):
|
||||||
|
|
||||||
|
- **exec-of-path** — spawn an arbitrary VFS path, not a named ramdisk binary;
|
||||||
|
- **argv/envp** — carried through spawn onto the child's entry stack (env from
|
||||||
|
P1, argv new);
|
||||||
|
- **numeric exit status** — extend the exit record beyond the categorical
|
||||||
|
`ExitReason` (the gotcha the Zig roadmap flagged: `WEXITSTATUS` must be real);
|
||||||
|
- **fd inheritance + pipes** — a kernel or service pipe (a character device by
|
||||||
|
the tty note's definition) and spawn-time fd mapping.
|
||||||
|
|
||||||
|
Then, in order: `subprocess` enabled in CPython (maps onto spawn + the
|
||||||
|
exit-notification endpoint — no fork, Windows-style); a **small Python shell** (a
|
||||||
|
few hundred lines over `subprocess` + the console: prompt, argv parsing, pipes,
|
||||||
|
cwd) as the forcing function that reveals what job control actually needs.
|
||||||
|
|
||||||
|
**Explicitly deferred past P5:** the pthread subset over `thread_spawn`/futex,
|
||||||
|
signals-in-libc via M17, termios job control (Ctrl-C to foreground child), and
|
||||||
|
**xonsh** — which wants all three and is the arc's endpoint, not a milestone.
|
||||||
|
|
||||||
|
**Tests.** QEMU: `spawn-argv-exit` (child echoes argv, exits 42, parent sees
|
||||||
|
42), `pipe-through` (parent → child → parent), `python-subprocess`, and a
|
||||||
|
scripted shell session.
|
||||||
|
|
||||||
|
**Exit:** the Python shell runs `program | program` typed at the terminal and
|
||||||
|
reports the exit status.
|
||||||
|
|
||||||
|
## Post-P5 outlook
|
||||||
|
|
||||||
|
Two tracks continue past this plan, each with its own design doc rather than a
|
||||||
|
P-number here:
|
||||||
|
|
||||||
|
- **Dynamic libraries** ([dynamic-libraries.md](dynamic-libraries.md), D1–D4) —
|
||||||
|
an application-layer facility (the OS stays static and lean): `dlopen` in the
|
||||||
|
libc, then libffi + `ctypes` + loadable extension modules, then shared
|
||||||
|
read-only mappings so N Python services hold one physical `libpython`.
|
||||||
|
- **The full C compatibility layer**
|
||||||
|
([c-library-compatibility.md](c-library-compatibility.md), stages 2–3) — the
|
||||||
|
standing rule that every system capability ships with its C spelling, draining
|
||||||
|
the absence table toward "portable C builds on danos"; `fork` is the one
|
||||||
|
permanent exception.
|
||||||
|
|
||||||
|
## Related
|
||||||
|
|
||||||
|
- [python-on-danos.md](python-on-danos.md) — the design note this executes.
|
||||||
|
- [c-library-compatibility.md](c-library-compatibility.md) — P0's design.
|
||||||
|
- [character-devices-and-tty.md](character-devices-and-tty.md) — P1's design.
|
||||||
|
- [zig-self-hosting.md](zig-self-hosting.md) — shares P1; its fork makes P0's
|
||||||
|
triple prettier but gates nothing here.
|
||||||
|
- [os-development/process-management.md](os-development/process-management.md) —
|
||||||
|
the spawn/exit surface P5 extends.
|
||||||
@@ -0,0 +1,261 @@
|
|||||||
|
# Python on danos: the CPython milestone
|
||||||
|
|
||||||
|
A design note (not built yet) on bringing **CPython** to danos, compiled with the Zig
|
||||||
|
toolchain (`zig cc`). Like [zig-self-hosting.md](zig-self-hosting.md), it is
|
||||||
|
forward-looking: it sets a direction and the decisions that follow from it.
|
||||||
|
|
||||||
|
## Why Python, and why now
|
||||||
|
|
||||||
|
The Zig self-hosting road is gated on a compiler fork and a long std-library seam.
|
||||||
|
Python is the **stop-gap that removes the wait**: a working CPython gives danos a way
|
||||||
|
to write programs — services, tools, application prototypes — *without* the Zig
|
||||||
|
compiler being self-hosted, and it brings the pure-Python package ecosystem along as
|
||||||
|
a bonus. The intended division of labour:
|
||||||
|
|
||||||
|
- **Zig** — the kernel, drivers, and anything on a data plane (interrupt paths,
|
||||||
|
DMA rings, block I/O). Unchanged.
|
||||||
|
- **Python** — the control plane and the prototyping surface: services that are
|
||||||
|
event loops over IPC, policy logic that changes often, application experiments,
|
||||||
|
and eventually the shell.
|
||||||
|
|
||||||
|
Python is also the scripting language for the terminal-and-shell arc: the first
|
||||||
|
real danos application is planned as a terminal, a terminal wants a shell, a shell
|
||||||
|
wants a scripting language — and [xonsh](https://xon.sh) (a shell written in
|
||||||
|
Python) marks where that road can end.
|
||||||
|
|
||||||
|
### Non-goals
|
||||||
|
|
||||||
|
- **No drivers in Python.** Interrupt handling, ring management, and DMA stay in
|
||||||
|
Zig. Python may *supervise and configure* drivers; it does not sit in their hot
|
||||||
|
paths (interpreter overhead and garbage-collection pauses in an interrupt path
|
||||||
|
are disqualifying).
|
||||||
|
- **No dynamic loading during bring-up, no `pip`.** The whole arc here ships
|
||||||
|
statically linked. Dynamic libraries are a real *later* milestone
|
||||||
|
([dynamic-libraries.md](dynamic-libraries.md)) — an application-layer
|
||||||
|
facility that unlocks `ctypes` and loadable extension modules; the operating
|
||||||
|
system itself stays static and lean regardless (the size doctrine below).
|
||||||
|
`pip` stays out either way until a networking track exists.
|
||||||
|
- **No fork.** `os.fork` will not exist. This costs almost nothing (see "The
|
||||||
|
spawn model fits").
|
||||||
|
|
||||||
|
## The realization that shapes everything: the compiler is not the obstacle
|
||||||
|
|
||||||
|
`zig cc` is a full Clang-based C cross-compiler, and CPython is portable C with
|
||||||
|
official precedent for stranger targets than danos — the WASI port is upstream
|
||||||
|
tier-2, and it runs **without fork, without dynamic loading, and without working
|
||||||
|
threads**. Every "CPython can't possibly run there" objection has already been
|
||||||
|
answered upstream by a target *more* constrained than danos.
|
||||||
|
|
||||||
|
What CPython actually needs is a **C environment**: headers and a `libc.a`. danos
|
||||||
|
has neither — and that is the whole project. In the language of the Zig roadmap's
|
||||||
|
three doors, this is the **door-2-shaped work** (the deferred "musl door"), not the
|
||||||
|
`std.os.danos` seam: CPython never touches Zig's std.
|
||||||
|
|
||||||
|
### The same surface, a third time
|
||||||
|
|
||||||
|
The Zig roadmap observed that door 1 (`std.os.danos`) and door 2 (a libc) implement
|
||||||
|
the *same* ~30 danos-facing operations at different layers. CPython consumes that
|
||||||
|
identical surface through C spellings. So nothing here is throwaway: the
|
||||||
|
danos-native operations backing `runtime.os` are the same ones the libc bottoms out
|
||||||
|
in, and the gaps this track must close (stdio byte streams, cwd, environment,
|
||||||
|
entropy) are **exactly the Phase-1 gaps the Zig roadmap already lists**. The two
|
||||||
|
tracks share a road until Python forks off at "build the libc."
|
||||||
|
|
||||||
|
## Where danos stands: coverage vs. the gaps
|
||||||
|
|
||||||
|
Judged against the minimal CPython configuration (static, WASI-like):
|
||||||
|
|
||||||
|
| CPython need | danos today | Gap |
|
||||||
|
|--------------|-------------|-----|
|
||||||
|
| open/read/write/close/lseek, readdir | VFS + FAT via `runtime.fs` | none — wrap in C |
|
||||||
|
| mkdir / unlink / rename / truncate | done (self-hosting Phase 2) | none |
|
||||||
|
| stat with mtime | done (`wall_clock` + FAT mtime) | none |
|
||||||
|
| mmap/munmap (object allocator) | native syscalls | none |
|
||||||
|
| monotonic + wall clock | `clock` + `wall_clock` syscalls | none |
|
||||||
|
| a place for `Lib/` | FAT boot image | none — better than WASI has it |
|
||||||
|
| fork / exec | not needed (subprocess disabled at first) | — |
|
||||||
|
| dynamic loading | not needed (static extension modules) | — |
|
||||||
|
| getcwd / chdir | — | **missing** (shared with Zig Phase 1) |
|
||||||
|
| environment variables | `Init` has no env | **missing** (can start empty) |
|
||||||
|
| entropy | — | **missing** (hash seed; `PYTHONHASHSEED` pins it meanwhile) |
|
||||||
|
| byte-stream stdin/stdout (fd 0/1/2) | `debug_write` out; structured `InputEvent` in | **missing** (shared with Zig Phase 1; the REPL needs it) |
|
||||||
|
| signals | — | stubs suffice (WASI precedent); M17 signals-over-IPC maps on later |
|
||||||
|
| threads | native `thread_spawn`/futex | build threadless first; a pthread subset later (xonsh needs it) |
|
||||||
|
|
||||||
|
The clustering repeats the Zig roadmap's: **files, memory, and time are done; the
|
||||||
|
work is the C packaging plus the small seam pieces** (tty bytes, cwd, env, entropy).
|
||||||
|
|
||||||
|
## The libc decision: hand-rolled in Zig, computation lifted from musl
|
||||||
|
|
||||||
|
Two viable shapes were considered:
|
||||||
|
|
||||||
|
| Option | What it is | Verdict |
|
||||||
|
|--------|-----------|---------|
|
||||||
|
| **Mini-libc in Zig** | C-ABI-exporting Zig library over `runtime.os`/`runtime.fs`, shipped as headers + `libc.a`. | **Take this.** Reuses the danos-native surface directly; no Linux assumptions to fight. |
|
||||||
|
| **Port musl** | Full musl with a danos syscall backend. | Defer, again. musl assumes Linux syscall semantics in places; heavier than the need. |
|
||||||
|
|
||||||
|
The trick that makes the mini-libc tractable: musl's `string/`, `math/` (libm —
|
||||||
|
CPython needs essentially all of it), and number-conversion layers are **pure
|
||||||
|
computation with no syscalls**. Lift those wholesale (MIT-licensed, designed to
|
||||||
|
compile standalone) and hand-write only:
|
||||||
|
|
||||||
|
- the OS-facing bottom: fds, `mmap`, clocks, `exit`, `getcwd` — thin C-ABI wrappers
|
||||||
|
over `runtime.os`;
|
||||||
|
- a `FILE*` stdio layer (buffered, over the fd layer);
|
||||||
|
- `malloc` over danos `mmap` (a simple allocator is fine; CPython does its own
|
||||||
|
small-object arena management above it);
|
||||||
|
- the headers (`stdio.h`, `stdlib.h`, `string.h`, `math.h`, `errno.h`, …).
|
||||||
|
|
||||||
|
Estimate: **100–150 functions**, of which the hard 40% (libm, string, printf/strtod
|
||||||
|
cores) are lifted, not written. Correctness hot spots are `strtod`/`dtoa` (Python's
|
||||||
|
float repr round-trips through them) — another reason to lift musl's, not improvise.
|
||||||
|
|
||||||
|
## C interop: static extension modules, not ctypes
|
||||||
|
|
||||||
|
"Python can interface with C libraries" is true on danos with one important
|
||||||
|
correction: **`ctypes` does not work at first** — it is built on `dlopen` + libffi,
|
||||||
|
both of which arrive only with the [dynamic-libraries](dynamic-libraries.md)
|
||||||
|
milestone (D2). Until then the interop story is the other, older one:
|
||||||
|
|
||||||
|
- **Extension modules statically linked into the interpreter** via CPython's
|
||||||
|
`Modules/Setup` mechanism (the standard route for embedded/static builds).
|
||||||
|
- **Zig speaks C ABI natively**, so danos extension modules are written in Zig
|
||||||
|
against `Python.h` — no C required. Two modules are planned from the start:
|
||||||
|
- **`danos`** — the system module: endpoints, send/receive, capability passing,
|
||||||
|
spawn, exit notification. This is what makes a Python *service* possible: an
|
||||||
|
event loop over IPC, speaking the same wire protocols as Zig services.
|
||||||
|
- **UI toolkit bindings** — the in-progress danos UI toolkit exposed to Python,
|
||||||
|
so application prototypes drive real windows.
|
||||||
|
|
||||||
|
The package story follows: **pure-Python packages work** (unpack into
|
||||||
|
`Lib/site-packages` on the FAT image); packages with C extensions must be
|
||||||
|
cross-compiled and baked into the interpreter — a curated set chosen per image,
|
||||||
|
not `pip install`. That is the honest shape of the stop-gap.
|
||||||
|
|
||||||
|
## The roadmap
|
||||||
|
|
||||||
|
### Phase 0 — Toolchain + libc bring-up
|
||||||
|
|
||||||
|
`zig cc -target x86_64-freestanding-none` plus `-isystem` the danos headers and the
|
||||||
|
mini-libc archive. No compiler fork required — this track deliberately avoids the
|
||||||
|
Zig roadmap's Phase-0 gate (if the fork lands first, the triple becomes a clean
|
||||||
|
`x86_64-danos`; nothing else changes). Exit criterion: a **hello-world C program**
|
||||||
|
compiles on the host and runs on danos, printing via the libc's `write`.
|
||||||
|
|
||||||
|
### Phase 1 — The shared seam pieces
|
||||||
|
|
||||||
|
The same list as Zig self-hosting Phase 1, closed once for both tracks:
|
||||||
|
|
||||||
|
- fd 0/1/2 as console **byte** streams (output exists as `debug_write`; input is a
|
||||||
|
new small thing — cooked line input first, raw mode when the REPL wants editing);
|
||||||
|
- `getcwd`/`chdir`;
|
||||||
|
- environment variables (an empty block is a valid start);
|
||||||
|
- an entropy syscall or service (until then, builds pin `PYTHONHASHSEED`).
|
||||||
|
|
||||||
|
### Phase 2 — Cross-compile CPython, minimal configuration
|
||||||
|
|
||||||
|
Pin one CPython release (3.13 — strongest WASI-era cross-compile support). The
|
||||||
|
mechanics are well-trodden upstream since 3.11:
|
||||||
|
|
||||||
|
- a same-version **build-Python on the host** (`--with-build-python`);
|
||||||
|
- a `config.site` cache answering what configure cannot probe cross
|
||||||
|
(`ac_cv_file__dev_ptmx=no` and friends);
|
||||||
|
- a `config.sub` patch so `x86_64-unknown-danos` parses;
|
||||||
|
- `--disable-shared`, static `Modules/Setup` with a minimal module set
|
||||||
|
(`posix`, `errno`, `_io`, `_codecs`, `time`, `math`, …);
|
||||||
|
- `Lib/` shipped on the FAT image; `PYTHONHOME` pointed at it.
|
||||||
|
|
||||||
|
Exit criterion: `python -c 'print(2**100)'` runs on danos under QEMU.
|
||||||
|
|
||||||
|
### Phase 3 — Terminal + REPL: the first real application
|
||||||
|
|
||||||
|
Depends on the display track's font rendering (already its stated next step) and
|
||||||
|
Phase 1's tty. A terminal emulator drawing a `python` REPL is the milestone demo:
|
||||||
|
interactive, self-evidently real, and it needs **zero** process-control machinery.
|
||||||
|
|
||||||
|
### Phase 4 — The `danos` module and Python services
|
||||||
|
|
||||||
|
Write the `danos` extension module and the UI-toolkit bindings; prototype one real
|
||||||
|
service in Python (a policy-shaped one — e.g. hot-plug policy or a settings
|
||||||
|
service) speaking the existing IPC protocols. This is the payoff phase for
|
||||||
|
"prototyping a service or application."
|
||||||
|
|
||||||
|
### Phase 5 — Process control, then the shell
|
||||||
|
|
||||||
|
The shell — any shell, in any language — forces the surface danos has deferred so
|
||||||
|
far: **exec-of-path, argv/envp passing, numeric exit status (`WEXITSTATUS`, not the
|
||||||
|
categorical `ExitReason`), fd inheritance, and pipes.** That is a kernel/VFS
|
||||||
|
milestone cluster of its own. Then, in order:
|
||||||
|
|
||||||
|
1. `subprocess` enabled in CPython (maps onto danos spawn — see below);
|
||||||
|
2. a **small Python shell** (a few hundred lines over `subprocess` + line input, no
|
||||||
|
job control) — the forcing function that reveals which process-control pieces
|
||||||
|
actually matter;
|
||||||
|
3. **explicitly deferred:** a pthread subset over `thread_spawn`/futex
|
||||||
|
(create/join/mutex/condition/thread-locals), signals via M17 signals-over-IPC,
|
||||||
|
termios job control — and then **xonsh**, which wants all three.
|
||||||
|
|
||||||
|
### The spawn model fits
|
||||||
|
|
||||||
|
One genuinely good alignment: **CPython does not need fork.** `subprocess` maps
|
||||||
|
cleanly onto a posix_spawn-style model — exactly what danos has — and the existing
|
||||||
|
exit-notification-via-endpoint is a *better* fit for `Popen.wait` than Unix's
|
||||||
|
`wait` semantics. `os.fork` simply won't exist, as on Windows, and almost nothing
|
||||||
|
in practice cares.
|
||||||
|
|
||||||
|
## Risks and gotchas
|
||||||
|
|
||||||
|
- **Binary size — and the size doctrine that makes it acceptable.** danos's
|
||||||
|
leanness mandate applies to the **operating system**: the kernel and the system
|
||||||
|
services stay small (the kernel is measured in kilobytes, not megabytes), and
|
||||||
|
nothing in this track changes that — Python never enters the OS layer. An
|
||||||
|
**application** budget is different: a statically-linked CPython with its
|
||||||
|
module set will be tens of megabytes in ReleaseSafe (the measured ~2×
|
||||||
|
safety-check factor compounds it), and that is *allowed* — applications live
|
||||||
|
on the FAT image, not in the kernel's world. It still shapes the image, and it
|
||||||
|
means every Python service shares one interpreter binary + per-service
|
||||||
|
scripts, so the spawn model needs **argv** before "run this .py" works at all.
|
||||||
|
- **FAT mtime granularity is 2 seconds.** CPython's `.pyc` cache validation is
|
||||||
|
mtime-based by default; a rapid edit-run cycle can see stale bytecode. Use
|
||||||
|
hash-based `.pyc` invalidation (PEP 552, `--invalidation-mode checked-hash` at
|
||||||
|
freeze time) or accept the quirk during bring-up.
|
||||||
|
- **FAT name lookups are case-insensitive.** Long file names preserve case but
|
||||||
|
match insensitively — the same world Python inhabits on Windows/macOS, so
|
||||||
|
importlib copes, but two modules differing only by case cannot coexist on the
|
||||||
|
image.
|
||||||
|
- **`strtod`/float repr correctness.** Python's float round-tripping is exacting;
|
||||||
|
lift musl's conversions rather than writing them, and run CPython's float tests
|
||||||
|
early.
|
||||||
|
- **Threadless build is load-bearing, initially.** Like WASI, the first builds have
|
||||||
|
no working `threading`. The escape hatch is real (danos has native threads and
|
||||||
|
futexes; a pthread subset is Phase-5 work) but keep the configuration honestly
|
||||||
|
single-threaded until then.
|
||||||
|
- **The test suite is the porting harness.** CPython ships its own conformance
|
||||||
|
suite; getting `test_builtin`, `test_int`, `test_float`, `test_io` running on
|
||||||
|
danos early converts "it seems to work" into a checklist. Budget image space for
|
||||||
|
the test `Lib/` tree during bring-up.
|
||||||
|
- **Entropy before exposure.** `PYTHONHASHSEED=0` is fine for bring-up and wrong
|
||||||
|
forever; hash randomization exists because attacker-controlled dict keys are a
|
||||||
|
denial-of-service vector. Land the entropy source before any Python service
|
||||||
|
parses external input.
|
||||||
|
|
||||||
|
## Related
|
||||||
|
|
||||||
|
- [python-on-danos-milestones.md](python-on-danos-milestones.md) — the execution
|
||||||
|
plan (P0–P5) for this note.
|
||||||
|
- [c-library-compatibility.md](c-library-compatibility.md) — the mini-libc
|
||||||
|
(libdanos-c) design behind Phase 0.
|
||||||
|
- [character-devices-and-tty.md](character-devices-and-tty.md) — the stream-node /
|
||||||
|
console / no-pty design behind Phase 1.
|
||||||
|
- [zig-self-hosting.md](zig-self-hosting.md) — the sibling track; shares Phase 1,
|
||||||
|
diverges at the libc.
|
||||||
|
- [os-development/syscall.md](os-development/syscall.md) — the kernel ABI the
|
||||||
|
mini-libc bottoms out in.
|
||||||
|
- [os-development/vdso.md](os-development/vdso.md) — the public ABI boundary the
|
||||||
|
`danos` extension module wraps.
|
||||||
|
- [os-development/sysv.md](os-development/sysv.md) — the entry stack (argv/envp)
|
||||||
|
the spawn-argv work extends.
|
||||||
|
- [device-driver-development/ipc.md](device-driver-development/ipc.md) — the IPC
|
||||||
|
surface Python services speak.
|
||||||
|
- [file-system-development/file-system-hierarchy.md](file-system-development/file-system-hierarchy.md)
|
||||||
|
— where `Lib/` and `site-packages` land on the image.
|
||||||
@@ -0,0 +1,481 @@
|
|||||||
|
# Security track execution plan: paths, protocol namespace, SMEP/SMAP
|
||||||
|
|
||||||
|
The design is settled in
|
||||||
|
[communication.md](os-development/communication.md),
|
||||||
|
[protocol-namespace.md](os-development/protocol-namespace.md),
|
||||||
|
[file-system-hierarchy.md](file-system-development/file-system-hierarchy.md),
|
||||||
|
and [smep-smap.md](os-development/smep-smap.md). This file is the build order
|
||||||
|
— one phase at a time, each phase green before the next starts. Delete or
|
||||||
|
archive this file when the last milestone lands.
|
||||||
|
|
||||||
|
**Context a fresh session should read first:** the four design docs above,
|
||||||
|
then this plan's *Settled decisions* section — those decisions came out of a
|
||||||
|
full-code grounding pass (2026-07-31) and must not be re-derived or reopened.
|
||||||
|
|
||||||
|
**Definition of green, every phase:** `zig build` clean, `zig build test`
|
||||||
|
clean, `python3 test/qemu_test.py` passes (existing scenarios plus the
|
||||||
|
phase's new ones — record the suite count in the checkbox), and the relevant
|
||||||
|
design doc's status/known-gap lines updated in the same commit. Commit per
|
||||||
|
green phase, style `area: lower-case declarative summary`, **no co-author
|
||||||
|
trailers**. On a suite failure, read
|
||||||
|
`zig-out/qemu-test/<case>-failed-serial.log` before changing anything.
|
||||||
|
|
||||||
|
**Workflow:** work in a dedicated git worktree on feature branches cut from
|
||||||
|
`main` (one branch per milestone group as marked below); when a group's
|
||||||
|
phases are all green, merge to `main` and push. The loop marks a phase `[x]`
|
||||||
|
in the same commit that lands it.
|
||||||
|
|
||||||
|
**Numbering note:** milestones use the design docs' own names (PM, H1–H3,
|
||||||
|
HS, P1–P4) — the M-number sequence is left alone (M19–M22 are reserved by
|
||||||
|
the logging/USB-lifecycle track).
|
||||||
|
|
||||||
|
## Status
|
||||||
|
|
||||||
|
**Live state — updated on `main` after every phase, so this file read from a
|
||||||
|
plain `main` checkout always tells the truth about where the work is.**
|
||||||
|
|
||||||
|
| | |
|
||||||
|
|---|---|
|
||||||
|
| Working on | **P4a** — protocol rebase onto `envelope.Define` (next) |
|
||||||
|
| Branch carrying it | `feat/security-group-2` (pushed to origin) |
|
||||||
|
| On `main` | Phase 0, PM, H1, P1, P2, P3 (group 2 merged) |
|
||||||
|
| Awaiting merge | nothing — group 2 is on `main` |
|
||||||
|
| Suite | 109 cases, all passing |
|
||||||
|
| Last updated | 2026-08-01 |
|
||||||
|
|
||||||
|
A checkbox below means the phase met its definition of green and was
|
||||||
|
committed — on the branch named above, which reaches `main` at the next
|
||||||
|
group boundary.
|
||||||
|
|
||||||
|
- [x] **Phase 0** — baseline: suite green on `main` (106/106, 2026-07-31; `zig build` + `zig build test` clean at 9a32380), plan committed
|
||||||
|
- [x] **PM** — path-migration flag-day (`/etc`→`/system/configuration`, `/var/log`→`/system/logs`, `/mnt/usb`→`/volumes/usb`; vfs carve-out for the two writable `/system` subtrees, FAT's `/var` mount split in two; suite 106/106)
|
||||||
|
- [x] **H1** — the `user-memory` module; nine stragglers converted; leaf U/S+W checks (plus physmap-coverage confirmation, so an `mmio_map`'d buffer cannot fault ring 0 — this also closes the same hazard on the IPC path; `fs_resolve`'s out-capacity bound made overflow-safe; suite 107/107)
|
||||||
|
- [x] **merge** group 1 → main, push (f3bc23c, 2026-07-31)
|
||||||
|
- [x] **P1** — envelope module + `Define`; vfs `NodeKind.protocol` + open-reply-capability; client `Channel` (mechanics only, nothing converted; suite unchanged at 107)
|
||||||
|
- [x] **P2** — registry in init; `/protocol` reserved; ServiceId flag-day (11 binds, 17 lookups; `protocol.csv` grants, chain-attested identity, dead-owner rebind; the kernel's endpoint-death sweep generalized off the retired registry; suite 108/108). Three adversarial review rounds closed six defects a green suite had missed: a forged power event could shut the machine down; the ping path leaked a capability per call, first in init and then in the shared harness; supervisor attestation by name was defeated by a laundering deputy; and the kernel let any handle-holder bind signals, timers, exits and IRQs to an endpoint it did not own.
|
||||||
|
- [x] **P3** — open grants: `protocol.csv` enforcement, denial test. `onOpen`
|
||||||
|
consults the manifest with the same chain-attested identity a bind uses, and a
|
||||||
|
refused caller gets the *same* answer as one naming a contract nobody bound —
|
||||||
|
`-ENOENT`, no capability, the same reply bytes, no log line, and both questions
|
||||||
|
asked on every open so there is nothing to time. Twenty-seven `open` rows cover
|
||||||
|
the whole live client set. One wrinkle the plan had not foreseen: the driver
|
||||||
|
tree is three deep (device manager → PS/2 bus → keyboard/mouse) and attestation
|
||||||
|
is one hop, so a legitimate grandchild read exactly like a laundering deputy;
|
||||||
|
the manifest gained a third permission, `supervise`, which names an authorized
|
||||||
|
supervising task per contract and is deliberately **open-only**, leaving P2's
|
||||||
|
bind attestation and every refusal it makes untouched (suite 109/109)
|
||||||
|
- [x] **merge** group 2 → main, push
|
||||||
|
- [ ] **P4a** — clean protocols rebased onto `Define` (vfs, block, display, scanout, input)
|
||||||
|
- [ ] **P4b** — misfit protocols rebased (device-manager, power, usb-transfer)
|
||||||
|
- [ ] **P4c** — harness subscriber lift + badge-scoped per-client integers
|
||||||
|
- [ ] **merge** group 3 → main, push
|
||||||
|
- [ ] **H2** — SMEP on every core
|
||||||
|
- [ ] **HS** — SYSRET canonical-RIP guard
|
||||||
|
- [ ] **H3** — SMAP + boot-patched `clac`; `-cpu max` in the harness; negative tests
|
||||||
|
- [ ] **merge** group 4 → main, push
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Settled decisions (grounding pass, 2026-07-31 — do not reopen)
|
||||||
|
|
||||||
|
These resolve every open wrinkle the code inventory surfaced. Where one
|
||||||
|
amends a design doc, the amendment lands in the same commit as the phase
|
||||||
|
that implements it.
|
||||||
|
|
||||||
|
1. **Every packet — request, reply, and event — begins with the envelope
|
||||||
|
`Header`, exactly as the design says; the header is FOLDED, never
|
||||||
|
stacked.** It absorbs each protocol's existing operation/id fields
|
||||||
|
rather than sitting on top of them, so the two apparent 64-byte-limit
|
||||||
|
offenders fit: `ChildAdded` re-lays to 60 bytes (its packed operation
|
||||||
|
byte and `device_id` become `Header.operation`/`.target`);
|
||||||
|
`InterruptReport` puts `device_token` in `Header.target` and trims
|
||||||
|
inline data 48 → 40 bytes (largest real report today is 8). A
|
||||||
|
headerless-events variant was considered and REJECTED (2026-07-31): it
|
||||||
|
re-invents per-protocol mini-headers and breaks uniform tooling. No
|
||||||
|
design-doc amendment; `Define`'s event check stays ≤ 64 *including*
|
||||||
|
the header.
|
||||||
|
2. **Bind/open authorization is chain-attested identity: the
|
||||||
|
kernel-stamped binary name PLUS the supervision chain**, both read from
|
||||||
|
the kernel's process records (`ProcessDescriptor` carries `name` and
|
||||||
|
`supervisor`; init walks the chain with `process_enumerate` — no new
|
||||||
|
protocol). A grant row names the binary *and* the supervisor expected
|
||||||
|
in its chain, so a malicious process re-spawning a granted binary
|
||||||
|
(ungated `spawn`, hostile argv — the confused deputy) is refused: its
|
||||||
|
chain roots at the attacker, not at init or device-manager. Name alone
|
||||||
|
is NOT sufficient — that was considered and rejected (2026-07-31).
|
||||||
|
Pure delegation (device-manager forwarding driver binds as
|
||||||
|
capabilities — "option B") is deliberately deferred to P5, whose
|
||||||
|
spawner-wired namespaces subsume it. Amends protocol-namespace.md's
|
||||||
|
"Authorization" bullet in P2.
|
||||||
|
3. **Grants live in a new manifest, `/system/configuration/protocol.csv`**
|
||||||
|
(rows: `binary-path, supervisor, bind|open, protocol-name`, where
|
||||||
|
`supervisor` is the binary expected in the caller's supervision chain —
|
||||||
|
`init` for init's own children, `kernel` for harness-spawned fixtures),
|
||||||
|
not in extra init.csv columns — today every post-path init.csv field is
|
||||||
|
argv, and overloading that is ambiguous. init parses both files.
|
||||||
|
*(P2 spelling: the supervisor column carries the binary exactly as the
|
||||||
|
kernel stamped it, so init's own children say `/system/services/init` and
|
||||||
|
the drivers say `/system/services/device-manager`; `kernel` stays a bare
|
||||||
|
word because a kernel task has no binary. A trailing `*` on any field
|
||||||
|
matches a subtree, which is how decision 4's `/test/` rule is expressed.)*
|
||||||
|
*(Clarification, 2026-08-01: the supervisor column names **the authorized
|
||||||
|
supervising task, matched by identity** — the binary is how the row spells
|
||||||
|
it, but init checks the task id. `kernel` is satisfied only by supervisor
|
||||||
|
id 0 (which only the kernel confers — user `system_spawn` always stamps the
|
||||||
|
caller); init's own path only by this init's task id; any other path only by
|
||||||
|
a task init spawned itself or one the kernel spawned. Matching the supervisor
|
||||||
|
by *name* alone is defeated by a laundering deputy — an attacker runs its own
|
||||||
|
instance of `/system/services/init`, has that spawn `/system/services/input`,
|
||||||
|
and both stamped names satisfy the row while the chain is entirely the
|
||||||
|
attacker's. Walking to the root of the chain does not fix it either, since
|
||||||
|
the laundered chain still roots at the real PID 1.)*
|
||||||
|
*(P3 amendment: a third permission, `supervise`, joins `bind|open`. One-hop
|
||||||
|
attestation cannot express the one three-deep chain in the tree — the device
|
||||||
|
manager starts the PS/2 bus, and the bus starts the keyboard and mouse
|
||||||
|
drivers — and nothing structural tells that chain apart from the laundering
|
||||||
|
deputy, since both are a granted binary spawned by a granted binary. Only
|
||||||
|
policy can: a `supervise` row names the authorized supervising task the way
|
||||||
|
every other row names a claimant (binary, its own supervisor, the contract it
|
||||||
|
concerns), and an `open` row may then name that task in its supervisor
|
||||||
|
column. The delegate is itself attested the ordinary strict way, so the chain
|
||||||
|
still anchors in init or the kernel one hop above it and the recursion stops
|
||||||
|
there. It is **open-only** on purpose — a delegate may vouch for what its
|
||||||
|
children *reach*, never for what they *claim* — so the bind path is
|
||||||
|
byte-for-byte P2's and the laundering-deputy refusal is untouched.)*
|
||||||
|
4. **Test fixtures bind under `/protocol/test/...`**, granted to any
|
||||||
|
binary whose path starts `/test/` — the subtree-scoping rule from the
|
||||||
|
design doc, dogfooded. `shared_memory_test` (the borrowed-ServiceId
|
||||||
|
hack) becomes `/protocol/test/shared-memory`; process-test's child gets
|
||||||
|
`/protocol/test/process`.
|
||||||
|
5. **Rebind after provider death:** a `bind` hitting an existing binding
|
||||||
|
succeeds only if the current owner process is dead (init checks
|
||||||
|
liveness); otherwise `-EBUSY`. Init also unbinds in `restartChild`
|
||||||
|
before respawning its own children. This preserves collision-refusal
|
||||||
|
while making restart work for providers init does not supervise.
|
||||||
|
6. **Cross-thread service access** (the display mouse-listener's
|
||||||
|
per-thread self-lookup, `display.zig:512`): threads resolve and open
|
||||||
|
`/protocol/<name>` like any client — once, at thread startup. No
|
||||||
|
special mechanism.
|
||||||
|
7. **The envelope module is `library/protocol/envelope/envelope.zig`**
|
||||||
|
(module name `envelope`) — the one protocol-package module not ending
|
||||||
|
in `-protocol`, because it is not a protocol. Wired as a new
|
||||||
|
`addModule` row in `library/protocol/build.zig` with its host tests in
|
||||||
|
that package's test step.
|
||||||
|
8. **The QEMU harness gains `-cpu max`** (in `qemu_args`,
|
||||||
|
`test/qemu_test.py:66-83`) so TCG exposes SMEP/SMAP — without it the
|
||||||
|
enabled paths never execute in CI. Landed in H2 so the flag soaks
|
||||||
|
before H3 depends on it.
|
||||||
|
9. **Scenario fixtures that need the registry are init-driven.** Kernel
|
||||||
|
test cases that today spawn providers directly (shared-memory,
|
||||||
|
process-test) either spawn init first or move to init.csv-driven
|
||||||
|
scenario boots — resolved per-case in P2 with the suite as the
|
||||||
|
arbiter.
|
||||||
|
*(P2 resolution: init gained a `registry` argv role — it mounts
|
||||||
|
`/protocol`, reads the grants, and starts no services — and each affected
|
||||||
|
case calls `spawnRegistry(rd)` before its own providers. Every case keeps
|
||||||
|
its own spawn set, so no scenario had to be re-shaped.)*
|
||||||
|
10. **The capsule-staleness caveat is documented, not fixed.** On-volume
|
||||||
|
edits to `/system/configuration/*.csv` do not reach the initrd copy
|
||||||
|
the loader boots (capsule shadows tree). Same drift exists today with
|
||||||
|
`/etc`; PM adds the note to file-system-hierarchy.md and moves on.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## PM — path-migration flag-day
|
||||||
|
|
||||||
|
One commit, everything moves together. The authoritative site inventory is
|
||||||
|
the grounding pass; the checklist order:
|
||||||
|
|
||||||
|
1. Move repo `etc/` → `configuration/` sources; fix the three CSVs'
|
||||||
|
self-referencing headers (`etc/init.csv:1,12`, `etc/devices.csv:1`,
|
||||||
|
`etc/init-diagnose.csv:1`).
|
||||||
|
2. `build.zig:309-311`: bundled entries `etc/...` →
|
||||||
|
`system/configuration/...` (this alone re-shapes the image, manifest,
|
||||||
|
and capsule — `tools/make-fat-image.py` and the EFI loader need
|
||||||
|
nothing; the tree-walk fallback even starts picking the CSVs up, a
|
||||||
|
bonus fix).
|
||||||
|
3. `system/kernel/vfs.zig` `mountBackend` (`:332-340`): allow exactly
|
||||||
|
`/system/configuration` and `/system/logs` as backend prefixes beneath
|
||||||
|
the initrd `/system` mount; keep refusing everything else under
|
||||||
|
`/system` and `/test`.
|
||||||
|
4. `system/services/fat/fat.zig`: `mount_point` → `/volumes/usb` (`:25`);
|
||||||
|
replace the `/var` mount (`:155`) with two `mountRewritten` calls for
|
||||||
|
`/system/configuration` and `/system/logs`; update the mount log lines
|
||||||
|
(the harness matches them).
|
||||||
|
5. `system/services/init/init.zig:76` and
|
||||||
|
`system/services/device-manager/device-manager.zig:48`: open the new
|
||||||
|
CSV paths; update the message strings (`init.zig:77,92`,
|
||||||
|
`device-manager.zig:49,61-63,454`).
|
||||||
|
6. `system/services/logger/logger.zig:44`: `base = "/system/logs"`
|
||||||
|
(buffers derive from `base.len` comptime — nothing else changes).
|
||||||
|
7. `system/kernel/tests.zig:2808-2810`: exclude `/system/configuration/`
|
||||||
|
from the spawn-everything sweep (the CSVs are not programs).
|
||||||
|
8. Tests: `fat-test.zig` and `vfs-test.zig` `/mnt/usb` literals →
|
||||||
|
`/volumes/usb`; harness regexes `test/qemu_test.py:175,211,632,717`.
|
||||||
|
9. Comment sweep (init, device-manager, logger, fat, engine, vfs, abi,
|
||||||
|
file-system, csv, device, protocol/device-manager, drivers, acpi,
|
||||||
|
build.zig — full list in the grounding inventory); delete vestigial
|
||||||
|
repo `var/`.
|
||||||
|
|
||||||
|
**Test:** no new case — the existing 106 are the test, since fat/logger/
|
||||||
|
init/device-manager scenarios all assert the new paths through their
|
||||||
|
regexes. Suite stays 106.
|
||||||
|
|
||||||
|
## H1 — user-memory copy discipline
|
||||||
|
|
||||||
|
New kernel module `system/kernel/user-memory.zig`:
|
||||||
|
|
||||||
|
- `copyFromUser` moves from ipc-synchronous.zig (which re-exports or
|
||||||
|
imports it); new `copyToUser(user_as, user_va, source) bool` — the
|
||||||
|
mechanical mirror (kernel-source `copyAcross` already does this for IPC
|
||||||
|
replies at `ipc-synchronous.zig:431,460`).
|
||||||
|
- The page walk gains leaf U/S and writable checks: `paging.translateIn`
|
||||||
|
(`architecture/x86_64/paging.zig:513-525`) tests only `present` today —
|
||||||
|
add a flags-accumulating variant (2 MiB leaves included); reads require
|
||||||
|
U/S, writes require U/S+W. Closes the TODO at
|
||||||
|
`ipc-synchronous.zig:20-22`.
|
||||||
|
- Convert the nine stragglers (table in smep-smap.md). Read direction is
|
||||||
|
local to `process.zig`; the write direction restructures callees with
|
||||||
|
kernel bounce buffers: `scheduler.enumerate` (`scheduler.zig:1209`),
|
||||||
|
`devices_broker.enumerate` (`devices-broker.zig:136`), `log.readAt`
|
||||||
|
(`log.zig:209`), and the `fs_node` flows through
|
||||||
|
`vfs.nodeRead/nodeStatus/nodeReaddir` (`vfs.zig:257/269/289`).
|
||||||
|
|
||||||
|
**Test:** kernel unit coverage in `system/kernel/tests.zig` for
|
||||||
|
`copyToUser` bounds/permission refusals; one new QEMU case `user-memory` —
|
||||||
|
a fixture passes an unmapped-but-in-range buffer to `klog_read`,
|
||||||
|
`process_enumerate`, and `fs_resolve` and asserts `-EFAULT` returns with
|
||||||
|
the system still alive (today each would oops the kernel). Suite 107.
|
||||||
|
|
||||||
|
## P1 — envelope, vfs additions, Channel
|
||||||
|
|
||||||
|
- `library/protocol/envelope/envelope.zig`: `Header` {operation:u32, pad,
|
||||||
|
target:u64}, `Status`, reserved verbs (describe=0, enumerate=1,
|
||||||
|
subscribe=2, unsubscribe=3, protocol verbs from 16), `packet_maximum`
|
||||||
|
= 256 / `post_maximum` = 64 (the floor constants protocols compile
|
||||||
|
against — nothing exports them today), and comptime
|
||||||
|
`Define(.{name, version, operations, events})` generating request/reply
|
||||||
|
types, encode/decode, a provider dispatch table (automatic `describe`,
|
||||||
|
`-ENOSYS` for unknown verbs), and compile-time size checks:
|
||||||
|
request/reply ≤ 256, each `.events` entry ≤ 64 *including* its Header
|
||||||
|
(decision 1). Host unit tests in the protocol package's test step.
|
||||||
|
- `library/protocol/vfs/vfs-protocol.zig`: `NodeKind.protocol = 7`; the
|
||||||
|
open-reply-may-carry-capability convention documented in the module.
|
||||||
|
Rewrite the value-pinning unit test (`:108-117`) to pin the *new*
|
||||||
|
stable values.
|
||||||
|
- `library/kernel/file-system.zig` + a new `Channel` type in
|
||||||
|
`library/kernel` (or `library/client`): `open("/protocol/<name>")` →
|
||||||
|
resolve, vfs open, receive the reply capability → a `Channel` wrapping
|
||||||
|
the handle with `call`/typed helpers. Nothing uses it yet — P2 converts
|
||||||
|
the world.
|
||||||
|
- Docs: vfs-protocol.md's NodeKind table gains value 7 (no
|
||||||
|
protocol-namespace.md amendment — decision 1 conforms to it as written).
|
||||||
|
|
||||||
|
**Test:** host unit tests only (envelope round-trips, size-check compile
|
||||||
|
errors via `error` tests, Channel plumbing against a mock). Suite stays
|
||||||
|
107.
|
||||||
|
|
||||||
|
## P2 — the registry; ServiceId flag-day
|
||||||
|
|
||||||
|
The single biggest phase; one branch, may be several commits, green at the
|
||||||
|
end of each.
|
||||||
|
|
||||||
|
- **init as registry backend** (`system/services/init/init.zig`): a second
|
||||||
|
endpoint (the supervision endpoint's reply-empty loop is unsuitable for
|
||||||
|
a vfs backend); serve vfs `open`/`readdir` over `/protocol` plus the
|
||||||
|
`bind` operation (name payload + capability). Mount `/protocol` before
|
||||||
|
spawning children. Parse `/system/configuration/protocol.csv`
|
||||||
|
(decision 3). Authorization by chain-attested identity (decision 2):
|
||||||
|
badge → kernel process records → binary name **and** supervision chain
|
||||||
|
(walk `supervisor` links) checked against the grant row's expected
|
||||||
|
supervisor. Unbind on child death in `restartChild`; dead-owner rebind
|
||||||
|
rule (decision 5).
|
||||||
|
Provenance: readdir/diagnostics show name → pid → binary path.
|
||||||
|
- **Kernel:** reserve `/protocol` — `mountBackend` refuses mounts at or
|
||||||
|
under it once bound, `installMount`'s remount-replace path refuses it,
|
||||||
|
and `fs_unmount` refuses it (`vfs.zig:164-181,332-351`,
|
||||||
|
`process.zig:1879-1889`). First mount wins (init is PID 1).
|
||||||
|
- **Harness:** `library/kernel/service.zig` `Callbacks.service:
|
||||||
|
?abi.ServiceId` becomes a protocol name; the register call (`:49-51`)
|
||||||
|
becomes bind-with-retry via the registry.
|
||||||
|
- **Flag-day conversion** — all 11 registration sites and 17 lookup sites
|
||||||
|
from the grounding inventory: providers (input:123, ps2-bus:223,
|
||||||
|
device-manager:569, acpi:193, usb-xhci-bus:676, usb-storage:205,
|
||||||
|
fat:307, display:699, virtio-gpu:550, shared-memory-server:43,
|
||||||
|
process-test:130 → `/protocol/test/...` per decision 4); clients
|
||||||
|
(input-client:53, display-client:28, driver.zig:173, usb.zig:139,
|
||||||
|
block.zig:72+87, ps2-bus keyboard:35 + mouse:34, virtio-gpu:478,
|
||||||
|
display:314+512 (decision 6), acpi:212, init:218+245 — init
|
||||||
|
short-circuits its own registry, shared-memory-client:22,
|
||||||
|
process-test:85, device-list:22, crash-test:32). Retry loops keep their
|
||||||
|
cadence, wrapping resolve+open instead of lookup.
|
||||||
|
- **Delete:** `abi.zig:36-37` (syscall ids — leave holes),
|
||||||
|
`abi.zig:287-303` (enum), `process.zig:223-224,314-343`,
|
||||||
|
`ipc-synchronous.zig:41-43,646-664` and the registry sweep in
|
||||||
|
`:121-140`; the wrappers `library/kernel/ipc.zig:33-35,47-50`; comment
|
||||||
|
sweep (irq.zig:50, tests.zig:3744, vdso.md's syscall table, the docs
|
||||||
|
list in the inventory).
|
||||||
|
- Kernel-spawned test scenarios made init-driven where they need the
|
||||||
|
registry (decision 9).
|
||||||
|
|
||||||
|
**Test:** new QEMU case `protocol-registry`: a fixture asserts (a) bind of
|
||||||
|
an ungranted name → `-EPERM`, (b) bind collision with a live owner →
|
||||||
|
`-EBUSY`, (c) provider kill → re-resolve reaches the restarted instance.
|
||||||
|
Every existing scenario doubles as conversion proof. Suite 108.
|
||||||
|
|
||||||
|
## P3 — open grants (restriction stage one)
|
||||||
|
|
||||||
|
- `protocol.csv` `open` rows enforced in the registry's `open` handler,
|
||||||
|
same name-based identity as bind. Default rows grant what today's
|
||||||
|
clients need (from the P2 conversion table); a deliberate hole for the
|
||||||
|
test fixture.
|
||||||
|
- Docs: protocol-namespace.md stage-one section gets its "landed" line.
|
||||||
|
|
||||||
|
**Test:** new QEMU case `protocol-denied`: a fixture granted
|
||||||
|
`/protocol/test/shared-memory` but not `/protocol/display` asserts open of
|
||||||
|
the first succeeds and the second fails identically to not-found. Suite
|
||||||
|
109.
|
||||||
|
|
||||||
|
*Landed. Four things the plan did not foresee, recorded because P4 and P5
|
||||||
|
inherit them:*
|
||||||
|
|
||||||
|
- *`supervise` — decision 3's amendment. The PS/2 keyboard and mouse drivers
|
||||||
|
are started by the PS/2 bus driver, which the device manager started: the
|
||||||
|
tree's one three-deep chain, and one hop deeper than attestation reaches.
|
||||||
|
Nothing structural separates it from the laundering deputy, so the manifest
|
||||||
|
says which delegate is authorized, per contract. Open-only, so P2's bind
|
||||||
|
attestation is unchanged.*
|
||||||
|
- *Indistinguishability is a claim about work, not only about bytes. `onOpen`
|
||||||
|
refreshes the process table, identifies the caller, scans the grants and
|
||||||
|
scans the bindings on **every** open and forms one verdict at the end; and
|
||||||
|
it logs nothing on any branch, because `klog_read` is ungated (a line
|
||||||
|
written on one branch is a line the refused caller can read) and a serial
|
||||||
|
line is milliseconds it could time. The operator's diagnosis is the pair the
|
||||||
|
namespace publishes anyway: `readdir /protocol` for what is bound, the
|
||||||
|
manifest for who may reach it.*
|
||||||
|
- *The fixture is `protocol-denied-test`, and its scenario boots the **input
|
||||||
|
service** so the forbidden name is genuinely bound — the fixture reads the
|
||||||
|
namespace listing to prove it before asking for it. Without a live provider
|
||||||
|
the case would be comparing two boot races and asserting nothing.*
|
||||||
|
- *Two channels stay open by design, named rather than papered over: `readdir`
|
||||||
|
over `/protocol` lists every bound name to anyone (deliberate — the tree is
|
||||||
|
diagnosable), and `/system/configuration/protocol.csv` is world-readable on
|
||||||
|
the `/system` mount. Stage one hides neither the set of contracts nor the
|
||||||
|
policy; what it removes is the **oracle in the reply**, which is what stage
|
||||||
|
two's parked and faked opens depend on.*
|
||||||
|
|
||||||
|
## P4a — clean protocols onto Define
|
||||||
|
|
||||||
|
vfs, block, display, scanout, input — the modules whose shapes map
|
||||||
|
directly (grounding inventory §1,3,4,6,8):
|
||||||
|
|
||||||
|
- vfs: `node` → `target`; `Reply.node` (open's result) moves to reply
|
||||||
|
payload — `library/kernel/file-system.zig` decoders change; readdir
|
||||||
|
stays a protocol verb.
|
||||||
|
- block: pure renumber; `attach`'s DMA cap rides the call as today.
|
||||||
|
- display: the overloaded 40-byte `Request` becomes per-operation structs
|
||||||
|
(attach_scanout's field abuse dies); `layer` → `target`; blit payload
|
||||||
|
grows to 224 bytes.
|
||||||
|
- scanout: renumber; drop its bogus `message_maximum=64` (sync floor is
|
||||||
|
256); fix virtio-gpu's hard-coded `service.run(256, …)` to the
|
||||||
|
generated constant.
|
||||||
|
- input: subscribe merges into reserved subscribe; publish renumbers;
|
||||||
|
the event re-lays onto the Header folded (operation = event kind,
|
||||||
|
target = 0; 16 + 28-byte payload = 44 ≤ 64); **input moves onto the
|
||||||
|
service harness** (it is the last hand-rolled loop, no ping/terminate
|
||||||
|
compliance today).
|
||||||
|
|
||||||
|
**Test:** new QEMU case `protocol-conformance`: a fixture opens every
|
||||||
|
registered protocol and asserts `describe` answers (name, version) and an
|
||||||
|
unknown verb returns `-ENOSYS`. Existing input/display/fat scenarios prove
|
||||||
|
the rebase. Suite 110.
|
||||||
|
|
||||||
|
## P4b — misfit protocols onto Define
|
||||||
|
|
||||||
|
device-manager, power, usb-transfer (inventory §2,5,7 — the u8-operation
|
||||||
|
re-layouts and raw-offset readers):
|
||||||
|
|
||||||
|
- device-manager: u8 operations → Header; its enumerate=4/subscribe=5
|
||||||
|
merge into the reserved verbs; `ChildAdded` splits its dual role —
|
||||||
|
request struct and event, both Header-first (folded to 60 B ≤ 64);
|
||||||
|
`ChildRemoved`'s (parent, bus_address) addressing stays payload.
|
||||||
|
- power: u8 operations → Header; subscribe merges; **init's raw
|
||||||
|
byte-offset event parsing (`init.zig:171-173`) and acpi's
|
||||||
|
`message[0]` dispatch (`acpi.zig:435-467`) are rewritten against the
|
||||||
|
generated types** — the two silent-breakage sites, called out so the
|
||||||
|
loop treats them as first-class conversions, not collateral.
|
||||||
|
- usb-transfer: `device_token` → `target` (already layout-identical);
|
||||||
|
`InterruptReport` re-lays onto the Header (`device_token` → `target`,
|
||||||
|
inline data trimmed 48 → 40 — largest real report is 8); control/bulk
|
||||||
|
budgets re-verified by `Define` (Status absorbs `actual_length`).
|
||||||
|
|
||||||
|
**Test:** existing scenarios are the proof (device hot-add, power button,
|
||||||
|
USB storage/HID all exercise these wires); the conformance case now covers
|
||||||
|
three more providers. Suite 110.
|
||||||
|
|
||||||
|
## P4c — harness subscriber lift + badge scoping
|
||||||
|
|
||||||
|
- `library/kernel/service.zig` grows the subscriber table, exit-
|
||||||
|
notification sweep, and fan-out loop declared via `Define(.events)`;
|
||||||
|
input (:33-116), acpi (:67-68,393-406), and device-manager (:155-166)
|
||||||
|
delete their hand-rolled variants. One sweep idiom: exit notifications
|
||||||
|
(fat's pattern), replacing input's process-list polling and acpi's
|
||||||
|
none-at-all.
|
||||||
|
- Badge-scoped per-client integers (the guessable-id holes): fat node ids
|
||||||
|
gain owner checks on every operation (`fat.zig:72-76`), xhci device
|
||||||
|
tokens validate sender and sweep on exit (`usb-xhci-bus.zig:66-88,479`),
|
||||||
|
display layers gain an owner field.
|
||||||
|
|
||||||
|
**Test:** extend the fat scenario: a second fixture guesses the first's
|
||||||
|
node id and asserts refusal; kernel-side unit test for the harness sweep.
|
||||||
|
Suite 111.
|
||||||
|
|
||||||
|
## H2 — SMEP
|
||||||
|
|
||||||
|
- Generalize the cpuid helper (`apic.zig:351-365`, private, subleaf-0) to
|
||||||
|
a shared probe; gate on `cpuid(0).eax >= 7`.
|
||||||
|
- Set CR4 bit 20 in `per-cpu.zig:initSystemCall` (or a sibling called
|
||||||
|
from both `cpu.zig:148` and `smp.zig:181` — the one path both BSP and
|
||||||
|
every AP already execute). Log enabled/absent (fail-open, IOMMU style).
|
||||||
|
- Harness: add `-cpu max` to `qemu_args` (decision 8).
|
||||||
|
|
||||||
|
**Test:** new QEMU case `fault-smep` mirroring the `fault-*` injector
|
||||||
|
pattern (`tests.zig:3906-3938`): ring-0 call through a pointer into a
|
||||||
|
user-mapped page; expect `page fault (vector 14)` + `error code : 0x11` +
|
||||||
|
kernel-half IP, machine reports the exception (deliberate-exception cases
|
||||||
|
put the text in `expect`, per `qemu_test.py:189`). Suite 112.
|
||||||
|
|
||||||
|
## HS — SYSRET canonical-RIP guard
|
||||||
|
|
||||||
|
- `isr.s` syscall exit (`:256`): validate RCX canonicality before
|
||||||
|
`sysretq`; non-canonical → `iretq` fallback (or kill), per the hazard
|
||||||
|
note at `isr.s:192-194`.
|
||||||
|
|
||||||
|
**Test:** kernel unit case driving a thread whose return RIP is forged
|
||||||
|
non-canonical via the syscall path if constructible cheaply; otherwise the
|
||||||
|
review-level proof plus the existing fault cases regression. Suite 112.
|
||||||
|
|
||||||
|
## H3 — SMAP
|
||||||
|
|
||||||
|
- `clac` patch site at `isr_common` (`isr.s:367`, before the CPL test —
|
||||||
|
ring-0 nesting inherits AC too): assemble a 3-byte NOP, patch to `clac`
|
||||||
|
at boot through the physmap (the `process.zig:1990-1995` /
|
||||||
|
`smp.zig:79-111` precedent), BSP-only before AP bring-up.
|
||||||
|
- Set CR4 bit 21 in the same per-CPU init as SMEP.
|
||||||
|
- Coding standards: kernel code touches user memory only through
|
||||||
|
`user-memory`; no `stac` anywhere, ever.
|
||||||
|
|
||||||
|
**Test:** new QEMU case `fault-smap`: ring-0 deliberate read of a mapped
|
||||||
|
user page; expect vector 14 + `error code : 0x1` + kernel IP. And the
|
||||||
|
whole suite becomes the tripwire — any missed straggler now fails loudly.
|
||||||
|
Suite 113.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
**Explicitly out of scope** (own tracks, after this plan): P5 restriction
|
||||||
|
stage two (spawn's initial capability, namespace views, parked replies,
|
||||||
|
dedicated killable channels — needs a design session on the spawn
|
||||||
|
contract), file-path namespacing, trusted UI (display track), pipes/FIFOs
|
||||||
|
(Python track), `/applications` and its storage, `fs_mount`/`spawn`/
|
||||||
|
`klog_read` gating beyond the `/protocol` reserved prefix, KPTI, IPC
|
||||||
|
priority inheritance.
|
||||||
@@ -347,7 +347,7 @@ Two current decisions fall out of this roadmap:
|
|||||||
- [syscall.md](os-development/syscall.md) — the kernel↔runtime ABI `runtime.os` is built on.
|
- [syscall.md](os-development/syscall.md) — the kernel↔runtime ABI `runtime.os` is built on.
|
||||||
- [sysv.md](os-development/sysv.md) — the entry stack (`argc/argv/envp/auxv`) danos already constructs.
|
- [sysv.md](os-development/sysv.md) — the entry stack (`argc/argv/envp/auxv`) danos already constructs.
|
||||||
- [ipc.md](device-driver-development/ipc.md) — the IPC the VFS/FAT operations travel over.
|
- [ipc.md](device-driver-development/ipc.md) — the IPC the VFS/FAT operations travel over.
|
||||||
- [danos-file-system-hierarchy-FSH.md](file-system-development/danos-file-system-hierarchy-FSH.md) — the
|
- [file-system-hierarchy.md](file-system-development/file-system-hierarchy.md) — the
|
||||||
filesystem layout the file surface serves.
|
filesystem layout the file surface serves.
|
||||||
- [coding-standards.md](coding-standards.md) — danos naming (why the compat spellings
|
- [coding-standards.md](coding-standards.md) — danos naming (why the compat spellings
|
||||||
are confined, and now retired).
|
are confined, and now retired).
|
||||||
|
|||||||
@@ -12,10 +12,15 @@ pub fn build(b: *std.Build) void {
|
|||||||
|
|
||||||
const ipc = kernel.module("ipc");
|
const ipc = kernel.module("ipc");
|
||||||
const time = kernel.module("time");
|
const time = kernel.module("time");
|
||||||
|
// Every client reaches its service by name now: resolve `/protocol/<name>`,
|
||||||
|
// open it, and take the provider's endpoint out of the reply
|
||||||
|
// (docs/os-development/protocol-namespace.md).
|
||||||
|
const channel = kernel.module("channel");
|
||||||
|
|
||||||
_ = b.addModule("display-client", .{
|
_ = b.addModule("display-client", .{
|
||||||
.root_source_file = b.path("display/display-client.zig"),
|
.root_source_file = b.path("display/display-client.zig"),
|
||||||
.imports = &.{
|
.imports = &.{
|
||||||
|
.{ .name = "channel", .module = channel },
|
||||||
.{ .name = "ipc", .module = ipc },
|
.{ .name = "ipc", .module = ipc },
|
||||||
.{ .name = "time", .module = time },
|
.{ .name = "time", .module = time },
|
||||||
.{ .name = "display-protocol", .module = protocol.module("display-protocol") },
|
.{ .name = "display-protocol", .module = protocol.module("display-protocol") },
|
||||||
@@ -24,6 +29,7 @@ pub fn build(b: *std.Build) void {
|
|||||||
_ = b.addModule("input-client", .{
|
_ = b.addModule("input-client", .{
|
||||||
.root_source_file = b.path("input/input-client.zig"),
|
.root_source_file = b.path("input/input-client.zig"),
|
||||||
.imports = &.{
|
.imports = &.{
|
||||||
|
.{ .name = "channel", .module = channel },
|
||||||
.{ .name = "ipc", .module = ipc },
|
.{ .name = "ipc", .module = ipc },
|
||||||
.{ .name = "time", .module = time },
|
.{ .name = "time", .module = time },
|
||||||
.{ .name = "input-protocol", .module = protocol.module("input-protocol") },
|
.{ .name = "input-protocol", .module = protocol.module("input-protocol") },
|
||||||
|
|||||||
@@ -1,9 +1,10 @@
|
|||||||
//! User-space display client: talk to the display service (query the mode, and — from D3
|
//! User-space display client: talk to the display service (query the mode, and — from D3
|
||||||
//! — create layers, draw, and present) without hand-rolling the IPC. The `runtime.block`
|
//! — create layers, draw, and present) without hand-rolling the IPC. The `runtime.block`
|
||||||
//! shape: a cached `.display` lookup with a boot-race retry, then extern-struct request/
|
//! shape: a cached `/protocol/display` open with a boot-race retry, then extern-struct request/
|
||||||
//! reply marshalling. See system/services/display/ and docs/display.md.
|
//! reply marshalling. See system/services/display/ and docs/display.md.
|
||||||
|
|
||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
|
const channel = @import("channel");
|
||||||
const ipc = @import("ipc");
|
const ipc = @import("ipc");
|
||||||
const time = @import("time");
|
const time = @import("time");
|
||||||
const display_protocol = @import("display-protocol");
|
const display_protocol = @import("display-protocol");
|
||||||
@@ -19,13 +20,13 @@ pub const Info = struct {
|
|||||||
/// The service endpoint, looked up once and cached.
|
/// The service endpoint, looked up once and cached.
|
||||||
var handle: ?ipc.Handle = null;
|
var handle: ?ipc.Handle = null;
|
||||||
|
|
||||||
/// Look up the display service, retrying while it comes up (a client races its
|
/// Open `/protocol/display`, retrying while it comes up (a client races the
|
||||||
/// registration at boot). Returns the endpoint, or null if it never appears.
|
/// service's bind at boot). Returns the endpoint, or null if it never appears.
|
||||||
fn service() ?ipc.Handle {
|
fn service() ?ipc.Handle {
|
||||||
if (handle) |h| return h;
|
if (handle) |h| return h;
|
||||||
var attempts: usize = 0;
|
var attempts: usize = 0;
|
||||||
while (attempts < 100) : (attempts += 1) {
|
while (attempts < 100) : (attempts += 1) {
|
||||||
if (ipc.lookup(.display)) |h| {
|
if (channel.openEndpoint("display")) |h| {
|
||||||
handle = h;
|
handle = h;
|
||||||
return h;
|
return h;
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -22,7 +22,7 @@
|
|||||||
//! }
|
//! }
|
||||||
|
|
||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
const abi = @import("abi");
|
const channel = @import("channel");
|
||||||
const ipc = @import("ipc");
|
const ipc = @import("ipc");
|
||||||
const time = @import("time");
|
const time = @import("time");
|
||||||
const input_protocol = @import("input-protocol");
|
const input_protocol = @import("input-protocol");
|
||||||
@@ -44,13 +44,13 @@ pub const device_mouse = input_protocol.device_mouse;
|
|||||||
pub const device_joystick = input_protocol.device_joystick;
|
pub const device_joystick = input_protocol.device_joystick;
|
||||||
pub const device_all = input_protocol.device_all;
|
pub const device_all = input_protocol.device_all;
|
||||||
|
|
||||||
/// Look up the input service, retrying while it is still coming up. Both a subscriber and
|
/// Open `/protocol/input`, retrying while it is still coming up. Both a subscriber and
|
||||||
/// a source race the service's registration at boot, so both wait for it here rather than
|
/// a source race the service's bind at boot, so both wait for it here rather than
|
||||||
/// failing. Returns the service endpoint handle, or null if it never appears.
|
/// failing. Returns the provider's endpoint handle, or null if it never appears.
|
||||||
fn lookupService() ?ipc.Handle {
|
fn lookupService() ?ipc.Handle {
|
||||||
var attempts: usize = 0;
|
var attempts: usize = 0;
|
||||||
while (attempts < 100) : (attempts += 1) {
|
while (attempts < 100) : (attempts += 1) {
|
||||||
if (ipc.lookup(.input)) |handle| return handle;
|
if (channel.openEndpoint("input")) |handle| return handle;
|
||||||
time.sleepMillis(50);
|
time.sleepMillis(50);
|
||||||
}
|
}
|
||||||
return null;
|
return null;
|
||||||
|
|||||||
@@ -1,5 +1,5 @@
|
|||||||
//! The "csv" library domain: shared CSV helpers (comment stripping, field
|
//! The "csv" library domain: shared CSV helpers (comment stripping, field
|
||||||
//! iteration) for the /etc/*.csv config files — the device registry and the
|
//! iteration) for the /system/configuration/*.csv config files — the device registry and the
|
||||||
//! init service list both parse them.
|
//! init service list both parse them.
|
||||||
|
|
||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
|
|||||||
+2
-2
@@ -1,5 +1,5 @@
|
|||||||
//! Minimal CSV helpers shared by the `/etc/*.csv` config files — the device
|
//! Minimal CSV helpers shared by the `/system/configuration/*.csv` config files — the device
|
||||||
//! registry (`/etc/devices.csv`) and the init service list (`/etc/init.csv`).
|
//! registry (`/system/configuration/devices.csv`) and the init service list (`/system/configuration/init.csv`).
|
||||||
//! Freestanding, no allocator: returned fields are slices into the source line,
|
//! Freestanding, no allocator: returned fields are slices into the source line,
|
||||||
//! so the source must outlive them. `#` starts a comment (whole-line or trailing);
|
//! so the source must outlive them. `#` starts a comment (whole-line or trailing);
|
||||||
//! whitespace around a field is trimmed, so columns may be padded for alignment.
|
//! whitespace around a field is trimmed, so columns may be padded for alignment.
|
||||||
|
|||||||
@@ -8,6 +8,7 @@
|
|||||||
//! limit — the same handoff usb-storage uses toward the controller.
|
//! limit — the same handoff usb-storage uses toward the controller.
|
||||||
|
|
||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
|
const channel = @import("channel");
|
||||||
const ipc = @import("ipc");
|
const ipc = @import("ipc");
|
||||||
const time = @import("time");
|
const time = @import("time");
|
||||||
const block_protocol = @import("block-protocol");
|
const block_protocol = @import("block-protocol");
|
||||||
@@ -66,14 +67,14 @@ pub const Device = struct {
|
|||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
/// One lookup attempt, no waiting — for a server that retries on its own
|
/// One open attempt, no waiting — for a server that retries on its own
|
||||||
/// timer (the fat service) instead of blocking its harness in here.
|
/// timer (the fat service) instead of blocking its harness in here.
|
||||||
pub fn tryOpen() ?Device {
|
pub fn tryOpen() ?Device {
|
||||||
if (ipc.lookup(.block)) |handle| return .{ .endpoint = handle };
|
if (channel.openEndpoint("block")) |handle| return .{ .endpoint = handle };
|
||||||
return null;
|
return null;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Look up the block device, retrying generously while the USB storage chain
|
/// Open `/protocol/block`, retrying generously while the USB storage chain
|
||||||
/// (controller reset, enumeration, mass-storage bring-up) comes up.
|
/// (controller reset, enumeration, mass-storage bring-up) comes up.
|
||||||
pub fn open() ?Device {
|
pub fn open() ?Device {
|
||||||
// Patient: the whole USB storage chain (firmware discovery, xHCI reset and
|
// Patient: the whole USB storage chain (firmware discovery, xHCI reset and
|
||||||
@@ -84,7 +85,7 @@ pub fn open() ?Device {
|
|||||||
// completed at ~24 s); a machine whose stick genuinely failed setup should
|
// completed at ~24 s); a machine whose stick genuinely failed setup should
|
||||||
// not sit a further minute pretending otherwise.
|
// not sit a further minute pretending otherwise.
|
||||||
while (attempts < 600) : (attempts += 1) {
|
while (attempts < 600) : (attempts += 1) {
|
||||||
if (ipc.lookup(.block)) |handle| return .{ .endpoint = handle };
|
if (channel.openEndpoint("block")) |handle| return .{ .endpoint = handle };
|
||||||
time.sleepMillis(50);
|
time.sleepMillis(50);
|
||||||
}
|
}
|
||||||
return null;
|
return null;
|
||||||
|
|||||||
@@ -14,6 +14,10 @@ pub fn build(b: *std.Build) void {
|
|||||||
const system_call = kernel.module("system-call");
|
const system_call = kernel.module("system-call");
|
||||||
const ipc = kernel.module("ipc");
|
const ipc = kernel.module("ipc");
|
||||||
const time = kernel.module("time");
|
const time = kernel.module("time");
|
||||||
|
// A driver finds the bus it attaches to by name — `/protocol/device-manager`,
|
||||||
|
// `/protocol/usb-transfer`, `/protocol/block`
|
||||||
|
// (docs/os-development/protocol-namespace.md).
|
||||||
|
const channel = kernel.module("channel");
|
||||||
|
|
||||||
// The devices sub-project's public interface (the flat wire types),
|
// The devices sub-project's public interface (the flat wire types),
|
||||||
// importable by user space, unlike the kernel-internal device model it
|
// importable by user space, unlike the kernel-internal device model it
|
||||||
@@ -55,6 +59,7 @@ pub fn build(b: *std.Build) void {
|
|||||||
.root_source_file = b.path("driver/driver.zig"),
|
.root_source_file = b.path("driver/driver.zig"),
|
||||||
.imports = &.{
|
.imports = &.{
|
||||||
.{ .name = "abi", .module = abi },
|
.{ .name = "abi", .module = abi },
|
||||||
|
.{ .name = "channel", .module = channel },
|
||||||
.{ .name = "device-abi", .module = device_abi },
|
.{ .name = "device-abi", .module = device_abi },
|
||||||
.{ .name = "system-call", .module = system_call },
|
.{ .name = "system-call", .module = system_call },
|
||||||
.{ .name = "ipc", .module = ipc },
|
.{ .name = "ipc", .module = ipc },
|
||||||
@@ -81,6 +86,7 @@ pub fn build(b: *std.Build) void {
|
|||||||
_ = b.addModule("usb", .{
|
_ = b.addModule("usb", .{
|
||||||
.root_source_file = b.path("usb/usb.zig"),
|
.root_source_file = b.path("usb/usb.zig"),
|
||||||
.imports = &.{
|
.imports = &.{
|
||||||
|
.{ .name = "channel", .module = channel },
|
||||||
.{ .name = "ipc", .module = ipc },
|
.{ .name = "ipc", .module = ipc },
|
||||||
.{ .name = "time", .module = time },
|
.{ .name = "time", .module = time },
|
||||||
.{ .name = "usb-transfer-protocol", .module = protocol.module("usb-transfer-protocol") },
|
.{ .name = "usb-transfer-protocol", .module = protocol.module("usb-transfer-protocol") },
|
||||||
@@ -92,12 +98,13 @@ pub fn build(b: *std.Build) void {
|
|||||||
_ = b.addModule("block", .{
|
_ = b.addModule("block", .{
|
||||||
.root_source_file = b.path("block/block.zig"),
|
.root_source_file = b.path("block/block.zig"),
|
||||||
.imports = &.{
|
.imports = &.{
|
||||||
|
.{ .name = "channel", .module = channel },
|
||||||
.{ .name = "ipc", .module = ipc },
|
.{ .name = "ipc", .module = ipc },
|
||||||
.{ .name = "time", .module = time },
|
.{ .name = "time", .module = time },
|
||||||
.{ .name = "block-protocol", .module = protocol.module("block-protocol") },
|
.{ .name = "block-protocol", .module = protocol.module("block-protocol") },
|
||||||
},
|
},
|
||||||
});
|
});
|
||||||
// The device registry: parse /etc/devices.csv into match rules and bind a
|
// The device registry: parse /system/configuration/devices.csv into match rules and bind a
|
||||||
// reported device to a driver. Pure logic (no hardware, no syscalls), so it
|
// reported device to a driver. Pure logic (no hardware, no syscalls), so it
|
||||||
// unit-tests on the host; the device manager imports it.
|
// unit-tests on the host; the device manager imports it.
|
||||||
_ = b.addModule("device-registry", .{
|
_ = b.addModule("device-registry", .{
|
||||||
|
|||||||
@@ -8,7 +8,7 @@
|
|||||||
.kernel = .{ .path = "../kernel" },
|
.kernel = .{ .path = "../kernel" },
|
||||||
// driver speaks device-manager-protocol; block/usb their transfer protocols.
|
// driver speaks device-manager-protocol; block/usb their transfer protocols.
|
||||||
.protocol = .{ .path = "../protocol" },
|
.protocol = .{ .path = "../protocol" },
|
||||||
// device-registry parses /etc/devices.csv with the shared csv helpers.
|
// device-registry parses /system/configuration/devices.csv with the shared csv helpers.
|
||||||
.csv = .{ .path = "../csv" },
|
.csv = .{ .path = "../csv" },
|
||||||
},
|
},
|
||||||
.paths = .{""},
|
.paths = .{""},
|
||||||
|
|||||||
@@ -7,6 +7,7 @@ const std = @import("std");
|
|||||||
const abi = @import("abi");
|
const abi = @import("abi");
|
||||||
const device_abi = @import("device-abi");
|
const device_abi = @import("device-abi");
|
||||||
const sc = @import("system-call");
|
const sc = @import("system-call");
|
||||||
|
const channel = @import("channel");
|
||||||
const ipc = @import("ipc");
|
const ipc = @import("ipc");
|
||||||
const time = @import("time");
|
const time = @import("time");
|
||||||
const device_manager_protocol = @import("device-manager-protocol");
|
const device_manager_protocol = @import("device-manager-protocol");
|
||||||
@@ -170,7 +171,7 @@ const lookup_pause_ms: u64 = 20;
|
|||||||
pub fn hello(role: Role, device_id: u64) ?ipc.Handle {
|
pub fn hello(role: Role, device_id: u64) ?ipc.Handle {
|
||||||
var attempts: u32 = 0;
|
var attempts: u32 = 0;
|
||||||
const manager = while (attempts < lookup_attempts) : (attempts += 1) {
|
const manager = while (attempts < lookup_attempts) : (attempts += 1) {
|
||||||
if (ipc.lookup(.device_manager)) |handle| break handle;
|
if (channel.openEndpoint("device-manager")) |handle| break handle;
|
||||||
time.sleepMillis(lookup_pause_ms);
|
time.sleepMillis(lookup_pause_ms);
|
||||||
} else {
|
} else {
|
||||||
std.log.info("no device manager to hello", .{});
|
std.log.info("no device manager to hello", .{});
|
||||||
|
|||||||
@@ -126,7 +126,7 @@ pub const DeviceDescriptor = extern struct {
|
|||||||
// names with the pci-class module.
|
// names with the pci-class module.
|
||||||
pci_class: u64,
|
pci_class: u64,
|
||||||
// Numeric identity beyond the class triple, mirrored in the bus report's
|
// Numeric identity beyond the class triple, mirrored in the bus report's
|
||||||
// ChildAdded so /etc/devices.csv can bind on it: `vendor`/`device` are the PCI
|
// ChildAdded so /system/configuration/devices.csv can bind on it: `vendor`/`device` are the PCI
|
||||||
// vendor/device (or USB idVendor/idProduct), `subsystem` is the PCI subsystem id
|
// vendor/device (or USB idVendor/idProduct), `subsystem` is the PCI subsystem id
|
||||||
// packed `(subsystem_vendor << 16) | subsystem_device`. Zero where the bus has no
|
// packed `(subsystem_vendor << 16) | subsystem_device`. Zero where the bus has no
|
||||||
// such concept. Defaulted so existing descriptor literals keep compiling and lay
|
// such concept. Defaulted so existing descriptor literals keep compiling and lay
|
||||||
|
|||||||
@@ -1,4 +1,4 @@
|
|||||||
//! The device registry: parse `/etc/devices.csv` into match rules and bind a
|
//! The device registry: parse `/system/configuration/devices.csv` into match rules and bind a
|
||||||
//! reported device to a driver. This is the data-driven replacement for the
|
//! reported device to a driver. This is the data-driven replacement for the
|
||||||
//! device manager's three hand-written `switch` tables (`pciDriverForIdentity`,
|
//! device manager's three hand-written `switch` tables (`pciDriverForIdentity`,
|
||||||
//! `hidDriverFor`, `usbDriverForIdentity`); the registry is now **authoritative**
|
//! `hidDriverFor`, `usbDriverForIdentity`); the registry is now **authoritative**
|
||||||
@@ -11,7 +11,7 @@
|
|||||||
//! That keeps this module freestanding and unit-testable with plain `zig test`.
|
//! That keeps this module freestanding and unit-testable with plain `zig test`.
|
||||||
//!
|
//!
|
||||||
//! The file format (docs/device-driver-development/device-manager.md, and the
|
//! The file format (docs/device-driver-development/device-manager.md, and the
|
||||||
//! `/etc/devices.csv` header itself): one rule per line, nine comma-separated
|
//! `/system/configuration/devices.csv` header itself): one rule per line, nine comma-separated
|
||||||
//! fields, `#` starts a comment (whole-line or trailing), blank lines ignored.
|
//! fields, `#` starts a comment (whole-line or trailing), blank lines ignored.
|
||||||
//!
|
//!
|
||||||
//! bus, base, class, prog_if, vendor, device, subsystem, hid, driver
|
//! bus, base, class, prog_if, vendor, device, subsystem, hid, driver
|
||||||
@@ -226,7 +226,7 @@ fn parseLine(line: []const u8) Line {
|
|||||||
} };
|
} };
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Parse a whole `/etc/devices.csv` into `out_rules`. The string fields of the
|
/// Parse a whole `/system/configuration/devices.csv` into `out_rules`. The string fields of the
|
||||||
/// returned rules point into `source`, which must outlive them.
|
/// returned rules point into `source`, which must outlive them.
|
||||||
pub fn parse(source: []const u8, out_rules: []Rule) ParseResult {
|
pub fn parse(source: []const u8, out_rules: []Rule) ParseResult {
|
||||||
var result: ParseResult = .{ .count = 0, .malformed = 0, .truncated = false };
|
var result: ParseResult = .{ .count = 0, .malformed = 0, .truncated = false };
|
||||||
|
|||||||
@@ -16,6 +16,7 @@
|
|||||||
//! the service harness drops buffered-message payloads — see service.zig).
|
//! the service harness drops buffered-message payloads — see service.zig).
|
||||||
|
|
||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
|
const channel = @import("channel");
|
||||||
const ipc = @import("ipc");
|
const ipc = @import("ipc");
|
||||||
const time = @import("time");
|
const time = @import("time");
|
||||||
const usb_transfer_protocol = @import("usb-transfer-protocol");
|
const usb_transfer_protocol = @import("usb-transfer-protocol");
|
||||||
@@ -130,13 +131,15 @@ pub const Device = struct {
|
|||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
/// Look up the USB bus and open the device with the assigned id, handing over a
|
/// Open `/protocol/usb-transfer` and, on that channel, open the device with the
|
||||||
/// freshly created endpoint for asynchronous interrupt reports. Retries while the
|
/// assigned id, handing over a freshly created endpoint for asynchronous interrupt
|
||||||
/// bus is still coming up (a class driver races the bus driver at boot).
|
/// reports. Retries while the bus is still coming up (a class driver races the bus
|
||||||
|
/// driver at boot). Two opens, deliberately: the first names the contract, the
|
||||||
|
/// second names an object within it.
|
||||||
pub fn open(device_id: u64) ?Device {
|
pub fn open(device_id: u64) ?Device {
|
||||||
var attempts: usize = 0;
|
var attempts: usize = 0;
|
||||||
const bus = while (attempts < 100) : (attempts += 1) {
|
const bus = while (attempts < 100) : (attempts += 1) {
|
||||||
if (ipc.lookup(.usb_bus)) |handle| break handle;
|
if (channel.openEndpoint("usb-transfer")) |handle| break handle;
|
||||||
time.sleepMillis(20);
|
time.sleepMillis(20);
|
||||||
} else return null;
|
} else return null;
|
||||||
|
|
||||||
|
|||||||
@@ -52,7 +52,7 @@ pub fn build(b: *std.Build) void {
|
|||||||
.{ .name = "time", .module = time },
|
.{ .name = "time", .module = time },
|
||||||
},
|
},
|
||||||
});
|
});
|
||||||
_ = b.addModule("file-system", .{
|
const file_system = b.addModule("file-system", .{
|
||||||
.root_source_file = b.path("file-system.zig"),
|
.root_source_file = b.path("file-system.zig"),
|
||||||
.imports = &.{
|
.imports = &.{
|
||||||
.{ .name = "abi", .module = abi },
|
.{ .name = "abi", .module = abi },
|
||||||
@@ -61,6 +61,20 @@ pub fn build(b: *std.Build) void {
|
|||||||
.{ .name = "vfs-protocol", .module = protocol.module("vfs-protocol") },
|
.{ .name = "vfs-protocol", .module = protocol.module("vfs-protocol") },
|
||||||
},
|
},
|
||||||
});
|
});
|
||||||
|
// The channel is the L1 concept made concrete (docs/os-development/communication.md):
|
||||||
|
// it needs the namespace (file-system, to resolve a /protocol name) and the
|
||||||
|
// transport (ipc) both, which is why it lives here rather than in a protocol
|
||||||
|
// module — those import nothing.
|
||||||
|
const channel = b.addModule("channel", .{
|
||||||
|
.root_source_file = b.path("channel.zig"),
|
||||||
|
.imports = &.{
|
||||||
|
.{ .name = "ipc", .module = ipc },
|
||||||
|
.{ .name = "time", .module = time },
|
||||||
|
.{ .name = "file-system", .module = file_system },
|
||||||
|
.{ .name = "vfs-protocol", .module = protocol.module("vfs-protocol") },
|
||||||
|
.{ .name = "envelope", .module = protocol.module("envelope") },
|
||||||
|
},
|
||||||
|
});
|
||||||
_ = b.addModule("memory", .{
|
_ = b.addModule("memory", .{
|
||||||
.root_source_file = b.path("memory/memory.zig"),
|
.root_source_file = b.path("memory/memory.zig"),
|
||||||
.imports = &.{
|
.imports = &.{
|
||||||
@@ -70,10 +84,13 @@ pub fn build(b: *std.Build) void {
|
|||||||
.{ .name = "thread", .module = thread },
|
.{ .name = "thread", .module = thread },
|
||||||
},
|
},
|
||||||
});
|
});
|
||||||
|
// The harness binds the service's contract name at startup, which is a
|
||||||
|
// conversation with the registry — hence channel (and time, for the patience
|
||||||
|
// a provider that beat init to the mount needs).
|
||||||
_ = b.addModule("service", .{
|
_ = b.addModule("service", .{
|
||||||
.root_source_file = b.path("service.zig"),
|
.root_source_file = b.path("service.zig"),
|
||||||
.imports = &.{
|
.imports = &.{
|
||||||
.{ .name = "abi", .module = abi },
|
.{ .name = "channel", .module = channel },
|
||||||
.{ .name = "ipc", .module = ipc },
|
.{ .name = "ipc", .module = ipc },
|
||||||
.{ .name = "process", .module = process },
|
.{ .name = "process", .module = process },
|
||||||
},
|
},
|
||||||
@@ -101,4 +118,22 @@ pub fn build(b: *std.Build) void {
|
|||||||
});
|
});
|
||||||
test_step.dependOn(&b.addRunArtifact(kernel_tests).step);
|
test_step.dependOn(&b.addRunArtifact(kernel_tests).step);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// channel needs its whole import set to compile at all; only its framing is
|
||||||
|
// host-runnable (the syscall seams are x86_64-only, and unreferenced from
|
||||||
|
// the tests), so that is what it tests.
|
||||||
|
const channel_tests = b.addTest(.{
|
||||||
|
.root_module = b.createModule(.{
|
||||||
|
.root_source_file = b.path("channel.zig"),
|
||||||
|
.target = b.resolveTargetQuery(.{}),
|
||||||
|
.imports = &.{
|
||||||
|
.{ .name = "ipc", .module = ipc },
|
||||||
|
.{ .name = "time", .module = time },
|
||||||
|
.{ .name = "file-system", .module = file_system },
|
||||||
|
.{ .name = "vfs-protocol", .module = protocol.module("vfs-protocol") },
|
||||||
|
.{ .name = "envelope", .module = protocol.module("envelope") },
|
||||||
|
},
|
||||||
|
}),
|
||||||
|
});
|
||||||
|
test_step.dependOn(&b.addRunArtifact(channel_tests).step);
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,343 @@
|
|||||||
|
//! `Channel` — layer L1 of the communication stack
|
||||||
|
//! (docs/os-development/communication.md) made concrete. A program holds a
|
||||||
|
//! channel that speaks a protocol; it does not hold a raw handle and marshal
|
||||||
|
//! bytes at one. The channel is the answer to "who am I talking to", decided
|
||||||
|
//! once at establishment, so nothing after that ever routes a party again:
|
||||||
|
//! every packet's `target` addresses an *object* within the peer already chosen.
|
||||||
|
//!
|
||||||
|
//! **Possession of the Channel is the connection.** There is no connect step, no
|
||||||
|
//! session id, no reconnect handshake — the endpoint capability inside is the
|
||||||
|
//! whole of the relationship, and it cannot be forged, only handed over. Which
|
||||||
|
//! also means a channel is a resource: `close` it, or it occupies a handle-table
|
||||||
|
//! slot for the life of the process.
|
||||||
|
//!
|
||||||
|
//! **A dead provider surfaces as `-EPEER`, and the recovery is to re-open.**
|
||||||
|
//! When the process on the other end exits, the kernel fails calls on its
|
||||||
|
//! endpoint rather than blocking forever; `call` returns null. The client does
|
||||||
|
//! not repair the channel — it discards it and opens the name again, which
|
||||||
|
//! reaches whatever instance the registry now points at. The restart story
|
||||||
|
//! falls out of the naming layer for free; no protocol needs a reconnect verb.
|
||||||
|
//!
|
||||||
|
//! `open` resolves a `/protocol/<name>` path through the kernel VFS router and
|
||||||
|
//! takes the provider's endpoint from the open reply's capability. The registry
|
||||||
|
//! answering it is init, PID 1, which mounts `/protocol` before it spawns anyone
|
||||||
|
//! (docs/os-development/protocol-namespace.md); `bind` below is the other half —
|
||||||
|
//! how a provider claims the name in the first place.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const ipc = @import("ipc");
|
||||||
|
const time = @import("time");
|
||||||
|
const file_system = @import("file-system");
|
||||||
|
const vfs_protocol = @import("vfs-protocol");
|
||||||
|
const envelope = @import("envelope");
|
||||||
|
|
||||||
|
/// Longest `/protocol/...` path this client marshals. The registry's names are
|
||||||
|
/// short by construction (a contract leaf, not a file path), and the buffer is
|
||||||
|
/// on the stack of whoever opens.
|
||||||
|
pub const path_maximum: usize = 224;
|
||||||
|
|
||||||
|
/// Where the protocol namespace is rooted — the one path prefix in the system
|
||||||
|
/// that names contracts rather than files. Spelled once, here, so no caller
|
||||||
|
/// builds it by hand (docs/file-system-development/file-system-hierarchy.md).
|
||||||
|
pub const root: []const u8 = "/protocol";
|
||||||
|
|
||||||
|
/// Longest contract name — the part after `/protocol/`. Short by construction:
|
||||||
|
/// a leaf like `display`, or a subtree leaf like `test/shared-memory`.
|
||||||
|
pub const name_maximum: usize = 64;
|
||||||
|
|
||||||
|
/// What a `call` came back with: the provider's status, the reply payload (the
|
||||||
|
/// bytes after the `Status`, in the caller's own buffer), and any capability the
|
||||||
|
/// reply carried.
|
||||||
|
pub const Response = struct {
|
||||||
|
status: envelope.Status,
|
||||||
|
payload: []u8,
|
||||||
|
capability: ?ipc.Handle,
|
||||||
|
|
||||||
|
/// Whether the provider answered success. A negative status is its refusal
|
||||||
|
/// (`-ENOSYS` for a verb it does not implement, and so on).
|
||||||
|
pub fn succeeded(self: Response) bool {
|
||||||
|
return self.status.status == 0;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// An open conversation with one provider, speaking one protocol.
|
||||||
|
pub const Channel = struct {
|
||||||
|
/// The provider's endpoint. Sending into it is the only thing this handle
|
||||||
|
/// can do — an endpoint is a mailbox owned by its creator, and that
|
||||||
|
/// direction never reverses.
|
||||||
|
endpoint: ipc.Handle,
|
||||||
|
|
||||||
|
/// Adopt an endpoint that arrived some other way — a capability delivered
|
||||||
|
/// in a reply, or one a supervisor wired in at spawn time (P5). The channel
|
||||||
|
/// takes ownership of the handle.
|
||||||
|
pub fn adopt(endpoint: ipc.Handle) Channel {
|
||||||
|
return .{ .endpoint = endpoint };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Establish a channel by name: resolve `/protocol/<name>` to the registry
|
||||||
|
/// backend, `open` the contract there, and take the provider's endpoint from
|
||||||
|
/// the reply's capability. Null if the path does not resolve, the registry
|
||||||
|
/// refuses (an ungranted name is refused *as* not-found), or the reply
|
||||||
|
/// carries no capability.
|
||||||
|
///
|
||||||
|
/// The path is spoken exactly once, here. Everything afterwards is integers
|
||||||
|
/// in the packet header.
|
||||||
|
pub fn open(path: []const u8) ?Channel {
|
||||||
|
return .{ .endpoint = openPath(path) orelse return null };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Establish a channel by contract name — `open` with `/protocol/` supplied,
|
||||||
|
/// which is how every caller in the system spells it.
|
||||||
|
pub fn connect(name: []const u8) ?Channel {
|
||||||
|
return .{ .endpoint = openEndpoint(name) orelse return null };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Send one request packet and block for the reply: `[Header][request]` out,
|
||||||
|
/// `[Status][reply]` back. `request` is the bytes *after* the header — the
|
||||||
|
/// protocol's fixed part plus any tail — because the header is this call's
|
||||||
|
/// to lay down. The reply's payload lands in `into`.
|
||||||
|
///
|
||||||
|
/// Null means the transport failed, which today means one of: a dead
|
||||||
|
/// provider (`-EPEER` — discard this channel and `open` the name again), an
|
||||||
|
/// oversized packet, or a bad handle. A provider that answered *and refused*
|
||||||
|
/// is not a failure here: it comes back with a negative `Response.status`.
|
||||||
|
pub fn call(self: Channel, header: envelope.Header, request: []const u8, into: []u8) ?Response {
|
||||||
|
return self.callCapability(header, request, into, null);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// As `call`, handing the provider a capability with the request — the only
|
||||||
|
/// direction-crossing move kernel-ipc offers, and how `subscribe` delivers
|
||||||
|
/// the subscriber's own endpoint.
|
||||||
|
pub fn callCapability(
|
||||||
|
self: Channel,
|
||||||
|
header: envelope.Header,
|
||||||
|
request: []const u8,
|
||||||
|
into: []u8,
|
||||||
|
capability: ?ipc.Handle,
|
||||||
|
) ?Response {
|
||||||
|
var packet: [envelope.packet_maximum]u8 = undefined;
|
||||||
|
const framed = frame(header, request, &packet) orelse return null;
|
||||||
|
|
||||||
|
var reply: [envelope.packet_maximum]u8 = undefined;
|
||||||
|
const answer = ipc.callCap(self.endpoint, framed, &reply, capability) catch return null;
|
||||||
|
const status = envelope.statusOf(reply[0..answer.len]) orelse return null;
|
||||||
|
const available = @min(answer.len - envelope.prefix_size, @as(usize, status.len));
|
||||||
|
const taken = @min(available, into.len);
|
||||||
|
@memcpy(into[0..taken], reply[envelope.prefix_size..][0..taken]);
|
||||||
|
return .{ .status = status, .payload = into[0..taken], .capability = answer.cap };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Push one event packet and return immediately — no reply owed, and a slow
|
||||||
|
/// or dead peer can never stall the sender. Bounded by `post_maximum`: an
|
||||||
|
/// event that does not fit is refused here rather than split, because a
|
||||||
|
/// packet is never fragmented.
|
||||||
|
pub fn send(self: Channel, header: envelope.Header, payload: []const u8) bool {
|
||||||
|
var packet: [envelope.post_maximum]u8 = undefined;
|
||||||
|
const framed = frame(header, payload, &packet) orelse return false;
|
||||||
|
return ipc.send(self.endpoint, framed);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Ask the provider what it is: the reserved `describe` verb, answered by
|
||||||
|
/// every protocol built through `envelope.Define`. The name and version come
|
||||||
|
/// back in `into`, which the returned `Described` borrows.
|
||||||
|
pub fn describe(self: Channel, into: []u8) ?envelope.Described {
|
||||||
|
var request: [envelope.packet_maximum]u8 = undefined;
|
||||||
|
const packet = envelope.encodeDescribe(&request) orelse return null;
|
||||||
|
|
||||||
|
var reply: [envelope.packet_maximum]u8 = undefined;
|
||||||
|
const answer = ipc.callCap(self.endpoint, packet, &reply, null) catch return null;
|
||||||
|
const taken = @min(answer.len, into.len);
|
||||||
|
@memcpy(into[0..taken], reply[0..taken]);
|
||||||
|
return envelope.decodeDescribe(into[0..taken]);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Drop the provider's endpoint and free the handle-table slot. The
|
||||||
|
/// conversation is over the moment the capability is gone — there is nothing
|
||||||
|
/// else holding it open.
|
||||||
|
pub fn close(self: Channel) void {
|
||||||
|
_ = ipc.close(self.endpoint);
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
// --- the namespace: resolving, opening, and claiming a contract name ---------
|
||||||
|
|
||||||
|
/// Where a `/protocol/...` path routed: the registry's endpoint, plus the path
|
||||||
|
/// rewritten mount-relative (`/display` for `/protocol/display`). The handle is
|
||||||
|
/// deduplicated by the kernel across resolves and shared with every other user
|
||||||
|
/// of that mount, so it is never ours to close.
|
||||||
|
const Registry = struct {
|
||||||
|
handle: ipc.Handle,
|
||||||
|
relative: [path_maximum]u8,
|
||||||
|
relative_len: usize,
|
||||||
|
|
||||||
|
fn path(self: *const Registry) []const u8 {
|
||||||
|
return self.relative[0..self.relative_len];
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Route `path` to whatever backend serves it. Null when nothing is mounted
|
||||||
|
/// there — under `/protocol` that means the registry is not up yet, which is a
|
||||||
|
/// *retry*, not a refusal. A kernel-served route (the read-only `/system` tree)
|
||||||
|
/// is the wrong path, not a channel, and is refused here.
|
||||||
|
fn reach(path: []const u8) ?Registry {
|
||||||
|
var out: Registry = .{ .handle = 0, .relative = undefined, .relative_len = 0 };
|
||||||
|
const route = file_system.fsResolve(path, 0, &out.relative) orelse return null;
|
||||||
|
switch (route) {
|
||||||
|
.kernel => return null,
|
||||||
|
.backend => |b| {
|
||||||
|
out.handle = b.handle;
|
||||||
|
out.relative_len = b.path_len;
|
||||||
|
return out;
|
||||||
|
},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// One vfs-protocol round trip at a backend: fixed header, inline payload, and
|
||||||
|
/// an optional capability in each direction.
|
||||||
|
fn transact(
|
||||||
|
handle: ipc.Handle,
|
||||||
|
operation: vfs_protocol.Operation,
|
||||||
|
payload: []const u8,
|
||||||
|
send_capability: ?ipc.Handle,
|
||||||
|
) ?struct { reply: vfs_protocol.Reply, capability: ?ipc.Handle } {
|
||||||
|
var request: [vfs_protocol.message_maximum]u8 = undefined;
|
||||||
|
if (vfs_protocol.request_size + payload.len > request.len) return null;
|
||||||
|
const header = vfs_protocol.Request{
|
||||||
|
.operation = operation,
|
||||||
|
.node = 0,
|
||||||
|
.offset = 0,
|
||||||
|
.len = @intCast(payload.len),
|
||||||
|
.flags = 0,
|
||||||
|
};
|
||||||
|
@memcpy(request[0..vfs_protocol.request_size], std.mem.asBytes(&header));
|
||||||
|
@memcpy(request[vfs_protocol.request_size..][0..payload.len], payload);
|
||||||
|
|
||||||
|
var reply: [vfs_protocol.message_maximum]u8 = undefined;
|
||||||
|
const answer = ipc.callCap(handle, request[0 .. vfs_protocol.request_size + payload.len], &reply, send_capability) catch return null;
|
||||||
|
if (answer.len < vfs_protocol.reply_size) return null;
|
||||||
|
return .{
|
||||||
|
.reply = std.mem.bytesToValue(vfs_protocol.Reply, reply[0..vfs_protocol.reply_size]),
|
||||||
|
.capability = answer.cap,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Resolve an absolute `/protocol/...` path and take the provider's endpoint out
|
||||||
|
/// of the open reply's capability.
|
||||||
|
fn openPath(path: []const u8) ?ipc.Handle {
|
||||||
|
const registry = reach(path) orelse return null;
|
||||||
|
const answered = transact(registry.handle, .open, registry.path(), null) orelse return null;
|
||||||
|
if (answered.reply.status != 0) return null;
|
||||||
|
// The capability *is* the channel — an open that succeeds without one was
|
||||||
|
// answered by a file backend, which does not speak protocols.
|
||||||
|
return answered.capability;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The provider's raw endpoint behind `/protocol/<name>`. The transitional form,
|
||||||
|
/// for the clients that still marshal their protocol's bytes by hand; P4 moves
|
||||||
|
/// them onto `Channel` proper and this shrinks back to `connect`.
|
||||||
|
///
|
||||||
|
/// Null covers both "no such contract" and "you may not have it" — deliberately
|
||||||
|
/// the same answer (protocol-namespace.md: enforcement is absence), and also
|
||||||
|
/// "the registry is not mounted yet", which is why every caller retries.
|
||||||
|
pub fn openEndpoint(name: []const u8) ?ipc.Handle {
|
||||||
|
var path: [path_maximum]u8 = undefined;
|
||||||
|
const full = join(name, &path) orelse return null;
|
||||||
|
return openPath(full);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Claim `/protocol/<name>` for `endpoint`: the registry records the name
|
||||||
|
/// against this process and hands the endpoint to whoever opens it afterwards.
|
||||||
|
/// The endpoint rides the call as its capability, the one direction-crossing
|
||||||
|
/// move kernel-ipc offers.
|
||||||
|
///
|
||||||
|
/// Three-valued on purpose. **Null** is "the registry could not be reached" —
|
||||||
|
/// it is not mounted yet, which happens when a provider starts before init has
|
||||||
|
/// finished coming up, and the answer is to retry. A **value** is the registry's
|
||||||
|
/// verdict and is final: 0 bound, `-EPERM` this binary is not granted that name,
|
||||||
|
/// `-EBUSY` a live provider already holds it.
|
||||||
|
pub fn bind(name: []const u8, endpoint: ipc.Handle) ?i32 {
|
||||||
|
const registry = reach(root) orelse return null;
|
||||||
|
const answered = transact(registry.handle, .bind, name, endpoint) orelse return null;
|
||||||
|
return answered.reply.status;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// How long a provider keeps offering itself before giving up. The registry is
|
||||||
|
/// init, which mounts `/protocol` before it spawns anyone, so in a normal boot
|
||||||
|
/// the first try lands; a provider the kernel test harness starts may well beat
|
||||||
|
/// init to the mount, which is what the patience is for. Four seconds of 20 ms
|
||||||
|
/// tries — the same cadence every client in the tree spends finding a service.
|
||||||
|
const bind_attempts: u32 = 200;
|
||||||
|
const bind_retry_ms: u64 = 20;
|
||||||
|
|
||||||
|
/// `bind`, waiting out a registry that is not mounted yet. Only unreachability
|
||||||
|
/// is retried: a registry that *answered* has decided, and asking again cannot
|
||||||
|
/// change its mind. True when the name is ours.
|
||||||
|
pub fn bindPatiently(name: []const u8, endpoint: ipc.Handle) bool {
|
||||||
|
var attempt: u32 = 0;
|
||||||
|
while (attempt < bind_attempts) : (attempt += 1) {
|
||||||
|
if (bind(name, endpoint)) |status| return status == 0;
|
||||||
|
time.sleepMillis(bind_retry_ms);
|
||||||
|
}
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `/protocol/` + `name`, in the caller's buffer. Null if the name is empty or
|
||||||
|
/// longer than the namespace admits.
|
||||||
|
fn join(name: []const u8, buffer: []u8) ?[]u8 {
|
||||||
|
if (name.len == 0 or name.len > name_maximum) return null;
|
||||||
|
const total = root.len + 1 + name.len;
|
||||||
|
if (total > buffer.len) return null;
|
||||||
|
@memcpy(buffer[0..root.len], root);
|
||||||
|
buffer[root.len] = '/';
|
||||||
|
@memcpy(buffer[root.len + 1 ..][0..name.len], name);
|
||||||
|
return buffer[0..total];
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Lay a packet down: the folded header first, then the protocol's bytes. Null
|
||||||
|
/// when it would not fit the buffer — the same rule as `envelope`'s framing,
|
||||||
|
/// applied where the buffer is the transport's, not the protocol's.
|
||||||
|
fn frame(header: envelope.Header, body: []const u8, buffer: []u8) ?[]u8 {
|
||||||
|
const total = envelope.prefix_size + body.len;
|
||||||
|
if (total > buffer.len) return null;
|
||||||
|
@memcpy(buffer[0..envelope.prefix_size], std.mem.asBytes(&header));
|
||||||
|
@memcpy(buffer[envelope.prefix_size..][0..body.len], body);
|
||||||
|
return buffer[0..total];
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- tests ------------------------------------------------------------------
|
||||||
|
//
|
||||||
|
// The syscall half cannot run on the host, and there is no registry to reach
|
||||||
|
// until P2 — so what is testable here is the framing, which is the part with
|
||||||
|
// arithmetic in it.
|
||||||
|
|
||||||
|
const testing = std.testing;
|
||||||
|
|
||||||
|
test "a framed packet is the header followed by the protocol's bytes" {
|
||||||
|
var buffer: [envelope.packet_maximum]u8 = undefined;
|
||||||
|
const header = envelope.Header{ .operation = envelope.first_protocol_operation, .target = 9 };
|
||||||
|
const packet = frame(header, "body", &buffer).?;
|
||||||
|
|
||||||
|
try testing.expectEqual(envelope.prefix_size + "body".len, packet.len);
|
||||||
|
const decoded = envelope.headerOf(packet).?;
|
||||||
|
try testing.expectEqual(envelope.first_protocol_operation, decoded.operation);
|
||||||
|
try testing.expectEqual(@as(u64, 9), decoded.target);
|
||||||
|
try testing.expectEqualStrings("body", packet[envelope.prefix_size..]);
|
||||||
|
}
|
||||||
|
|
||||||
|
test "a contract name joins the namespace root exactly once" {
|
||||||
|
var buffer: [path_maximum]u8 = undefined;
|
||||||
|
try testing.expectEqualStrings("/protocol/display", join("display", &buffer).?);
|
||||||
|
try testing.expectEqualStrings("/protocol/test/shared-memory", join("test/shared-memory", &buffer).?);
|
||||||
|
try testing.expect(join("", &buffer) == null);
|
||||||
|
try testing.expect(join("x" ** (name_maximum + 1), &buffer) == null);
|
||||||
|
}
|
||||||
|
|
||||||
|
test "framing refuses a packet that would not fit rather than truncating it" {
|
||||||
|
var post: [envelope.post_maximum]u8 = undefined;
|
||||||
|
const header = envelope.Header{ .operation = envelope.first_protocol_operation };
|
||||||
|
const body = [_]u8{0} ** (envelope.post_maximum - envelope.prefix_size);
|
||||||
|
const one_too_many = body ++ [_]u8{0};
|
||||||
|
|
||||||
|
try testing.expect(frame(header, &body, &post) != null);
|
||||||
|
try testing.expect(frame(header, &one_too_many, &post) == null);
|
||||||
|
}
|
||||||
@@ -39,6 +39,7 @@ fn kindFromWire(value: u32) Kind {
|
|||||||
@intFromEnum(Kind.symbolic_link) => .symbolic_link,
|
@intFromEnum(Kind.symbolic_link) => .symbolic_link,
|
||||||
@intFromEnum(Kind.fifo) => .fifo,
|
@intFromEnum(Kind.fifo) => .fifo,
|
||||||
@intFromEnum(Kind.socket) => .socket,
|
@intFromEnum(Kind.socket) => .socket,
|
||||||
|
@intFromEnum(Kind.protocol) => .protocol,
|
||||||
else => .regular,
|
else => .regular,
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
@@ -305,7 +306,7 @@ pub fn makePath(path: []const u8) bool {
|
|||||||
while (end < path.len and path[end] != '/') end += 1;
|
while (end < path.len and path[end] != '/') end += 1;
|
||||||
const prefix = path[0..end];
|
const prefix = path[0..end];
|
||||||
if (prefix.len == 0 or (prefix.len == 1 and prefix[0] == '/')) continue;
|
if (prefix.len == 0 or (prefix.len == 1 and prefix[0] == '/')) continue;
|
||||||
// Best-effort per prefix: components at or above a mount point ("/mnt")
|
// Best-effort per prefix: components at or above a mount point ("/volumes")
|
||||||
// are router names, not filesystem nodes — they neither exist as nodes
|
// are router names, not filesystem nodes — they neither exist as nodes
|
||||||
// nor accept mkdir, and that is fine. Only the final verdict counts.
|
// nor accept mkdir, and that is fine. Only the final verdict counts.
|
||||||
if (!exists(prefix)) _ = makeDirectory(prefix);
|
if (!exists(prefix)) _ = makeDirectory(prefix);
|
||||||
@@ -348,8 +349,9 @@ pub fn mount(target: []const u8, backend: ipc.Handle) bool {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// As `mount`, with a backend-side rewrite prefix: a path under `target` reaches
|
/// As `mount`, with a backend-side rewrite prefix: a path under `target` reaches
|
||||||
/// the backend as `rewrite` + the mount-relative tail. How one volume serves two
|
/// the backend as `rewrite` + the mount-relative tail. How one volume serves
|
||||||
/// mounts ("/mnt/usb" from its root, "/var" from its /var subtree).
|
/// several mounts ("/volumes/usb" from its root, "/system/logs" from its
|
||||||
|
/// /system/logs subtree).
|
||||||
pub fn mountRewritten(target: []const u8, backend: ipc.Handle, rewrite: []const u8) bool {
|
pub fn mountRewritten(target: []const u8, backend: ipc.Handle, rewrite: []const u8) bool {
|
||||||
return fsMount(target, backend, rewrite);
|
return fsMount(target, backend, rewrite);
|
||||||
}
|
}
|
||||||
|
|||||||
+54
-11
@@ -29,10 +29,10 @@ pub fn createIpcEndpoint() ?Handle {
|
|||||||
return if (failed(r)) null else r;
|
return if (failed(r)) null else r;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Publish endpoint `h` under a well-known service id so other processes find it.
|
// `register`/`lookup` lived here — the two wrappers over the flat ServiceId
|
||||||
pub fn register(id: abi.ServiceId, h: Handle) bool {
|
// registry. Naming is not a system call any more: a provider binds its contract
|
||||||
return !failed(sc.systemCall2(.ipc_register, @intFromEnum(id), h));
|
// name at the registry and a client resolves and opens `/protocol/<name>`, both
|
||||||
}
|
// through `channel` (docs/os-development/protocol-namespace.md).
|
||||||
|
|
||||||
/// Drop a capability handle (endpoint, shared-memory, or DMA-region) and free its table
|
/// Drop a capability handle (endpoint, shared-memory, or DMA-region) and free its table
|
||||||
/// slot. A forwarding hop closes a cap it passed on; a binder closes a DMA-region cap
|
/// slot. A forwarding hop closes a cap it passed on; a binder closes a DMA-region cap
|
||||||
@@ -42,13 +42,6 @@ pub fn close(h: Handle) bool {
|
|||||||
return !failed(sc.systemCall1(.handle_close, h));
|
return !failed(sc.systemCall1(.handle_close, h));
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Find the endpoint published under `id`, installing a handle to it in this
|
|
||||||
/// process.
|
|
||||||
pub fn lookup(id: abi.ServiceId) ?Handle {
|
|
||||||
const r = sc.systemCall1(.ipc_lookup, @intFromEnum(id));
|
|
||||||
return if (failed(r)) null else r;
|
|
||||||
}
|
|
||||||
|
|
||||||
pub const CallError = error{Failed};
|
pub const CallError = error{Failed};
|
||||||
|
|
||||||
/// The result of a capability-passing `callCap`: the reply length, and the handle of
|
/// The result of a capability-passing `callCap`: the reply length, and the handle of
|
||||||
@@ -178,6 +171,56 @@ pub const Received = struct {
|
|||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
|
/// A capability that arrived with one turn of a receive loop, and the ownership
|
||||||
|
/// rule for it: **the turn owns it until a handler takes it, and closes whatever
|
||||||
|
/// is left.**
|
||||||
|
///
|
||||||
|
/// The kernel installs a sent capability in the receiver's handle table whenever
|
||||||
|
/// the caller attached one, *independent of the message's length or kind*
|
||||||
|
/// (system/kernel/ipc-synchronous.zig `replyWait`), so every path out of a loop
|
||||||
|
/// has to dispose of one — including the paths that never look at the message.
|
||||||
|
/// The table is thirty-two slots, and `ipc_call` does not dedupe, so a client
|
||||||
|
/// looping on `callCap(server, &.{}, endpoint)` spends one slot per call: about
|
||||||
|
/// thirty-two zero-length pings and the service can never accept another
|
||||||
|
/// capability, which means no subscribe and no shared-memory handover, for the
|
||||||
|
/// rest of the boot. It is unauthenticated and it is two lines to write.
|
||||||
|
///
|
||||||
|
/// So ownership is structural rather than a close per branch — the per-branch
|
||||||
|
/// version has already failed twice in this tree, in PID 1's ping path and in
|
||||||
|
/// every `service.run` callback that simply ignored its capability argument.
|
||||||
|
/// Written this way, forgetting **closes**, and *keeping* a capability is the
|
||||||
|
/// thing a handler has to say out loud:
|
||||||
|
///
|
||||||
|
/// ```zig
|
||||||
|
/// var arrived: ipc.Arrival = .{ .handle = got.cap };
|
||||||
|
/// defer arrived.release(); // every exit path, including `continue`
|
||||||
|
/// ...
|
||||||
|
/// const kept = arrived.take().?; // claimed: mine to hold or close
|
||||||
|
/// ```
|
||||||
|
pub const Arrival = struct {
|
||||||
|
handle: ?Handle = null,
|
||||||
|
|
||||||
|
/// Look without claiming — a handler that may still refuse wants no close of
|
||||||
|
/// its own on the refusal paths.
|
||||||
|
pub fn peek(self: *const Arrival) ?Handle {
|
||||||
|
return self.handle;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Claim ownership: from here the capability is the taker's to keep or close,
|
||||||
|
/// and the turn will not touch it.
|
||||||
|
pub fn take(self: *Arrival) ?Handle {
|
||||||
|
defer self.handle = null;
|
||||||
|
return self.handle;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Close whatever nobody claimed. Idempotent, so it is safe as a `defer` next
|
||||||
|
/// to any number of `take`s.
|
||||||
|
pub fn release(self: *Arrival) void {
|
||||||
|
if (self.handle) |handle| _ = close(handle);
|
||||||
|
self.handle = null;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
/// Server side of IPC_ReplyWait: deliver `reply` to the client last received (if any,
|
/// Server side of IPC_ReplyWait: deliver `reply` to the client last received (if any,
|
||||||
/// optionally handing it `send_cap`), then block until the next request arrives in
|
/// optionally handing it `send_cap`), then block until the next request arrives in
|
||||||
/// `receive`. Returns its length, the sender badge, and any capability the request
|
/// `receive`. Returns its length, the sender badge, and any capability the request
|
||||||
|
|||||||
@@ -142,6 +142,18 @@ pub fn subscribeExits(endpoint: usize) bool {
|
|||||||
/// snapshot buffer without importing `abi` itself.
|
/// snapshot buffer without importing `abi` itself.
|
||||||
pub const ProcessDescriptor = abi.ProcessDescriptor;
|
pub const ProcessDescriptor = abi.ProcessDescriptor;
|
||||||
|
|
||||||
|
/// The calling task's own kernel id — its row in the process table, and the value
|
||||||
|
/// every other process sees as this one's `supervisor` after it spawns them. For a
|
||||||
|
/// single-threaded program that is its process id; in a threaded one it is the
|
||||||
|
/// calling thread's id (`Thread.getCurrentId` is the same system call, named for
|
||||||
|
/// the threading vocabulary). Ids are monotonic and never reused
|
||||||
|
/// (system/kernel/process.zig), which is what makes comparing one an identity
|
||||||
|
/// test where comparing a *name* is only a resemblance test — the registrar in
|
||||||
|
/// init leans on exactly that.
|
||||||
|
pub fn taskId() u32 {
|
||||||
|
return @intCast(sc.systemCall0(.thread_self));
|
||||||
|
}
|
||||||
|
|
||||||
/// Give up the rest of this quantum.
|
/// Give up the rest of this quantum.
|
||||||
pub fn yield() void {
|
pub fn yield() void {
|
||||||
_ = sc.systemCall0(.yield);
|
_ = sc.systemCall0(.yield);
|
||||||
|
|||||||
+50
-16
@@ -6,13 +6,19 @@
|
|||||||
//! loop chose, never on a hijacked stack — the whole reason signals are
|
//! loop chose, never on a hijacked stack — the whole reason signals are
|
||||||
//! messages.
|
//! messages.
|
||||||
//!
|
//!
|
||||||
|
//! One rule a service author does have to know, and it is stated on
|
||||||
|
//! `Callbacks.on_message`: **a capability that arrives belongs to the turn** —
|
||||||
|
//! the loop closes it unless the callback claims it with `take()`. Forgetting is
|
||||||
|
//! therefore safe, and keeping is explicit; the opposite arrangement quietly
|
||||||
|
//! spends a handle-table slot per request.
|
||||||
|
//!
|
||||||
//! The liveness probe: a **zero-length request is the universal ping**, answered
|
//! The liveness probe: a **zero-length request is the universal ping**, answered
|
||||||
//! with a zero-length reply by the harness itself. No protocol's requests start
|
//! with a zero-length reply by the harness itself. No protocol's requests start
|
||||||
//! at length zero, so the encoding cannot collide, and there is nothing for a
|
//! at length zero, so the encoding cannot collide, and there is nothing for a
|
||||||
//! service author to implement — a wedged service simply fails to answer, which
|
//! service author to implement — a wedged service simply fails to answer, which
|
||||||
//! is the diagnosis (see docs/ipc.md).
|
//! is the diagnosis (see docs/ipc.md).
|
||||||
|
|
||||||
const abi = @import("abi");
|
const channel = @import("channel");
|
||||||
const ipc = @import("ipc");
|
const ipc = @import("ipc");
|
||||||
const process = @import("process");
|
const process = @import("process");
|
||||||
|
|
||||||
@@ -22,10 +28,22 @@ pub const Callbacks = struct {
|
|||||||
/// Return false to abort startup (the process exits).
|
/// Return false to abort startup (the process exits).
|
||||||
init: ?*const fn (endpoint: ipc.Handle) bool = null,
|
init: ?*const fn (endpoint: ipc.Handle) bool = null,
|
||||||
/// One protocol request from `sender` (a task id): write the reply into
|
/// One protocol request from `sender` (a task id): write the reply into
|
||||||
/// `reply`, return its length. `capability` is the handle the request
|
/// `reply`, return its length. The zero-length ping never reaches this.
|
||||||
/// carried, if any (M13 cap passing — how a subscriber hands over its
|
///
|
||||||
/// endpoint). The zero-length ping never reaches this.
|
/// `arrived` is the capability the request carried (M13 cap passing — how a
|
||||||
on_message: *const fn (message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Handle) usize,
|
/// subscriber hands over its endpoint), and it comes with **an ownership
|
||||||
|
/// rule: the turn owns it, and a handler that wants to keep it must say so
|
||||||
|
/// with `take()`.** Whatever is left when this returns, the loop closes.
|
||||||
|
/// `peek()` reads it without claiming, which is what a handler that may
|
||||||
|
/// still refuse wants — no close of its own on the refusal paths.
|
||||||
|
///
|
||||||
|
/// The rule is stated here, in the contract, because the alternative has
|
||||||
|
/// failed in practice: an implementation that simply ignored a `?ipc.Handle`
|
||||||
|
/// argument leaked a handle table slot per request, and every operation
|
||||||
|
/// except a subscribe ignores it. Thirty-two such requests — zero-length
|
||||||
|
/// pings will do, and they need no authorization — and the service can never
|
||||||
|
/// accept another capability for the rest of the boot. See `ipc.Arrival`.
|
||||||
|
on_message: *const fn (message: []const u8, reply: []u8, sender: u32, arrived: *ipc.Arrival) usize,
|
||||||
/// A notification that is not a signal — a subscribed exit event, a bound
|
/// A notification that is not a signal — a subscribed exit event, a bound
|
||||||
/// IRQ, a timer landing. The raw badge; decode with the ipc helpers.
|
/// IRQ, a timer landing. The raw badge; decode with the ipc helpers.
|
||||||
on_notification: ?*const fn (badge: u64) void = null,
|
on_notification: ?*const fn (badge: u64) void = null,
|
||||||
@@ -35,19 +53,25 @@ pub const Callbacks = struct {
|
|||||||
/// the return itself — never put *necessary* work here (iron rule 1: a kill
|
/// the return itself — never put *necessary* work here (iron rule 1: a kill
|
||||||
/// arrives with no warning; this is for graceful extras only).
|
/// arrives with no warning; this is for graceful extras only).
|
||||||
on_terminate: ?*const fn () void = null,
|
on_terminate: ?*const fn () void = null,
|
||||||
/// Publish the endpoint under a well-known service id at startup.
|
/// The contract this service provides: a name under `/protocol`, mirroring
|
||||||
service: ?abi.ServiceId = null,
|
/// the `library/protocol/` module that defines the wire format — a program
|
||||||
|
/// imports `display-protocol` and the provider binds `"display"`
|
||||||
|
/// (docs/os-development/protocol-namespace.md). Bound at startup, before
|
||||||
|
/// `init` runs, so the service is reachable the moment it serves. A refusal
|
||||||
|
/// (not granted, or a live provider already holds the name) aborts startup.
|
||||||
|
service: ?[]const u8 = null,
|
||||||
};
|
};
|
||||||
|
|
||||||
/// Run the service: create and (optionally) register the endpoint, bind signals
|
/// Run the service: create the endpoint, bind it under the service's contract
|
||||||
/// to it, call `init`, then serve until `terminate` arrives — at which point the
|
/// name (if it has one), bind signals to it, call `init`, then serve until
|
||||||
/// loop returns and main's return is the clean exit the supervisor reads as
|
/// `terminate` arrives — at which point the loop returns and main's return is
|
||||||
/// `ExitReason.exited`. `maximum_message` sizes the receive and reply buffers
|
/// the clean exit the supervisor reads as `ExitReason.exited`.
|
||||||
/// (a service passes its protocol's message maximum).
|
/// `maximum_message` sizes the receive and reply buffers (a service passes its
|
||||||
|
/// protocol's message maximum).
|
||||||
pub fn run(comptime maximum_message: usize, callbacks: Callbacks) void {
|
pub fn run(comptime maximum_message: usize, callbacks: Callbacks) void {
|
||||||
const endpoint = ipc.createIpcEndpoint() orelse return;
|
const endpoint = ipc.createIpcEndpoint() orelse return;
|
||||||
if (callbacks.service) |id| {
|
if (callbacks.service) |name| {
|
||||||
if (!ipc.register(id, endpoint)) return;
|
if (!channel.bindPatiently(name, endpoint)) return;
|
||||||
}
|
}
|
||||||
_ = process.bindSignals(endpoint);
|
_ = process.bindSignals(endpoint);
|
||||||
if (callbacks.init) |initialise| {
|
if (callbacks.init) |initialise| {
|
||||||
@@ -59,6 +83,16 @@ pub fn run(comptime maximum_message: usize, callbacks: Callbacks) void {
|
|||||||
var receive: [maximum_message]u8 = undefined;
|
var receive: [maximum_message]u8 = undefined;
|
||||||
while (true) {
|
while (true) {
|
||||||
const got = ipc.replyWait(endpoint, reply_buffer[0..reply_len], &receive, null);
|
const got = ipc.replyWait(endpoint, reply_buffer[0..reply_len], &receive, null);
|
||||||
|
// Whatever capability came with this turn is the turn's, and the turn
|
||||||
|
// closes it unless a callback claims it (`ipc.Arrival`). Structural
|
||||||
|
// rather than a close per branch, because the branches are exactly what
|
||||||
|
// gets forgotten: the ping's `continue` below, and every `on_message`
|
||||||
|
// that has no use for a capability — which is every operation but a
|
||||||
|
// subscribe. A `defer` in a loop body runs on `continue` and on the
|
||||||
|
// `return` that ends the loop, so this covers all four exits.
|
||||||
|
var arrived: ipc.Arrival = .{ .handle = got.cap };
|
||||||
|
defer arrived.release();
|
||||||
|
|
||||||
if (got.isNotification()) {
|
if (got.isNotification()) {
|
||||||
reply_len = 0; // nothing owed for a notification
|
reply_len = 0; // nothing owed for a notification
|
||||||
if (process.signalsFrom(got.badge)) |signals| {
|
if (process.signalsFrom(got.badge)) |signals| {
|
||||||
@@ -76,8 +110,8 @@ pub fn run(comptime maximum_message: usize, callbacks: Callbacks) void {
|
|||||||
}
|
}
|
||||||
if (got.len == 0) {
|
if (got.len == 0) {
|
||||||
reply_len = 0; // the universal ping: a zero-length reply, from the harness
|
reply_len = 0; // the universal ping: a zero-length reply, from the harness
|
||||||
continue;
|
continue; // any capability it carried goes out through the turn's `defer`
|
||||||
}
|
}
|
||||||
reply_len = callbacks.on_message(receive[0..got.len], &reply_buffer, got.senderTaskId(), got.cap);
|
reply_len = callbacks.on_message(receive[0..got.len], &reply_buffer, got.senderTaskId(), &arrived);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -3,6 +3,10 @@
|
|||||||
//! every conversation depend on the contract by name; neither reaches into the
|
//! every conversation depend on the contract by name; neither reaches into the
|
||||||
//! other's files. Pure flat wire types: no protocol module imports anything.
|
//! other's files. Pure flat wire types: no protocol module imports anything.
|
||||||
//!
|
//!
|
||||||
|
//! One module here is not a protocol but the shape the others are written in:
|
||||||
|
//!
|
||||||
|
//! envelope : the packet prefix + comptime Define (docs/os-development/protocol-namespace.md)
|
||||||
|
//!
|
||||||
//! vfs-protocol : the VFS server <-> the file layer (unistd/stdio)
|
//! vfs-protocol : the VFS server <-> the file layer (unistd/stdio)
|
||||||
//! input-protocol : the input fan-out service <-> sources + subscribers
|
//! input-protocol : the input fan-out service <-> sources + subscribers
|
||||||
//! block-protocol : a filesystem <-> a block driver (usb-storage)
|
//! block-protocol : a filesystem <-> a block driver (usb-storage)
|
||||||
@@ -16,6 +20,9 @@ const std = @import("std");
|
|||||||
|
|
||||||
pub fn build(b: *std.Build) void {
|
pub fn build(b: *std.Build) void {
|
||||||
for ([_]struct { name: []const u8, root: []const u8 }{
|
for ([_]struct { name: []const u8, root: []const u8 }{
|
||||||
|
// Not a protocol, hence not `-protocol`: the envelope is what a
|
||||||
|
// protocol is defined *through*.
|
||||||
|
.{ .name = "envelope", .root = "envelope/envelope.zig" },
|
||||||
.{ .name = "vfs-protocol", .root = "vfs/vfs-protocol.zig" },
|
.{ .name = "vfs-protocol", .root = "vfs/vfs-protocol.zig" },
|
||||||
.{ .name = "input-protocol", .root = "input/input-protocol.zig" },
|
.{ .name = "input-protocol", .root = "input/input-protocol.zig" },
|
||||||
.{ .name = "block-protocol", .root = "block/block-protocol.zig" },
|
.{ .name = "block-protocol", .root = "block/block-protocol.zig" },
|
||||||
@@ -32,6 +39,7 @@ pub fn build(b: *std.Build) void {
|
|||||||
// its aggregate test step.
|
// its aggregate test step.
|
||||||
const test_step = b.step("test", "Run the protocol unit tests");
|
const test_step = b.step("test", "Run the protocol unit tests");
|
||||||
for ([_][]const u8{
|
for ([_][]const u8{
|
||||||
|
"envelope/envelope.zig", // framing round trips, verb numbering, dispatch, the floors
|
||||||
"vfs/vfs-protocol.zig", // NodeKind / DirectoryEntry sizes + op values
|
"vfs/vfs-protocol.zig", // NodeKind / DirectoryEntry sizes + op values
|
||||||
"display/display-protocol.zig", // pack(): native pixel encoding per format
|
"display/display-protocol.zig", // pack(): native pixel encoding per format
|
||||||
}) |root| {
|
}) |root| {
|
||||||
|
|||||||
@@ -12,7 +12,7 @@
|
|||||||
pub const version: u16 = 1;
|
pub const version: u16 = 1;
|
||||||
|
|
||||||
/// Which bus a `child_added` came from — stated by the reporting bus driver so
|
/// Which bus a `child_added` came from — stated by the reporting bus driver so
|
||||||
/// the manager's /etc/devices.csv matcher knows how to read the report's identity
|
/// the manager's /system/configuration/devices.csv matcher knows how to read the report's identity
|
||||||
/// (a PCI class triple vs a USB class triple are the same 24 bits but different
|
/// (a PCI class triple vs a USB class triple are the same 24 bits but different
|
||||||
/// namespaces) and which `bus` column a rule must name to bind it. `unknown` is
|
/// namespaces) and which `bus` column a rule must name to bind it. `unknown` is
|
||||||
/// the zero default, so an un-upgraded reporter fails to match rather than
|
/// the zero default, so an un-upgraded reporter fails to match rather than
|
||||||
@@ -96,7 +96,7 @@ pub const ChildAdded = extern struct {
|
|||||||
/// for an unregistered leaf (a USB port before the descriptor track).
|
/// for an unregistered leaf (a USB port before the descriptor track).
|
||||||
device_id: u64 = no_device,
|
device_id: u64 = no_device,
|
||||||
/// The vendor id (PCI vendor / USB idVendor), or 0 when the bus has no such
|
/// The vendor id (PCI vendor / USB idVendor), or 0 when the bus has no such
|
||||||
/// concept (ACPI). Carried so the manager's /etc/devices.csv matcher can bind
|
/// concept (ACPI). Carried so the manager's /system/configuration/devices.csv matcher can bind
|
||||||
/// on vendor — a level the bus-native `identity` (a class triple) cannot express.
|
/// on vendor — a level the bus-native `identity` (a class triple) cannot express.
|
||||||
vendor: u16 = 0,
|
vendor: u16 = 0,
|
||||||
/// The device id (PCI device / USB idProduct), or 0. The most specific numeric
|
/// The device id (PCI device / USB idProduct), or 0. The most specific numeric
|
||||||
|
|||||||
@@ -0,0 +1,924 @@
|
|||||||
|
//! The envelope — the fixed prefix every danos packet begins with, and the
|
||||||
|
//! comptime `Define` that builds a protocol out of it. Layer L2 of
|
||||||
|
//! [communication.md](../../../docs/os-development/communication.md); the
|
||||||
|
//! authoritative description is
|
||||||
|
//! [protocol-namespace.md](../../../docs/os-development/protocol-namespace.md).
|
||||||
|
//!
|
||||||
|
//! This is the one module in the protocol domain that is not itself a protocol:
|
||||||
|
//! it is the shape every protocol is expressed in. A protocol module hands
|
||||||
|
//! `Define` its verbs and events and gets back numbered operations (never
|
||||||
|
//! colliding with the reserved range), typed encode/decode helpers, a
|
||||||
|
//! provider-side dispatch table that answers `describe` on its own — and, the
|
||||||
|
//! point of the exercise, compile-time proof that none of its packets can
|
||||||
|
//! exceed the transport floor. Errors that used to surface as runtime
|
||||||
|
//! truncation are compile errors, and "packets never fragment" is enforced at
|
||||||
|
//! the source rather than by review.
|
||||||
|
//!
|
||||||
|
//! **The prefix is folded, never stacked.** A `request`, `reply`, or `payload`
|
||||||
|
//! type names the bytes that follow the prefix — never the whole packet. The
|
||||||
|
//! verb and the object being addressed live in the prefix, so a protocol type
|
||||||
|
//! carries neither an `operation` field of its own nor a nested `Header`;
|
||||||
|
//! `rejectStacking` refuses one at compile time, and every size check adds
|
||||||
|
//! `prefix_size` exactly once.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
|
||||||
|
// --- the prefix -------------------------------------------------------------
|
||||||
|
|
||||||
|
/// Every packet a danos protocol transmits begins with this header — requests
|
||||||
|
/// on the synchronous call path, event packets on the asynchronous push path.
|
||||||
|
/// A reply spends the same 16 bytes on `Status` instead.
|
||||||
|
pub const Header = extern struct {
|
||||||
|
/// The verb. Values below `first_protocol_operation` are the reserved
|
||||||
|
/// universal verbs, which mean the same thing in every protocol.
|
||||||
|
operation: u32,
|
||||||
|
_padding: u32 = 0,
|
||||||
|
/// **Object** addressing within the peer, never party addressing: which of
|
||||||
|
/// the peer's objects this packet operates on — a volume, a layer, a node,
|
||||||
|
/// a device. `0` addresses the provider itself, and a protocol with no
|
||||||
|
/// objects never uses the field. *Which* party is at the other end was
|
||||||
|
/// decided once, when the channel was opened, and *who sent this* is the
|
||||||
|
/// kernel-stamped badge; neither is ever written here, which is what keeps
|
||||||
|
/// the source unforgeable.
|
||||||
|
target: u64 = 0,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Every reply begins with this. `len` counts the bytes that follow: the
|
||||||
|
/// reply's fixed part plus whatever variable tail the operation defines.
|
||||||
|
pub const Status = extern struct {
|
||||||
|
status: i32, // 0, or a negative errno
|
||||||
|
_padding: u32 = 0,
|
||||||
|
len: u32 = 0,
|
||||||
|
_padding2: u32 = 0,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// The fixed prefix every packet spends — `Header` on a request or an event,
|
||||||
|
/// `Status` on a reply. One constant, because the two are deliberately the same
|
||||||
|
/// width: the packet budget does not depend on the direction.
|
||||||
|
pub const prefix_size: usize = @sizeOf(Header);
|
||||||
|
|
||||||
|
comptime {
|
||||||
|
if (@sizeOf(Header) != 16 or @sizeOf(Status) != 16)
|
||||||
|
@compileError("the envelope prefix is 16 bytes in both directions");
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- the reserved verbs -----------------------------------------------------
|
||||||
|
|
||||||
|
/// Reserved verbs, answered by every provider. `Define` numbers a protocol's
|
||||||
|
/// own verbs from `first_protocol_operation`, so no protocol can reach in here.
|
||||||
|
pub const operation_describe: u32 = 0; // -> protocol name, version, target kinds
|
||||||
|
pub const operation_enumerate: u32 = 1; // -> the current targets, one per reply page
|
||||||
|
pub const operation_subscribe: u32 = 2; // capability = the subscriber's endpoint
|
||||||
|
pub const operation_unsubscribe: u32 = 3;
|
||||||
|
pub const first_protocol_operation: u32 = 16;
|
||||||
|
|
||||||
|
/// The `describe` reply's fixed part, followed inline by `name_len` bytes of the
|
||||||
|
/// protocol's name. This is the version handshake: the version is asked for
|
||||||
|
/// once, at connect time, rather than re-carried by every packet out of a
|
||||||
|
/// 256-byte budget.
|
||||||
|
pub const Description = extern struct {
|
||||||
|
version: u32,
|
||||||
|
operation_count: u32,
|
||||||
|
event_count: u32,
|
||||||
|
name_len: u32,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Longest protocol name a `describe` reply can carry.
|
||||||
|
pub const name_maximum: usize = packet_maximum - prefix_size - @sizeOf(Description);
|
||||||
|
|
||||||
|
// --- the transport floor ----------------------------------------------------
|
||||||
|
|
||||||
|
/// The packet budget every protocol may assume on *any* transport. These are
|
||||||
|
/// the kernel-ipc transport's limits — `MESSAGE_MAXIMUM` and `POST_MAXIMUM` in
|
||||||
|
/// system/kernel/ipc-synchronous.zig — restated here because the kernel keeps
|
||||||
|
/// them private and a protocol has to compile against something. A fatter
|
||||||
|
/// transport raises its own ceiling; the floor does not move, so a protocol
|
||||||
|
/// that fits here fits everywhere (communication.md: ceilings are transport
|
||||||
|
/// properties, the floor is the protocol's contract).
|
||||||
|
pub const packet_maximum: usize = 256; // one request or one reply (ipc_call)
|
||||||
|
pub const post_maximum: usize = 64; // one event packet (ipc_send)
|
||||||
|
|
||||||
|
/// Whether a request or reply whose fixed part is `T` fits the call floor once
|
||||||
|
/// the prefix is counted. The folded rule in one line: `prefix_size` is added
|
||||||
|
/// exactly once, because `T` describes only what follows it. Exported so the
|
||||||
|
/// rule itself is testable — `Define` enforces it as a compile error.
|
||||||
|
pub fn fitsPacket(comptime T: type) bool {
|
||||||
|
return prefix_size + @sizeOf(T) <= packet_maximum;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The same, against the much smaller push floor an event packet lives within.
|
||||||
|
pub fn fitsPost(comptime T: type) bool {
|
||||||
|
return prefix_size + @sizeOf(T) <= post_maximum;
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- reply statuses the envelope itself produces -----------------------------
|
||||||
|
|
||||||
|
/// Continued from the kernel's danos-native errno numbering
|
||||||
|
/// (system/kernel/ipc-synchronous.zig, which ends at `EPERM` = 9), so a client
|
||||||
|
/// reads one vocabulary whether the number came from the kernel or a provider.
|
||||||
|
/// Positive here, sent negated in `Status.status`, as the kernel spells it.
|
||||||
|
pub const ENOSYS: i32 = 10; // this protocol has no such operation
|
||||||
|
pub const EPROTO: i32 = 11; // malformed packet: shorter than the verb it names
|
||||||
|
pub const EBUSY: i32 = 12; // the thing asked for is held by someone still alive
|
||||||
|
|
||||||
|
/// Restated from the kernel's half of the numbering, because a provider refuses
|
||||||
|
/// too and userspace has no other place to read these from: `ENOENT` is "no such
|
||||||
|
/// name", `EPERM` "not permitted". The protocol registry answers an ungranted
|
||||||
|
/// bind with the second and a name a live provider already holds with `EBUSY`.
|
||||||
|
pub const ENOENT: i32 = 4;
|
||||||
|
pub const ENOSPC: i32 = 5;
|
||||||
|
pub const EPERM: i32 = 9;
|
||||||
|
|
||||||
|
// --- framing ----------------------------------------------------------------
|
||||||
|
|
||||||
|
/// The header of a received packet, or null when it is too short to have one.
|
||||||
|
pub fn headerOf(packet: []const u8) ?Header {
|
||||||
|
if (packet.len < prefix_size) return null;
|
||||||
|
return std.mem.bytesToValue(Header, packet[0..prefix_size]);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The status of a received reply, or null when it is too short to have one.
|
||||||
|
pub fn statusOf(packet: []const u8) ?Status {
|
||||||
|
if (packet.len < prefix_size) return null;
|
||||||
|
return std.mem.bytesToValue(Status, packet[0..prefix_size]);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Frame a bare `describe` request. Protocol-independent: the reserved verbs
|
||||||
|
/// are asked the same way of every provider.
|
||||||
|
pub fn encodeDescribe(buffer: []u8) ?[]u8 {
|
||||||
|
const header = Header{ .operation = operation_describe };
|
||||||
|
return frame(std.mem.asBytes(&header), &.{}, &.{}, buffer);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A decoded `describe` reply: the fixed part, plus the name that follows it.
|
||||||
|
pub const Described = struct {
|
||||||
|
description: Description,
|
||||||
|
name: []const u8,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Decode a `describe` reply packet. Null if it failed, was truncated, or is
|
||||||
|
/// not a description at all.
|
||||||
|
pub fn decodeDescribe(packet: []const u8) ?Described {
|
||||||
|
const status = statusOf(packet) orelse return null;
|
||||||
|
if (status.status != 0) return null;
|
||||||
|
const body = packet[prefix_size..];
|
||||||
|
if (body.len < @sizeOf(Description)) return null;
|
||||||
|
const description = std.mem.bytesToValue(Description, body[0..@sizeOf(Description)]);
|
||||||
|
const name = body[@sizeOf(Description)..];
|
||||||
|
if (name.len < description.name_len) return null;
|
||||||
|
return .{ .description = description, .name = name[0..description.name_len] };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The single framing point: prefix, then the fixed part, then the variable
|
||||||
|
/// tail, contiguous in one buffer. Null when the packet would not fit — a
|
||||||
|
/// packet is never split, so not fitting is a failure, not a continuation.
|
||||||
|
fn frame(prefix: []const u8, fixed: []const u8, tail: []const u8, buffer: []u8) ?[]u8 {
|
||||||
|
const total = prefix.len + fixed.len + tail.len;
|
||||||
|
if (total > buffer.len) return null;
|
||||||
|
@memcpy(buffer[0..prefix.len], prefix);
|
||||||
|
@memcpy(buffer[prefix.len..][0..fixed.len], fixed);
|
||||||
|
@memcpy(buffer[prefix.len + fixed.len ..][0..tail.len], tail);
|
||||||
|
return buffer[0..total];
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The bytes of a fixed part — empty for `void`, which is how an operation says
|
||||||
|
/// "nothing but the verb".
|
||||||
|
fn bytesOf(comptime T: type, value: *const T) []const u8 {
|
||||||
|
if (@sizeOf(T) == 0) return &.{};
|
||||||
|
return @as([*]const u8, @ptrCast(value))[0..@sizeOf(T)];
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Read a fixed part out of a packet body. A zero-sized part always succeeds
|
||||||
|
/// (there is nothing to be short of); anything else needs its full width.
|
||||||
|
fn valueOf(comptime T: type, body: []const u8) ?T {
|
||||||
|
if (@sizeOf(T) == 0) return @as(T, undefined);
|
||||||
|
if (body.len < @sizeOf(T)) return null;
|
||||||
|
return std.mem.bytesToValue(T, body[0..@sizeOf(T)]);
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- the specification ------------------------------------------------------
|
||||||
|
|
||||||
|
/// One verb of a protocol. `request` and `reply` describe the bytes *after* the
|
||||||
|
/// prefix; either may be `void`, meaning the verb (and its target) says it all.
|
||||||
|
pub const OperationSpecification = struct {
|
||||||
|
name: []const u8,
|
||||||
|
request: type = void,
|
||||||
|
reply: type = void,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// One event a provider pushes to its subscribers. `payload` is the bytes after
|
||||||
|
/// the `Header`, and the whole packet must fit the push floor.
|
||||||
|
pub const EventSpecification = struct {
|
||||||
|
name: []const u8,
|
||||||
|
payload: type = void,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// What `Define` is given: the contract, whole.
|
||||||
|
pub const Specification = struct {
|
||||||
|
/// The contract's name — the same word as its `/protocol/<name>` leaf and
|
||||||
|
/// its `library/protocol/` module.
|
||||||
|
name: []const u8,
|
||||||
|
version: u32,
|
||||||
|
operations: []const OperationSpecification = &.{},
|
||||||
|
events: []const EventSpecification = &.{},
|
||||||
|
};
|
||||||
|
|
||||||
|
const reserved_names = [_][]const u8{ "describe", "enumerate", "subscribe", "unsubscribe" };
|
||||||
|
|
||||||
|
fn isReservedName(comptime name: []const u8) bool {
|
||||||
|
for (reserved_names) |reserved| {
|
||||||
|
if (std.mem.eql(u8, reserved, name)) return true;
|
||||||
|
}
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Refuse a protocol type that carries the prefix inside itself. The header is
|
||||||
|
/// folded into every packet, so a type that also holds one would send it twice
|
||||||
|
/// and re-invent per-protocol addressing — the mistake the envelope exists to
|
||||||
|
/// prevent.
|
||||||
|
fn rejectStacking(comptime protocol: []const u8, comptime verb: []const u8, comptime T: type) void {
|
||||||
|
switch (@typeInfo(T)) {
|
||||||
|
.@"struct" => |info| for (info.fields) |field| {
|
||||||
|
if (field.type == Header or field.type == Status) @compileError(std.fmt.comptimePrint(
|
||||||
|
"protocol '{s}', verb '{s}': the envelope prefix is folded, not stacked — " ++
|
||||||
|
"drop the {s} field '{s}' and use the packet's own Header.operation / Header.target",
|
||||||
|
.{ protocol, verb, @typeName(field.type), field.name },
|
||||||
|
));
|
||||||
|
},
|
||||||
|
else => {},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn nullDefault(comptime T: type) *const anyopaque {
|
||||||
|
const empty: ?T = null;
|
||||||
|
return @ptrCast(&empty);
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- Define -----------------------------------------------------------------
|
||||||
|
|
||||||
|
/// Build a protocol from its specification. Everything below happens at compile
|
||||||
|
/// time; the generated type is what both sides of the conversation import.
|
||||||
|
///
|
||||||
|
/// ```zig
|
||||||
|
/// pub const Protocol = envelope.Define(.{
|
||||||
|
/// .name = "display",
|
||||||
|
/// .version = 1,
|
||||||
|
/// .operations = &.{
|
||||||
|
/// .{ .name = "configure_layer", .request = ConfigureLayer, .reply = void },
|
||||||
|
/// .{ .name = "blit", .request = Blit, .reply = void },
|
||||||
|
/// },
|
||||||
|
/// .events = &.{
|
||||||
|
/// .{ .name = "layer_lost", .payload = LayerLost },
|
||||||
|
/// },
|
||||||
|
/// });
|
||||||
|
/// ```
|
||||||
|
///
|
||||||
|
/// Refused at compile time, each with the protocol, the verb, and the numbers
|
||||||
|
/// named in the message:
|
||||||
|
///
|
||||||
|
/// - a request or reply that does not fit `packet_maximum` once `prefix_size`
|
||||||
|
/// is added (`.request = extern struct { bytes: [241]u8 }` — 241 + 16 = 257);
|
||||||
|
/// - an event payload that does not fit `post_maximum` the same way
|
||||||
|
/// (`.payload = extern struct { bytes: [49]u8 }` — 49 + 16 = 65);
|
||||||
|
/// - a type that stacks the prefix instead of folding it (a `Header` field);
|
||||||
|
/// - a verb named after a reserved one, or named twice.
|
||||||
|
///
|
||||||
|
/// A variable tail is bounded at *run* time instead, by `encodeRequest` and its
|
||||||
|
/// siblings, because only the caller knows how long it is.
|
||||||
|
pub fn Define(comptime specification: Specification) type {
|
||||||
|
comptime {
|
||||||
|
if (specification.name.len == 0) @compileError("a protocol needs a name");
|
||||||
|
if (specification.name.len > name_maximum) @compileError(std.fmt.comptimePrint(
|
||||||
|
"protocol '{s}': the name is {d} bytes, and a describe reply carries at most {d}",
|
||||||
|
.{ specification.name, specification.name.len, name_maximum },
|
||||||
|
));
|
||||||
|
|
||||||
|
for (specification.operations, 0..) |operation, index| {
|
||||||
|
if (isReservedName(operation.name)) @compileError(std.fmt.comptimePrint(
|
||||||
|
"protocol '{s}': '{s}' is a reserved universal verb — the envelope already answers it",
|
||||||
|
.{ specification.name, operation.name },
|
||||||
|
));
|
||||||
|
for (specification.operations[0..index]) |earlier| {
|
||||||
|
if (std.mem.eql(u8, earlier.name, operation.name)) @compileError(std.fmt.comptimePrint(
|
||||||
|
"protocol '{s}': operation '{s}' is declared twice",
|
||||||
|
.{ specification.name, operation.name },
|
||||||
|
));
|
||||||
|
}
|
||||||
|
rejectStacking(specification.name, operation.name, operation.request);
|
||||||
|
rejectStacking(specification.name, operation.name, operation.reply);
|
||||||
|
if (!fitsPacket(operation.request)) @compileError(std.fmt.comptimePrint(
|
||||||
|
"protocol '{s}', operation '{s}': the request is {d} bytes and the header {d}, " ++
|
||||||
|
"over the {d}-byte call floor — packets never fragment, so this has to shrink " ++
|
||||||
|
"or move its bulk to shared memory",
|
||||||
|
.{ specification.name, operation.name, @sizeOf(operation.request), prefix_size, packet_maximum },
|
||||||
|
));
|
||||||
|
if (!fitsPacket(operation.reply)) @compileError(std.fmt.comptimePrint(
|
||||||
|
"protocol '{s}', operation '{s}': the reply is {d} bytes and the status header {d}, " ++
|
||||||
|
"over the {d}-byte call floor",
|
||||||
|
.{ specification.name, operation.name, @sizeOf(operation.reply), prefix_size, packet_maximum },
|
||||||
|
));
|
||||||
|
}
|
||||||
|
|
||||||
|
for (specification.events, 0..) |event, index| {
|
||||||
|
for (specification.events[0..index]) |earlier| {
|
||||||
|
if (std.mem.eql(u8, earlier.name, event.name)) @compileError(std.fmt.comptimePrint(
|
||||||
|
"protocol '{s}': event '{s}' is declared twice",
|
||||||
|
.{ specification.name, event.name },
|
||||||
|
));
|
||||||
|
}
|
||||||
|
rejectStacking(specification.name, event.name, event.payload);
|
||||||
|
if (!fitsPost(event.payload)) @compileError(std.fmt.comptimePrint(
|
||||||
|
"protocol '{s}', event '{s}': the payload is {d} bytes and the header {d}, " ++
|
||||||
|
"over the {d}-byte push floor — an event carries the header too, so it is the " ++
|
||||||
|
"payload that has to give",
|
||||||
|
.{ specification.name, event.name, @sizeOf(event.payload), prefix_size, post_maximum },
|
||||||
|
));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
return struct {
|
||||||
|
pub const protocol_name: []const u8 = specification.name;
|
||||||
|
pub const version: u32 = specification.version;
|
||||||
|
|
||||||
|
/// What a provider sizes its receive and reply buffers to. A packet's
|
||||||
|
/// fixed part may be far smaller, but any caller may send up to the
|
||||||
|
/// floor and a short buffer truncates rather than refuses.
|
||||||
|
pub const message_maximum: usize = packet_maximum;
|
||||||
|
|
||||||
|
/// The widest packet this protocol's fixed parts can actually produce,
|
||||||
|
/// prefix included — a diagnostic, and what a test pins.
|
||||||
|
pub const request_maximum: usize = widest(specification.operations, .request);
|
||||||
|
pub const reply_maximum: usize = widest(specification.operations, .reply);
|
||||||
|
pub const event_maximum: usize = blk: {
|
||||||
|
var widest_event: usize = prefix_size;
|
||||||
|
for (specification.events) |event| widest_event = @max(widest_event, prefix_size + @sizeOf(event.payload));
|
||||||
|
break :blk widest_event;
|
||||||
|
};
|
||||||
|
|
||||||
|
/// This protocol's verbs, numbered from `first_protocol_operation` in
|
||||||
|
/// declaration order.
|
||||||
|
pub const Operation = numbered(specification.operations, "name");
|
||||||
|
|
||||||
|
/// This protocol's events, numbered from `first_protocol_operation` in
|
||||||
|
/// their **own** space. Events travel only provider → subscriber over
|
||||||
|
/// `ipc_send` and operations only client → provider over `ipc_call`, so
|
||||||
|
/// the direction already tells the two apart; separate spaces mean
|
||||||
|
/// appending an operation can never renumber a shipped event.
|
||||||
|
pub const Event = numbered(specification.events, "name");
|
||||||
|
|
||||||
|
/// The bytes after the `Header` on a request for `operation`.
|
||||||
|
pub fn RequestOf(comptime operation: Operation) type {
|
||||||
|
return specification.operations[indexOf(@intFromEnum(operation))].request;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The bytes after the `Status` on the reply to `operation`.
|
||||||
|
pub fn ReplyOf(comptime operation: Operation) type {
|
||||||
|
return specification.operations[indexOf(@intFromEnum(operation))].reply;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The bytes after the `Header` on an `event` packet.
|
||||||
|
pub fn PayloadOf(comptime event: Event) type {
|
||||||
|
return specification.events[indexOf(@intFromEnum(event))].payload;
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- client side ----------------------------------------------------
|
||||||
|
|
||||||
|
/// Frame `[Header][request][tail]`. `tail` is the variable part (a path,
|
||||||
|
/// write bytes); pass `&.{}` when the verb has none. Null if the packet
|
||||||
|
/// would exceed the buffer or the call floor.
|
||||||
|
pub fn encodeRequest(
|
||||||
|
comptime operation: Operation,
|
||||||
|
target: u64,
|
||||||
|
request: RequestOf(operation),
|
||||||
|
tail: []const u8,
|
||||||
|
buffer: []u8,
|
||||||
|
) ?[]u8 {
|
||||||
|
const header = Header{ .operation = @intFromEnum(operation), .target = target };
|
||||||
|
const packet = frame(std.mem.asBytes(&header), bytesOf(RequestOf(operation), &request), tail, buffer) orelse return null;
|
||||||
|
return if (packet.len > packet_maximum) null else packet;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Frame `[Status][reply][tail]` — the provider's answer, for a provider
|
||||||
|
/// that composes its own reply rather than using `Provider.dispatch`.
|
||||||
|
pub fn encodeReply(
|
||||||
|
comptime operation: Operation,
|
||||||
|
status: i32,
|
||||||
|
reply: ReplyOf(operation),
|
||||||
|
tail: []const u8,
|
||||||
|
buffer: []u8,
|
||||||
|
) ?[]u8 {
|
||||||
|
const fixed = bytesOf(ReplyOf(operation), &reply);
|
||||||
|
const head = Status{ .status = status, .len = @intCast(fixed.len + tail.len) };
|
||||||
|
const packet = frame(std.mem.asBytes(&head), fixed, tail, buffer) orelse return null;
|
||||||
|
return if (packet.len > packet_maximum) null else packet;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Frame `[Header][payload]` for an asynchronous push. Null if it would
|
||||||
|
/// exceed the buffer or the push floor — an event that does not fit is
|
||||||
|
/// dropped at the source, never split.
|
||||||
|
pub fn encodeEvent(
|
||||||
|
comptime event: Event,
|
||||||
|
target: u64,
|
||||||
|
payload: PayloadOf(event),
|
||||||
|
buffer: []u8,
|
||||||
|
) ?[]u8 {
|
||||||
|
const header = Header{ .operation = @intFromEnum(event), .target = target };
|
||||||
|
const packet = frame(std.mem.asBytes(&header), bytesOf(PayloadOf(event), &payload), &.{}, buffer) orelse return null;
|
||||||
|
return if (packet.len > post_maximum) null else packet;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Which of this protocol's verbs a packet names — null for a reserved
|
||||||
|
/// verb, or for a number this protocol does not define.
|
||||||
|
pub fn operationOf(packet: []const u8) ?Operation {
|
||||||
|
const header = headerOf(packet) orelse return null;
|
||||||
|
const index = header.operation -% first_protocol_operation;
|
||||||
|
if (header.operation < first_protocol_operation or index >= specification.operations.len) return null;
|
||||||
|
return @enumFromInt(header.operation);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Which of this protocol's events a pushed packet carries.
|
||||||
|
pub fn eventOf(packet: []const u8) ?Event {
|
||||||
|
const header = headerOf(packet) orelse return null;
|
||||||
|
const index = header.operation -% first_protocol_operation;
|
||||||
|
if (header.operation < first_protocol_operation or index >= specification.events.len) return null;
|
||||||
|
return @enumFromInt(header.operation);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The fixed request part of a packet already known to name `operation`.
|
||||||
|
pub fn decodeRequest(comptime operation: Operation, packet: []const u8) ?RequestOf(operation) {
|
||||||
|
if (packet.len < prefix_size) return null;
|
||||||
|
return valueOf(RequestOf(operation), packet[prefix_size..]);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The bytes after the fixed request part — empty when there are none.
|
||||||
|
pub fn requestTail(comptime operation: Operation, packet: []const u8) []const u8 {
|
||||||
|
const start = prefix_size + @sizeOf(RequestOf(operation));
|
||||||
|
return if (packet.len <= start) &.{} else packet[start..];
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The fixed reply part of a reply packet. Null on a short packet; the
|
||||||
|
/// caller checks `statusOf(packet).status` for the provider's verdict.
|
||||||
|
pub fn decodeReply(comptime operation: Operation, packet: []const u8) ?ReplyOf(operation) {
|
||||||
|
if (packet.len < prefix_size) return null;
|
||||||
|
return valueOf(ReplyOf(operation), packet[prefix_size..]);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The bytes after the fixed reply part, clipped to what `Status.len`
|
||||||
|
/// says actually arrived.
|
||||||
|
pub fn replyTail(comptime operation: Operation, packet: []const u8) []const u8 {
|
||||||
|
const status = statusOf(packet) orelse return &.{};
|
||||||
|
const start = prefix_size + @sizeOf(ReplyOf(operation));
|
||||||
|
const end = @min(packet.len, prefix_size + @as(usize, status.len));
|
||||||
|
return if (end <= start) &.{} else packet[start..end];
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The payload of a pushed packet already known to carry `event`.
|
||||||
|
pub fn decodeEvent(comptime event: Event, packet: []const u8) ?PayloadOf(event) {
|
||||||
|
if (packet.len < prefix_size) return null;
|
||||||
|
return valueOf(PayloadOf(event), packet[prefix_size..]);
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- provider side --------------------------------------------------
|
||||||
|
|
||||||
|
/// This protocol's dispatch table, bound to the provider's own state
|
||||||
|
/// type. `describe` is answered here, from the specification; every verb
|
||||||
|
/// this provider left null answers `-ENOSYS`, which is what makes the
|
||||||
|
/// reserved verbs mean the same thing at every provider in the system.
|
||||||
|
///
|
||||||
|
/// ```zig
|
||||||
|
/// const Serve = Protocol.Provider(*Server);
|
||||||
|
/// const handlers = Serve.Handlers{ .blit = onBlit, .configure_layer = onConfigureLayer };
|
||||||
|
/// const reply_len = Serve.dispatch(server, handlers, message, sender, capability, reply);
|
||||||
|
/// ```
|
||||||
|
///
|
||||||
|
/// A handler returns the number of `answer.tail()` bytes it wrote, or a
|
||||||
|
/// negative errno.
|
||||||
|
pub fn Provider(comptime Context: type) type {
|
||||||
|
return struct {
|
||||||
|
/// A reserved verb a provider chooses to implement itself.
|
||||||
|
/// `enumerate` writes its targets into the tail; `subscribe`
|
||||||
|
/// takes the subscriber's endpoint from `invocation.capability`.
|
||||||
|
pub const ReservedHandler = *const fn (Context, Invocation(void), Answer(void)) isize;
|
||||||
|
|
||||||
|
/// One optional handler per verb, named exactly as the verb,
|
||||||
|
/// plus the reserved verbs the envelope cannot answer alone.
|
||||||
|
pub const Handlers = handlerTable(Context);
|
||||||
|
|
||||||
|
/// Answer one received packet: writes `[Status][reply][tail]`
|
||||||
|
/// into `reply` and returns its length. Zero means the reply
|
||||||
|
/// buffer could not even hold a status, so nothing was written.
|
||||||
|
pub fn dispatch(
|
||||||
|
context: Context,
|
||||||
|
handlers: Handlers,
|
||||||
|
packet: []const u8,
|
||||||
|
sender: u32,
|
||||||
|
capability: ?usize,
|
||||||
|
reply: []u8,
|
||||||
|
) usize {
|
||||||
|
if (reply.len < prefix_size) return 0;
|
||||||
|
const header = headerOf(packet) orelse return refuse(reply, -EPROTO);
|
||||||
|
const body = packet[prefix_size..];
|
||||||
|
|
||||||
|
if (header.operation == operation_describe) return describeInto(reply);
|
||||||
|
|
||||||
|
inline for (specification.operations, 0..) |operation, index| {
|
||||||
|
if (header.operation == first_protocol_operation + index) {
|
||||||
|
return invoke(
|
||||||
|
Context,
|
||||||
|
operation.request,
|
||||||
|
operation.reply,
|
||||||
|
@field(handlers, operation.name),
|
||||||
|
context,
|
||||||
|
header.target,
|
||||||
|
body,
|
||||||
|
sender,
|
||||||
|
capability,
|
||||||
|
reply,
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
const reserved: ?ReservedHandler = switch (header.operation) {
|
||||||
|
operation_enumerate => handlers.enumerate,
|
||||||
|
operation_subscribe => handlers.subscribe,
|
||||||
|
operation_unsubscribe => handlers.unsubscribe,
|
||||||
|
else => null,
|
||||||
|
};
|
||||||
|
return invoke(Context, void, void, reserved, context, header.target, body, sender, capability, reply);
|
||||||
|
}
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
// Shared by the protocol verbs and the reserved ones: decode, hand the
|
||||||
|
// handler a typed invocation, stamp the status. One place, so a reserved
|
||||||
|
// verb and a protocol verb behave identically.
|
||||||
|
fn invoke(
|
||||||
|
comptime Context: type,
|
||||||
|
comptime RequestType: type,
|
||||||
|
comptime ReplyType: type,
|
||||||
|
handler: ?*const fn (Context, Invocation(RequestType), Answer(ReplyType)) isize,
|
||||||
|
context: Context,
|
||||||
|
target: u64,
|
||||||
|
body: []const u8,
|
||||||
|
sender: u32,
|
||||||
|
capability: ?usize,
|
||||||
|
reply: []u8,
|
||||||
|
) usize {
|
||||||
|
const call = handler orelse return refuse(reply, -ENOSYS);
|
||||||
|
const request = valueOf(RequestType, body) orelse return refuse(reply, -EPROTO);
|
||||||
|
if (reply.len < prefix_size + @sizeOf(ReplyType)) return refuse(reply, -EPROTO);
|
||||||
|
const produced = call(context, .{
|
||||||
|
.target = target,
|
||||||
|
.request = request,
|
||||||
|
.tail = body[@min(@sizeOf(RequestType), body.len)..],
|
||||||
|
.sender = sender,
|
||||||
|
.capability = capability,
|
||||||
|
}, .{ .buffer = reply[prefix_size..] });
|
||||||
|
if (produced < 0) return refuse(reply, @intCast(produced));
|
||||||
|
return succeed(reply, @sizeOf(ReplyType) + @as(usize, @intCast(produced)));
|
||||||
|
}
|
||||||
|
|
||||||
|
fn describeInto(reply: []u8) usize {
|
||||||
|
const description = Description{
|
||||||
|
.version = specification.version,
|
||||||
|
.operation_count = specification.operations.len,
|
||||||
|
.event_count = specification.events.len,
|
||||||
|
.name_len = specification.name.len,
|
||||||
|
};
|
||||||
|
const total = @sizeOf(Description) + specification.name.len;
|
||||||
|
if (reply.len < prefix_size + total) return refuse(reply, -EPROTO);
|
||||||
|
@memcpy(reply[prefix_size..][0..@sizeOf(Description)], std.mem.asBytes(&description));
|
||||||
|
@memcpy(reply[prefix_size + @sizeOf(Description) ..][0..specification.name.len], specification.name);
|
||||||
|
return succeed(reply, total);
|
||||||
|
}
|
||||||
|
|
||||||
|
// "describe" is answered by the envelope, so it is the one reserved verb
|
||||||
|
// with no slot in the table.
|
||||||
|
const implementable_reserved = [_][]const u8{ "enumerate", "subscribe", "unsubscribe" };
|
||||||
|
|
||||||
|
fn handlerTable(comptime Context: type) type {
|
||||||
|
const count = specification.operations.len + implementable_reserved.len;
|
||||||
|
var names: [count][]const u8 = undefined;
|
||||||
|
var types: [count]type = undefined;
|
||||||
|
var attributes: [count]std.builtin.Type.StructField.Attributes = undefined;
|
||||||
|
for (specification.operations, 0..) |operation, index| {
|
||||||
|
const Handler = *const fn (Context, Invocation(operation.request), Answer(operation.reply)) isize;
|
||||||
|
names[index] = operation.name;
|
||||||
|
types[index] = ?Handler;
|
||||||
|
attributes[index] = .{ .default_value_ptr = nullDefault(Handler) };
|
||||||
|
}
|
||||||
|
const Reserved = *const fn (Context, Invocation(void), Answer(void)) isize;
|
||||||
|
for (implementable_reserved, 0..) |name, offset| {
|
||||||
|
const slot = specification.operations.len + offset;
|
||||||
|
names[slot] = name;
|
||||||
|
types[slot] = ?Reserved;
|
||||||
|
attributes[slot] = .{ .default_value_ptr = nullDefault(Reserved) };
|
||||||
|
}
|
||||||
|
const frozen_names = names;
|
||||||
|
const frozen_types = types;
|
||||||
|
const frozen_attributes = attributes;
|
||||||
|
return @Struct(.auto, null, &frozen_names, &frozen_types, &frozen_attributes);
|
||||||
|
}
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- the provider's view of one packet --------------------------------------
|
||||||
|
|
||||||
|
/// What a provider's handler is given.
|
||||||
|
pub fn Invocation(comptime RequestType: type) type {
|
||||||
|
return struct {
|
||||||
|
/// Object addressing within this provider — the packet's `Header.target`.
|
||||||
|
target: u64,
|
||||||
|
/// The fixed request part, already decoded.
|
||||||
|
request: RequestType,
|
||||||
|
/// The bytes after it: a path, write data, a name.
|
||||||
|
tail: []const u8,
|
||||||
|
/// The kernel-stamped badge of the caller. The only source identity
|
||||||
|
/// there is — no protocol defines a sender field — so per-client state
|
||||||
|
/// is keyed on this.
|
||||||
|
sender: u32,
|
||||||
|
/// A capability the call carried: a subscriber's endpoint, a DMA
|
||||||
|
/// region. Only the synchronous call path can move one.
|
||||||
|
capability: ?usize,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Where a provider's handler writes its answer. The `Status` in front of it is
|
||||||
|
/// the dispatcher's to stamp — a handler never writes its own.
|
||||||
|
pub fn Answer(comptime ReplyType: type) type {
|
||||||
|
return struct {
|
||||||
|
buffer: []u8,
|
||||||
|
|
||||||
|
const fixed_size = @sizeOf(ReplyType);
|
||||||
|
|
||||||
|
/// Write the fixed reply part. A `void` reply writes nothing.
|
||||||
|
pub fn set(self: @This(), reply: ReplyType) void {
|
||||||
|
if (fixed_size == 0) return;
|
||||||
|
@memcpy(self.buffer[0..fixed_size], bytesOf(ReplyType, &reply));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Room for the variable tail; the handler returns how much of it it used.
|
||||||
|
pub fn tail(self: @This()) []u8 {
|
||||||
|
return self.buffer[fixed_size..];
|
||||||
|
}
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
fn refuse(reply: []u8, status: i32) usize {
|
||||||
|
const head = Status{ .status = status, .len = 0 };
|
||||||
|
@memcpy(reply[0..prefix_size], std.mem.asBytes(&head));
|
||||||
|
return prefix_size;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn succeed(reply: []u8, len: usize) usize {
|
||||||
|
const head = Status{ .status = 0, .len = @intCast(len) };
|
||||||
|
@memcpy(reply[0..prefix_size], std.mem.asBytes(&head));
|
||||||
|
return prefix_size + len;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The widest `[prefix][fixed]` over a set of operations, on one side.
|
||||||
|
fn widest(comptime operations: []const OperationSpecification, comptime side: enum { request, reply }) usize {
|
||||||
|
var found: usize = prefix_size;
|
||||||
|
for (operations) |operation| {
|
||||||
|
const size = switch (side) {
|
||||||
|
.request => @sizeOf(operation.request),
|
||||||
|
.reply => @sizeOf(operation.reply),
|
||||||
|
};
|
||||||
|
found = @max(found, prefix_size + size);
|
||||||
|
}
|
||||||
|
return found;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// An enum over `specifications`, tagged by their `name` field, numbered from
|
||||||
|
/// `first_protocol_operation` in declaration order.
|
||||||
|
fn numbered(comptime specifications: anytype, comptime field: []const u8) type {
|
||||||
|
var names: [specifications.len][]const u8 = undefined;
|
||||||
|
var values: [specifications.len]u32 = undefined;
|
||||||
|
for (specifications, 0..) |specification, index| {
|
||||||
|
names[index] = @field(specification, field);
|
||||||
|
values[index] = first_protocol_operation + index;
|
||||||
|
}
|
||||||
|
const frozen_names = names;
|
||||||
|
const frozen_values = values;
|
||||||
|
return @Enum(u32, .exhaustive, &frozen_names, &frozen_values);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A verb's position in its declaration list, from its wire number.
|
||||||
|
fn indexOf(comptime operation: u32) usize {
|
||||||
|
return operation - first_protocol_operation;
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- tests ------------------------------------------------------------------
|
||||||
|
|
||||||
|
const testing = std.testing;
|
||||||
|
|
||||||
|
const Produce = extern struct { count: u32, flags: u32 = 0 };
|
||||||
|
const Produced = extern struct { total: u64 };
|
||||||
|
const Changed = extern struct { kind: u32, value: u32 };
|
||||||
|
|
||||||
|
const Sample = Define(.{
|
||||||
|
.name = "sample",
|
||||||
|
.version = 3,
|
||||||
|
.operations = &.{
|
||||||
|
.{ .name = "produce", .request = Produce, .reply = Produced },
|
||||||
|
.{ .name = "reset" },
|
||||||
|
},
|
||||||
|
.events = &.{
|
||||||
|
.{ .name = "changed", .payload = Changed },
|
||||||
|
},
|
||||||
|
});
|
||||||
|
|
||||||
|
test "the prefix is 16 bytes in both directions" {
|
||||||
|
try testing.expectEqual(@as(usize, 16), @sizeOf(Header));
|
||||||
|
try testing.expectEqual(@as(usize, 16), @sizeOf(Status));
|
||||||
|
try testing.expectEqual(@as(usize, 16), prefix_size);
|
||||||
|
try testing.expectEqual(@as(usize, 256), packet_maximum);
|
||||||
|
try testing.expectEqual(@as(usize, 64), post_maximum);
|
||||||
|
}
|
||||||
|
|
||||||
|
test "verb numbering skips the reserved range" {
|
||||||
|
try testing.expectEqual(@as(u32, 16), first_protocol_operation);
|
||||||
|
try testing.expectEqual(@as(u32, 16), @intFromEnum(Sample.Operation.produce));
|
||||||
|
try testing.expectEqual(@as(u32, 17), @intFromEnum(Sample.Operation.reset));
|
||||||
|
// Events are numbered in their own space, so appending an operation can
|
||||||
|
// never renumber a shipped event.
|
||||||
|
try testing.expectEqual(@as(u32, 16), @intFromEnum(Sample.Event.changed));
|
||||||
|
for ([_]u32{ operation_describe, operation_enumerate, operation_subscribe, operation_unsubscribe }) |reserved| {
|
||||||
|
try testing.expect(reserved < first_protocol_operation);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
test "request round trip, header folded and tail carried" {
|
||||||
|
var buffer: [packet_maximum]u8 = undefined;
|
||||||
|
const packet = Sample.encodeRequest(.produce, 42, .{ .count = 7 }, "tail bytes", &buffer).?;
|
||||||
|
try testing.expectEqual(prefix_size + @sizeOf(Produce) + "tail bytes".len, packet.len);
|
||||||
|
|
||||||
|
const header = headerOf(packet).?;
|
||||||
|
try testing.expectEqual(@as(u32, 16), header.operation);
|
||||||
|
try testing.expectEqual(@as(u64, 42), header.target);
|
||||||
|
try testing.expectEqual(Sample.Operation.produce, Sample.operationOf(packet).?);
|
||||||
|
|
||||||
|
const request = Sample.decodeRequest(.produce, packet).?;
|
||||||
|
try testing.expectEqual(@as(u32, 7), request.count);
|
||||||
|
try testing.expectEqualStrings("tail bytes", Sample.requestTail(.produce, packet));
|
||||||
|
}
|
||||||
|
|
||||||
|
test "reply round trip" {
|
||||||
|
var buffer: [packet_maximum]u8 = undefined;
|
||||||
|
const packet = Sample.encodeReply(.produce, 0, .{ .total = 99 }, "more", &buffer).?;
|
||||||
|
const status = statusOf(packet).?;
|
||||||
|
try testing.expectEqual(@as(i32, 0), status.status);
|
||||||
|
try testing.expectEqual(@as(u32, @sizeOf(Produced) + "more".len), status.len);
|
||||||
|
try testing.expectEqual(@as(u64, 99), Sample.decodeReply(.produce, packet).?.total);
|
||||||
|
try testing.expectEqualStrings("more", Sample.replyTail(.produce, packet));
|
||||||
|
}
|
||||||
|
|
||||||
|
test "a void request and reply carry nothing but the verb" {
|
||||||
|
var buffer: [packet_maximum]u8 = undefined;
|
||||||
|
const packet = Sample.encodeRequest(.reset, 0, {}, &.{}, &buffer).?;
|
||||||
|
try testing.expectEqual(prefix_size, packet.len);
|
||||||
|
try testing.expectEqual(Sample.Operation.reset, Sample.operationOf(packet).?);
|
||||||
|
try testing.expectEqual(@as(usize, 0), Sample.requestTail(.reset, packet).len);
|
||||||
|
}
|
||||||
|
|
||||||
|
test "event round trip within the push floor" {
|
||||||
|
var buffer: [post_maximum]u8 = undefined;
|
||||||
|
const packet = Sample.encodeEvent(.changed, 0, .{ .kind = 1, .value = 2 }, &buffer).?;
|
||||||
|
try testing.expectEqual(prefix_size + @sizeOf(Changed), packet.len);
|
||||||
|
try testing.expect(packet.len <= post_maximum);
|
||||||
|
try testing.expectEqual(Sample.Event.changed, Sample.eventOf(packet).?);
|
||||||
|
try testing.expectEqual(@as(u32, 2), Sample.decodeEvent(.changed, packet).?.value);
|
||||||
|
}
|
||||||
|
|
||||||
|
// A provider over a trivial context, to drive the generated dispatch table.
|
||||||
|
const Counter = struct {
|
||||||
|
total: u64 = 0,
|
||||||
|
|
||||||
|
fn onProduce(self: *Counter, invocation: Invocation(Produce), answer: Answer(Produced)) isize {
|
||||||
|
self.total += invocation.request.count;
|
||||||
|
answer.set(.{ .total = self.total });
|
||||||
|
const note = "counted";
|
||||||
|
@memcpy(answer.tail()[0..note.len], note);
|
||||||
|
return note.len;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
const CounterProvider = Sample.Provider(*Counter);
|
||||||
|
|
||||||
|
test "dispatch reaches a handler and stamps the status" {
|
||||||
|
var counter = Counter{};
|
||||||
|
const handlers = CounterProvider.Handlers{ .produce = Counter.onProduce };
|
||||||
|
|
||||||
|
var request: [packet_maximum]u8 = undefined;
|
||||||
|
const packet = Sample.encodeRequest(.produce, 0, .{ .count = 5 }, &.{}, &request).?;
|
||||||
|
var reply: [packet_maximum]u8 = undefined;
|
||||||
|
const len = CounterProvider.dispatch(&counter, handlers, packet, 3, null, &reply);
|
||||||
|
|
||||||
|
const answered = reply[0..len];
|
||||||
|
try testing.expectEqual(@as(i32, 0), statusOf(answered).?.status);
|
||||||
|
try testing.expectEqual(@as(u64, 5), Sample.decodeReply(.produce, answered).?.total);
|
||||||
|
try testing.expectEqualStrings("counted", Sample.replyTail(.produce, answered));
|
||||||
|
}
|
||||||
|
|
||||||
|
test "describe is answered by the envelope, not the provider" {
|
||||||
|
var counter = Counter{};
|
||||||
|
const handlers = CounterProvider.Handlers{ .produce = Counter.onProduce };
|
||||||
|
|
||||||
|
var request: [packet_maximum]u8 = undefined;
|
||||||
|
const packet = encodeDescribe(&request).?;
|
||||||
|
var reply: [packet_maximum]u8 = undefined;
|
||||||
|
const len = CounterProvider.dispatch(&counter, handlers, packet, 3, null, &reply);
|
||||||
|
|
||||||
|
const described = decodeDescribe(reply[0..len]).?;
|
||||||
|
try testing.expectEqualStrings("sample", described.name);
|
||||||
|
try testing.expectEqual(@as(u32, 3), described.description.version);
|
||||||
|
try testing.expectEqual(@as(u32, 2), described.description.operation_count);
|
||||||
|
try testing.expectEqual(@as(u32, 1), described.description.event_count);
|
||||||
|
}
|
||||||
|
|
||||||
|
test "an unimplemented or unknown verb answers -ENOSYS" {
|
||||||
|
var counter = Counter{};
|
||||||
|
const handlers = CounterProvider.Handlers{ .produce = Counter.onProduce };
|
||||||
|
var reply: [packet_maximum]u8 = undefined;
|
||||||
|
|
||||||
|
// A verb this protocol declares but this provider left null.
|
||||||
|
var request: [packet_maximum]u8 = undefined;
|
||||||
|
const declared = Sample.encodeRequest(.reset, 0, {}, &.{}, &request).?;
|
||||||
|
var len = CounterProvider.dispatch(&counter, handlers, declared, 3, null, &reply);
|
||||||
|
try testing.expectEqual(@as(i32, -ENOSYS), statusOf(reply[0..len]).?.status);
|
||||||
|
|
||||||
|
// A number no verb of this protocol wears.
|
||||||
|
const stranger = Header{ .operation = first_protocol_operation + 900 };
|
||||||
|
len = CounterProvider.dispatch(&counter, handlers, std.mem.asBytes(&stranger), 3, null, &reply);
|
||||||
|
try testing.expectEqual(@as(i32, -ENOSYS), statusOf(reply[0..len]).?.status);
|
||||||
|
|
||||||
|
// A reserved verb the provider does not implement answers the same way.
|
||||||
|
const enumerate = Header{ .operation = operation_enumerate };
|
||||||
|
len = CounterProvider.dispatch(&counter, handlers, std.mem.asBytes(&enumerate), 3, null, &reply);
|
||||||
|
try testing.expectEqual(@as(i32, -ENOSYS), statusOf(reply[0..len]).?.status);
|
||||||
|
}
|
||||||
|
|
||||||
|
test "a truncated packet answers -EPROTO" {
|
||||||
|
var counter = Counter{};
|
||||||
|
const handlers = CounterProvider.Handlers{ .produce = Counter.onProduce };
|
||||||
|
var reply: [packet_maximum]u8 = undefined;
|
||||||
|
|
||||||
|
// Names `produce`, but stops before the request it promises.
|
||||||
|
const header = Header{ .operation = @intFromEnum(Sample.Operation.produce) };
|
||||||
|
const len = CounterProvider.dispatch(&counter, handlers, std.mem.asBytes(&header), 3, null, &reply);
|
||||||
|
try testing.expectEqual(@as(i32, -EPROTO), statusOf(reply[0..len]).?.status);
|
||||||
|
}
|
||||||
|
|
||||||
|
// The size rule, exercised directly. `Define` turns exactly these predicates
|
||||||
|
// into compile errors, which a test cannot catch — so the predicate is what the
|
||||||
|
// test pins, and the boundary protocol below proves the compile-time half from
|
||||||
|
// the other side. The negative example, spelled out: giving `Define` an
|
||||||
|
// operation with `.request = extern struct { bytes: [241]u8 }`, or an event with
|
||||||
|
// `.payload = extern struct { bytes: [49]u8 }`, fails to compile with the
|
||||||
|
// protocol, the verb, and the two numbers named in the message.
|
||||||
|
test "the floor counts the header once, and the boundary is exact" {
|
||||||
|
try testing.expect(fitsPacket(extern struct { bytes: [240]u8 }));
|
||||||
|
try testing.expect(!fitsPacket(extern struct { bytes: [241]u8 }));
|
||||||
|
try testing.expect(fitsPost(extern struct { bytes: [48]u8 }));
|
||||||
|
try testing.expect(!fitsPost(extern struct { bytes: [49]u8 }));
|
||||||
|
try testing.expect(fitsPacket(void));
|
||||||
|
try testing.expect(fitsPost(void));
|
||||||
|
}
|
||||||
|
|
||||||
|
const WidestRequest = extern struct { bytes: [packet_maximum - prefix_size]u8 };
|
||||||
|
const WidestEvent = extern struct { bytes: [post_maximum - prefix_size]u8 };
|
||||||
|
|
||||||
|
// A protocol sitting exactly on both floors. That this compiles at all is the
|
||||||
|
// positive half of the compile-time check.
|
||||||
|
const Boundary = Define(.{
|
||||||
|
.name = "boundary",
|
||||||
|
.version = 1,
|
||||||
|
.operations = &.{.{ .name = "fill", .request = WidestRequest, .reply = WidestRequest }},
|
||||||
|
.events = &.{.{ .name = "filled", .payload = WidestEvent }},
|
||||||
|
});
|
||||||
|
|
||||||
|
test "a protocol may sit exactly on the floor" {
|
||||||
|
try testing.expectEqual(packet_maximum, Boundary.request_maximum);
|
||||||
|
try testing.expectEqual(packet_maximum, Boundary.reply_maximum);
|
||||||
|
try testing.expectEqual(post_maximum, Boundary.event_maximum);
|
||||||
|
|
||||||
|
var buffer: [packet_maximum]u8 = undefined;
|
||||||
|
const packet = Boundary.encodeRequest(.fill, 0, .{ .bytes = @splat(0xAB) }, &.{}, &buffer).?;
|
||||||
|
try testing.expectEqual(packet_maximum, packet.len);
|
||||||
|
try testing.expectEqual(@as(u8, 0xAB), Boundary.decodeRequest(.fill, packet).?.bytes[239]);
|
||||||
|
|
||||||
|
// One byte of tail past the floor is refused at run time, not truncated.
|
||||||
|
try testing.expect(Boundary.encodeRequest(.fill, 0, .{ .bytes = @splat(0) }, "x", &buffer) == null);
|
||||||
|
|
||||||
|
var post: [post_maximum]u8 = undefined;
|
||||||
|
const event = Boundary.encodeEvent(.filled, 0, .{ .bytes = @splat(1) }, &post).?;
|
||||||
|
try testing.expectEqual(post_maximum, event.len);
|
||||||
|
}
|
||||||
|
|
||||||
|
test "a protocol's own sizes are reported prefix-included" {
|
||||||
|
try testing.expectEqual(prefix_size + @sizeOf(Produce), Sample.request_maximum);
|
||||||
|
try testing.expectEqual(prefix_size + @sizeOf(Produced), Sample.reply_maximum);
|
||||||
|
try testing.expectEqual(prefix_size + @sizeOf(Changed), Sample.event_maximum);
|
||||||
|
try testing.expectEqual(packet_maximum, Sample.message_maximum);
|
||||||
|
try testing.expectEqualStrings("sample", Sample.protocol_name);
|
||||||
|
}
|
||||||
@@ -1,7 +1,9 @@
|
|||||||
//! The power protocol (docs/power.md): system power's domain-named surface,
|
//! The power protocol (docs/power.md): system power's domain-named surface,
|
||||||
//! registered under `ServiceId.power`. On x86 the acpi service serves it; on
|
//! bound at `/protocol/power`. On x86 the acpi service provides it; on ARM a
|
||||||
//! ARM a PSCI/mailbox service will register the same id — subscribers never
|
//! PSCI/mailbox service will bind the same name — subscribers never learn which
|
||||||
//! learn which firmware they are on (docs/discovery.md — firmware neutrality).
|
//! firmware they are on (docs/discovery.md — firmware neutrality), which is the
|
||||||
|
//! whole point of naming the contract rather than the provider
|
||||||
|
//! (docs/os-development/protocol-namespace.md).
|
||||||
//! The vfs-protocol pattern: extern-struct messages, a version, reserved fields.
|
//! The vfs-protocol pattern: extern-struct messages, a version, reserved fields.
|
||||||
|
|
||||||
/// The protocol version a client states nowhere yet — reserved for the day a
|
/// The protocol version a client states nowhere yet — reserved for the day a
|
||||||
|
|||||||
@@ -13,8 +13,17 @@
|
|||||||
//! written against has retired — path routing moved into the kernel (system/kernel/vfs.zig,
|
//! written against has retired — path routing moved into the kernel (system/kernel/vfs.zig,
|
||||||
//! fs_resolve) — but the protocol module outlived it.
|
//! fs_resolve) — but the protocol module outlived it.
|
||||||
|
|
||||||
|
//! **An `open` reply may carry a capability.** The vfs `open` request rides
|
||||||
|
//! `ipc_call`, and the reply direction of a call can hand back an endpoint
|
||||||
|
//! (`ipc.callCap`'s `Reply.cap`). A file backend never uses it — FAT answers
|
||||||
|
//! with a node id and nothing else — but a *synthetic* backend does: opening a
|
||||||
|
//! `NodeKind.protocol` node under `/protocol` returns the provider's endpoint,
|
||||||
|
//! which is the channel (docs/os-development/protocol-namespace.md). The
|
||||||
|
//! convention is per-backend, not per-operation: a client that did not ask a
|
||||||
|
//! synthetic backend simply gets no capability back, exactly as today.
|
||||||
|
|
||||||
pub const Operation = enum(u32) {
|
pub const Operation = enum(u32) {
|
||||||
open, // open(path) -> node id
|
open, // open(path) -> node id (a synthetic backend may reply with a capability instead)
|
||||||
close, // close(node)
|
close, // close(node)
|
||||||
read, // read(node, offset, len) -> bytes
|
read, // read(node, offset, len) -> bytes
|
||||||
write, // write(node, offset, bytes) -> count
|
write, // write(node, offset, bytes) -> count
|
||||||
@@ -31,10 +40,16 @@ pub const Operation = enum(u32) {
|
|||||||
// rename: the payload is the old path, a single 0x00 separator, then the new
|
// rename: the payload is the old path, a single 0x00 separator, then the new
|
||||||
// path. Same-directory rename only (the router requires both under one mount).
|
// path. Same-directory rename only (the router requires both under one mount).
|
||||||
rename, // rename(old\0new payload) -> status
|
rename, // rename(old\0new payload) -> status
|
||||||
|
// Appended for the protocol namespace (P2). The registry is a synthetic
|
||||||
|
// backend mounted at /protocol: `open` establishes a channel and `readdir`
|
||||||
|
// lists the bound names like any directory, so those two verbs need nothing
|
||||||
|
// new — but *claiming* a name does. A file backend refuses it, alongside the
|
||||||
|
// router verbs it does not implement either; only the registry implements it.
|
||||||
|
bind, // bind(name payload, capability = the provider's endpoint) -> status
|
||||||
};
|
};
|
||||||
|
|
||||||
/// The type of a filesystem node, aligned to the FSH file-type table
|
/// The type of a filesystem node, aligned to the node-kind table
|
||||||
/// (docs/danos-file-system-hierarchy-FSH.md). Fills `FileStatus.kind` and
|
/// (docs/file-system-development/file-system-hierarchy.md). Fills `FileStatus.kind` and
|
||||||
/// `DirectoryEntry.kind`; `regular = 0` keeps the historical hardcoded value.
|
/// `DirectoryEntry.kind`; `regular = 0` keeps the historical hardcoded value.
|
||||||
pub const NodeKind = enum(u32) {
|
pub const NodeKind = enum(u32) {
|
||||||
regular = 0,
|
regular = 0,
|
||||||
@@ -44,6 +59,12 @@ pub const NodeKind = enum(u32) {
|
|||||||
symbolic_link = 4,
|
symbolic_link = 4,
|
||||||
fifo = 5,
|
fifo = 5,
|
||||||
socket = 6,
|
socket = 6,
|
||||||
|
/// A node that names a *contract*, not a file: opening it establishes a
|
||||||
|
/// channel to whatever process currently provides that protocol, delivered
|
||||||
|
/// as an endpoint capability in the reply rather than a node id. This is
|
||||||
|
/// what lives under `/protocol`; `readdir` lists these like any other node,
|
||||||
|
/// so the tree stays browsable for diagnosis.
|
||||||
|
protocol = 7,
|
||||||
};
|
};
|
||||||
|
|
||||||
/// One directory entry, returned by `readdir`: a fixed header followed inline in
|
/// One directory entry, returned by `readdir`: a fixed header followed inline in
|
||||||
@@ -109,9 +130,15 @@ test "protocol struct sizes and node kinds" {
|
|||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
try std.testing.expectEqual(@as(u32, 0), @intFromEnum(NodeKind.regular));
|
try std.testing.expectEqual(@as(u32, 0), @intFromEnum(NodeKind.regular));
|
||||||
try std.testing.expectEqual(@as(u32, 1), @intFromEnum(NodeKind.directory));
|
try std.testing.expectEqual(@as(u32, 1), @intFromEnum(NodeKind.directory));
|
||||||
|
// Appended with the protocol namespace; every earlier value keeps its own.
|
||||||
|
try std.testing.expectEqual(@as(u32, 6), @intFromEnum(NodeKind.socket));
|
||||||
|
try std.testing.expectEqual(@as(u32, 7), @intFromEnum(NodeKind.protocol));
|
||||||
try std.testing.expectEqual(@as(usize, 16), @sizeOf(DirectoryEntry));
|
try std.testing.expectEqual(@as(usize, 16), @sizeOf(DirectoryEntry));
|
||||||
// The appended operations keep the original values.
|
// The appended operations keep the original values.
|
||||||
try std.testing.expectEqual(@as(u32, 0), @intFromEnum(Operation.open));
|
try std.testing.expectEqual(@as(u32, 0), @intFromEnum(Operation.open));
|
||||||
try std.testing.expectEqual(@as(u32, 4), @intFromEnum(Operation.status));
|
try std.testing.expectEqual(@as(u32, 4), @intFromEnum(Operation.status));
|
||||||
try std.testing.expectEqual(@as(u32, 5), @intFromEnum(Operation.readdir));
|
try std.testing.expectEqual(@as(u32, 5), @intFromEnum(Operation.readdir));
|
||||||
|
try std.testing.expectEqual(@as(u32, 10), @intFromEnum(Operation.rename));
|
||||||
|
// The registry's claim verb, appended last with the protocol namespace.
|
||||||
|
try std.testing.expectEqual(@as(u32, 11), @intFromEnum(Operation.bind));
|
||||||
}
|
}
|
||||||
|
|||||||
+13
-20
@@ -1,6 +1,6 @@
|
|||||||
//! The **private kernel ↔ runtime** ABI: the raw system_call contract — the call
|
//! The **private kernel ↔ runtime** ABI: the raw system_call contract — the call
|
||||||
//! numbers, `mmap` protection flags, the page size those calls work in, and the IPC
|
//! numbers, `mmap` protection flags, the page size those calls work in, and the IPC
|
||||||
//! name-registry ids and notification bit. Shared by the kernel dispatcher
|
//! notification bits. Shared by the kernel dispatcher
|
||||||
//! (system/kernel/process.zig) and the user-space runtime library (library/runtime/),
|
//! (system/kernel/process.zig) and the user-space runtime library (library/runtime/),
|
||||||
//! so the two can never drift.
|
//! so the two can never drift.
|
||||||
//!
|
//!
|
||||||
@@ -33,8 +33,13 @@ pub const SystemCall = enum(u64) {
|
|||||||
mmap = 4, // mmap(len, prot) -> base: grant zeroed, page-aligned user pages
|
mmap = 4, // mmap(len, prot) -> base: grant zeroed, page-aligned user pages
|
||||||
munmap = 5, // munmap(base, len): release pages from a prior mmap
|
munmap = 5, // munmap(base, len): release pages from a prior mmap
|
||||||
create_ipc_endpoint = 6, // create_ipc_endpoint() -> handle: a new IPC endpoint
|
create_ipc_endpoint = 6, // create_ipc_endpoint() -> handle: a new IPC endpoint
|
||||||
ipc_register = 7, // ipc_register(service_id, handle): publish an endpoint by well-known id
|
// 7 and 8 were ipc_register/ipc_lookup — the flat ServiceId name registry,
|
||||||
ipc_lookup = 8, // ipc_lookup(service_id) -> handle: find a published endpoint
|
// retired with the protocol namespace (docs/os-development/protocol-namespace.md).
|
||||||
|
// A service now binds its name at the registry (init, over /protocol) and a
|
||||||
|
// client resolves and opens that path; neither is a system call any more. The
|
||||||
|
// numbers stay vacant rather than being reused: every other entry is
|
||||||
|
// position-fixed by an explicit value, so a hole costs nothing and a reused
|
||||||
|
// number would silently mean two things across a rebuild boundary.
|
||||||
ipc_call = 9, // ipc_call(h, message, len, reply, cap) -> reply_len: send + block for reply
|
ipc_call = 9, // ipc_call(h, message, len, reply, cap) -> reply_len: send + block for reply
|
||||||
ipc_reply_wait = 10, // ipc_reply_wait(h, reply, len, receive, cap) -> receive_len (+badge in rdx)
|
ipc_reply_wait = 10, // ipc_reply_wait(h, reply, len, receive, cap) -> receive_len (+badge in rdx)
|
||||||
device_enumerate = 11, // device_enumerate(buffer, maximum) -> count: snapshot the device table
|
device_enumerate = 11, // device_enumerate(buffer, maximum) -> count: snapshot the device table
|
||||||
@@ -284,23 +289,11 @@ pub const KlogStatus = extern struct {
|
|||||||
boot_unix_seconds: u64, // wall-clock time of boot (RTC anchor)
|
boot_unix_seconds: u64, // wall-clock time of boot (RTC anchor)
|
||||||
};
|
};
|
||||||
|
|
||||||
/// Well-known IPC service ids for the bootstrap name registry (create_ipc_endpoint +
|
// The `ServiceId` enum lived here: a flat, compile-time list of well-known
|
||||||
/// ipc_register/ipc_lookup). Small integers, so no string interning is needed
|
// service ids backed by a 16-slot kernel table. It is gone with the protocol
|
||||||
/// during bring-up. The VFS server registers under `vfs`; clients look it up.
|
// namespace — names are strings resolved under `/protocol` at run time, so a
|
||||||
pub const ServiceId = enum(u32) {
|
// third-party program can introduce a contract the ABI never heard of, and the
|
||||||
vfs = 1, // RETIRED: the router moved into the kernel (fs_resolve); the slot stays reserved
|
// registrar (init) decides who may claim one.
|
||||||
input = 2,
|
|
||||||
ps2_bus = 3, // the 8042 owner; child device drivers attach here for raw bytes
|
|
||||||
device_manager = 4, // the tree, the matcher, the supervisor (docs/device-manager.md)
|
|
||||||
power = 5, // system power: events (button, lid, battery) + shutdown (docs/power.md; domain-named per docs/discovery.md — the acpi service registers it on x86, a PSCI service will on ARM)
|
|
||||||
usb_bus = 6, // the xHCI host-controller driver's transfer endpoint; USB class drivers look it up and `callCap`-open their device to get a private per-device transfer channel (docs/driver-model.md)
|
|
||||||
block = 7, // a block-device driver (USB mass storage today): read/write of fixed-size blocks, the storage a filesystem sits on
|
|
||||||
fat = 8, // the FAT filesystem server; the VFS mounts it and forwards paths under its mount point (/mnt/usb) to it
|
|
||||||
display = 9, // the display service: owns the framebuffer, composites a layer stack, presents frames (docs/display.md)
|
|
||||||
shared_memory_test = 10, // the shared-memory test server (V2): a client passes it a shared-memory capability, it maps + verifies (docs/display-v2.md)
|
|
||||||
scanout = 11, // a native scanout driver (virtio-gpu): the compositor finds it here to upgrade off the GOP framebuffer (docs/display-v2.md)
|
|
||||||
_,
|
|
||||||
};
|
|
||||||
|
|
||||||
/// Protection flags for `mmap` (matching the usual C bit values).
|
/// Protection flags for `mmap` (matching the usual C bit values).
|
||||||
pub const prot_read: u64 = 1;
|
pub const prot_read: u64 = 1;
|
||||||
|
|||||||
@@ -1,4 +1,4 @@
|
|||||||
# /etc/devices.csv — the device→driver registry.
|
# /system/configuration/devices.csv — the device→driver registry.
|
||||||
#
|
#
|
||||||
# The device manager reads this at boot and binds each device a bus driver
|
# The device manager reads this at boot and binds each device a bus driver
|
||||||
# reports to the driver named here. It is AUTHORITATIVE: a device that no row
|
# reports to the driver named here. It is AUTHORITATIVE: a device that no row
|
||||||
|
@@ -1,9 +1,10 @@
|
|||||||
# /etc/init.csv — diagnose variant (-Ddiagnose), bundled at /etc/init.csv.
|
# /system/configuration/init.csv — diagnose variant (-Ddiagnose), bundled at
|
||||||
|
# /system/configuration/init.csv.
|
||||||
#
|
#
|
||||||
# The display stack (display, display-demo) is omitted so the kernel's timestamped
|
# The display stack (display, display-demo) is omitted so the kernel's timestamped
|
||||||
# on-screen boot transcript is never suppressed — the bring-up timeline (USB,
|
# on-screen boot transcript is never suppressed — the bring-up timeline (USB,
|
||||||
# storage, logger) stays readable on real hardware with no serial. See etc/init.csv
|
# storage, logger) stays readable on real hardware with no serial. See
|
||||||
# for the format; this file must otherwise track it.
|
# system/configuration/init.csv for the format; this file must otherwise track it.
|
||||||
#
|
#
|
||||||
# service args...
|
# service args...
|
||||||
/system/services/input
|
/system/services/input
|
||||||
|
@@ -1,4 +1,4 @@
|
|||||||
# /etc/init.csv — the services init (PID 1) starts at boot, in order.
|
# /system/configuration/init.csv — the services init (PID 1) starts at boot, in order.
|
||||||
#
|
#
|
||||||
# init reads this at startup and spawns each service supervised (restarting it on
|
# init reads this at startup and spawns each service supervised (restarting it on
|
||||||
# a crash, up to a cap). Startup order is top->bottom; shutdown is the reverse, so
|
# a crash, up to a cap). Startup order is top->bottom; shutdown is the reverse, so
|
||||||
@@ -9,7 +9,7 @@
|
|||||||
# '#' starts a comment (whole-line or trailing); blank lines are ignored. The
|
# '#' starts a comment (whole-line or trailing); blank lines are ignored. The
|
||||||
# first field is the service binary path; any fields after it are the service's
|
# first field is the service binary path; any fields after it are the service's
|
||||||
# argv. Drivers are absent on purpose — the device manager discovers hardware and
|
# argv. Drivers are absent on purpose — the device manager discovers hardware and
|
||||||
# spawns those (see /etc/devices.csv).
|
# spawns those (see /system/configuration/devices.csv).
|
||||||
#
|
#
|
||||||
# service args...
|
# service args...
|
||||||
/system/services/input
|
/system/services/input
|
||||||
|
@@ -0,0 +1,155 @@
|
|||||||
|
# /system/configuration/protocol.csv — who may claim, and who may reach, a name
|
||||||
|
# under /protocol (docs/os-development/protocol-namespace.md).
|
||||||
|
#
|
||||||
|
# init is the registrar: it serves /protocol, and every bind AND every open is
|
||||||
|
# checked against this file. It is AUTHORITATIVE — a name no row grants cannot be
|
||||||
|
# bound or reached, and a missing file means nothing may be bound or reached at
|
||||||
|
# all.
|
||||||
|
#
|
||||||
|
# A refused open is answered exactly as a name nobody bound is: -ENOENT, and no
|
||||||
|
# capability. That is not politeness, it is the model — the namespace IS the
|
||||||
|
# restriction, so what a process may not open simply does not exist for it, and
|
||||||
|
# there is no "permission denied" for it to tell apart from "no such contract".
|
||||||
|
# Which is why a missing row here shows up as a client retrying forever rather
|
||||||
|
# than as an error: check this file first, and `readdir /protocol` second.
|
||||||
|
#
|
||||||
|
# '#' starts a comment (whole-line or trailing); blank lines are ignored.
|
||||||
|
# Whitespace around a field is trimmed, so columns may be padded. Four
|
||||||
|
# comma-separated fields per row:
|
||||||
|
#
|
||||||
|
# binary the claimant's binary path, exactly as the kernel stamped it at
|
||||||
|
# spawn (argv[0]) — unforgeable, read from the process records
|
||||||
|
# supervisor the authorized supervising TASK, written as the binary it runs —
|
||||||
|
# the path init was started as for its own services, the device
|
||||||
|
# manager's path for the drivers it starts. The one word that is not
|
||||||
|
# a path is 'kernel', because a kernel task has no binary; that is
|
||||||
|
# what the test harness's direct spawns look like.
|
||||||
|
# Matched by IDENTITY, not by spelling. Name alone is not identity —
|
||||||
|
# spawn is ungated, so a hostile process can start a granted binary
|
||||||
|
# itself and inherit its grants; and it can equally start its own
|
||||||
|
# instance of the *supervisor's* binary and have that spawn the
|
||||||
|
# granted one, at which point both names read correctly (the
|
||||||
|
# laundering deputy). So init also asks which task the supervisor
|
||||||
|
# is: 'kernel' means supervisor id 0, which only the kernel can
|
||||||
|
# confer; init's own path means this init; any other path means a
|
||||||
|
# task init spawned itself or one the kernel spawned. Task ids are
|
||||||
|
# monotonic and never reused, so an id cannot be borrowed.
|
||||||
|
# permission bind (provide this contract) | open (speak to it) |
|
||||||
|
# supervise (stand in someone else's chain — see below)
|
||||||
|
# name the contract, relative to /protocol
|
||||||
|
#
|
||||||
|
# A trailing '*' on any field matches any tail — how a subtree is granted whole.
|
||||||
|
#
|
||||||
|
# 'supervise' exists because attestation is one hop deep and the driver tree is
|
||||||
|
# three: the device manager starts the PS/2 bus, and the bus starts the keyboard
|
||||||
|
# and mouse drivers. Init never met the bus, so it cannot vouch for it by
|
||||||
|
# acquaintance — and it must not vouch for it by name, or the laundering deputy
|
||||||
|
# walks straight in. A 'supervise' row is the manifest saying it: a task running
|
||||||
|
# this binary, under this supervisor, may be the supervising task an 'open' row
|
||||||
|
# names, for this contract and no other. It grants the delegate nothing itself,
|
||||||
|
# and it is deliberately open-only — a delegate may vouch for what its children
|
||||||
|
# REACH, never for what they CLAIM, so every bind refusal is untouched by it.
|
||||||
|
#
|
||||||
|
# binary supervisor permission name
|
||||||
|
|
||||||
|
# --- the services init spawns from init.csv ---------------------------------
|
||||||
|
/system/services/input, /system/services/init, bind, input
|
||||||
|
/system/services/device-manager, /system/services/init, bind, device-manager
|
||||||
|
/system/services/fat, /system/services/init, bind, vfs
|
||||||
|
/system/services/display, /system/services/init, bind, display
|
||||||
|
|
||||||
|
# The discovery service ships under one neutral name per firmware (docs/discovery.md);
|
||||||
|
# on x86 it is the acpi service, and what it provides is the power contract.
|
||||||
|
/system/services/discovery, /system/services/device-manager, bind, power
|
||||||
|
|
||||||
|
# --- the drivers, which the device manager spawns ---------------------------
|
||||||
|
/system/drivers/ps2-bus, /system/services/device-manager, bind, ps2-bus
|
||||||
|
/system/drivers/usb-xhci-bus, /system/services/device-manager, bind, usb-transfer
|
||||||
|
/system/drivers/usb-storage, /system/services/device-manager, bind, block
|
||||||
|
/system/drivers/virtio-gpu, /system/services/device-manager, bind, scanout
|
||||||
|
|
||||||
|
# --- the same providers when the kernel test harness starts them directly ---
|
||||||
|
# A scenario boot spawns its own providers instead of letting init do it
|
||||||
|
# (docs/security-track-plan.md, decision 9), so the same binaries appear with
|
||||||
|
# 'kernel' as the supervisor. Nothing else changes: the binary must still match.
|
||||||
|
/system/services/input, kernel, bind, input
|
||||||
|
/system/services/device-manager, kernel, bind, device-manager
|
||||||
|
/system/services/fat, kernel, bind, vfs
|
||||||
|
/system/services/display, kernel, bind, display
|
||||||
|
/system/services/discovery, kernel, bind, power
|
||||||
|
|
||||||
|
# --- test fixtures ----------------------------------------------------------
|
||||||
|
# The subtree rule, dogfooded: anything installed under /test may claim anything
|
||||||
|
# under /protocol/test, and nothing above it — whether the harness spawned it or
|
||||||
|
# another fixture did.
|
||||||
|
/test/*, kernel, bind, test/*
|
||||||
|
/test/*, /test/*, bind, test/*
|
||||||
|
|
||||||
|
|
||||||
|
# ============================================================================
|
||||||
|
# open — who may REACH each contract. One row per client per contract; a client
|
||||||
|
# with no row here simply finds the name absent, forever.
|
||||||
|
# ============================================================================
|
||||||
|
|
||||||
|
# --- init's own services ----------------------------------------------------
|
||||||
|
# fat reaches the block device behind the volume it mounts; the compositor
|
||||||
|
# reaches the scanout its driver announced, its own endpoint (the mouse-listener
|
||||||
|
# thread opens /protocol/display like any other client — threads share no
|
||||||
|
# handles), and the input stream that moves the cursor.
|
||||||
|
/system/services/fat, /system/services/init, open, block
|
||||||
|
/system/services/display, /system/services/init, open, scanout
|
||||||
|
/system/services/display, /system/services/init, open, display
|
||||||
|
/system/services/display, /system/services/init, open, input
|
||||||
|
/system/services/display-demo, /system/services/init, open, display
|
||||||
|
|
||||||
|
# --- the same two when the kernel test harness starts them directly ---------
|
||||||
|
/system/services/display, kernel, open, scanout
|
||||||
|
/system/services/display, kernel, open, display
|
||||||
|
/system/services/display, kernel, open, input
|
||||||
|
/system/services/display-demo, kernel, open, display
|
||||||
|
|
||||||
|
# --- the drivers, and the discovery service ---------------------------------
|
||||||
|
# Every driver says hello to the manager that started it — one row for the whole
|
||||||
|
# subtree, because that handshake is what being a driver means. The rest are per
|
||||||
|
# driver: the storage and HID class drivers talk to their controller, the HID
|
||||||
|
# drivers publish into the input stream, and the GPU driver announces its scanout
|
||||||
|
# to the compositor.
|
||||||
|
/system/drivers/*, /system/services/device-manager, open, device-manager
|
||||||
|
/system/services/discovery, /system/services/device-manager, open, device-manager
|
||||||
|
/system/drivers/usb-storage, /system/services/device-manager, open, usb-transfer
|
||||||
|
/system/drivers/usb-hid-keyboard, /system/services/device-manager, open, usb-transfer
|
||||||
|
/system/drivers/usb-hid-keyboard, /system/services/device-manager, open, input
|
||||||
|
/system/drivers/usb-hid-mouse, /system/services/device-manager, open, usb-transfer
|
||||||
|
/system/drivers/usb-hid-mouse, /system/services/device-manager, open, input
|
||||||
|
/system/drivers/virtio-gpu, /system/services/device-manager, open, display
|
||||||
|
|
||||||
|
# --- the PS/2 child drivers, one hop further down ---------------------------
|
||||||
|
# The keyboard and mouse drivers are started by the BUS driver, not by the
|
||||||
|
# device manager — the one three-deep chain in the tree. Init cannot vouch for
|
||||||
|
# the bus by acquaintance (it never started it), so the manifest authorizes it
|
||||||
|
# explicitly, and only for the two contracts its children need.
|
||||||
|
/system/drivers/ps2-bus, /system/services/device-manager, supervise, ps2-bus
|
||||||
|
/system/drivers/ps2-bus, /system/services/device-manager, supervise, input
|
||||||
|
/system/drivers/ps2-keyboard, /system/drivers/ps2-bus, open, ps2-bus
|
||||||
|
/system/drivers/ps2-keyboard, /system/drivers/ps2-bus, open, input
|
||||||
|
/system/drivers/ps2-mouse, /system/drivers/ps2-bus, open, ps2-bus
|
||||||
|
/system/drivers/ps2-mouse, /system/drivers/ps2-bus, open, input
|
||||||
|
|
||||||
|
# --- test fixtures ----------------------------------------------------------
|
||||||
|
# The /protocol/test subtree is theirs whole, the way the bind rows give it to
|
||||||
|
# them. Everything ABOVE that subtree is named one fixture at a time, so a
|
||||||
|
# fixture reaches a system contract only where a scenario needs it — which is
|
||||||
|
# what leaves the rest genuinely absent for the rest of them (the protocol-denied
|
||||||
|
# case asks for one it was not given, and is told there is no such thing).
|
||||||
|
/test/*, kernel, open, test/*
|
||||||
|
/test/*, /test/*, open, test/*
|
||||||
|
/test/*, kernel, open, device-manager
|
||||||
|
/test/*, /system/services/device-manager, open, device-manager
|
||||||
|
/test/system/services/input-source, kernel, open, input
|
||||||
|
/test/system/services/input-test, kernel, open, input
|
||||||
|
|
||||||
|
# The laundering-deputy probe (test/system/services/protocol-registry-test) runs
|
||||||
|
# a grandchild whose supervisor is a fixture nobody authorized — that is the
|
||||||
|
# point of it, and its bind must stay refused. It still has to report the verdict
|
||||||
|
# it got, so its reporting channel, and nothing else, is delegated.
|
||||||
|
/test/*, /test/*, supervise, test/verdict
|
||||||
|
Can't render this file because it contains an unexpected character in line 12 and column 15.
|
@@ -1,6 +1,6 @@
|
|||||||
//! The pci-bus driver as a binary package (docs/build-packages-plan.md):
|
//! The pci-bus driver as a binary package (docs/build-packages-plan.md):
|
||||||
//! this file names the binary and EXACTLY the modules its source imports —
|
//! this file names the binary and EXACTLY the modules its source imports —
|
||||||
//! the shared recipe and the module-to-domain map live in build-support.
|
//! build-support resolves each name from the domains this zon declares.
|
||||||
|
|
||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
const build_support = @import("build-support");
|
const build_support = @import("build-support");
|
||||||
|
|||||||
@@ -21,7 +21,7 @@ const logging = @import("logging");
|
|||||||
const device_manager_protocol = @import("device-manager-protocol");
|
const device_manager_protocol = @import("device-manager-protocol");
|
||||||
const pci_class = @import("pci-class");
|
const pci_class = @import("pci-class");
|
||||||
|
|
||||||
/// Log a discovered function as its would-be /etc/devices.csv columns (bus, base,
|
/// Log a discovered function as its would-be /system/configuration/devices.csv columns (bus, base,
|
||||||
/// class, prog_if, vendor, device, subsystem) followed by the human-readable
|
/// class, prog_if, vendor, device, subsystem) followed by the human-readable
|
||||||
/// class/subclass/prog-IF names — so a row for a new driver reads straight off the
|
/// class/subclass/prog-IF names — so a row for a new driver reads straight off the
|
||||||
/// boot log. `subsystem` prints as `*` when the function has none, matching the CSV
|
/// boot log. `subsystem` prints as `*` when the function has none, matching the CSV
|
||||||
@@ -154,7 +154,7 @@ fn registerAndReport(bus: u64, dev: u64, function: u64, class_triple: u32) void
|
|||||||
descriptor.class = @intFromEnum(device.DeviceClass.pci_device);
|
descriptor.class = @intFromEnum(device.DeviceClass.pci_device);
|
||||||
descriptor.pci_class = class_triple;
|
descriptor.pci_class = class_triple;
|
||||||
// Vendor/device from the first config dword (0x00): low half vendor, high half
|
// Vendor/device from the first config dword (0x00): low half vendor, high half
|
||||||
// device. These carry to the manager's /etc/devices.csv matcher so a function
|
// device. These carry to the manager's /system/configuration/devices.csv matcher so a function
|
||||||
// can bind on its exact 1AF4:1050 identity, not just its class triple.
|
// can bind on its exact 1AF4:1050 identity, not just its class triple.
|
||||||
const vendor_device = configRead(bus, dev, function, 0x00);
|
const vendor_device = configRead(bus, dev, function, 0x00);
|
||||||
descriptor.vendor = @truncate(vendor_device);
|
descriptor.vendor = @truncate(vendor_device);
|
||||||
@@ -245,11 +245,11 @@ fn registerAndReport(bus: u64, dev: u64, function: u64, class_triple: u32) void
|
|||||||
};
|
};
|
||||||
}
|
}
|
||||||
|
|
||||||
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Handle) usize {
|
fn onMessage(message: []const u8, reply: []u8, sender: u32, arrived: *ipc.Arrival) usize {
|
||||||
_ = message;
|
_ = message;
|
||||||
_ = reply;
|
_ = reply;
|
||||||
_ = sender;
|
_ = sender;
|
||||||
_ = capability;
|
_ = arrived; // nothing here takes a capability: the harness closes what arrives
|
||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
//! The ps2-bus driver as a binary package (docs/build-packages-plan.md):
|
//! The ps2-bus driver as a binary package (docs/build-packages-plan.md):
|
||||||
//! this file names the binary and EXACTLY the modules its source imports —
|
//! this file names the binary and EXACTLY the modules its source imports —
|
||||||
//! the shared recipe and the module-to-domain map live in build-support.
|
//! build-support resolves each name from the domains this zon declares.
|
||||||
|
|
||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
const build_support = @import("build-support");
|
const build_support = @import("build-support");
|
||||||
@@ -9,7 +9,7 @@ pub fn build(b: *std.Build) void {
|
|||||||
const ps2_bus_exe = build_support.userBinary(b, .{
|
const ps2_bus_exe = build_support.userBinary(b, .{
|
||||||
.name = "ps2-bus",
|
.name = "ps2-bus",
|
||||||
.root_source_file = b.path("ps2-bus.zig"),
|
.root_source_file = b.path("ps2-bus.zig"),
|
||||||
.imports = &.{ "acpi-ids", "driver", "ipc", "logging", "memory", "process", "service", "time" },
|
.imports = &.{ "acpi-ids", "channel", "driver", "ipc", "logging", "memory", "process", "service", "time" },
|
||||||
});
|
});
|
||||||
b.installArtifact(ps2_bus_exe);
|
b.installArtifact(ps2_bus_exe);
|
||||||
|
|
||||||
@@ -17,8 +17,8 @@ pub fn build(b: *std.Build) void {
|
|||||||
.name = "ps2-keyboard",
|
.name = "ps2-keyboard",
|
||||||
.root_source_file = b.path("keyboard.zig"),
|
.root_source_file = b.path("keyboard.zig"),
|
||||||
.imports = &.{
|
.imports = &.{
|
||||||
"acpi-ids", "driver", "input-client", "input-protocol", "ipc", "logging", "memory",
|
"acpi-ids", "channel", "driver", "input-client", "input-protocol", "ipc",
|
||||||
"process", "time", "xkeyboard-config",
|
"logging", "memory", "process", "time", "xkeyboard-config",
|
||||||
},
|
},
|
||||||
});
|
});
|
||||||
b.installArtifact(ps2_keyboard_exe);
|
b.installArtifact(ps2_keyboard_exe);
|
||||||
@@ -27,8 +27,8 @@ pub fn build(b: *std.Build) void {
|
|||||||
.name = "ps2-mouse",
|
.name = "ps2-mouse",
|
||||||
.root_source_file = b.path("mouse.zig"),
|
.root_source_file = b.path("mouse.zig"),
|
||||||
.imports = &.{
|
.imports = &.{
|
||||||
"acpi-ids", "driver", "input-client", "input-protocol", "ipc", "logging", "memory",
|
"acpi-ids", "channel", "driver", "input-client", "input-protocol", "ipc", "logging",
|
||||||
"process", "time",
|
"memory", "process", "time",
|
||||||
},
|
},
|
||||||
});
|
});
|
||||||
b.installArtifact(ps2_mouse_exe);
|
b.installArtifact(ps2_mouse_exe);
|
||||||
|
|||||||
@@ -16,6 +16,7 @@
|
|||||||
|
|
||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
const device = @import("driver");
|
const device = @import("driver");
|
||||||
|
const channel = @import("channel");
|
||||||
const ipc = @import("ipc");
|
const ipc = @import("ipc");
|
||||||
const process = @import("process");
|
const process = @import("process");
|
||||||
const time = @import("time");
|
const time = @import("time");
|
||||||
@@ -27,12 +28,12 @@ const ps2 = @import("ps2-library.zig");
|
|||||||
const scancode = @import("scancode.zig");
|
const scancode = @import("scancode.zig");
|
||||||
const input_protocol = @import("input-protocol");
|
const input_protocol = @import("input-protocol");
|
||||||
|
|
||||||
/// Look up the ps2-bus service, retrying while the bus (which spawned us before
|
/// Open `/protocol/ps2-bus`, retrying while the bus (which spawned us before
|
||||||
/// registering) is still coming up.
|
/// binding) is still coming up.
|
||||||
fn lookupBus() ?ipc.Handle {
|
fn lookupBus() ?ipc.Handle {
|
||||||
var attempts: usize = 0;
|
var attempts: usize = 0;
|
||||||
while (attempts < 100) : (attempts += 1) {
|
while (attempts < 100) : (attempts += 1) {
|
||||||
if (ipc.lookup(.ps2_bus)) |handle| return handle;
|
if (channel.openEndpoint("ps2-bus")) |handle| return handle;
|
||||||
time.sleepMillis(50);
|
time.sleepMillis(50);
|
||||||
}
|
}
|
||||||
return null;
|
return null;
|
||||||
|
|||||||
@@ -16,6 +16,7 @@
|
|||||||
|
|
||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
const device = @import("driver");
|
const device = @import("driver");
|
||||||
|
const channel = @import("channel");
|
||||||
const ipc = @import("ipc");
|
const ipc = @import("ipc");
|
||||||
const process = @import("process");
|
const process = @import("process");
|
||||||
const time = @import("time");
|
const time = @import("time");
|
||||||
@@ -26,12 +27,12 @@ const ps2 = @import("ps2-library.zig");
|
|||||||
const mouse_packet = @import("mouse-packet.zig");
|
const mouse_packet = @import("mouse-packet.zig");
|
||||||
const input_protocol = @import("input-protocol");
|
const input_protocol = @import("input-protocol");
|
||||||
|
|
||||||
/// Look up the ps2-bus service, retrying while the bus (which spawned us before
|
/// Open `/protocol/ps2-bus`, retrying while the bus (which spawned us before
|
||||||
/// registering) is still coming up.
|
/// binding) is still coming up.
|
||||||
fn lookupBus() ?ipc.Handle {
|
fn lookupBus() ?ipc.Handle {
|
||||||
var attempts: usize = 0;
|
var attempts: usize = 0;
|
||||||
while (attempts < 100) : (attempts += 1) {
|
while (attempts < 100) : (attempts += 1) {
|
||||||
if (ipc.lookup(.ps2_bus)) |handle| return handle;
|
if (channel.openEndpoint("ps2-bus")) |handle| return handle;
|
||||||
time.sleepMillis(50);
|
time.sleepMillis(50);
|
||||||
}
|
}
|
||||||
return null;
|
return null;
|
||||||
|
|||||||
@@ -11,6 +11,7 @@
|
|||||||
//! - irq 0xc len 0x1
|
//! - irq 0xc len 0x1
|
||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
const device = @import("driver");
|
const device = @import("driver");
|
||||||
|
const channel = @import("channel");
|
||||||
const ipc = @import("ipc");
|
const ipc = @import("ipc");
|
||||||
const process = @import("process");
|
const process = @import("process");
|
||||||
const service = @import("service");
|
const service = @import("service");
|
||||||
@@ -62,7 +63,13 @@ var port_device_types = [_]?ps2.DeviceType{ null, null };
|
|||||||
/// Handle a child driver's `AttachRequest`: record the endpoint capability it
|
/// Handle a child driver's `AttachRequest`: record the endpoint capability it
|
||||||
/// passed as the forwarding target for the port whose device matches its type.
|
/// passed as the forwarding target for the port whose device matches its type.
|
||||||
/// Writes an `AttachReply` into `out` and returns its length.
|
/// Writes an `AttachReply` into `out` and returns its length.
|
||||||
fn handleAttach(message: []const u8, got: ipc.Received, out: []u8) usize {
|
/// A child driver's AttachRequest. The endpoint it hands over arrives under the
|
||||||
|
/// same ownership rule the service harness states (`ipc.Arrival`): the turn owns
|
||||||
|
/// it, and only the path that records it in `port_endpoints` says `take`. Every
|
||||||
|
/// refusal here simply returns, and the loop closes what arrived — otherwise a
|
||||||
|
/// stranger (this is a named contract, reachable by anyone) spends one of this
|
||||||
|
/// driver's thirty-two handle slots per malformed attach.
|
||||||
|
fn handleAttach(message: []const u8, out: []u8, arrived: *ipc.Arrival) usize {
|
||||||
const reply = struct {
|
const reply = struct {
|
||||||
fn write(buffer: []u8, status: ps2.AttachStatus) usize {
|
fn write(buffer: []u8, status: ps2.AttachStatus) usize {
|
||||||
const header = ps2.AttachReply{ .status = @intFromEnum(status) };
|
const header = ps2.AttachReply{ .status = @intFromEnum(status) };
|
||||||
@@ -73,12 +80,17 @@ fn handleAttach(message: []const u8, got: ipc.Received, out: []u8) usize {
|
|||||||
|
|
||||||
if (message.len < @sizeOf(ps2.AttachRequest)) return reply.write(out, .invalid_request);
|
if (message.len < @sizeOf(ps2.AttachRequest)) return reply.write(out, .invalid_request);
|
||||||
const request = std.mem.bytesToValue(ps2.AttachRequest, message[0..@sizeOf(ps2.AttachRequest)]);
|
const request = std.mem.bytesToValue(ps2.AttachRequest, message[0..@sizeOf(ps2.AttachRequest)]);
|
||||||
const endpoint = got.cap orelse return reply.write(out, .missing_endpoint);
|
const endpoint = arrived.peek() orelse return reply.write(out, .missing_endpoint);
|
||||||
|
|
||||||
for (&port_device_types, 0..) |maybe_type, port_index| {
|
for (&port_device_types, 0..) |maybe_type, port_index| {
|
||||||
const device_type = maybe_type orelse continue;
|
const device_type = maybe_type orelse continue;
|
||||||
if (@intFromEnum(device_type) != request.device_type) continue;
|
if (@intFromEnum(device_type) != request.device_type) continue;
|
||||||
port_endpoints[port_index] = endpoint;
|
// Claimed. A re-attach supersedes the previous driver's endpoint, and the
|
||||||
|
// one it displaces is closed: the slot holds exactly one reference.
|
||||||
|
if (port_endpoints[port_index]) |previous| {
|
||||||
|
if (previous != endpoint) _ = ipc.close(previous);
|
||||||
|
}
|
||||||
|
port_endpoints[port_index] = arrived.take();
|
||||||
std.log.info("{s} driver attached", .{@tagName(device_type)});
|
std.log.info("{s} driver attached", .{@tagName(device_type)});
|
||||||
return reply.write(out, .ok);
|
return reply.write(out, .ok);
|
||||||
}
|
}
|
||||||
@@ -213,15 +225,16 @@ pub fn main() void {
|
|||||||
return;
|
return;
|
||||||
};
|
};
|
||||||
|
|
||||||
// The endpoint the child drivers attach to and IRQ1 wakes. Registered under a
|
// The endpoint the child drivers attach to and IRQ1 wakes. Bound as the
|
||||||
// well-known id so the children can find it, the way input subscribers find
|
// `ps2-bus` contract so the children can find it by name, the way input
|
||||||
// the input service.
|
// subscribers find the input service. This driver runs its own loop rather
|
||||||
|
// than the service harness, so it binds by hand — same call the harness makes.
|
||||||
const endpoint = ipc.createIpcEndpoint() orelse {
|
const endpoint = ipc.createIpcEndpoint() orelse {
|
||||||
_ = logging.write("/system/drivers/ps2-bus: no endpoint\n");
|
_ = logging.write("/system/drivers/ps2-bus: no endpoint\n");
|
||||||
return;
|
return;
|
||||||
};
|
};
|
||||||
if (!ipc.register(.ps2_bus, endpoint)) {
|
if (!channel.bindPatiently("ps2-bus", endpoint)) {
|
||||||
_ = logging.write("/system/drivers/ps2-bus: register failed\n");
|
_ = logging.write("/system/drivers/ps2-bus: could not bind /protocol/ps2-bus\n");
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -273,6 +286,13 @@ pub fn main() void {
|
|||||||
var receive: [@sizeOf(ps2.AttachRequest)]u8 = undefined;
|
var receive: [@sizeOf(ps2.AttachRequest)]u8 = undefined;
|
||||||
while (true) {
|
while (true) {
|
||||||
const got = ipc.replyWait(endpoint, reply_buffer[0..reply_len], &receive, null);
|
const got = ipc.replyWait(endpoint, reply_buffer[0..reply_len], &receive, null);
|
||||||
|
// The turn owns whatever capability arrived and closes it unless
|
||||||
|
// `handleAttach` claims it (`ipc.Arrival`) — the kernel installs one
|
||||||
|
// whatever the message's length or kind, so this covers the notification
|
||||||
|
// path and every refusal below it.
|
||||||
|
var arrived: ipc.Arrival = .{ .handle = got.cap };
|
||||||
|
defer arrived.release();
|
||||||
|
|
||||||
if (got.isNotification()) {
|
if (got.isNotification()) {
|
||||||
reply_len = 0;
|
reply_len = 0;
|
||||||
if (got.isMessage() or got.isChildExit()) continue; // nothing sends us these
|
if (got.isMessage() or got.isChildExit()) continue; // nothing sends us these
|
||||||
@@ -301,6 +321,6 @@ pub fn main() void {
|
|||||||
}
|
}
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
reply_len = handleAttach(receive[0..got.len], got, &reply_buffer);
|
reply_len = handleAttach(receive[0..got.len], &reply_buffer, &arrived);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
//! The usb-hid driver as a binary package (docs/build-packages-plan.md):
|
//! The usb-hid driver as a binary package (docs/build-packages-plan.md):
|
||||||
//! this file names the binary and EXACTLY the modules its source imports —
|
//! this file names the binary and EXACTLY the modules its source imports —
|
||||||
//! the shared recipe and the module-to-domain map live in build-support.
|
//! build-support resolves each name from the domains this zon declares.
|
||||||
|
|
||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
const build_support = @import("build-support");
|
const build_support = @import("build-support");
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
//! The usb-storage driver as a binary package (docs/build-packages-plan.md):
|
//! The usb-storage driver as a binary package (docs/build-packages-plan.md):
|
||||||
//! this file names the binary and EXACTLY the modules its source imports —
|
//! this file names the binary and EXACTLY the modules its source imports —
|
||||||
//! the shared recipe and the module-to-domain map live in build-support.
|
//! build-support resolves each name from the domains this zon declares.
|
||||||
|
|
||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
const build_support = @import("build-support");
|
const build_support = @import("build-support");
|
||||||
|
|||||||
@@ -146,17 +146,18 @@ fn initialise(endpoint: ipc.Handle) bool {
|
|||||||
|
|
||||||
/// Serve the block protocol: geometry, and whole-block read/write to/from the
|
/// Serve the block protocol: geometry, and whole-block read/write to/from the
|
||||||
/// caller's DMA buffer (named by physical address).
|
/// caller's DMA buffer (named by physical address).
|
||||||
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Handle) usize {
|
fn onMessage(message: []const u8, reply: []u8, sender: u32, arrived: *ipc.Arrival) usize {
|
||||||
_ = sender;
|
_ = sender;
|
||||||
if (message.len < block_protocol.request_size) return 0;
|
if (message.len < block_protocol.request_size) return 0;
|
||||||
const request = std.mem.bytesToValue(block_protocol.Request, message[0..block_protocol.request_size]);
|
const request = std.mem.bytesToValue(block_protocol.Request, message[0..block_protocol.request_size]);
|
||||||
switch (request.operation) {
|
switch (request.operation) {
|
||||||
@intFromEnum(block_protocol.Operation.attach) => {
|
@intFromEnum(block_protocol.Operation.attach) => {
|
||||||
// The filesystem's DMA buffer: forward its capability to the controller so
|
// The filesystem's DMA buffer: forward its capability to the controller
|
||||||
// the device can reach it, then release our copy (the binding holds a ref).
|
// so the device can reach it. Never claimed — the binding holds its own
|
||||||
const handle = capability orelse return writeReply(reply, .{ .status = -1, .block_size = 0, .block_count = 0 });
|
// reference, so our copy is the turn's to close, on this path and on the
|
||||||
|
// refusal above it alike.
|
||||||
|
const handle = arrived.peek() orelse return writeReply(reply, .{ .status = -1, .block_size = 0, .block_count = 0 });
|
||||||
const ok = device.attachDma(handle);
|
const ok = device.attachDma(handle);
|
||||||
_ = ipc.close(handle);
|
|
||||||
return writeReply(reply, .{ .status = if (ok) 0 else -1, .block_size = 0, .block_count = 0 });
|
return writeReply(reply, .{ .status = if (ok) 0 else -1, .block_size = 0, .block_count = 0 });
|
||||||
},
|
},
|
||||||
@intFromEnum(block_protocol.Operation.geometry) => {
|
@intFromEnum(block_protocol.Operation.geometry) => {
|
||||||
@@ -202,7 +203,7 @@ pub fn main(init: process.Init) void {
|
|||||||
return;
|
return;
|
||||||
};
|
};
|
||||||
service.run(block_protocol.message_maximum, .{
|
service.run(block_protocol.message_maximum, .{
|
||||||
.service = .block,
|
.service = "block",
|
||||||
.init = initialise,
|
.init = initialise,
|
||||||
.on_message = onMessage,
|
.on_message = onMessage,
|
||||||
});
|
});
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
//! The usb-xhci-bus driver as a binary package (docs/build-packages-plan.md):
|
//! The usb-xhci-bus driver as a binary package (docs/build-packages-plan.md):
|
||||||
//! this file names the binary and EXACTLY the modules its source imports —
|
//! this file names the binary and EXACTLY the modules its source imports —
|
||||||
//! the shared recipe and the module-to-domain map live in build-support.
|
//! build-support resolves each name from the domains this zon declares.
|
||||||
|
|
||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
const build_support = @import("build-support");
|
const build_support = @import("build-support");
|
||||||
@@ -10,9 +10,10 @@ pub fn build(b: *std.Build) void {
|
|||||||
.name = "usb-xhci-bus",
|
.name = "usb-xhci-bus",
|
||||||
.root_source_file = b.path("usb-xhci-bus.zig"),
|
.root_source_file = b.path("usb-xhci-bus.zig"),
|
||||||
.imports = &.{
|
.imports = &.{
|
||||||
"device-manager-protocol", "driver", "input-client", "ipc", "logging", "memory",
|
"channel", "device-manager-protocol", "driver", "input-client",
|
||||||
"mmio", "pci", "process", "service", "time", "usb-abi", "usb-ids",
|
"ipc", "logging", "memory", "mmio",
|
||||||
"usb-transfer-protocol",
|
"pci", "process", "service", "time",
|
||||||
|
"usb-abi", "usb-ids", "usb-transfer-protocol",
|
||||||
},
|
},
|
||||||
});
|
});
|
||||||
b.installArtifact(exe);
|
b.installArtifact(exe);
|
||||||
|
|||||||
@@ -15,6 +15,7 @@
|
|||||||
|
|
||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
const device = @import("driver");
|
const device = @import("driver");
|
||||||
|
const channel = @import("channel");
|
||||||
const ipc = @import("ipc");
|
const ipc = @import("ipc");
|
||||||
const process = @import("process");
|
const process = @import("process");
|
||||||
const service = @import("service");
|
const service = @import("service");
|
||||||
@@ -68,19 +69,25 @@ const Open = struct {
|
|||||||
};
|
};
|
||||||
var opens = [_]Open{.{}} ** 16;
|
var opens = [_]Open{.{}} ** 16;
|
||||||
|
|
||||||
fn recordOpen(device_token: u64, report_endpoint: usize) void {
|
/// Remember (or replace) the endpoint that reports for `device_token`. Returns
|
||||||
|
/// whether the table kept the handle — false means the caller still owns it and
|
||||||
|
/// must dispose of it. A re-open supersedes the previous endpoint, and the one
|
||||||
|
/// it displaced is closed here: the table holds exactly one reference per slot.
|
||||||
|
fn recordOpen(device_token: u64, report_endpoint: usize) bool {
|
||||||
for (&opens) |*open| {
|
for (&opens) |*open| {
|
||||||
if (open.used and open.device_token == device_token) {
|
if (open.used and open.device_token == device_token) {
|
||||||
|
if (open.report_endpoint != report_endpoint) _ = ipc.close(open.report_endpoint);
|
||||||
open.report_endpoint = report_endpoint;
|
open.report_endpoint = report_endpoint;
|
||||||
return;
|
return true;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
for (&opens) |*open| {
|
for (&opens) |*open| {
|
||||||
if (!open.used) {
|
if (!open.used) {
|
||||||
open.* = .{ .used = true, .device_token = device_token, .report_endpoint = report_endpoint };
|
open.* = .{ .used = true, .device_token = device_token, .report_endpoint = report_endpoint };
|
||||||
return;
|
return true;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
return false; // table full: not kept
|
||||||
}
|
}
|
||||||
|
|
||||||
fn reportEndpointFor(device_token: u64) ?usize {
|
fn reportEndpointFor(device_token: u64) ?usize {
|
||||||
@@ -97,6 +104,22 @@ var controller_id: u64 = device_manager_protocol.no_device;
|
|||||||
/// manager reads as "meant to stop" — a missing assignment is not a crash loop.
|
/// manager reads as "meant to stop" — a missing assignment is not a crash loop.
|
||||||
fn initialise(endpoint: ipc.Handle) bool {
|
fn initialise(endpoint: ipc.Handle) bool {
|
||||||
service_endpoint = endpoint;
|
service_endpoint = endpoint;
|
||||||
|
|
||||||
|
// The transfer contract, bound by hand rather than through the harness's
|
||||||
|
// `.service`, because **losing it is not fatal here**. One machine can carry
|
||||||
|
// several xHCI controllers and the driver model spawns one process per
|
||||||
|
// controller, so several processes provide the same contract for different
|
||||||
|
// hardware — and `/protocol` holds exactly one name, deliberately (addressing
|
||||||
|
// lives inside the protocol, never in the path). Whoever binds first is the
|
||||||
|
// one clients reach by name; a later instance still owns its controller,
|
||||||
|
// enumerates its bus, and reports its children to the device manager, so it
|
||||||
|
// keeps running. **Known gap:** a class driver behind a second controller
|
||||||
|
// cannot reach it — the transfer protocol has no controller field for
|
||||||
|
// `target`, and the fix is either one process multiplexing every controller
|
||||||
|
// or the spawner wiring the child's channel (P5), not a second name.
|
||||||
|
if (!channel.bindPatiently("usb-transfer", endpoint))
|
||||||
|
_ = logging.write("/system/drivers/usb-xhci-bus: /protocol/usb-transfer is another controller's; serving mine unnamed\n");
|
||||||
|
|
||||||
if (!device.claim(controller_id)) {
|
if (!device.claim(controller_id)) {
|
||||||
std.log.info("unable to claim controller device {d}", .{controller_id});
|
std.log.info("unable to claim controller device {d}", .{controller_id});
|
||||||
return false;
|
return false;
|
||||||
@@ -459,7 +482,7 @@ fn reportInterface(manager: ipc.Handle, port: u32, interface: library.InterfaceI
|
|||||||
return null;
|
return null;
|
||||||
};
|
};
|
||||||
// The devices.csv columns (bus=usb, and the class triple as base/class/prog_if)
|
// The devices.csv columns (bus=usb, and the class triple as base/class/prog_if)
|
||||||
// then the human-readable interface name — a would-be /etc/devices.csv row read
|
// then the human-readable interface name — a would-be /system/configuration/devices.csv row read
|
||||||
// straight off the boot log.
|
// straight off the boot log.
|
||||||
std.log.info("port {d} interface {d} bus=usb base={X:0>2} class={X:0>2} prog_if={X:0>2} — {s} registered as device {d}", .{
|
std.log.info("port {d} interface {d} bus=usb base={X:0>2} class={X:0>2} prog_if={X:0>2} — {s} registered as device {d}", .{
|
||||||
port,
|
port,
|
||||||
@@ -475,28 +498,29 @@ fn reportInterface(manager: ipc.Handle, port: u32, interface: library.InterfaceI
|
|||||||
|
|
||||||
/// Serve the USB transfer protocol: a class driver opens its device, then issues
|
/// Serve the USB transfer protocol: a class driver opens its device, then issues
|
||||||
/// control / interrupt-subscribe / bulk requests against it.
|
/// control / interrupt-subscribe / bulk requests against it.
|
||||||
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Handle) usize {
|
fn onMessage(message: []const u8, reply: []u8, sender: u32, arrived: *ipc.Arrival) usize {
|
||||||
_ = sender;
|
_ = sender;
|
||||||
if (message.len < 4) return 0;
|
if (message.len < 4) return 0;
|
||||||
const operation = std.mem.readInt(u32, message[0..4], .little);
|
const operation = std.mem.readInt(u32, message[0..4], .little);
|
||||||
return switch (operation) {
|
return switch (operation) {
|
||||||
@intFromEnum(usb_transfer_protocol.Operation.open) => handleOpen(message, reply, capability),
|
@intFromEnum(usb_transfer_protocol.Operation.open) => handleOpen(message, reply, arrived),
|
||||||
@intFromEnum(usb_transfer_protocol.Operation.control) => handleControl(message, reply),
|
@intFromEnum(usb_transfer_protocol.Operation.control) => handleControl(message, reply),
|
||||||
@intFromEnum(usb_transfer_protocol.Operation.interrupt_subscribe) => handleSubscribe(message, reply),
|
@intFromEnum(usb_transfer_protocol.Operation.interrupt_subscribe) => handleSubscribe(message, reply),
|
||||||
@intFromEnum(usb_transfer_protocol.Operation.bulk) => handleBulk(message, reply),
|
@intFromEnum(usb_transfer_protocol.Operation.bulk) => handleBulk(message, reply),
|
||||||
@intFromEnum(usb_transfer_protocol.Operation.dma_attach) => handleDmaAttach(message, reply, capability),
|
@intFromEnum(usb_transfer_protocol.Operation.dma_attach) => handleDmaAttach(message, reply, arrived),
|
||||||
else => 0,
|
else => 0,
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
|
|
||||||
/// dma_attach: bind the class driver's DMA-region capability into the controller's IOMMU
|
/// dma_attach: bind the class driver's DMA-region capability into the controller's IOMMU
|
||||||
/// domain, so the controller may DMA to the physical addresses inside that buffer. The
|
/// domain, so the controller may DMA to the physical addresses inside that buffer. The
|
||||||
/// binding holds its own kernel reference, so the forwarded capability is closed here.
|
/// binding holds its own kernel reference, so this never claims the arriving handle —
|
||||||
fn handleDmaAttach(message: []const u8, reply: []u8, capability: ?ipc.Handle) usize {
|
/// the turn's `defer` in the harness is the close, on the failure paths as well as this
|
||||||
|
/// one.
|
||||||
|
fn handleDmaAttach(message: []const u8, reply: []u8, arrived: *ipc.Arrival) usize {
|
||||||
if (message.len < @sizeOf(usb_transfer_protocol.DmaAttachRequest)) return writeReply(reply, usb_transfer_protocol.DmaAttachReply{ .status = -1 });
|
if (message.len < @sizeOf(usb_transfer_protocol.DmaAttachRequest)) return writeReply(reply, usb_transfer_protocol.DmaAttachReply{ .status = -1 });
|
||||||
const handle = capability orelse return writeReply(reply, usb_transfer_protocol.DmaAttachReply{ .status = -1 });
|
const handle = arrived.peek() orelse return writeReply(reply, usb_transfer_protocol.DmaAttachReply{ .status = -1 });
|
||||||
const ok = device.dmaBind(controller_id, handle);
|
const ok = device.dmaBind(controller_id, handle);
|
||||||
_ = ipc.close(handle);
|
|
||||||
return writeReply(reply, usb_transfer_protocol.DmaAttachReply{ .status = if (ok) 0 else -1 });
|
return writeReply(reply, usb_transfer_protocol.DmaAttachReply{ .status = if (ok) 0 else -1 });
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -509,13 +533,17 @@ fn writeReply(reply: []u8, value: anytype) usize {
|
|||||||
/// open: resolve the assigned device id to an interface, remember the caller's
|
/// open: resolve the assigned device id to an interface, remember the caller's
|
||||||
/// endpoint (for interrupt reports), and answer with a device token + the
|
/// endpoint (for interrupt reports), and answer with a device token + the
|
||||||
/// interface's endpoints so the class driver need not re-read the config.
|
/// interface's endpoints so the class driver need not re-read the config.
|
||||||
fn handleOpen(message: []const u8, reply: []u8, capability: ?ipc.Handle) usize {
|
fn handleOpen(message: []const u8, reply: []u8, arrived: *ipc.Arrival) usize {
|
||||||
if (message.len < @sizeOf(usb_transfer_protocol.OpenRequest)) return writeReply(reply, usb_transfer_protocol.OpenReply{ .status = -1, .endpoint_count = 0, .device_token = 0, .interface_class = 0, .interface_subclass = 0, .interface_protocol = 0, .interface_number = 0 });
|
if (message.len < @sizeOf(usb_transfer_protocol.OpenRequest)) return writeReply(reply, usb_transfer_protocol.OpenReply{ .status = -1, .endpoint_count = 0, .device_token = 0, .interface_class = 0, .interface_subclass = 0, .interface_protocol = 0, .interface_number = 0 });
|
||||||
const request = std.mem.bytesToValue(usb_transfer_protocol.OpenRequest, message[0..@sizeOf(usb_transfer_protocol.OpenRequest)]);
|
const request = std.mem.bytesToValue(usb_transfer_protocol.OpenRequest, message[0..@sizeOf(usb_transfer_protocol.OpenRequest)]);
|
||||||
const engine = if (controller) |*c| c else return writeReply(reply, usb_transfer_protocol.OpenReply{ .status = -1, .endpoint_count = 0, .device_token = 0, .interface_class = 0, .interface_subclass = 0, .interface_protocol = 0, .interface_number = 0 });
|
const engine = if (controller) |*c| c else return writeReply(reply, usb_transfer_protocol.OpenReply{ .status = -1, .endpoint_count = 0, .device_token = 0, .interface_class = 0, .interface_subclass = 0, .interface_protocol = 0, .interface_number = 0 });
|
||||||
const found = engine.findInterface(request.device_id) orelse return writeReply(reply, usb_transfer_protocol.OpenReply{ .status = -1, .endpoint_count = 0, .device_token = 0, .interface_class = 0, .interface_subclass = 0, .interface_protocol = 0, .interface_number = 0 });
|
const found = engine.findInterface(request.device_id) orelse return writeReply(reply, usb_transfer_protocol.OpenReply{ .status = -1, .endpoint_count = 0, .device_token = 0, .interface_class = 0, .interface_subclass = 0, .interface_protocol = 0, .interface_number = 0 });
|
||||||
|
|
||||||
if (capability) |endpoint| recordOpen(request.device_id, endpoint);
|
// The report endpoint is claimed only if the open table actually keeps it;
|
||||||
|
// a full table leaves it to the turn to close.
|
||||||
|
if (arrived.peek()) |endpoint| {
|
||||||
|
if (recordOpen(request.device_id, endpoint)) _ = arrived.take();
|
||||||
|
}
|
||||||
|
|
||||||
var open_reply = usb_transfer_protocol.OpenReply{
|
var open_reply = usb_transfer_protocol.OpenReply{
|
||||||
.status = 0,
|
.status = 0,
|
||||||
@@ -673,7 +701,8 @@ pub fn main(init: process.Init) void {
|
|||||||
return;
|
return;
|
||||||
};
|
};
|
||||||
service.run(usb_transfer_protocol.message_maximum, .{
|
service.run(usb_transfer_protocol.message_maximum, .{
|
||||||
.service = .usb_bus,
|
// No `.service`: the contract is bound inside `initialise`, where losing
|
||||||
|
// it to another controller's driver is survivable rather than fatal.
|
||||||
.init = initialise,
|
.init = initialise,
|
||||||
.on_message = onMessage,
|
.on_message = onMessage,
|
||||||
.on_notification = onNotification,
|
.on_notification = onNotification,
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
//! The virtio-gpu driver as a binary package (docs/build-packages-plan.md):
|
//! The virtio-gpu driver as a binary package (docs/build-packages-plan.md):
|
||||||
//! this file names the binary and EXACTLY the modules its source imports —
|
//! this file names the binary and EXACTLY the modules its source imports —
|
||||||
//! the shared recipe and the module-to-domain map live in build-support.
|
//! build-support resolves each name from the domains this zon declares.
|
||||||
|
|
||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
const build_support = @import("build-support");
|
const build_support = @import("build-support");
|
||||||
@@ -10,8 +10,8 @@ pub fn build(b: *std.Build) void {
|
|||||||
.name = "virtio-gpu",
|
.name = "virtio-gpu",
|
||||||
.root_source_file = b.path("virtio-gpu.zig"),
|
.root_source_file = b.path("virtio-gpu.zig"),
|
||||||
.imports = &.{
|
.imports = &.{
|
||||||
"display-protocol", "driver", "ipc", "logging", "memory", "mmio", "pci", "process",
|
"channel", "display-protocol", "driver", "ipc", "logging", "memory", "mmio", "pci",
|
||||||
"scanout-protocol", "service", "time",
|
"process", "scanout-protocol", "service", "time",
|
||||||
},
|
},
|
||||||
});
|
});
|
||||||
b.installArtifact(exe);
|
b.installArtifact(exe);
|
||||||
|
|||||||
@@ -15,6 +15,7 @@
|
|||||||
|
|
||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
const device = @import("driver");
|
const device = @import("driver");
|
||||||
|
const channel = @import("channel");
|
||||||
const ipc = @import("ipc");
|
const ipc = @import("ipc");
|
||||||
const process = @import("process");
|
const process = @import("process");
|
||||||
const service = @import("service");
|
const service = @import("service");
|
||||||
@@ -205,7 +206,7 @@ fn initialise(endpoint: ipc.Handle) bool {
|
|||||||
return false;
|
return false;
|
||||||
};
|
};
|
||||||
|
|
||||||
// Config space is resource 0. The registry (/etc/devices.csv) bound this driver by the
|
// Config space is resource 0. The registry (/system/configuration/devices.csv) bound this driver by the
|
||||||
// exact virtio-gpu identity (vendor 0x1AF4 / device 0x1050), so there is no re-confirm to
|
// exact virtio-gpu identity (vendor 0x1AF4 / device 0x1050), so there is no re-confirm to
|
||||||
// do here any more — map config space and enable memory-space decode + bus mastering (the
|
// do here any more — map config space and enable memory-space decode + bus mastering (the
|
||||||
// device DMAs the ring and backing out of RAM; pci-bus only preserves whatever the firmware
|
// device DMAs the ring and backing out of RAM; pci-bus only preserves whatever the firmware
|
||||||
@@ -475,7 +476,7 @@ fn presentFull() bool {
|
|||||||
fn announce() void {
|
fn announce() void {
|
||||||
var tries: u32 = 0;
|
var tries: u32 = 0;
|
||||||
const display = while (tries < 50) : (tries += 1) {
|
const display = while (tries < 50) : (tries += 1) {
|
||||||
if (ipc.lookup(.display)) |h| break h;
|
if (channel.openEndpoint("display")) |h| break h;
|
||||||
time.sleepMillis(20);
|
time.sleepMillis(20);
|
||||||
} else {
|
} else {
|
||||||
std.log.info("no display service to announce to (scanout-only)", .{});
|
std.log.info("no display service to announce to (scanout-only)", .{});
|
||||||
@@ -507,9 +508,9 @@ fn scanoutStatus(reply: []u8, ok: bool) usize {
|
|||||||
/// The `.scanout` service: the compositor drives present / mode queries here. The pixels are
|
/// The `.scanout` service: the compositor drives present / mode queries here. The pixels are
|
||||||
/// already in the shared surface, so a present is a transfer-to-host + fenced flush; a mode
|
/// already in the shared surface, so a present is a transfer-to-host + fenced flush; a mode
|
||||||
/// change just re-points the scanout rectangle (the surface is sized to the largest mode).
|
/// change just re-points the scanout rectangle (the surface is sized to the largest mode).
|
||||||
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Handle) usize {
|
fn onMessage(message: []const u8, reply: []u8, sender: u32, arrived: *ipc.Arrival) usize {
|
||||||
_ = sender;
|
_ = sender;
|
||||||
_ = capability;
|
_ = arrived; // nothing here takes a capability: the harness closes what arrives
|
||||||
if (message.len < scanout_protocol.request_size) return 0;
|
if (message.len < scanout_protocol.request_size) return 0;
|
||||||
const request = std.mem.bytesToValue(scanout_protocol.Request, message[0..scanout_protocol.request_size]);
|
const request = std.mem.bytesToValue(scanout_protocol.Request, message[0..scanout_protocol.request_size]);
|
||||||
switch (request.operation) {
|
switch (request.operation) {
|
||||||
@@ -547,7 +548,7 @@ pub fn main(init: process.Init) void {
|
|||||||
return;
|
return;
|
||||||
};
|
};
|
||||||
service.run(256, .{
|
service.run(256, .{
|
||||||
.service = .scanout,
|
.service = "scanout",
|
||||||
.init = initialise,
|
.init = initialise,
|
||||||
.on_message = onMessage,
|
.on_message = onMessage,
|
||||||
});
|
});
|
||||||
|
|||||||
@@ -257,6 +257,16 @@ pub fn translate(root: u64, virtual: u64) ?u64 {
|
|||||||
return paging.translateIn(root, virtual);
|
return paging.translateIn(root, virtual);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// `translate` for an address the kernel is about to touch *on a process's
|
||||||
|
/// behalf*: the walk additionally demands the permission ring 3 would need — the
|
||||||
|
/// leaf user-accessible (U/S set at every level), and writable (R/W at every
|
||||||
|
/// level) when `for_write`. Null means "the process itself could not do this",
|
||||||
|
/// which the checked copy layer (system/kernel/user-memory.zig) turns into
|
||||||
|
/// -EFAULT instead of a kernel dereference.
|
||||||
|
pub fn translateUser(root: u64, virtual: u64, for_write: bool) ?u64 {
|
||||||
|
return paging.translateUserIn(root, virtual, for_write);
|
||||||
|
}
|
||||||
|
|
||||||
/// Map a page accessible from ring 3 (U/S bit at every level). The caller keeps
|
/// Map a page accessible from ring 3 (U/S bit at every level). The caller keeps
|
||||||
/// W^X: code read-only + executable, data writable + no-execute.
|
/// W^X: code read-only + executable, data writable + no-execute.
|
||||||
pub fn mapUserPage(virtual: u64, physical: u64, writable: bool, executable: bool) void {
|
pub fn mapUserPage(virtual: u64, physical: u64, writable: bool, executable: bool) void {
|
||||||
|
|||||||
@@ -524,6 +524,58 @@ pub fn translateIn(pml4: u64, virtual: u64) ?u64 {
|
|||||||
return (pte & address_mask) | (virtual & (page_size - 1));
|
return (pte & address_mask) | (virtual & (page_size - 1));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// `translateIn` with the ring-3 permission bits enforced: the walk accumulates
|
||||||
|
/// the protection flags of every level it descends through and refuses the
|
||||||
|
/// translation unless the *effective* permission allows the access ring 3 would
|
||||||
|
/// be allowed — U/S set at every level, and (for `for_write`) R/W set at every
|
||||||
|
/// level too. A bit cleared anywhere on the path denies, which is exactly how
|
||||||
|
/// the MMU combines them, so a checked kernel copy sees the same permissions the
|
||||||
|
/// process itself does.
|
||||||
|
///
|
||||||
|
/// This is the walk `system/kernel/user-memory.zig` copies through, and the
|
||||||
|
/// reason a kernel copy can never be steered at a kernel-only mapping or made to
|
||||||
|
/// write a read-only user page (a process's own text, say).
|
||||||
|
///
|
||||||
|
/// Huge pages: a 2 MiB PDE leaf resolves like `translateIn`, with its own U/S and
|
||||||
|
/// R/W folded into the accumulator first. A PDPTE with PS set (a 1 GiB leaf) is
|
||||||
|
/// refused rather than descended into — danos never builds one, and denying is
|
||||||
|
/// the safe direction for a permission-checked walk.
|
||||||
|
pub fn translateUserIn(pml4: u64, virtual: u64, for_write: bool) ?u64 {
|
||||||
|
// Start all-ones and AND in each level: a cleared bit at any level denies.
|
||||||
|
var effective: u64 = ~@as(u64, 0);
|
||||||
|
|
||||||
|
const pml4e = tableAt(pml4)[(virtual >> 39) & 0x1FF];
|
||||||
|
if (pml4e & present == 0) return null;
|
||||||
|
effective &= pml4e;
|
||||||
|
|
||||||
|
const pdpte = tableAt(pml4e & address_mask)[(virtual >> 30) & 0x1FF];
|
||||||
|
if (pdpte & present == 0) return null;
|
||||||
|
if (pdpte & page_size_bit != 0) return null; // 1 GiB leaf: never built here, refuse
|
||||||
|
effective &= pdpte;
|
||||||
|
|
||||||
|
const pde = tableAt(pdpte & address_mask)[(virtual >> 21) & 0x1FF];
|
||||||
|
if (pde & present == 0) return null;
|
||||||
|
effective &= pde;
|
||||||
|
if (pde & page_size_bit != 0) { // 2 MiB huge leaf: frame base is bits 51:21
|
||||||
|
if (!permits(effective, for_write)) return null;
|
||||||
|
return (pde & address_mask & ~@as(u64, huge_page_size - 1)) | (virtual & (huge_page_size - 1));
|
||||||
|
}
|
||||||
|
|
||||||
|
const pte = tableAt(pde & address_mask)[(virtual >> 12) & 0x1FF];
|
||||||
|
if (pte & present == 0) return null;
|
||||||
|
effective &= pte;
|
||||||
|
if (!permits(effective, for_write)) return null;
|
||||||
|
return (pte & address_mask) | (virtual & (page_size - 1));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether accumulated walk flags allow a ring-3 access: user-accessible always,
|
||||||
|
/// and writable when the access is a store.
|
||||||
|
fn permits(effective: u64, for_write: bool) bool {
|
||||||
|
if (effective & user == 0) return false;
|
||||||
|
if (for_write and effective & writable == 0) return false;
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
fn invalidate(virtual: u64) void {
|
fn invalidate(virtual: u64) void {
|
||||||
// invlpg needs its operand via a register-indirect memory reference that Zig
|
// invlpg needs its operand via a register-indirect memory reference that Zig
|
||||||
// inline asm won't form directly, so stage the address in a register first.
|
// inline asm won't form directly, so stage the address in a register first.
|
||||||
|
|||||||
@@ -132,13 +132,31 @@ fn record(node: *platform.Device, parent_id: u64) u64 {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Copy up to `out.len` device descriptors into `out`; returns the total count
|
/// Copy up to `out.len` device descriptors into `out`; returns the total count
|
||||||
/// available (which may exceed `out.len`).
|
/// available (which may exceed `out.len`). For kernel callers with a buffer big
|
||||||
|
/// enough to take the whole table in one go.
|
||||||
pub fn enumerate(out: []device_abi.DeviceDescriptor) usize {
|
pub fn enumerate(out: []device_abi.DeviceDescriptor) usize {
|
||||||
const n = @min(count, out.len);
|
_ = enumerateFrom(0, out);
|
||||||
@memcpy(out[0..n], devices[0..n]);
|
|
||||||
return count;
|
return count;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// How many devices the table holds — the total `device_enumerate` reports back
|
||||||
|
/// however few of them fit in the caller's buffer.
|
||||||
|
pub fn deviceCount() usize {
|
||||||
|
return count;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Copy up to `out.len` descriptors starting at table index `start`, returning how
|
||||||
|
/// many were filled (0 once `start` reaches the end). The chunked form: the
|
||||||
|
/// `device_enumerate` system call bounces the table out through a small kernel
|
||||||
|
/// buffer, one chunk at a time, because a descriptor is far too big to stage a
|
||||||
|
/// whole user-requested array of them on a 16 KiB kernel stack.
|
||||||
|
pub fn enumerateFrom(start: usize, out: []device_abi.DeviceDescriptor) usize {
|
||||||
|
if (start >= count) return 0;
|
||||||
|
const n = @min(count - start, out.len);
|
||||||
|
@memcpy(out[0..n], devices[start..][0..n]);
|
||||||
|
return n;
|
||||||
|
}
|
||||||
|
|
||||||
/// Take exclusive ownership of device `id` for task `owner`. Fails if the id is
|
/// Take exclusive ownership of device `id` for task `owner`. Fails if the id is
|
||||||
/// out of range or already claimed.
|
/// out of range or already claimed.
|
||||||
pub fn claim(id: u64, owner: u32) bool {
|
pub fn claim(id: u64, owner: u32) bool {
|
||||||
|
|||||||
@@ -17,18 +17,20 @@
|
|||||||
//! endpoint's own FIFO (threaded through the otherwise-idle `Task.next`); servers
|
//! endpoint's own FIFO (threaded through the otherwise-idle `Task.next`); servers
|
||||||
//! waiting for work use a normal WaitQueue.
|
//! waiting for work use a normal WaitQueue.
|
||||||
//!
|
//!
|
||||||
//! Trust model (bring-up): copies honour only page presence and a user-half bound,
|
//! Trust model: every side of a copy that names a *user* address space goes
|
||||||
//! not the leaf U/S or R/W bits and not SMAP — a #PF-tolerant, permission-checked
|
//! through system/kernel/user-memory.zig — user-half bound, page presence, and
|
||||||
//! copy is a later security-track item, matching the existing debug_write gap.
|
//! the leaf permissions ring 3 itself would face (U/S to read, U/S + R/W to
|
||||||
|
//! write). A kernel-side buffer is trusted and translated as-is. An unmapped or
|
||||||
|
//! wrongly-permissioned page fails the operation; it never faults ring 0.
|
||||||
|
|
||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
const boot_handoff = @import("boot-handoff");
|
const boot_handoff = @import("boot-handoff");
|
||||||
const abi = @import("abi");
|
const abi = @import("abi");
|
||||||
const architecture = @import("architecture");
|
|
||||||
const scheduler = @import("scheduler.zig");
|
const scheduler = @import("scheduler.zig");
|
||||||
const sync = @import("sync.zig");
|
const sync = @import("sync.zig");
|
||||||
const heap = @import("heap.zig");
|
const heap = @import("heap.zig");
|
||||||
const pmm = @import("pmm.zig");
|
const pmm = @import("pmm.zig");
|
||||||
|
const user_memory = @import("user-memory.zig");
|
||||||
|
|
||||||
const page_size = abi.page_size;
|
const page_size = abi.page_size;
|
||||||
const Task = scheduler.Task;
|
const Task = scheduler.Task;
|
||||||
@@ -38,16 +40,13 @@ const Task = scheduler.Task;
|
|||||||
pub const MESSAGE_MAXIMUM: usize = 256;
|
pub const MESSAGE_MAXIMUM: usize = 256;
|
||||||
|
|
||||||
pub const maximum_handles = scheduler.ipc_maximum_handles;
|
pub const maximum_handles = scheduler.ipc_maximum_handles;
|
||||||
// The name registry is indexed directly by ServiceId, so this must exceed the
|
|
||||||
// largest id (currently fat = 8). Sized with headroom for new services.
|
|
||||||
pub const maximum_services = 16;
|
|
||||||
|
|
||||||
/// Errno-style failures, returned as `-value` in the system_call result register.
|
/// Errno-style failures, returned as `-value` in the system_call result register.
|
||||||
pub const EBADF: i64 = 1; // bad handle
|
pub const EBADF: i64 = 1; // bad handle
|
||||||
pub const E2BIG: i64 = 2; // message exceeds MESSAGE_MAXIMUM
|
pub const E2BIG: i64 = 2; // message exceeds MESSAGE_MAXIMUM
|
||||||
pub const EFAULT: i64 = 3; // buffer unmapped / out of the user half
|
pub const EFAULT: i64 = 3; // buffer unmapped / out of the user half
|
||||||
pub const ENOENT: i64 = 4; // no such registered service
|
pub const ENOENT: i64 = 4; // no such name
|
||||||
pub const ENOSPC: i64 = 5; // handle table or registry full
|
pub const ENOSPC: i64 = 5; // handle table full
|
||||||
pub const ENOMEM: i64 = 6; // out of memory
|
pub const ENOMEM: i64 = 6; // out of memory
|
||||||
pub const EPEER: i64 = 7; // peer died before replying (its process exited or was killed)
|
pub const EPEER: i64 = 7; // peer died before replying (its process exited or was killed)
|
||||||
pub const ESRCH: i64 = 8; // no such process (process_kill of an unknown/dead id)
|
pub const ESRCH: i64 = 8; // no such process (process_kill of an unknown/dead id)
|
||||||
@@ -83,16 +82,26 @@ const PostSlot = struct {
|
|||||||
bytes: [POST_MAXIMUM]u8 = undefined,
|
bytes: [POST_MAXIMUM]u8 = undefined,
|
||||||
};
|
};
|
||||||
|
|
||||||
/// End of the user (low) canonical half — user buffers must lie below it.
|
/// End of the user (low) canonical half — user buffers must lie below it. One
|
||||||
const user_half_end: u64 = 0x0000_8000_0000_0000;
|
/// definition, in the module that owns the user-memory contract.
|
||||||
|
const user_half_end: u64 = user_memory.user_half_end;
|
||||||
|
|
||||||
/// A rendezvous endpoint. Allocated from the kernel heap; referenced by handle
|
/// A rendezvous endpoint. Allocated from the kernel heap; referenced by handle
|
||||||
/// (per process) and/or by a registry slot, counted by `refcount`.
|
/// (per process) and by whoever a capability was passed to, counted by `refcount`.
|
||||||
pub const Endpoint = struct {
|
pub const Endpoint = struct {
|
||||||
refcount: u32 = 1,
|
refcount: u32 = 1,
|
||||||
|
/// Next in the list of every live endpoint. Endpoints are otherwise reachable
|
||||||
|
/// only through the handle tables that name them, and the death path has to
|
||||||
|
/// find a dying task's endpoints without one — see `live_endpoints`.
|
||||||
|
next_live: ?*Endpoint = null,
|
||||||
// The task that created it. When that task dies, the endpoint is marked `dead` so a caller
|
// The task that created it. When that task dies, the endpoint is marked `dead` so a caller
|
||||||
// gets -EPEER instead of blocking forever on a service that will never reply again (V6).
|
// gets -EPEER instead of blocking forever on a service that will never reply again (V6).
|
||||||
owner: u32 = 0,
|
owner: u32 = 0,
|
||||||
|
// The *process* that created it — `owner`'s leader, snapshotted at creation so the
|
||||||
|
// answer survives the creating thread. `owner` alone cannot answer "is this mine?"
|
||||||
|
// for a threaded service, and the question has to be answerable after that thread is
|
||||||
|
// gone; see `ownedBy`.
|
||||||
|
owner_leader: u32 = 0,
|
||||||
dead: bool = false,
|
dead: bool = false,
|
||||||
// Callers blocked in `call`, awaiting receive, in FIFO order (threaded via
|
// Callers blocked in `call`, awaiting receive, in FIFO order (threaded via
|
||||||
// Task.next; each such task is .blocked and in no scheduler queue).
|
// Task.next; each such task is .blocked and in no scheduler queue).
|
||||||
@@ -112,29 +121,78 @@ pub const Endpoint = struct {
|
|||||||
post_tail: u16 = 0,
|
post_tail: u16 = 0,
|
||||||
};
|
};
|
||||||
|
|
||||||
|
/// Every live endpoint, singly linked through `next_live`. The list exists for
|
||||||
|
/// exactly one purpose: the death path must mark a dying task's endpoints dead,
|
||||||
|
/// and a handle table only answers the other question (which endpoints does this
|
||||||
|
/// task *hold*). Mutated under the big kernel lock, like every other IPC global.
|
||||||
|
var live_endpoints: ?*Endpoint = null;
|
||||||
|
|
||||||
pub fn createIpcEndpoint() ?*Endpoint {
|
pub fn createIpcEndpoint() ?*Endpoint {
|
||||||
|
const creator = scheduler.current();
|
||||||
const endpoint = heap.allocator().create(Endpoint) catch return null;
|
const endpoint = heap.allocator().create(Endpoint) catch return null;
|
||||||
endpoint.* = .{ .owner = scheduler.currentId() };
|
endpoint.* = .{ .owner = creator.id, .owner_leader = creator.leader, .next_live = live_endpoints };
|
||||||
|
live_endpoints = endpoint;
|
||||||
return endpoint;
|
return endpoint;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// A task is dying: kill the endpoints it registered as services. Mark each `dead` (so a later
|
/// Whether `t` may have the kernel post **notifications** — signals, timer
|
||||||
/// `call` returns -EPEER rather than blocking on a reply that will never come), wake anyone
|
/// landings, exit notices, interrupts — into `endpoint`: whether the endpoint is
|
||||||
/// already parked sending to it with that error, and vacate its registry slot. Only *registered*
|
/// its process's own.
|
||||||
/// endpoints are reachable from here; unregistered ones drop with the task's handle table. The
|
///
|
||||||
/// caller holds the big kernel lock (this runs on the death path). See docs/display-v2.md (V6).
|
/// Holding a *handle* to an endpoint is not ownership of it. `fs_resolve`
|
||||||
|
/// installs a mounted backend's capability in any caller's table
|
||||||
|
/// (`installHandleDeduped`), and any capability may be passed along a call, so a
|
||||||
|
/// sendable handle means only "you may talk to this". A kernel notification is
|
||||||
|
/// different in kind: it makes the kernel speak *into* someone else's mailbox
|
||||||
|
/// with a badge that receiver cannot distinguish from one it asked for — a
|
||||||
|
/// genuine signal badge, a genuine timer landing. That is how a forged
|
||||||
|
/// `terminate` reached PID 1's shutdown path: the attacker aimed **its own**
|
||||||
|
/// signal delivery at init's endpoint with `signal_bind` and then signalled
|
||||||
|
/// itself, and every bit the kernel stamped was authentic. Refusing the *bind*
|
||||||
|
/// is the only place the distinction still exists.
|
||||||
|
///
|
||||||
|
/// Threads: ownership is the **process's**, not the task's, so any thread may
|
||||||
|
/// bind an endpoint a sibling created — the same normalization `process_signal`
|
||||||
|
/// and `process_kill` perform when they resolve a member to its leader. The
|
||||||
|
/// creating task's own id is honoured too, which is what keeps kernel tasks
|
||||||
|
/// (leader 0) from being treated as one process.
|
||||||
|
pub fn ownedBy(endpoint: *const Endpoint, t: *const Task) bool {
|
||||||
|
if (endpoint.owner == t.id) return true;
|
||||||
|
return t.leader != 0 and endpoint.owner_leader == t.leader;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Unlink a freed endpoint from the live list. O(n) in the number of live
|
||||||
|
/// endpoints, which is tens.
|
||||||
|
fn forgetEndpoint(endpoint: *Endpoint) void {
|
||||||
|
var link = &live_endpoints;
|
||||||
|
while (link.*) |current| {
|
||||||
|
if (current == endpoint) {
|
||||||
|
link.* = current.next_live;
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
link = ¤t.next_live;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A task is dying: kill every endpoint it created. Mark each `dead` (so a later
|
||||||
|
/// `call` returns -EPEER rather than blocking on a reply that will never come) and wake
|
||||||
|
/// anyone already parked sending to it with that error. This is what makes a provider's
|
||||||
|
/// death visible to the clients holding its capability — the naming layer's restart
|
||||||
|
/// story (a client re-resolves on -EPEER) rests on it, as does the VFS router's lazy
|
||||||
|
/// unmount of a backend that died. The endpoint object itself lives until the last
|
||||||
|
/// handle naming it drops. The caller holds the big kernel lock (this runs on the death
|
||||||
|
/// path). See docs/display-v2.md (V6).
|
||||||
pub fn killOwnedEndpointsLocked(task_id: u32) void {
|
pub fn killOwnedEndpointsLocked(task_id: u32) void {
|
||||||
for (®istry) |*slot| {
|
var current = live_endpoints;
|
||||||
const endpoint = slot.* orelse continue;
|
while (current) |endpoint| {
|
||||||
if (endpoint.owner != task_id) continue;
|
current = endpoint.next_live;
|
||||||
|
if (endpoint.owner != task_id or endpoint.dead) continue;
|
||||||
endpoint.dead = true;
|
endpoint.dead = true;
|
||||||
while (dequeueSender(endpoint)) |sender| {
|
while (dequeueSender(endpoint)) |sender| {
|
||||||
sender.ipc_status = -EPEER;
|
sender.ipc_status = -EPEER;
|
||||||
sender.ipc_received_cap = abi.no_cap;
|
sender.ipc_received_cap = abi.no_cap;
|
||||||
scheduler.readyLocked(sender);
|
scheduler.readyLocked(sender);
|
||||||
}
|
}
|
||||||
slot.* = null;
|
|
||||||
dropRef(endpoint);
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -144,6 +202,7 @@ pub fn dropRef(endpoint: *Endpoint) void {
|
|||||||
if (endpoint.refcount > 1) {
|
if (endpoint.refcount > 1) {
|
||||||
endpoint.refcount -= 1;
|
endpoint.refcount -= 1;
|
||||||
} else {
|
} else {
|
||||||
|
forgetEndpoint(endpoint);
|
||||||
heap.allocator().destroy(endpoint);
|
heap.allocator().destroy(endpoint);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -300,18 +359,22 @@ pub fn abandonSenderLocked(t: *Task) void {
|
|||||||
/// Copy `len` bytes from `source_va` in address space `source_as` to `destination_va` in
|
/// Copy `len` bytes from `source_va` in address space `source_as` to `destination_va` in
|
||||||
/// `destination_as`, walking each side's page tables through the physmap (no CR3 switch).
|
/// `destination_as`, walking each side's page tables through the physmap (no CR3 switch).
|
||||||
/// `*_as == 0` means the kernel address space (for kernel-task endpoints). User
|
/// `*_as == 0` means the kernel address space (for kernel-task endpoints). User
|
||||||
/// buffers must lie in the low half. Returns false — never #PFs — if any page is
|
/// buffers must lie in the low half and carry the permission ring 3 would need for
|
||||||
/// unmapped or out of range. Handles page-straddling buffers.
|
/// their side of the copy — readable to send from, writable to receive into.
|
||||||
|
/// Returns false — never #PFs — if any page is unmapped, out of range, or
|
||||||
|
/// wrongly permissioned. Handles page-straddling buffers.
|
||||||
|
///
|
||||||
|
/// This is the process↔process case, which `user-memory` deliberately does not
|
||||||
|
/// cover (it knows one user address space at a time); both sides resolve through
|
||||||
|
/// `user_memory.resolve`, so the permission rules are the same ones.
|
||||||
fn copyAcross(source_as: u64, source_va: u64, destination_as: u64, destination_va: u64, len: usize) bool {
|
fn copyAcross(source_as: u64, source_va: u64, destination_as: u64, destination_va: u64, len: usize) bool {
|
||||||
const source_root = if (source_as != 0) source_as else architecture.kernelPageTable();
|
if (source_as != 0 and !user_memory.userRangeOk(source_va, len)) return false;
|
||||||
const destination_root = if (destination_as != 0) destination_as else architecture.kernelPageTable();
|
if (destination_as != 0 and !user_memory.userRangeOk(destination_va, len)) return false;
|
||||||
if (source_as != 0 and (source_va >= user_half_end or source_va + len > user_half_end)) return false;
|
|
||||||
if (destination_as != 0 and (destination_va >= user_half_end or destination_va + len > user_half_end)) return false;
|
|
||||||
|
|
||||||
var off: usize = 0;
|
var off: usize = 0;
|
||||||
while (off < len) {
|
while (off < len) {
|
||||||
const s = architecture.translate(source_root, source_va + off) orelse return false;
|
const s = user_memory.resolve(source_as, source_va + off, false) orelse return false;
|
||||||
const d = architecture.translate(destination_root, destination_va + off) orelse return false;
|
const d = user_memory.resolve(destination_as, destination_va + off, true) orelse return false;
|
||||||
const s_left = page_size - ((source_va + off) & (page_size - 1));
|
const s_left = page_size - ((source_va + off) & (page_size - 1));
|
||||||
const d_left = page_size - ((destination_va + off) & (page_size - 1));
|
const d_left = page_size - ((destination_va + off) & (page_size - 1));
|
||||||
const n = @min(@min(s_left, d_left), len - off);
|
const n = @min(@min(s_left, d_left), len - off);
|
||||||
@@ -323,27 +386,10 @@ fn copyAcross(source_as: u64, source_va: u64, destination_as: u64, destination_v
|
|||||||
return true;
|
return true;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Copy `destination.len` bytes from `user_va` in address space `user_as` into the kernel
|
/// The checked copy-in, re-exported from its home in `user-memory` so the many
|
||||||
/// buffer `destination`, walking the user page tables through the physmap. Returns false if
|
/// `ipc.copyFromUser` call sites keep reading naturally. New code should reach
|
||||||
/// the range escapes the user half or any source page is unmapped — so a bad user
|
/// for `user-memory` directly — it is where the write direction lives too.
|
||||||
/// pointer *fails the system_call* rather than faulting the kernel (danos has no
|
pub const copyFromUser = user_memory.copyFromUser;
|
||||||
/// fault-recovering copy-in, so a raw dereference of an unmapped user page would halt
|
|
||||||
/// the machine). The correct way to pull a fixed-size struct in from user space, and
|
|
||||||
/// a single fetch: no TOCTOU against a hostile pointer.
|
|
||||||
pub fn copyFromUser(user_as: u64, user_va: u64, destination: []u8) bool {
|
|
||||||
if (user_as == 0) return false; // not a user address space
|
|
||||||
if (user_va >= user_half_end or user_va + destination.len > user_half_end) return false;
|
|
||||||
var off: usize = 0;
|
|
||||||
while (off < destination.len) {
|
|
||||||
const s = architecture.translate(user_as, user_va + off) orelse return false;
|
|
||||||
const s_left = page_size - ((user_va + off) & (page_size - 1));
|
|
||||||
const n = @min(s_left, destination.len - off);
|
|
||||||
const source: [*]const u8 = @ptrFromInt(boot_handoff.physicalToVirtual(s));
|
|
||||||
@memcpy(destination[off..][0..n], source[0..n]);
|
|
||||||
off += n;
|
|
||||||
}
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- the two IPC operations -------------------------------------------------
|
// --- the two IPC operations -------------------------------------------------
|
||||||
|
|
||||||
@@ -643,22 +689,9 @@ fn dropEntry(entry: scheduler.HandleObject) void {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
var registry: [maximum_services]?*Endpoint = .{null} ** maximum_services;
|
// The flat `ServiceId` registry lived here — a 16-slot table any process could
|
||||||
|
// write, indexed by a compile-time enum. Naming is user-space's job now: init
|
||||||
/// Publish `endpoint` under well-known `id` (takes a reference). Returns 0 or -errno.
|
// serves `/protocol` and decides who may claim a name
|
||||||
pub fn register(id: u32, endpoint: *Endpoint) i64 {
|
// (docs/os-development/protocol-namespace.md). The kernel keeps only what is
|
||||||
if (id >= maximum_services) return -ENOENT;
|
// genuinely kernel work — moving capabilities and telling clients their provider
|
||||||
if (registry[id]) |old| dropRef(old);
|
// died (`killOwnedEndpointsLocked`).
|
||||||
endpoint.refcount += 1;
|
|
||||||
registry[id] = endpoint;
|
|
||||||
return 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Find the endpoint published under `id`, taking a reference for the caller to
|
|
||||||
/// install in its handle table. Null if nothing is registered there.
|
|
||||||
pub fn lookup(id: u32) ?*Endpoint {
|
|
||||||
if (id >= maximum_services) return null;
|
|
||||||
const endpoint = registry[id] orelse return null;
|
|
||||||
endpoint.refcount += 1;
|
|
||||||
return endpoint;
|
|
||||||
}
|
|
||||||
|
|||||||
@@ -47,8 +47,8 @@ pub const maximum_gsi = 24;
|
|||||||
var bound: [maximum_gsi]?*ipc_sync.Endpoint = .{null} ** maximum_gsi;
|
var bound: [maximum_gsi]?*ipc_sync.Endpoint = .{null} ** maximum_gsi;
|
||||||
|
|
||||||
/// Task that owns each binding. Teardown is keyed on *this*, not on the endpoint
|
/// Task that owns each binding. Teardown is keyed on *this*, not on the endpoint
|
||||||
/// pointer: an endpoint can be shared between processes (ipc_register/ipc_lookup hand
|
/// pointer: an endpoint can be shared between processes (a capability passed in a message
|
||||||
/// out extra references), so "every GSI pointing at this endpoint" is not the same
|
/// hands out extra references), so "every GSI pointing at this endpoint" is not the same
|
||||||
/// set as "every GSI this process bound", and releasing the former on exit would mask
|
/// set as "every GSI this process bound", and releasing the former on exit would mask
|
||||||
/// a live sibling's device line.
|
/// a live sibling's device line.
|
||||||
var bound_owner: [maximum_gsi]u32 = .{0} ** maximum_gsi;
|
var bound_owner: [maximum_gsi]u32 = .{0} ** maximum_gsi;
|
||||||
|
|||||||
@@ -161,7 +161,7 @@ test "append/read round trip" {
|
|||||||
defer std.testing.allocator.destroy(ring);
|
defer std.testing.allocator.destroy(ring);
|
||||||
ring.* = .{};
|
ring.* = .{};
|
||||||
|
|
||||||
_ = ring.append(7, "/system/services/fat", .info, 123, "mounted /mnt/usb", false);
|
_ = ring.append(7, "/system/services/fat", .info, 123, "mounted /volumes/usb", false);
|
||||||
_ = ring.append(0, "kernel", .raw, 456, "wall clock online", false);
|
_ = ring.append(0, "kernel", .raw, 456, "wall clock online", false);
|
||||||
|
|
||||||
const first = parseAt(ring, ring.tail);
|
const first = parseAt(ring, ring.tail);
|
||||||
@@ -169,7 +169,7 @@ test "append/read round trip" {
|
|||||||
try std.testing.expectEqual(abi.KlogLevel.info, first.header.level);
|
try std.testing.expectEqual(abi.KlogLevel.info, first.header.level);
|
||||||
try std.testing.expectEqual(@as(u64, 123), first.header.timestamp_ns);
|
try std.testing.expectEqual(@as(u64, 123), first.header.timestamp_ns);
|
||||||
try std.testing.expectEqualStrings("/system/services/fat", first.nameSlice());
|
try std.testing.expectEqualStrings("/system/services/fat", first.nameSlice());
|
||||||
try std.testing.expectEqualStrings("mounted /mnt/usb", first.messageSlice());
|
try std.testing.expectEqualStrings("mounted /volumes/usb", first.messageSlice());
|
||||||
|
|
||||||
const second = parseAt(ring, first.next(ring.tail));
|
const second = parseAt(ring, first.next(ring.tail));
|
||||||
try std.testing.expectEqual(@as(u32, 0), second.header.pid);
|
try std.testing.expectEqual(@as(u32, 0), second.header.pid);
|
||||||
|
|||||||
+203
-83
@@ -31,6 +31,7 @@ const scheduler = @import("scheduler.zig");
|
|||||||
const console = @import("console.zig");
|
const console = @import("console.zig");
|
||||||
const sync = @import("sync.zig");
|
const sync = @import("sync.zig");
|
||||||
const ipc = @import("ipc-synchronous.zig");
|
const ipc = @import("ipc-synchronous.zig");
|
||||||
|
const user_memory = @import("user-memory.zig");
|
||||||
const devices_broker = @import("devices-broker.zig");
|
const devices_broker = @import("devices-broker.zig");
|
||||||
const irq = @import("irq.zig");
|
const irq = @import("irq.zig");
|
||||||
const iommu = @import("iommu.zig");
|
const iommu = @import("iommu.zig");
|
||||||
@@ -110,6 +111,14 @@ pub const maximum_arguments = 8;
|
|||||||
/// Ceiling on the `system_spawn` extra-arguments blob (argv[1..], NUL-separated).
|
/// Ceiling on the `system_spawn` extra-arguments blob (argv[1..], NUL-separated).
|
||||||
pub const maximum_argument_bytes = 256;
|
pub const maximum_argument_bytes = 256;
|
||||||
|
|
||||||
|
/// Longest path `fs_resolve` accepts, longest prefix `fs_mount`/`fs_unmount`
|
||||||
|
/// accept, and longest backend rewrite prefix. Each is also the size of the
|
||||||
|
/// kernel staging buffer the argument is copied into, which is why they are
|
||||||
|
/// named here rather than spelled as literals at the check.
|
||||||
|
pub const maximum_resolve_path = 224;
|
||||||
|
pub const maximum_mount_prefix = 64;
|
||||||
|
pub const maximum_mount_rewrite = 32;
|
||||||
|
|
||||||
/// Auxiliary-vector entry types (System V AMD64 process entry). Only what the
|
/// Auxiliary-vector entry types (System V AMD64 process entry). Only what the
|
||||||
/// kernel emits today; a C runtime scans the vector until the null terminator.
|
/// kernel emits today; a C runtime scans the vector until the null terminator.
|
||||||
const auxiliary_vector_null: u64 = 0; // AT_NULL — end of the vector
|
const auxiliary_vector_null: u64 = 0; // AT_NULL — end of the vector
|
||||||
@@ -220,8 +229,6 @@ fn system_call(state: *architecture.CpuState) void {
|
|||||||
.mmap => systemMmap(state),
|
.mmap => systemMmap(state),
|
||||||
.munmap => systemMunmap(state),
|
.munmap => systemMunmap(state),
|
||||||
.create_ipc_endpoint => systemCreateIpcEndpoint(state),
|
.create_ipc_endpoint => systemCreateIpcEndpoint(state),
|
||||||
.ipc_register => systemIpcRegister(state),
|
|
||||||
.ipc_lookup => systemIpcLookup(state),
|
|
||||||
.ipc_call => systemIpcCall(state),
|
.ipc_call => systemIpcCall(state),
|
||||||
.ipc_reply_wait => systemIpcReplyWait(state),
|
.ipc_reply_wait => systemIpcReplyWait(state),
|
||||||
.ipc_send => systemIpcSend(state),
|
.ipc_send => systemIpcSend(state),
|
||||||
@@ -311,36 +318,6 @@ fn systemCreateIpcEndpoint(state: *architecture.CpuState) void {
|
|||||||
architecture.setSystemCallResult(state, @intCast(h));
|
architecture.setSystemCallResult(state, @intCast(h));
|
||||||
}
|
}
|
||||||
|
|
||||||
/// ipc_register(service_id, handle): publish the caller's endpoint under a
|
|
||||||
/// well-known id so other processes can find it.
|
|
||||||
fn systemIpcRegister(state: *architecture.CpuState) void {
|
|
||||||
// Under the big kernel lock: mutates the global service registry and endpoint
|
|
||||||
// refcounts, which threads of the same (or another) process can race.
|
|
||||||
const flags = sync.enter();
|
|
||||||
defer sync.leave(flags);
|
|
||||||
const id: u32 = @truncate(architecture.systemCallArg(state, 0));
|
|
||||||
const endpoint = ipc.resolveHandle(scheduler.current(), architecture.systemCallArg(state, 1)) orelse return failErr(state, ipc.EBADF);
|
|
||||||
architecture.setSystemCallResult(state, @bitCast(ipc.register(id, endpoint)));
|
|
||||||
}
|
|
||||||
|
|
||||||
/// ipc_lookup(service_id) -> handle: find a published endpoint and install a
|
|
||||||
/// handle to it in the caller.
|
|
||||||
fn systemIpcLookup(state: *architecture.CpuState) void {
|
|
||||||
// Under the big kernel lock: reads the global registry, takes an endpoint reference,
|
|
||||||
// and installs a handle — all racy against concurrent threads (this is the path the
|
|
||||||
// display's mouse-listener thread takes to reach the compositor endpoint).
|
|
||||||
const flags = sync.enter();
|
|
||||||
defer sync.leave(flags);
|
|
||||||
const id: u32 = @truncate(architecture.systemCallArg(state, 0));
|
|
||||||
const endpoint = ipc.lookup(id) orelse return failErr(state, ipc.ENOENT);
|
|
||||||
const h = ipc.installHandle(scheduler.current(), endpoint);
|
|
||||||
if (h < 0) {
|
|
||||||
ipc.dropRef(endpoint);
|
|
||||||
return failErr(state, ipc.ENOSPC);
|
|
||||||
}
|
|
||||||
architecture.setSystemCallResult(state, @intCast(h));
|
|
||||||
}
|
|
||||||
|
|
||||||
/// ipc_call(handle, message_ptr, message_len, reply_ptr, reply_cap) -> reply_len.
|
/// ipc_call(handle, message_ptr, message_len, reply_ptr, reply_cap) -> reply_len.
|
||||||
/// Blocks until the server replies; the trap frame lives on this task's kernel
|
/// Blocks until the server replies; the trap frame lives on this task's kernel
|
||||||
/// stack, so it survives the block and receives the result on resume.
|
/// stack, so it survives the block and receives the result on resume.
|
||||||
@@ -355,7 +332,14 @@ fn systemIpcCall(state: *architecture.CpuState) void {
|
|||||||
/// ipc_reply_wait(handle, reply_ptr, reply_len, receive_ptr, receive_cap) -> receive_len,
|
/// ipc_reply_wait(handle, reply_ptr, reply_len, receive_ptr, receive_cap) -> receive_len,
|
||||||
/// with the sender's badge in the secondary result register (rdx).
|
/// with the sender's badge in the secondary result register (rdx).
|
||||||
fn systemIpcReplyWait(state: *architecture.CpuState) void {
|
fn systemIpcReplyWait(state: *architecture.CpuState) void {
|
||||||
const endpoint = ipc.resolveHandle(scheduler.current(), architecture.systemCallArg(state, 0)) orelse return failErr(state, ipc.EBADF);
|
const t = scheduler.current();
|
||||||
|
const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 0)) orelse return failErr(state, ipc.EBADF);
|
||||||
|
// Receiving is the owner's privilege, the same rule the notification binders
|
||||||
|
// enforce: a sendable handle means only "you may talk to this". Anything
|
||||||
|
// else and a mount's backend endpoint — which `fs_resolve` installs in every
|
||||||
|
// caller's table — would let a stranger dequeue the requests meant for the
|
||||||
|
// server, taking the capabilities they carry and answering in its name.
|
||||||
|
if (!ipc.ownedBy(endpoint, t)) return failErr(state, ipc.EPERM);
|
||||||
var badge: u64 = 0;
|
var badge: u64 = 0;
|
||||||
var received_cap: u64 = abi.no_cap;
|
var received_cap: u64 = abi.no_cap;
|
||||||
const r = ipc.replyWait(endpoint, architecture.systemCallArg(state, 1), architecture.systemCallArg(state, 2), architecture.systemCallArg(state, 3), architecture.systemCallArg(state, 4), architecture.systemCallArg(state, 5), &badge, &received_cap);
|
const r = ipc.replyWait(endpoint, architecture.systemCallArg(state, 1), architecture.systemCallArg(state, 2), architecture.systemCallArg(state, 3), architecture.systemCallArg(state, 4), architecture.systemCallArg(state, 5), &badge, &received_cap);
|
||||||
@@ -378,6 +362,12 @@ fn systemIpcSend(state: *architecture.CpuState) void {
|
|||||||
|
|
||||||
/// device_enumerate(buffer, maximum) -> total: snapshot the device table into the caller's
|
/// device_enumerate(buffer, maximum) -> total: snapshot the device table into the caller's
|
||||||
/// buffer (up to `maximum` entries), returning the total device count.
|
/// buffer (up to `maximum` entries), returning the total device count.
|
||||||
|
///
|
||||||
|
/// The broker fills a small kernel chunk which `copyToUser` then places in the
|
||||||
|
/// caller's buffer: the kernel never stores through a user pointer, so a bad one
|
||||||
|
/// is -EFAULT instead of a ring-0 page fault. A DeviceDescriptor is a few hundred
|
||||||
|
/// bytes, so the chunk is deliberately tiny — the 16 KiB kernel stack could not
|
||||||
|
/// hold a whole user-requested array of them.
|
||||||
fn systemDeviceEnumerate(state: *architecture.CpuState) void {
|
fn systemDeviceEnumerate(state: *architecture.CpuState) void {
|
||||||
const buffer_ptr = architecture.systemCallArg(state, 0);
|
const buffer_ptr = architecture.systemCallArg(state, 0);
|
||||||
const maximum = architecture.systemCallArg(state, 1);
|
const maximum = architecture.systemCallArg(state, 1);
|
||||||
@@ -385,8 +375,20 @@ fn systemDeviceEnumerate(state: *architecture.CpuState) void {
|
|||||||
if (t.address_space == 0 or buffer_ptr >= user_half_end) return fail(state);
|
if (t.address_space == 0 or buffer_ptr >= user_half_end) return fail(state);
|
||||||
const sz = @sizeOf(device_abi.DeviceDescriptor);
|
const sz = @sizeOf(device_abi.DeviceDescriptor);
|
||||||
const cap = @min(maximum, (user_half_end - buffer_ptr) / sz); // clamp to the user half
|
const cap = @min(maximum, (user_half_end - buffer_ptr) / sz); // clamp to the user half
|
||||||
const out: [*]device_abi.DeviceDescriptor = @ptrFromInt(buffer_ptr);
|
|
||||||
architecture.setSystemCallResult(state, devices_broker.enumerate(out[0..@intCast(cap)]));
|
var chunk: [2]device_abi.DeviceDescriptor = undefined;
|
||||||
|
var copied: u64 = 0;
|
||||||
|
var start: usize = 0;
|
||||||
|
while (copied < cap) {
|
||||||
|
const filled = devices_broker.enumerateFrom(start, &chunk);
|
||||||
|
if (filled == 0) break;
|
||||||
|
start += filled;
|
||||||
|
const take = @min(@as(u64, filled), cap - copied);
|
||||||
|
const bytes = std.mem.sliceAsBytes(chunk[0..@intCast(take)]);
|
||||||
|
if (!user_memory.copyToUser(t.address_space, buffer_ptr + copied * sz, bytes)) return failErr(state, ipc.EFAULT);
|
||||||
|
copied += take;
|
||||||
|
}
|
||||||
|
architecture.setSystemCallResult(state, devices_broker.deviceCount());
|
||||||
}
|
}
|
||||||
|
|
||||||
/// device_claim(id) -> 0/-1: take exclusive ownership of a device for this process.
|
/// device_claim(id) -> 0/-1: take exclusive ownership of a device for this process.
|
||||||
@@ -962,14 +964,31 @@ fn systemSpawn(state: *architecture.CpuState) void {
|
|||||||
if (len == 0 or len > scheduler.maximum_task_name or ptr >= user_half_end or ptr + len > user_half_end) return fail(state);
|
if (len == 0 or len > scheduler.maximum_task_name or ptr >= user_half_end or ptr + len > user_half_end) return fail(state);
|
||||||
if (arguments_len > maximum_argument_bytes) return fail(state);
|
if (arguments_len > maximum_argument_bytes) return fail(state);
|
||||||
if (arguments_len != 0 and (arguments_ptr >= user_half_end or arguments_ptr + arguments_len > user_half_end)) return fail(state);
|
if (arguments_len != 0 and (arguments_ptr >= user_half_end or arguments_ptr + arguments_len > user_half_end)) return fail(state);
|
||||||
|
// The exit endpoint is a notification binding like signal_bind's and
|
||||||
|
// timer_bind's, so it obeys the same rule: the caller's own mailbox, never a
|
||||||
|
// stranger's. Otherwise any process could have the kernel post child-exit
|
||||||
|
// badges into PID 1 by spawning throwaway children against init's endpoint.
|
||||||
const exit_endpoint: ?*ipc.Endpoint = if (exit_handle == abi.no_cap)
|
const exit_endpoint: ?*ipc.Endpoint = if (exit_handle == abi.no_cap)
|
||||||
null
|
null
|
||||||
else
|
else block: {
|
||||||
ipc.resolveHandle(t, exit_handle) orelse return failErr(state, ipc.EBADF);
|
const endpoint = ipc.resolveHandle(t, exit_handle) orelse return failErr(state, ipc.EBADF);
|
||||||
|
if (!ipc.ownedBy(endpoint, t)) return failErr(state, ipc.EPERM);
|
||||||
|
break :block endpoint;
|
||||||
|
};
|
||||||
const image = ramdisk_image orelse return fail(state);
|
const image = ramdisk_image orelse return fail(state);
|
||||||
const rd = initial_ramdisk.Reader.init(image) orelse return fail(state);
|
const rd = initial_ramdisk.Reader.init(image) orelse return fail(state);
|
||||||
|
|
||||||
const name = @as([*]const u8, @ptrFromInt(ptr))[0..len];
|
// Both buffers come in through the checked copy layer, once. The lengths are
|
||||||
|
// already bounded above, so the staging arrays are small and fixed — and
|
||||||
|
// because the bytes are now the kernel's own, nothing below can be changed
|
||||||
|
// by another thread of the caller between validation and use.
|
||||||
|
var name_storage: [scheduler.maximum_task_name]u8 = undefined;
|
||||||
|
const name = name_storage[0..@intCast(len)];
|
||||||
|
if (!user_memory.copyFromUser(t.address_space, ptr, name)) return failErr(state, ipc.EFAULT);
|
||||||
|
var argument_storage: [maximum_argument_bytes]u8 = undefined;
|
||||||
|
const arguments = argument_storage[0..@intCast(arguments_len)];
|
||||||
|
if (arguments_len != 0 and !user_memory.copyFromUser(t.address_space, arguments_ptr, arguments)) return failErr(state, ipc.EFAULT);
|
||||||
|
|
||||||
// Exact path first, basename fallback second; either way argv[0] (and hence
|
// Exact path first, basename fallback second; either way argv[0] (and hence
|
||||||
// the task name, and the log ring's attribution) is the stored full path.
|
// the task name, and the log ring's attribution) is the stored full path.
|
||||||
const item = rd.find(name) orelse return fail(state); // no bundled binary by that name
|
const item = rd.find(name) orelse return fail(state); // no bundled binary by that name
|
||||||
@@ -977,8 +996,7 @@ fn systemSpawn(state: *architecture.CpuState) void {
|
|||||||
argv[0] = item.name;
|
argv[0] = item.name;
|
||||||
var argc: usize = 1;
|
var argc: usize = 1;
|
||||||
if (arguments_len != 0) {
|
if (arguments_len != 0) {
|
||||||
const blob = @as([*]const u8, @ptrFromInt(arguments_ptr))[0..arguments_len];
|
var pieces = std.mem.tokenizeScalar(u8, arguments, 0);
|
||||||
var pieces = std.mem.tokenizeScalar(u8, blob, 0);
|
|
||||||
while (pieces.next()) |piece| {
|
while (pieces.next()) |piece| {
|
||||||
if (argc == maximum_arguments) return fail(state);
|
if (argc == maximum_arguments) return fail(state);
|
||||||
argv[argc] = piece;
|
argv[argc] = piece;
|
||||||
@@ -1004,11 +1022,15 @@ fn systemThreadSpawn(state: *architecture.CpuState) void {
|
|||||||
if (t.address_space == 0) return fail(state); // kernel tasks own no address space to share
|
if (t.address_space == 0) return fail(state); // kernel tasks own no address space to share
|
||||||
if (entry == 0 or entry >= user_half_end) return fail(state);
|
if (entry == 0 or entry >= user_half_end) return fail(state);
|
||||||
if (stack_top == 0 or stack_top > user_half_end) return fail(state);
|
if (stack_top == 0 or stack_top > user_half_end) return fail(state);
|
||||||
// The endpoint the thread notifies on exit (how join waits), or none.
|
// The endpoint the thread notifies on exit (how join waits), or none — the
|
||||||
|
// caller's own, like every other notification binding.
|
||||||
const exit_endpoint: ?*ipc.Endpoint = if (exit_handle == abi.no_cap)
|
const exit_endpoint: ?*ipc.Endpoint = if (exit_handle == abi.no_cap)
|
||||||
null
|
null
|
||||||
else
|
else block: {
|
||||||
ipc.resolveHandle(t, exit_handle) orelse return failErr(state, ipc.EBADF);
|
const endpoint = ipc.resolveHandle(t, exit_handle) orelse return failErr(state, ipc.EBADF);
|
||||||
|
if (!ipc.ownedBy(endpoint, t)) return failErr(state, ipc.EPERM);
|
||||||
|
break :block endpoint;
|
||||||
|
};
|
||||||
const tid = spawnThreadSupervised(t.address_space, entry, stack_top, arg, t.priority, t.id, exit_endpoint, t.leader);
|
const tid = spawnThreadSupervised(t.address_space, entry, stack_top, arg, t.priority, t.id, exit_endpoint, t.leader);
|
||||||
if (tid == -ipc.ESRCH) return failErr(state, ipc.ESRCH); // dying group admits no member
|
if (tid == -ipc.ESRCH) return failErr(state, ipc.ESRCH); // dying group admits no member
|
||||||
if (tid < 0) return fail(state);
|
if (tid < 0) return fail(state);
|
||||||
@@ -1129,8 +1151,27 @@ fn systemProcessEnumerate(state: *architecture.CpuState) void {
|
|||||||
if (t.address_space == 0 or buffer_ptr >= user_half_end) return fail(state);
|
if (t.address_space == 0 or buffer_ptr >= user_half_end) return fail(state);
|
||||||
const sz = @sizeOf(abi.ProcessDescriptor);
|
const sz = @sizeOf(abi.ProcessDescriptor);
|
||||||
const cap = @min(maximum, (user_half_end - buffer_ptr) / sz); // clamp to the user half
|
const cap = @min(maximum, (user_half_end - buffer_ptr) / sz); // clamp to the user half
|
||||||
const out: [*]abi.ProcessDescriptor = @ptrFromInt(buffer_ptr);
|
|
||||||
architecture.setSystemCallResult(state, scheduler.enumerate(out[0..@intCast(cap)]));
|
// The scheduler describes a chunk of the table into kernel memory, then
|
||||||
|
// `copyToUser` places it — the kernel never stores through the user pointer.
|
||||||
|
// Once the caller's buffer is full the walk continues with an empty chunk,
|
||||||
|
// because the result is the true live count, not what fitted.
|
||||||
|
var chunk: [8]abi.ProcessDescriptor = undefined;
|
||||||
|
var copied: u64 = 0;
|
||||||
|
var total: u64 = 0;
|
||||||
|
var cursor: usize = 0;
|
||||||
|
while (true) {
|
||||||
|
const room: []abi.ProcessDescriptor = if (copied < cap) chunk[0..@intCast(@min(chunk.len, cap - copied))] else chunk[0..0];
|
||||||
|
const found = scheduler.enumerateFrom(&cursor, room);
|
||||||
|
total += found.live;
|
||||||
|
if (found.filled != 0) {
|
||||||
|
const bytes = std.mem.sliceAsBytes(chunk[0..found.filled]);
|
||||||
|
if (!user_memory.copyToUser(t.address_space, buffer_ptr + copied * sz, bytes)) return failErr(state, ipc.EFAULT);
|
||||||
|
copied += found.filled;
|
||||||
|
}
|
||||||
|
if (found.done) break;
|
||||||
|
}
|
||||||
|
architecture.setSystemCallResult(state, total);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// process_kill(id) -> 0 / -ESRCH / -EPERM: end the process `id`. Only its
|
/// process_kill(id) -> 0 / -ESRCH / -EPERM: end the process `id`. Only its
|
||||||
@@ -1483,13 +1524,20 @@ const exit_subscriber_capacity = 8;
|
|||||||
const ExitSubscriber = struct { endpoint: *ipc.Endpoint, owner: u32 };
|
const ExitSubscriber = struct { endpoint: *ipc.Endpoint, owner: u32 };
|
||||||
var exit_subscribers: [exit_subscriber_capacity]?ExitSubscriber = .{null} ** exit_subscriber_capacity;
|
var exit_subscribers: [exit_subscriber_capacity]?ExitSubscriber = .{null} ** exit_subscriber_capacity;
|
||||||
|
|
||||||
/// process_subscribe(endpoint): subscribe the caller's endpoint to published exit
|
/// process_subscribe(endpoint): subscribe the **caller's own** endpoint to
|
||||||
/// events. Ungated, like process_enumerate — what is running (and dying) is not a
|
/// published exit events. *Which* deaths one may hear of is ungated, like
|
||||||
/// secret between cooperating processes. -ENOSPC when the table is full.
|
/// process_enumerate — what is running (and dying) is not a secret between
|
||||||
|
/// cooperating processes. *Whose mailbox* they land in is not: the endpoint must
|
||||||
|
/// be the caller's (`ipc.ownedBy`), or any process could aim the firehose at a
|
||||||
|
/// stranger — filling PID 1's mailbox with exit notices it reads as its own
|
||||||
|
/// children's, and spending the eight-slot table so the services that need
|
||||||
|
/// deaths (the VFS's handle sweep) cannot subscribe at all. -EPERM otherwise,
|
||||||
|
/// -ENOSPC when the table is full.
|
||||||
fn systemProcessSubscribe(state: *architecture.CpuState) void {
|
fn systemProcessSubscribe(state: *architecture.CpuState) void {
|
||||||
const t = scheduler.current();
|
const t = scheduler.current();
|
||||||
if (t.address_space == 0) return fail(state);
|
if (t.address_space == 0) return fail(state);
|
||||||
const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 0)) orelse return failErr(state, ipc.EBADF);
|
const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 0)) orelse return failErr(state, ipc.EBADF);
|
||||||
|
if (!ipc.ownedBy(endpoint, t)) return failErr(state, ipc.EPERM);
|
||||||
const flags = sync.enter();
|
const flags = sync.enter();
|
||||||
defer sync.leave(flags);
|
defer sync.leave(flags);
|
||||||
for (&exit_subscribers) |*slot| {
|
for (&exit_subscribers) |*slot| {
|
||||||
@@ -1506,10 +1554,19 @@ fn systemProcessSubscribe(state: *architecture.CpuState) void {
|
|||||||
/// IRQ-as-IPC pattern a fourth time (docs/process-lifecycle.md). Replacing a
|
/// IRQ-as-IPC pattern a fourth time (docs/process-lifecycle.md). Replacing a
|
||||||
/// binding drops the old reference; signals that pended while unbound are
|
/// binding drops the old reference; signals that pended while unbound are
|
||||||
/// delivered immediately on bind, coalesced into one notification.
|
/// delivered immediately on bind, coalesced into one notification.
|
||||||
|
///
|
||||||
|
/// The endpoint must be the caller's own (`ipc.ownedBy`), or `signal_bind`
|
||||||
|
/// becomes a signal *forgery* primitive: `process_signal` is deliberately loose
|
||||||
|
/// about the target (a task may always signal itself) because the delivery point
|
||||||
|
/// was assumed to be the target's own mailbox. Aim it elsewhere and a stranger
|
||||||
|
/// signalling itself makes the kernel stamp a genuine `terminate` badge into
|
||||||
|
/// somebody else's queue — which is a shutdown request PID 1 has no way to
|
||||||
|
/// disbelieve.
|
||||||
fn systemSignalBind(state: *architecture.CpuState) void {
|
fn systemSignalBind(state: *architecture.CpuState) void {
|
||||||
const t = scheduler.current();
|
const t = scheduler.current();
|
||||||
if (t.address_space == 0) return fail(state);
|
if (t.address_space == 0) return fail(state);
|
||||||
const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 0)) orelse return failErr(state, ipc.EBADF);
|
const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 0)) orelse return failErr(state, ipc.EBADF);
|
||||||
|
if (!ipc.ownedBy(endpoint, t)) return failErr(state, ipc.EPERM);
|
||||||
const flags = sync.enter();
|
const flags = sync.enter();
|
||||||
defer sync.leave(flags);
|
defer sync.leave(flags);
|
||||||
if (t.signal_endpoint) |raw| ipc.dropRef(@ptrCast(@alignCast(raw)));
|
if (t.signal_endpoint) |raw| ipc.dropRef(@ptrCast(@alignCast(raw)));
|
||||||
@@ -1578,11 +1635,19 @@ fn timerSweepLocked() void {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// timer_bind(endpoint, ms): arm a one-shot timer. -ENOSPC when the table is full.
|
/// timer_bind(endpoint, ms): arm a one-shot timer on an endpoint of the caller's
|
||||||
|
/// own (`ipc.ownedBy`; -EPERM otherwise). A timer landing carries no identity —
|
||||||
|
/// that is the whole reason a service may keep exactly one in flight — so a
|
||||||
|
/// timer armed on someone else's endpoint is indistinguishable from one they
|
||||||
|
/// armed themselves, and a loop that re-arms on every landing (init's heartbeat)
|
||||||
|
/// multiplies: N forged timers leave N+1 self-perpetuating beats. The
|
||||||
|
/// sixteen-slot table is a shared resource on top of that. -ENOSPC when it is
|
||||||
|
/// full.
|
||||||
fn systemTimerBind(state: *architecture.CpuState) void {
|
fn systemTimerBind(state: *architecture.CpuState) void {
|
||||||
const t = scheduler.current();
|
const t = scheduler.current();
|
||||||
if (t.address_space == 0) return fail(state);
|
if (t.address_space == 0) return fail(state);
|
||||||
const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 0)) orelse return failErr(state, ipc.EBADF);
|
const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 0)) orelse return failErr(state, ipc.EBADF);
|
||||||
|
if (!ipc.ownedBy(endpoint, t)) return failErr(state, ipc.EPERM);
|
||||||
const ms = architecture.systemCallArg(state, 1);
|
const ms = architecture.systemCallArg(state, 1);
|
||||||
const flags = sync.enter();
|
const flags = sync.enter();
|
||||||
defer sync.leave(flags);
|
defer sync.leave(flags);
|
||||||
@@ -1622,12 +1687,17 @@ fn ownedGsi(t: *scheduler.Task, device_id: u64, resource_index: u64) ?u32 {
|
|||||||
/// irq_bind(device_id, resource_index, endpoint) -> 0/-1: deliver that device's IRQ to the
|
/// irq_bind(device_id, resource_index, endpoint) -> 0/-1: deliver that device's IRQ to the
|
||||||
/// endpoint as an asynchronous IPC notification. The driver then blocks in
|
/// endpoint as an asynchronous IPC notification. The driver then blocks in
|
||||||
/// IPC_ReplyWait and is woken by the ISR; see system/kernel/irq.zig for the cycle.
|
/// IPC_ReplyWait and is woken by the ISR; see system/kernel/irq.zig for the cycle.
|
||||||
|
/// Two gates, both necessary: the device must be *claimed* by the caller
|
||||||
|
/// (`ownedGsi`), and the endpoint must be the caller's own (`ipc.ownedBy`) — a
|
||||||
|
/// claim entitles a driver to its own interrupts, not to post them into a
|
||||||
|
/// stranger's mailbox.
|
||||||
fn systemIrqBind(state: *architecture.CpuState) void {
|
fn systemIrqBind(state: *architecture.CpuState) void {
|
||||||
const t = scheduler.current();
|
const t = scheduler.current();
|
||||||
if (t.address_space == 0) return fail(state);
|
if (t.address_space == 0) return fail(state);
|
||||||
const gsi = ownedGsi(t, architecture.systemCallArg(state, 0), architecture.systemCallArg(state, 1)) orelse
|
const gsi = ownedGsi(t, architecture.systemCallArg(state, 0), architecture.systemCallArg(state, 1)) orelse
|
||||||
return fail(state);
|
return fail(state);
|
||||||
const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 2)) orelse return fail(state);
|
const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 2)) orelse return fail(state);
|
||||||
|
if (!ipc.ownedBy(endpoint, t)) return failErr(state, ipc.EPERM);
|
||||||
|
|
||||||
const flags = sync.enter();
|
const flags = sync.enter();
|
||||||
defer sync.leave(flags);
|
defer sync.leave(flags);
|
||||||
@@ -1648,6 +1718,7 @@ fn systemMsiBind(state: *architecture.CpuState) void {
|
|||||||
const owner = devices_broker.ownerOf(device_id) orelse return fail(state);
|
const owner = devices_broker.ownerOf(device_id) orelse return fail(state);
|
||||||
if (owner != t.id) return fail(state); // not claimed by this process
|
if (owner != t.id) return fail(state); // not claimed by this process
|
||||||
const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 1)) orelse return failErr(state, ipc.EBADF);
|
const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 1)) orelse return failErr(state, ipc.EBADF);
|
||||||
|
if (!ipc.ownedBy(endpoint, t)) return failErr(state, ipc.EPERM); // interrupts land in your own mailbox
|
||||||
|
|
||||||
const flags = sync.enter();
|
const flags = sync.enter();
|
||||||
defer sync.leave(flags);
|
defer sync.leave(flags);
|
||||||
@@ -1682,9 +1753,10 @@ fn systemIrqAck(state: *architecture.CpuState) void {
|
|||||||
/// The pointer must lie in the user (low) half, so kernel addresses and
|
/// The pointer must lie in the user (low) half, so kernel addresses and
|
||||||
/// non-canonical values fall outside it and the read below can't be steered at
|
/// non-canonical values fall outside it and the read below can't be steered at
|
||||||
/// kernel data. Length is checked first so the upper-bound add can't overflow.
|
/// kernel data. Length is checked first so the upper-bound add can't overflow.
|
||||||
/// Known gap (fine for trusted user code): a pointer into an *unmapped* hole in
|
/// The message then comes in ONCE through the checked copy layer: an unmapped
|
||||||
/// the user half passes the check and the read #PFs -> on_fault halts — a
|
/// hole in the user half is -EFAULT rather than a kernel fault, and the bytes the
|
||||||
/// self-DoS, not an isolation break. Fault-recovering copy-in is a later item.
|
/// log stamps are the same bytes that were validated (the old code read the user
|
||||||
|
/// buffer twice — once to stage it, once again inside `log.append`).
|
||||||
///
|
///
|
||||||
/// The emit runs under the kernel lock, so a message is atomic on the wire — two
|
/// The emit runs under the kernel lock, so a message is atomic on the wire — two
|
||||||
/// processes writing from different cores can interleave *messages*, never bytes.
|
/// processes writing from different cores can interleave *messages*, never bytes.
|
||||||
@@ -1697,7 +1769,6 @@ fn systemDebugWrite(state: *architecture.CpuState) void {
|
|||||||
const len = architecture.systemCallArg(state, 1);
|
const len = architecture.systemCallArg(state, 1);
|
||||||
const level_raw = architecture.systemCallArg(state, 2);
|
const level_raw = architecture.systemCallArg(state, 2);
|
||||||
if (len <= write_buffer.len and ptr < user_half_end and ptr + len <= user_half_end) {
|
if (len <= write_buffer.len and ptr < user_half_end and ptr + len <= user_half_end) {
|
||||||
const source: [*]const u8 = @ptrFromInt(ptr);
|
|
||||||
// Levels above the enum range clamp to raw — old two-arg callers land
|
// Levels above the enum range clamp to raw — old two-arg callers land
|
||||||
// there naturally (garbage in arg 2 stays harmless).
|
// there naturally (garbage in arg 2 stays harmless).
|
||||||
const level: abi.KlogLevel = if (level_raw <= @intFromEnum(abi.KlogLevel.raw))
|
const level: abi.KlogLevel = if (level_raw <= @intFromEnum(abi.KlogLevel.raw))
|
||||||
@@ -1707,14 +1778,16 @@ fn systemDebugWrite(state: *architecture.CpuState) void {
|
|||||||
const t = scheduler.current();
|
const t = scheduler.current();
|
||||||
const flags = sync.enter();
|
const flags = sync.enter();
|
||||||
defer sync.leave(flags);
|
defer sync.leave(flags);
|
||||||
@memcpy(write_buffer[0..len], source[0..len]); // keep the latest message
|
// One copy in, under the lock; `write_buffer` (the latest message, which
|
||||||
|
// the kernel tests assert on) doubles as the staging buffer the log reads.
|
||||||
|
if (!user_memory.copyFromUser(t.address_space, ptr, write_buffer[0..len])) return failErr(state, ipc.EFAULT);
|
||||||
write_len = len;
|
write_len = len;
|
||||||
write_from_user = architecture.fromUser(state);
|
write_from_user = architecture.fromUser(state);
|
||||||
write_count += 1;
|
write_count += 1;
|
||||||
// The kernel stamps the sender's identity — attribution is structural,
|
// The kernel stamps the sender's identity — attribution is structural,
|
||||||
// not a prefix convention the payload could forge (and it is stamped
|
// not a prefix convention the payload could forge (and it is stamped
|
||||||
// per line inside log.append).
|
// per line inside log.append).
|
||||||
log.append(t.id, t.name(), level, source[0..len]);
|
log.append(t.id, t.name(), level, write_buffer[0..len]);
|
||||||
architecture.setSystemCallResult(state, len);
|
architecture.setSystemCallResult(state, len);
|
||||||
} else {
|
} else {
|
||||||
fail(state);
|
fail(state);
|
||||||
@@ -1729,18 +1802,35 @@ fn systemDebugWrite(state: *architecture.CpuState) void {
|
|||||||
/// [KlogRecordHeader][name][message] frames out of the byte stream (abi.zig).
|
/// [KlogRecordHeader][name][message] frames out of the byte stream (abi.zig).
|
||||||
///
|
///
|
||||||
/// The mirror of `debug_write`: the same overflow-safe user-half bounds check,
|
/// The mirror of `debug_write`: the same overflow-safe user-half bounds check,
|
||||||
/// but the copy runs kernel -> user, under the log lock (inside log.readAt) so
|
/// but the copy runs kernel -> user. The ring is drained a chunk at a time into a
|
||||||
/// the stream can't move underneath the copy. A read-only diagnostic.
|
/// kernel staging buffer (each chunk read under the log lock, so the stream can't
|
||||||
|
/// move underneath it) and each chunk is then placed with `copyToUser` — a
|
||||||
|
/// reader may ask for a megabyte, and the kernel stack is 16 KiB. A partial
|
||||||
|
/// result is honest: the reader advances its cursor by what it got. A read-only
|
||||||
|
/// diagnostic.
|
||||||
fn systemKlogRead(state: *architecture.CpuState) void {
|
fn systemKlogRead(state: *architecture.CpuState) void {
|
||||||
const offset = architecture.systemCallArg(state, 0);
|
const offset = architecture.systemCallArg(state, 0);
|
||||||
const ptr = architecture.systemCallArg(state, 1);
|
const ptr = architecture.systemCallArg(state, 1);
|
||||||
const len = architecture.systemCallArg(state, 2);
|
const len = architecture.systemCallArg(state, 2);
|
||||||
|
const t = scheduler.current();
|
||||||
// Confine the whole destination span to the user (low) half. `len <=
|
// Confine the whole destination span to the user (low) half. `len <=
|
||||||
// user_half_end - ptr` bounds the length without an overflowing add.
|
// user_half_end - ptr` bounds the length without an overflowing add.
|
||||||
if (ptr < user_half_end and len <= user_half_end - ptr) {
|
if (ptr < user_half_end and len <= user_half_end - ptr) {
|
||||||
const dest: [*]u8 = @ptrFromInt(ptr);
|
var chunk: [512]u8 = undefined;
|
||||||
const n = log.readAt(offset, dest[0..len]) orelse return fail(state);
|
var done: u64 = 0;
|
||||||
architecture.setSystemCallResult(state, n);
|
while (done < len) {
|
||||||
|
const want = @min(@as(u64, chunk.len), len - done);
|
||||||
|
const n = log.readAt(offset + done, chunk[0..@intCast(want)]) orelse {
|
||||||
|
// The cursor fell behind the ring's tail mid-drain. What was
|
||||||
|
// already placed stands; only a first-chunk miss fails the call.
|
||||||
|
if (done == 0) return fail(state);
|
||||||
|
break;
|
||||||
|
};
|
||||||
|
if (n == 0) break; // caught up
|
||||||
|
if (!user_memory.copyToUser(t.address_space, ptr + done, chunk[0..n])) return failErr(state, ipc.EFAULT);
|
||||||
|
done += n;
|
||||||
|
}
|
||||||
|
architecture.setSystemCallResult(state, done);
|
||||||
} else {
|
} else {
|
||||||
fail(state);
|
fail(state);
|
||||||
}
|
}
|
||||||
@@ -1753,10 +1843,10 @@ fn systemKlogRead(state: *architecture.CpuState) void {
|
|||||||
fn systemKlogStatus(state: *architecture.CpuState) void {
|
fn systemKlogStatus(state: *architecture.CpuState) void {
|
||||||
const ptr = architecture.systemCallArg(state, 0);
|
const ptr = architecture.systemCallArg(state, 0);
|
||||||
const size = @sizeOf(abi.KlogStatus);
|
const size = @sizeOf(abi.KlogStatus);
|
||||||
|
const t = scheduler.current();
|
||||||
if (ptr < user_half_end and size <= user_half_end - ptr) {
|
if (ptr < user_half_end and size <= user_half_end - ptr) {
|
||||||
var status = log.status();
|
var status = log.status();
|
||||||
const dest: [*]u8 = @ptrFromInt(ptr);
|
if (!user_memory.copyValueToUser(t.address_space, ptr, &status)) return failErr(state, ipc.EFAULT);
|
||||||
@memcpy(dest[0..size], std.mem.asBytes(&status)[0..size]);
|
|
||||||
architecture.setSystemCallResult(state, 0);
|
architecture.setSystemCallResult(state, 0);
|
||||||
} else {
|
} else {
|
||||||
fail(state);
|
fail(state);
|
||||||
@@ -1775,10 +1865,12 @@ fn systemFsResolve(state: *architecture.CpuState) void {
|
|||||||
const flags = architecture.systemCallArg(state, 2);
|
const flags = architecture.systemCallArg(state, 2);
|
||||||
const out_ptr = architecture.systemCallArg(state, 3);
|
const out_ptr = architecture.systemCallArg(state, 3);
|
||||||
const out_cap = architecture.systemCallArg(state, 4);
|
const out_cap = architecture.systemCallArg(state, 4);
|
||||||
if (path_len == 0 or path_len > 224 or path_ptr >= user_half_end or path_ptr + path_len > user_half_end) return fail(state);
|
if (path_len == 0 or path_len > maximum_resolve_path or path_ptr >= user_half_end or path_ptr + path_len > user_half_end) return fail(state);
|
||||||
if (out_cap != 0 and (out_ptr >= user_half_end or out_ptr + out_cap > user_half_end)) return fail(state);
|
if (out_cap != 0 and !user_memory.userRangeOk(out_ptr, @intCast(out_cap))) return fail(state);
|
||||||
const path = @as([*]const u8, @ptrFromInt(path_ptr))[0..path_len];
|
|
||||||
const t = scheduler.current();
|
const t = scheduler.current();
|
||||||
|
var path_storage: [maximum_resolve_path]u8 = undefined;
|
||||||
|
const path = path_storage[0..@intCast(path_len)];
|
||||||
|
if (!user_memory.copyFromUser(t.address_space, path_ptr, path)) return failErr(state, ipc.EFAULT);
|
||||||
|
|
||||||
const flags_lock = sync.enter();
|
const flags_lock = sync.enter();
|
||||||
defer sync.leave(flags_lock);
|
defer sync.leave(flags_lock);
|
||||||
@@ -1790,14 +1882,15 @@ fn systemFsResolve(state: *architecture.CpuState) void {
|
|||||||
.backend => |*backend| {
|
.backend => |*backend| {
|
||||||
// The rewritten path goes back in the out buffer behind a u16
|
// The rewritten path goes back in the out buffer behind a u16
|
||||||
// length prefix (a third result register would collide with r8's
|
// length prefix (a third result register would collide with r8's
|
||||||
// argument role in the userspace stub).
|
// argument role in the userspace stub). Both halves are placed with
|
||||||
|
// the checked copy, and *before* the handle is installed, so an
|
||||||
|
// -EFAULT never strands a capability in the caller's table.
|
||||||
if (backend.path_len + 2 > out_cap) return fail(state);
|
if (backend.path_len + 2 > out_cap) return fail(state);
|
||||||
|
const prefix = [2]u8{ @intCast(backend.path_len & 0xFF), @intCast(backend.path_len >> 8) };
|
||||||
|
if (!user_memory.copyToUser(t.address_space, out_ptr, &prefix)) return failErr(state, ipc.EFAULT);
|
||||||
|
if (!user_memory.copyToUser(t.address_space, out_ptr + 2, backend.path[0..backend.path_len])) return failErr(state, ipc.EFAULT);
|
||||||
const handle = ipc.installHandleDeduped(t, backend.endpoint);
|
const handle = ipc.installHandleDeduped(t, backend.endpoint);
|
||||||
if (handle < 0) return fail(state);
|
if (handle < 0) return fail(state);
|
||||||
const destination: [*]u8 = @ptrFromInt(out_ptr);
|
|
||||||
destination[0] = @intCast(backend.path_len & 0xFF);
|
|
||||||
destination[1] = @intCast(backend.path_len >> 8);
|
|
||||||
@memcpy(destination[2..][0..backend.path_len], backend.path[0..backend.path_len]);
|
|
||||||
architecture.setSystemCallResult(state, abi.fs_route_backend);
|
architecture.setSystemCallResult(state, abi.fs_route_backend);
|
||||||
architecture.setSystemCallResult2(state, @intCast(handle));
|
architecture.setSystemCallResult2(state, @intCast(handle));
|
||||||
},
|
},
|
||||||
@@ -1809,6 +1902,12 @@ fn systemFsResolve(state: *architecture.CpuState) void {
|
|||||||
/// kernel-backed node. read copies file bytes; status copies a FileAttributes;
|
/// kernel-backed node. read copies file bytes; status copies a FileAttributes;
|
||||||
/// readdir copies [DirectoryEntryHeader][name] for the `offset`th child. Reads
|
/// readdir copies [DirectoryEntryHeader][name] for the `offset`th child. Reads
|
||||||
/// of the immutable initrd never take the kernel lock.
|
/// of the immutable initrd never take the kernel lock.
|
||||||
|
///
|
||||||
|
/// Every result reaches the caller through `copyToUser`, never a store through
|
||||||
|
/// the user pointer. `read` stages the file bytes a chunk at a time — a caller
|
||||||
|
/// may ask for the 64 KiB ceiling, which no kernel stack could hold — so a
|
||||||
|
/// mid-way -EFAULT is possible; the call fails and the already-placed prefix is
|
||||||
|
/// meaningless, exactly as a failed read should be.
|
||||||
fn systemFsNode(state: *architecture.CpuState) void {
|
fn systemFsNode(state: *architecture.CpuState) void {
|
||||||
const operation = architecture.systemCallArg(state, 0);
|
const operation = architecture.systemCallArg(state, 0);
|
||||||
const node_token = architecture.systemCallArg(state, 1);
|
const node_token = architecture.systemCallArg(state, 1);
|
||||||
@@ -1817,16 +1916,24 @@ fn systemFsNode(state: *architecture.CpuState) void {
|
|||||||
const buf_len = architecture.systemCallArg(state, 4);
|
const buf_len = architecture.systemCallArg(state, 4);
|
||||||
if (buf_ptr >= user_half_end or buf_len > user_half_end - buf_ptr) return fail(state);
|
if (buf_ptr >= user_half_end or buf_len > user_half_end - buf_ptr) return fail(state);
|
||||||
const capped = @min(buf_len, 64 * 1024); // bound any single copy
|
const capped = @min(buf_len, 64 * 1024); // bound any single copy
|
||||||
const destination: [*]u8 = @ptrFromInt(buf_ptr);
|
const t = scheduler.current();
|
||||||
switch (operation) {
|
switch (operation) {
|
||||||
abi.fs_node_read => {
|
abi.fs_node_read => {
|
||||||
const n = vfs.nodeRead(node_token, offset, destination[0..capped]) orelse return fail(state);
|
var chunk: [512]u8 = undefined;
|
||||||
architecture.setSystemCallResult(state, n);
|
var done: u64 = 0;
|
||||||
|
while (done < capped) {
|
||||||
|
const want = @min(@as(u64, chunk.len), capped - done);
|
||||||
|
const n = vfs.nodeRead(node_token, offset + done, chunk[0..@intCast(want)]) orelse return fail(state);
|
||||||
|
if (n == 0) break; // end of file
|
||||||
|
if (!user_memory.copyToUser(t.address_space, buf_ptr + done, chunk[0..n])) return failErr(state, ipc.EFAULT);
|
||||||
|
done += n;
|
||||||
|
}
|
||||||
|
architecture.setSystemCallResult(state, done);
|
||||||
},
|
},
|
||||||
abi.fs_node_status => {
|
abi.fs_node_status => {
|
||||||
var attributes = vfs.nodeStatus(node_token) orelse return fail(state);
|
var attributes = vfs.nodeStatus(node_token) orelse return fail(state);
|
||||||
if (capped < @sizeOf(abi.FileAttributes)) return fail(state);
|
if (capped < @sizeOf(abi.FileAttributes)) return fail(state);
|
||||||
@memcpy(destination[0..@sizeOf(abi.FileAttributes)], std.mem.asBytes(&attributes));
|
if (!user_memory.copyValueToUser(t.address_space, buf_ptr, &attributes)) return failErr(state, ipc.EFAULT);
|
||||||
architecture.setSystemCallResult(state, @sizeOf(abi.FileAttributes));
|
architecture.setSystemCallResult(state, @sizeOf(abi.FileAttributes));
|
||||||
},
|
},
|
||||||
abi.fs_node_readdir => {
|
abi.fs_node_readdir => {
|
||||||
@@ -1837,10 +1944,13 @@ fn systemFsNode(state: *architecture.CpuState) void {
|
|||||||
architecture.setSystemCallResult(state, 0); // past the end
|
architecture.setSystemCallResult(state, 0); // past the end
|
||||||
return;
|
return;
|
||||||
};
|
};
|
||||||
var header = result.header;
|
// Header and name are staged contiguously so one entry is one copy.
|
||||||
|
var entry: [@sizeOf(abi.DirectoryEntryHeader) + name_buffer.len]u8 = undefined;
|
||||||
|
const header = result.header;
|
||||||
const total = header_size + @min(result.name_len, capped - header_size);
|
const total = header_size + @min(result.name_len, capped - header_size);
|
||||||
@memcpy(destination[0..header_size], std.mem.asBytes(&header));
|
@memcpy(entry[0..header_size], std.mem.asBytes(&header));
|
||||||
@memcpy(destination[header_size..total], name_buffer[0 .. total - header_size]);
|
@memcpy(entry[header_size..total], name_buffer[0 .. total - header_size]);
|
||||||
|
if (!user_memory.copyToUser(t.address_space, buf_ptr, entry[0..total])) return failErr(state, ipc.EFAULT);
|
||||||
architecture.setSystemCallResult(state, total);
|
architecture.setSystemCallResult(state, total);
|
||||||
},
|
},
|
||||||
else => fail(state),
|
else => fail(state),
|
||||||
@@ -1857,12 +1967,19 @@ fn systemFsMount(state: *architecture.CpuState) void {
|
|||||||
const backend_handle = architecture.systemCallArg(state, 2);
|
const backend_handle = architecture.systemCallArg(state, 2);
|
||||||
const rewrite_ptr = architecture.systemCallArg(state, 3);
|
const rewrite_ptr = architecture.systemCallArg(state, 3);
|
||||||
const rewrite_len = architecture.systemCallArg(state, 4);
|
const rewrite_len = architecture.systemCallArg(state, 4);
|
||||||
if (prefix_len == 0 or prefix_len > 64 or prefix_ptr >= user_half_end or prefix_ptr + prefix_len > user_half_end) return fail(state);
|
if (prefix_len == 0 or prefix_len > maximum_mount_prefix or prefix_ptr >= user_half_end or prefix_ptr + prefix_len > user_half_end) return fail(state);
|
||||||
if (rewrite_len > 32) return fail(state);
|
if (rewrite_len > maximum_mount_rewrite) return fail(state);
|
||||||
if (rewrite_len != 0 and (rewrite_ptr >= user_half_end or rewrite_ptr + rewrite_len > user_half_end)) return fail(state);
|
if (rewrite_len != 0 and (rewrite_ptr >= user_half_end or rewrite_ptr + rewrite_len > user_half_end)) return fail(state);
|
||||||
const t = scheduler.current();
|
const t = scheduler.current();
|
||||||
const prefix = @as([*]const u8, @ptrFromInt(prefix_ptr))[0..prefix_len];
|
// Both strings come in through the checked copy; `mountBackend` copies them
|
||||||
const rewrite = if (rewrite_len == 0) "" else @as([*]const u8, @ptrFromInt(rewrite_ptr))[0..rewrite_len];
|
// again into the mount table, so these staging buffers only need to outlive
|
||||||
|
// this call.
|
||||||
|
var prefix_storage: [maximum_mount_prefix]u8 = undefined;
|
||||||
|
const prefix = prefix_storage[0..@intCast(prefix_len)];
|
||||||
|
if (!user_memory.copyFromUser(t.address_space, prefix_ptr, prefix)) return failErr(state, ipc.EFAULT);
|
||||||
|
var rewrite_storage: [maximum_mount_rewrite]u8 = undefined;
|
||||||
|
const rewrite = rewrite_storage[0..@intCast(rewrite_len)];
|
||||||
|
if (rewrite_len != 0 and !user_memory.copyFromUser(t.address_space, rewrite_ptr, rewrite)) return failErr(state, ipc.EFAULT);
|
||||||
|
|
||||||
const flags = sync.enter();
|
const flags = sync.enter();
|
||||||
defer sync.leave(flags);
|
defer sync.leave(flags);
|
||||||
@@ -1879,8 +1996,11 @@ fn systemFsMount(state: *architecture.CpuState) void {
|
|||||||
fn systemFsUnmount(state: *architecture.CpuState) void {
|
fn systemFsUnmount(state: *architecture.CpuState) void {
|
||||||
const prefix_ptr = architecture.systemCallArg(state, 0);
|
const prefix_ptr = architecture.systemCallArg(state, 0);
|
||||||
const prefix_len = architecture.systemCallArg(state, 1);
|
const prefix_len = architecture.systemCallArg(state, 1);
|
||||||
if (prefix_len == 0 or prefix_len > 64 or prefix_ptr >= user_half_end or prefix_ptr + prefix_len > user_half_end) return fail(state);
|
if (prefix_len == 0 or prefix_len > maximum_mount_prefix or prefix_ptr >= user_half_end or prefix_ptr + prefix_len > user_half_end) return fail(state);
|
||||||
const prefix = @as([*]const u8, @ptrFromInt(prefix_ptr))[0..prefix_len];
|
const t = scheduler.current();
|
||||||
|
var prefix_storage: [maximum_mount_prefix]u8 = undefined;
|
||||||
|
const prefix = prefix_storage[0..@intCast(prefix_len)];
|
||||||
|
if (!user_memory.copyFromUser(t.address_space, prefix_ptr, prefix)) return failErr(state, ipc.EFAULT);
|
||||||
const flags = sync.enter();
|
const flags = sync.enter();
|
||||||
defer sync.leave(flags);
|
defer sync.leave(flags);
|
||||||
if (!vfs.unmount(prefix)) return fail(state);
|
if (!vfs.unmount(prefix)) return fail(state);
|
||||||
|
|||||||
+48
-11
@@ -1203,18 +1203,56 @@ pub fn destroyTaskLocked(t: *Task) void {
|
|||||||
/// Snapshot the task table into `out` (up to its length), returning the total
|
/// Snapshot the task table into `out` (up to its length), returning the total
|
||||||
/// number of live tasks — the kernel half of `process_enumerate`, mirroring
|
/// number of live tasks — the kernel half of `process_enumerate`, mirroring
|
||||||
/// devices_broker.enumerate. Kernel tasks are included (empty name, supervisor 0):
|
/// devices_broker.enumerate. Kernel tasks are included (empty name, supervisor 0):
|
||||||
/// an honest `ps` shows the idle tasks too. `out` may be user memory: the caller's
|
/// an honest `ps` shows the idle tasks too. `out` is always KERNEL memory: the
|
||||||
/// address space is loaded during its system call, and the same bring-up trust
|
/// system call bounces it out to the caller through the checked copy layer
|
||||||
/// applies as for device_enumerate (an unmapped user page faults the kernel).
|
/// (system/kernel/user-memory.zig), so a bad user pointer fails the call instead
|
||||||
|
/// of faulting ring 0.
|
||||||
pub fn enumerate(out: []abi.ProcessDescriptor) u64 {
|
pub fn enumerate(out: []abi.ProcessDescriptor) u64 {
|
||||||
|
var total: u64 = 0;
|
||||||
|
var cursor: usize = 0;
|
||||||
|
while (true) {
|
||||||
|
// Past the buffer, keep walking with an empty chunk: the total is the
|
||||||
|
// whole live count, however few descriptors the caller had room for.
|
||||||
|
const room = if (total < out.len) out[@intCast(total)..] else out[out.len..];
|
||||||
|
const chunk = enumerateFrom(&cursor, room);
|
||||||
|
total += chunk.live;
|
||||||
|
if (chunk.done) return total;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// What one chunk of the task-table walk found.
|
||||||
|
pub const TaskChunk = struct {
|
||||||
|
/// Live tasks passed in this chunk, whether or not they fit in `out` — this
|
||||||
|
/// is what the running total (and hence `process_enumerate`'s result) counts.
|
||||||
|
live: usize,
|
||||||
|
/// How many of those were written into `out` (`@min(live, out.len)`).
|
||||||
|
filled: usize,
|
||||||
|
/// The cursor reached the end of the table: this was the last chunk.
|
||||||
|
done: bool,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// One chunk of the task table: starting at slot `cursor` (advanced past
|
||||||
|
/// everything scanned), describe up to `out.len` live tasks into `out` — or, with
|
||||||
|
/// an empty `out`, just count the rest. `cursor == tasks.len` ends the walk.
|
||||||
|
///
|
||||||
|
/// The chunked form exists so `process_enumerate` can stage each chunk in a small
|
||||||
|
/// kernel buffer and copy it out with `user_memory.copyToUser`, rather than
|
||||||
|
/// handing a user pointer to the kernel's own stores. A *slot* cursor, rather
|
||||||
|
/// than a "skip the first N live tasks" count, keeps chunks from duplicating or
|
||||||
|
/// losing an entry when a task exits between them.
|
||||||
|
pub fn enumerateFrom(cursor: *usize, out: []abi.ProcessDescriptor) TaskChunk {
|
||||||
const flags = sync.enter();
|
const flags = sync.enter();
|
||||||
defer sync.leave(flags);
|
defer sync.leave(flags);
|
||||||
var total: u64 = 0;
|
var live: usize = 0;
|
||||||
for (&tasks) |*t| {
|
var filled: usize = 0;
|
||||||
|
const limit = if (out.len == 0) tasks.len else out.len; // always makes progress
|
||||||
|
while (cursor.* < tasks.len and live < limit) {
|
||||||
|
const t = &tasks[cursor.*];
|
||||||
|
cursor.* += 1;
|
||||||
if (t.state == .free or t.state == .reaping) continue; // reaping = already exited
|
if (t.state == .free or t.state == .reaping) continue; // reaping = already exited
|
||||||
if (total < out.len) {
|
live += 1;
|
||||||
const d = &out[total];
|
if (filled == out.len) continue;
|
||||||
d.* = .{
|
out[filled] = .{
|
||||||
.id = t.id,
|
.id = t.id,
|
||||||
.supervisor = t.supervisor,
|
.supervisor = t.supervisor,
|
||||||
.leader = t.leader,
|
.leader = t.leader,
|
||||||
@@ -1228,10 +1266,9 @@ pub fn enumerate(out: []abi.ProcessDescriptor) u64 {
|
|||||||
.name_length = t.name_length,
|
.name_length = t.name_length,
|
||||||
.name = t.name_buffer,
|
.name = t.name_buffer,
|
||||||
};
|
};
|
||||||
|
filled += 1;
|
||||||
}
|
}
|
||||||
total += 1;
|
return .{ .live = live, .filled = filled, .done = cursor.* >= tasks.len };
|
||||||
}
|
|
||||||
return total;
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Whether the running task is a user process (has its own address space).
|
/// Whether the running task is a user process (has its own address space).
|
||||||
|
|||||||
+286
-28
@@ -29,6 +29,7 @@ const process = @import("process.zig");
|
|||||||
const initial_ramdisk = @import("initial-ramdisk");
|
const initial_ramdisk = @import("initial-ramdisk");
|
||||||
const kernel_log = @import("log.zig");
|
const kernel_log = @import("log.zig");
|
||||||
const kernel_vfs = @import("vfs.zig");
|
const kernel_vfs = @import("vfs.zig");
|
||||||
|
const user_memory = @import("user-memory.zig");
|
||||||
|
|
||||||
/// Formatted test-marker write. Goes through the kernel log (not straight to
|
/// Formatted test-marker write. Goes through the kernel log (not straight to
|
||||||
/// serial): the log lock is what keeps marker lines from interleaving with
|
/// serial): the log lock is what keeps marker lines from interleaving with
|
||||||
@@ -141,6 +142,8 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void {
|
|||||||
faultNull();
|
faultNull();
|
||||||
} else if (eql(case, "usermem")) {
|
} else if (eql(case, "usermem")) {
|
||||||
userMemTest();
|
userMemTest();
|
||||||
|
} else if (eql(case, "user-memory")) {
|
||||||
|
userMemoryTest(boot_information);
|
||||||
} else if (eql(case, "user-pf")) {
|
} else if (eql(case, "user-pf")) {
|
||||||
userPfTest();
|
userPfTest();
|
||||||
} else if (eql(case, "fault-recovery")) {
|
} else if (eql(case, "fault-recovery")) {
|
||||||
@@ -243,6 +246,10 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void {
|
|||||||
containmentTest();
|
containmentTest();
|
||||||
} else if (eql(case, "device-manager")) {
|
} else if (eql(case, "device-manager")) {
|
||||||
deviceManagerTest(boot_information);
|
deviceManagerTest(boot_information);
|
||||||
|
} else if (eql(case, "protocol-registry")) {
|
||||||
|
protocolRegistryTest(boot_information);
|
||||||
|
} else if (eql(case, "protocol-denied")) {
|
||||||
|
protocolDeniedTest(boot_information);
|
||||||
} else if (eql(case, "reboot")) {
|
} else if (eql(case, "reboot")) {
|
||||||
rebootTest();
|
rebootTest();
|
||||||
} else {
|
} else {
|
||||||
@@ -1023,6 +1030,103 @@ fn userMemTest() void {
|
|||||||
result();
|
result();
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// The checked copy layer (system/kernel/user-memory.zig), against a scratch
|
||||||
|
/// address space built here rather than a live process — so the refusals can be
|
||||||
|
/// provoked exactly: a kernel-half address, an unmapped user page, and a user
|
||||||
|
/// page mapped read-only. Then the fixture proves the same refusals reach ring 3
|
||||||
|
/// as -errno instead of a kernel fault.
|
||||||
|
fn userMemoryTest(boot_information: *const BootInformation) void {
|
||||||
|
log("DANOS-TEST-BEGIN: user-memory\n", .{});
|
||||||
|
const base_free = pmm.stats().free_frames;
|
||||||
|
|
||||||
|
const address_space = architecture.createAddressSpace() orelse {
|
||||||
|
check("created a scratch address space", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
check("created a scratch address space", address_space != 0);
|
||||||
|
|
||||||
|
// Three consecutive pages: two writable, the third read-only — so a copy that
|
||||||
|
// straddles into the third proves the write check applies per page, not just
|
||||||
|
// to the first one the walk touches.
|
||||||
|
const writable_pages = 2;
|
||||||
|
const total_pages = 3;
|
||||||
|
const arena = process.heap_arena_base;
|
||||||
|
var frames: [total_pages]u64 = undefined;
|
||||||
|
var mapped: usize = 0;
|
||||||
|
while (mapped < total_pages) : (mapped += 1) {
|
||||||
|
frames[mapped] = pmm.alloc() orelse break;
|
||||||
|
architecture.mapUserPageInto(address_space, arena + mapped * abi.page_size, frames[mapped], mapped < writable_pages, false);
|
||||||
|
}
|
||||||
|
check("mapped two writable and one read-only user page", mapped == total_pages);
|
||||||
|
if (mapped == total_pages) {
|
||||||
|
const read_only = arena + writable_pages * abi.page_size;
|
||||||
|
const unmapped = arena + total_pages * abi.page_size;
|
||||||
|
const kernel_half: u64 = 0xFFFF_8000_0000_0000;
|
||||||
|
|
||||||
|
var out: [16]u8 = undefined;
|
||||||
|
const pattern = [_]u8{ 0xC0, 0xDE, 0xF0, 0x0D, 0xBA, 0xAD, 0xF0, 0x0D };
|
||||||
|
|
||||||
|
// A round trip through the writable page: what copyToUser placed is what
|
||||||
|
// copyFromUser brings back, and the frame really holds it.
|
||||||
|
const wrote = user_memory.copyToUser(address_space, arena + 32, &pattern);
|
||||||
|
const read_back = user_memory.copyFromUser(address_space, arena + 32, out[0..pattern.len]);
|
||||||
|
const frame_view: [*]const u8 = @ptrFromInt(boot_handoff.physicalToVirtual(frames[0] + 32));
|
||||||
|
check("copyToUser/copyFromUser round trip", wrote and read_back and
|
||||||
|
eql(out[0..pattern.len], &pattern) and eql(frame_view[0..pattern.len], &pattern));
|
||||||
|
|
||||||
|
// Straddling the 4 KiB boundary between the two writable pages.
|
||||||
|
const straddle = arena + abi.page_size - 4;
|
||||||
|
check("a page-straddling round trip", user_memory.copyToUser(address_space, straddle, &pattern) and
|
||||||
|
user_memory.copyFromUser(address_space, straddle, out[0..pattern.len]) and
|
||||||
|
eql(out[0..pattern.len], &pattern));
|
||||||
|
|
||||||
|
// Kernel-half addresses are refused by the range check, before any walk.
|
||||||
|
check("copyToUser refuses a kernel-half address", !user_memory.copyToUser(address_space, kernel_half, &pattern));
|
||||||
|
check("copyFromUser refuses a kernel-half address", !user_memory.copyFromUser(address_space, kernel_half, out[0..pattern.len]));
|
||||||
|
check("a range running off the end of the user half is refused", !user_memory.copyToUser(address_space, user_memory.user_half_end - 4, &pattern));
|
||||||
|
|
||||||
|
// An unmapped-but-in-range page: the latent kernel fault H1 exists to kill.
|
||||||
|
check("copyToUser refuses an unmapped user page", !user_memory.copyToUser(address_space, unmapped, &pattern));
|
||||||
|
check("copyFromUser refuses an unmapped user page", !user_memory.copyFromUser(address_space, unmapped, out[0..pattern.len]));
|
||||||
|
|
||||||
|
// The leaf permission bits: a read-only user page may be read, never written.
|
||||||
|
check("copyToUser refuses a read-only user mapping", !user_memory.copyToUser(address_space, read_only, &pattern));
|
||||||
|
check("copyFromUser accepts a read-only user mapping", user_memory.copyFromUser(address_space, read_only, out[0..pattern.len]));
|
||||||
|
check("a write straddling into a read-only page is refused", !user_memory.copyToUser(address_space, read_only - 4, &pattern));
|
||||||
|
|
||||||
|
// The kernel's own address space is not a user address space.
|
||||||
|
check("copyToUser refuses address space 0", !user_memory.copyToUser(0, arena, &pattern));
|
||||||
|
check("copyFromUser refuses address space 0", !user_memory.copyFromUser(0, arena, out[0..pattern.len]));
|
||||||
|
}
|
||||||
|
|
||||||
|
var i: usize = 0;
|
||||||
|
while (i < mapped) : (i += 1) {
|
||||||
|
const va = arena + i * abi.page_size;
|
||||||
|
architecture.unmapUserPageInto(address_space, va);
|
||||||
|
pmm.free(frames[i]);
|
||||||
|
}
|
||||||
|
architecture.destroyAddressSpace(address_space);
|
||||||
|
check("no frames leaked (free count restored)", pmm.stats().free_frames == base_free);
|
||||||
|
|
||||||
|
// Now the ring-3 half: the fixture aims bad pointers at the converted system
|
||||||
|
// calls and must get failures back with the machine still running.
|
||||||
|
if (boot_information.initial_ramdisk_len == 0) {
|
||||||
|
check("bootloader handed over an initial_ramdisk", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||||
|
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||||
|
check("initial_ramdisk image is valid", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
process.setInitialRamdisk(image);
|
||||||
|
check("user-memory-test spawned", spawnNamed(rd, "user-memory-test"));
|
||||||
|
result();
|
||||||
|
}
|
||||||
|
|
||||||
// --- synchronous IPC --------------------------------------------------------
|
// --- synchronous IPC --------------------------------------------------------
|
||||||
|
|
||||||
var ipc_endpoint: *ipcsync.Endpoint = undefined;
|
var ipc_endpoint: *ipcsync.Endpoint = undefined;
|
||||||
@@ -1970,7 +2074,7 @@ fn initTest(boot_information: *const BootInformation) void {
|
|||||||
check("init loaded and spawned as a process", spawned);
|
check("init loaded and spawned as a process", spawned);
|
||||||
|
|
||||||
// Wait (real time) until the LAST write is a heartbeat — proving init got
|
// Wait (real time) until the LAST write is a heartbeat — proving init got
|
||||||
// through its boot chatter (heap ok, the /etc/init.csv lookup) and settled
|
// through its boot chatter (heap ok, the /system/configuration/init.csv lookup) and settled
|
||||||
// into its beat-and-sleep loop (~1 s between beats). Waiting on the text
|
// into its beat-and-sleep loop (~1 s between beats). Waiting on the text
|
||||||
// rather than a raw write count: the boot chatter alone satisfies a count,
|
// rather than a raw write count: the boot chatter alone satisfies a count,
|
||||||
// which is exactly the too-early check that used to fail here.
|
// which is exactly the too-early check that used to fail here.
|
||||||
@@ -2109,7 +2213,7 @@ fn processKillTest(boot_information: *const BootInformation) void {
|
|||||||
while (i < rd.count) : (i += 1) {
|
while (i < rd.count) : (i += 1) {
|
||||||
const item = rd.entry(i) orelse continue;
|
const item = rd.entry(i) orelse continue;
|
||||||
if (!eql(initial_ramdisk.basename(item.name), "process-test")) continue;
|
if (!eql(initial_ramdisk.basename(item.name), "process-test")) continue;
|
||||||
spinner = process.spawnProcessSupervised(item.blob, 4, &.{ "process-test", "spinner" }, me, endpoint) catch 0;
|
spinner = process.spawnProcessSupervised(item.blob, 4, &.{ item.name, "spinner" }, me, endpoint) catch 0;
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
check("process-test spawned as the supervised spinner victim", spinner != 0);
|
check("process-test spawned as the supervised spinner victim", spinner != 0);
|
||||||
@@ -2220,7 +2324,7 @@ fn vfsClientDeathTest(boot_information: *const BootInformation) void {
|
|||||||
};
|
};
|
||||||
|
|
||||||
process.write_count = 0;
|
process.write_count = 0;
|
||||||
// The full tree: the storage chain must come up for /mnt/usb to exist —
|
// The full tree: the storage chain must come up for /volumes/usb to exist —
|
||||||
// the fat server (not a router) now owns client file state and its sweep.
|
// the fat server (not a router) now owns client file state and its sweep.
|
||||||
process.setInitialRamdisk(image);
|
process.setInitialRamdisk(image);
|
||||||
const init_ok = if (process.spawnBundled("/system/services/init")) true else |_| false;
|
const init_ok = if (process.spawnBundled("/system/services/init")) true else |_| false;
|
||||||
@@ -2295,13 +2399,14 @@ fn signalsTest(boot_information: *const BootInformation) void {
|
|||||||
};
|
};
|
||||||
|
|
||||||
process.setInitialRamdisk(image); // the parent system_spawns its children by name
|
process.setInitialRamdisk(image); // the parent system_spawns its children by name
|
||||||
|
_ = spawnRegistry(rd); // the service child binds /protocol/test/process
|
||||||
process.write_count = 0;
|
process.write_count = 0;
|
||||||
var runner: u32 = 0;
|
var runner: u32 = 0;
|
||||||
var i: u32 = 0;
|
var i: u32 = 0;
|
||||||
while (i < rd.count) : (i += 1) {
|
while (i < rd.count) : (i += 1) {
|
||||||
const item = rd.entry(i) orelse continue;
|
const item = rd.entry(i) orelse continue;
|
||||||
if (!eql(initial_ramdisk.basename(item.name), "process-test")) continue;
|
if (!eql(initial_ramdisk.basename(item.name), "process-test")) continue;
|
||||||
runner = process.spawnProcessSupervised(item.blob, 4, &.{ "process-test", "signal-run" }, scheduler.currentId(), null) catch 0;
|
runner = process.spawnProcessSupervised(item.blob, 4, &.{ item.name, "signal-run" }, scheduler.currentId(), null) catch 0;
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
check("signal-run parent spawned", runner != 0);
|
check("signal-run parent spawned", runner != 0);
|
||||||
@@ -2343,13 +2448,14 @@ fn driverRestartTest(boot_information: *const BootInformation) void {
|
|||||||
};
|
};
|
||||||
|
|
||||||
process.setInitialRamdisk(image); // the manager system_spawns drivers by name
|
process.setInitialRamdisk(image); // the manager system_spawns drivers by name
|
||||||
|
_ = spawnRegistry(rd); // the drivers bind their contracts
|
||||||
process.write_count = 0;
|
process.write_count = 0;
|
||||||
var manager: u32 = 0;
|
var manager: u32 = 0;
|
||||||
var i: u32 = 0;
|
var i: u32 = 0;
|
||||||
while (i < rd.count) : (i += 1) {
|
while (i < rd.count) : (i += 1) {
|
||||||
const item = rd.entry(i) orelse continue;
|
const item = rd.entry(i) orelse continue;
|
||||||
if (!eql(initial_ramdisk.basename(item.name), "device-manager")) continue;
|
if (!eql(initial_ramdisk.basename(item.name), "device-manager")) continue;
|
||||||
manager = process.spawnProcessSupervised(item.blob, 4, &.{ "device-manager", "test-restart" }, scheduler.currentId(), null) catch 0;
|
manager = process.spawnProcessSupervised(item.blob, 4, &.{ item.name, "test-restart" }, scheduler.currentId(), null) catch 0;
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
check("device-manager spawned in test-restart mode", manager != 0);
|
check("device-manager spawned in test-restart mode", manager != 0);
|
||||||
@@ -2382,12 +2488,13 @@ fn usbReportTest(boot_information: *const BootInformation) void {
|
|||||||
};
|
};
|
||||||
|
|
||||||
process.setInitialRamdisk(image);
|
process.setInitialRamdisk(image);
|
||||||
|
_ = spawnRegistry(rd); // the xhci driver binds /protocol/usb-transfer
|
||||||
var manager: u32 = 0;
|
var manager: u32 = 0;
|
||||||
var i: u32 = 0;
|
var i: u32 = 0;
|
||||||
while (i < rd.count) : (i += 1) {
|
while (i < rd.count) : (i += 1) {
|
||||||
const item = rd.entry(i) orelse continue;
|
const item = rd.entry(i) orelse continue;
|
||||||
if (!eql(initial_ramdisk.basename(item.name), "device-manager")) continue;
|
if (!eql(initial_ramdisk.basename(item.name), "device-manager")) continue;
|
||||||
manager = process.spawnProcessSupervised(item.blob, 4, &.{ "device-manager", "test-usb-restart" }, scheduler.currentId(), null) catch 0;
|
manager = process.spawnProcessSupervised(item.blob, 4, &.{ item.name, "test-usb-restart" }, scheduler.currentId(), null) catch 0;
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
check("device-manager spawned in test-usb-restart mode", manager != 0);
|
check("device-manager spawned in test-usb-restart mode", manager != 0);
|
||||||
@@ -2414,12 +2521,13 @@ fn deviceListTest(boot_information: *const BootInformation) void {
|
|||||||
};
|
};
|
||||||
|
|
||||||
process.setInitialRamdisk(image);
|
process.setInitialRamdisk(image);
|
||||||
|
_ = spawnRegistry(rd); // the fixture opens /protocol/device-manager
|
||||||
var manager: u32 = 0;
|
var manager: u32 = 0;
|
||||||
var i: u32 = 0;
|
var i: u32 = 0;
|
||||||
while (i < rd.count) : (i += 1) {
|
while (i < rd.count) : (i += 1) {
|
||||||
const item = rd.entry(i) orelse continue;
|
const item = rd.entry(i) orelse continue;
|
||||||
if (!eql(initial_ramdisk.basename(item.name), "device-manager")) continue;
|
if (!eql(initial_ramdisk.basename(item.name), "device-manager")) continue;
|
||||||
manager = process.spawnProcessSupervised(item.blob, 4, &.{ "device-manager", "test-usb-restart" }, scheduler.currentId(), null) catch 0;
|
manager = process.spawnProcessSupervised(item.blob, 4, &.{ item.name, "test-usb-restart" }, scheduler.currentId(), null) catch 0;
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
check("device-manager spawned in test-usb-restart mode", manager != 0);
|
check("device-manager spawned in test-usb-restart mode", manager != 0);
|
||||||
@@ -2448,13 +2556,14 @@ fn pciCapsTest(boot_information: *const BootInformation) void {
|
|||||||
};
|
};
|
||||||
|
|
||||||
process.setInitialRamdisk(image);
|
process.setInitialRamdisk(image);
|
||||||
|
_ = spawnRegistry(rd); // the manager binds /protocol/device-manager
|
||||||
// Plain mode — no restart drill, whose kill would race the fixture's claim.
|
// Plain mode — no restart drill, whose kill would race the fixture's claim.
|
||||||
var manager: u32 = 0;
|
var manager: u32 = 0;
|
||||||
var i: u32 = 0;
|
var i: u32 = 0;
|
||||||
while (i < rd.count) : (i += 1) {
|
while (i < rd.count) : (i += 1) {
|
||||||
const item = rd.entry(i) orelse continue;
|
const item = rd.entry(i) orelse continue;
|
||||||
if (!eql(initial_ramdisk.basename(item.name), "device-manager")) continue;
|
if (!eql(initial_ramdisk.basename(item.name), "device-manager")) continue;
|
||||||
manager = process.spawnProcessSupervised(item.blob, 4, &.{"device-manager"}, scheduler.currentId(), null) catch 0;
|
manager = process.spawnProcessSupervised(item.blob, 4, &.{item.name}, scheduler.currentId(), null) catch 0;
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
check("device-manager spawned", manager != 0);
|
check("device-manager spawned", manager != 0);
|
||||||
@@ -2482,12 +2591,13 @@ fn iommuFaultTest(boot_information: *const BootInformation) void {
|
|||||||
|
|
||||||
check("IOMMU enabled for the enforcement test", iommu.enabled());
|
check("IOMMU enabled for the enforcement test", iommu.enabled());
|
||||||
process.setInitialRamdisk(image);
|
process.setInitialRamdisk(image);
|
||||||
|
_ = spawnRegistry(rd); // the manager binds /protocol/device-manager
|
||||||
var manager: u32 = 0;
|
var manager: u32 = 0;
|
||||||
var i: u32 = 0;
|
var i: u32 = 0;
|
||||||
while (i < rd.count) : (i += 1) {
|
while (i < rd.count) : (i += 1) {
|
||||||
const item = rd.entry(i) orelse continue;
|
const item = rd.entry(i) orelse continue;
|
||||||
if (!eql(initial_ramdisk.basename(item.name), "device-manager")) continue;
|
if (!eql(initial_ramdisk.basename(item.name), "device-manager")) continue;
|
||||||
manager = process.spawnProcessSupervised(item.blob, 4, &.{"device-manager"}, scheduler.currentId(), null) catch 0;
|
manager = process.spawnProcessSupervised(item.blob, 4, &.{item.name}, scheduler.currentId(), null) catch 0;
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
check("device-manager spawned", manager != 0);
|
check("device-manager spawned", manager != 0);
|
||||||
@@ -2523,12 +2633,13 @@ fn pciScanTest(boot_information: *const BootInformation) void {
|
|||||||
check("the kernel seeded no PCI functions (the walk retired)", brokerPciCount(&buffer) == 0);
|
check("the kernel seeded no PCI functions (the walk retired)", brokerPciCount(&buffer) == 0);
|
||||||
|
|
||||||
process.setInitialRamdisk(image);
|
process.setInitialRamdisk(image);
|
||||||
|
_ = spawnRegistry(rd); // the manager binds /protocol/device-manager
|
||||||
var manager: u32 = 0;
|
var manager: u32 = 0;
|
||||||
var i: u32 = 0;
|
var i: u32 = 0;
|
||||||
while (i < rd.count) : (i += 1) {
|
while (i < rd.count) : (i += 1) {
|
||||||
const item = rd.entry(i) orelse continue;
|
const item = rd.entry(i) orelse continue;
|
||||||
if (!eql(initial_ramdisk.basename(item.name), "device-manager")) continue;
|
if (!eql(initial_ramdisk.basename(item.name), "device-manager")) continue;
|
||||||
manager = process.spawnProcessSupervised(item.blob, 4, &.{ "device-manager", "test-pci-restart" }, scheduler.currentId(), null) catch 0;
|
manager = process.spawnProcessSupervised(item.blob, 4, &.{ item.name, "test-pci-restart" }, scheduler.currentId(), null) catch 0;
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
check("device-manager spawned (test-pci-restart mode)", manager != 0);
|
check("device-manager spawned (test-pci-restart mode)", manager != 0);
|
||||||
@@ -2606,7 +2717,7 @@ fn usbStorageTest(boot_information: *const BootInformation) void {
|
|||||||
|
|
||||||
/// The FAT mount chain: boot the full tree (init spawns the fat server, which
|
/// The FAT mount chain: boot the full tree (init spawns the fat server, which
|
||||||
/// brings up the USB storage chain, mounts the FAT volume, and mounts itself into
|
/// brings up the USB storage chain, mounts the FAT volume, and mounts itself into
|
||||||
/// the VFS at /mnt/usb), then spawn a fat-test client that lists and reads through
|
/// the VFS at /volumes/usb), then spawn a fat-test client that lists and reads through
|
||||||
/// the mount. The harness attaches a usb-storage device; the expect regex requires
|
/// the mount. The harness attaches a usb-storage device; the expect regex requires
|
||||||
/// the fat mount and the client's success.
|
/// the fat mount and the client's success.
|
||||||
fn fatMountTest(boot_information: *const BootInformation) void {
|
fn fatMountTest(boot_information: *const BootInformation) void {
|
||||||
@@ -2676,12 +2787,13 @@ fn acpiReportTest(boot_information: *const BootInformation) void {
|
|||||||
return;
|
return;
|
||||||
};
|
};
|
||||||
process.setInitialRamdisk(image);
|
process.setInitialRamdisk(image);
|
||||||
|
_ = spawnRegistry(rd); // the manager and the acpi service bind theirs
|
||||||
var spawned = false;
|
var spawned = false;
|
||||||
var i: u32 = 0;
|
var i: u32 = 0;
|
||||||
while (i < rd.count) : (i += 1) {
|
while (i < rd.count) : (i += 1) {
|
||||||
const item = rd.entry(i) orelse continue;
|
const item = rd.entry(i) orelse continue;
|
||||||
if (!eql(initial_ramdisk.basename(item.name), "device-manager")) continue;
|
if (!eql(initial_ramdisk.basename(item.name), "device-manager")) continue;
|
||||||
_ = process.spawnProcessSupervised(item.blob, 4, &.{"device-manager"}, scheduler.currentId(), null) catch 0;
|
_ = process.spawnProcessSupervised(item.blob, 4, &.{item.name}, scheduler.currentId(), null) catch 0;
|
||||||
spawned = true;
|
spawned = true;
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
@@ -2719,7 +2831,7 @@ fn acpiParseTest(boot_information: *const BootInformation) void {
|
|||||||
while (i < rd.count) : (i += 1) {
|
while (i < rd.count) : (i += 1) {
|
||||||
const item = rd.entry(i) orelse continue;
|
const item = rd.entry(i) orelse continue;
|
||||||
if (!eql(initial_ramdisk.basename(item.name), "discovery")) continue;
|
if (!eql(initial_ramdisk.basename(item.name), "discovery")) continue;
|
||||||
_ = process.spawnProcessSupervised(item.blob, 4, &.{ "discovery", "1" }, scheduler.currentId(), null) catch 0;
|
_ = process.spawnProcessSupervised(item.blob, 4, &.{ item.name, "1" }, scheduler.currentId(), null) catch 0;
|
||||||
spawned = true;
|
spawned = true;
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
@@ -2754,7 +2866,7 @@ fn supervisionTest(boot_information: *const BootInformation) void {
|
|||||||
while (i < rd.count) : (i += 1) {
|
while (i < rd.count) : (i += 1) {
|
||||||
const item = rd.entry(i) orelse continue;
|
const item = rd.entry(i) orelse continue;
|
||||||
if (!eql(initial_ramdisk.basename(item.name), "process-test")) continue;
|
if (!eql(initial_ramdisk.basename(item.name), "process-test")) continue;
|
||||||
started = if (process.spawnProcess(item.blob, 4, &.{ "process-test", "run" })) true else |_| false;
|
started = if (process.spawnProcess(item.blob, 4, &.{ item.name, "run" })) true else |_| false;
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
check("process-test spawned as the user-space supervisor", started);
|
check("process-test spawned as the user-space supervisor", started);
|
||||||
@@ -2801,11 +2913,13 @@ fn initialRamdiskTest(boot_information: *const BootInformation) void {
|
|||||||
var i: u32 = 0;
|
var i: u32 = 0;
|
||||||
while (i < rd.count) : (i += 1) {
|
while (i < rd.count) : (i += 1) {
|
||||||
const item = rd.entry(i) orelse continue;
|
const item = rd.entry(i) orelse continue;
|
||||||
// The FHS boot tree ferries data files too (/etc/devices.csv,
|
// The boot tree ferries data files too (/system/configuration/devices.csv,
|
||||||
// /etc/init.csv — served read-only by the kernel VFS, never spawned);
|
// /system/configuration/init.csv — served read-only by the kernel VFS,
|
||||||
// only the /system and /test trees hold programs, so only those count
|
// never spawned); only the /system and /test trees hold programs, and
|
||||||
// toward the spawn-everything sweep.
|
// /system/configuration holds none, so only the rest counts toward the
|
||||||
const is_program = std.mem.startsWith(u8, item.name, "/system/") or
|
// spawn-everything sweep.
|
||||||
|
const is_program = (std.mem.startsWith(u8, item.name, "/system/") and
|
||||||
|
!std.mem.startsWith(u8, item.name, "/system/configuration/")) or
|
||||||
std.mem.startsWith(u8, item.name, "/test/");
|
std.mem.startsWith(u8, item.name, "/test/");
|
||||||
if (!is_program) continue;
|
if (!is_program) continue;
|
||||||
programs += 1;
|
programs += 1;
|
||||||
@@ -2894,6 +3008,9 @@ fn inputTest(boot_information: *const BootInformation) void {
|
|||||||
|
|
||||||
process.write_count = 0;
|
process.write_count = 0;
|
||||||
process.write_from_user = false;
|
process.write_from_user = false;
|
||||||
|
// init (the registry, below) reads its manifests through the kernel VFS.
|
||||||
|
process.setInitialRamdisk(image);
|
||||||
|
_ = spawnRegistry(rd); // the input service binds /protocol/input
|
||||||
_ = spawnNamed(rd, "input"); // the fan-out service
|
_ = spawnNamed(rd, "input"); // the fan-out service
|
||||||
_ = spawnNamed(rd, "input-source"); // a synthetic keyboard publishing events
|
_ = spawnNamed(rd, "input-source"); // a synthetic keyboard publishing events
|
||||||
_ = spawnNamed(rd, "input-test"); // the subscriber whose "ok" line is the marker
|
_ = spawnNamed(rd, "input-test"); // the subscriber whose "ok" line is the marker
|
||||||
@@ -2935,6 +3052,10 @@ fn displayServiceTest(boot_information: *const BootInformation) void {
|
|||||||
return;
|
return;
|
||||||
};
|
};
|
||||||
|
|
||||||
|
// init (the registry) reads its manifests through the kernel VFS.
|
||||||
|
process.setInitialRamdisk(image);
|
||||||
|
_ = spawnRegistry(rd); // the compositor binds /protocol/display
|
||||||
|
|
||||||
// Spawn the compositor and hand it the core. Its own serial heartbeats — `display:
|
// Spawn the compositor and hand it the core. Its own serial heartbeats — `display:
|
||||||
// online WxH` and `display: presented frame 0` — are what the harness matches (it
|
// online WxH` and `display: presented frame 0` — are what the harness matches (it
|
||||||
// reads serial directly, like the fault cases). We don't poll for them in-kernel: a
|
// reads serial directly, like the fault cases). We don't poll for them in-kernel: a
|
||||||
@@ -2971,6 +3092,9 @@ fn displayCursorTest(boot_information: *const BootInformation) void {
|
|||||||
return;
|
return;
|
||||||
};
|
};
|
||||||
|
|
||||||
|
// init (the registry, below) reads its manifests through the kernel VFS.
|
||||||
|
process.setInitialRamdisk(image);
|
||||||
|
_ = spawnRegistry(rd); // input and display bind theirs
|
||||||
if (!spawnNamed(rd, "input")) {
|
if (!spawnNamed(rd, "input")) {
|
||||||
log("display-cursor: could not spawn the input service\n", .{});
|
log("display-cursor: could not spawn the input service\n", .{});
|
||||||
result();
|
result();
|
||||||
@@ -3011,6 +3135,9 @@ fn displayDemoTest(boot_information: *const BootInformation) void {
|
|||||||
return;
|
return;
|
||||||
};
|
};
|
||||||
|
|
||||||
|
// init (the registry, below) reads its manifests through the kernel VFS.
|
||||||
|
process.setInitialRamdisk(image);
|
||||||
|
_ = spawnRegistry(rd); // the compositor binds /protocol/display
|
||||||
if (!spawnNamed(rd, "display")) {
|
if (!spawnNamed(rd, "display")) {
|
||||||
log("display-demo: could not spawn the display service\n", .{});
|
log("display-demo: could not spawn the display service\n", .{});
|
||||||
result();
|
result();
|
||||||
@@ -3046,6 +3173,9 @@ fn sharedMemoryTest(boot_information: *const BootInformation) void {
|
|||||||
return;
|
return;
|
||||||
};
|
};
|
||||||
|
|
||||||
|
// init (the registry, below) reads its manifests through the kernel VFS.
|
||||||
|
process.setInitialRamdisk(image);
|
||||||
|
_ = spawnRegistry(rd); // the server binds /protocol/test/shared-memory
|
||||||
if (!spawnNamed(rd, "shared-memory-server")) {
|
if (!spawnNamed(rd, "shared-memory-server")) {
|
||||||
log("shared-memory: could not spawn shared-memory-server\n", .{});
|
log("shared-memory: could not spawn shared-memory-server\n", .{});
|
||||||
result();
|
result();
|
||||||
@@ -3083,12 +3213,13 @@ fn virtioGpuTest(boot_information: *const BootInformation) void {
|
|||||||
// from the kernel device tree, spawns pci-bus, and matches the virtio-gpu class triple to
|
// from the kernel device tree, spawns pci-bus, and matches the virtio-gpu class triple to
|
||||||
// spawn our driver with the function's device id as argv[1].
|
// spawn our driver with the function's device id as argv[1].
|
||||||
process.setInitialRamdisk(image);
|
process.setInitialRamdisk(image);
|
||||||
|
_ = spawnRegistry(rd); // the driver binds /protocol/scanout
|
||||||
var manager: u32 = 0;
|
var manager: u32 = 0;
|
||||||
var i: u32 = 0;
|
var i: u32 = 0;
|
||||||
while (i < rd.count) : (i += 1) {
|
while (i < rd.count) : (i += 1) {
|
||||||
const item = rd.entry(i) orelse continue;
|
const item = rd.entry(i) orelse continue;
|
||||||
if (!eql(initial_ramdisk.basename(item.name), "device-manager")) continue;
|
if (!eql(initial_ramdisk.basename(item.name), "device-manager")) continue;
|
||||||
manager = process.spawnProcessSupervised(item.blob, 4, &.{"device-manager"}, scheduler.currentId(), null) catch 0;
|
manager = process.spawnProcessSupervised(item.blob, 4, &.{item.name}, scheduler.currentId(), null) catch 0;
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
if (manager == 0) {
|
if (manager == 0) {
|
||||||
@@ -3124,12 +3255,13 @@ fn displayNativeTest(boot_information: *const BootInformation) void {
|
|||||||
};
|
};
|
||||||
|
|
||||||
process.setInitialRamdisk(image);
|
process.setInitialRamdisk(image);
|
||||||
|
_ = spawnRegistry(rd); // display, the manager, and the driver bind theirs
|
||||||
var manager: u32 = 0;
|
var manager: u32 = 0;
|
||||||
var i: u32 = 0;
|
var i: u32 = 0;
|
||||||
while (i < rd.count) : (i += 1) {
|
while (i < rd.count) : (i += 1) {
|
||||||
const item = rd.entry(i) orelse continue;
|
const item = rd.entry(i) orelse continue;
|
||||||
if (!eql(initial_ramdisk.basename(item.name), "device-manager")) continue;
|
if (!eql(initial_ramdisk.basename(item.name), "device-manager")) continue;
|
||||||
manager = process.spawnProcessSupervised(item.blob, 4, &.{"device-manager"}, scheduler.currentId(), null) catch 0;
|
manager = process.spawnProcessSupervised(item.blob, 4, &.{item.name}, scheduler.currentId(), null) catch 0;
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
if (manager == 0) {
|
if (manager == 0) {
|
||||||
@@ -3168,12 +3300,13 @@ fn displayReattachTest(boot_information: *const BootInformation) void {
|
|||||||
};
|
};
|
||||||
|
|
||||||
process.setInitialRamdisk(image);
|
process.setInitialRamdisk(image);
|
||||||
|
_ = spawnRegistry(rd); // display and the restarted driver bind theirs
|
||||||
var manager: u32 = 0;
|
var manager: u32 = 0;
|
||||||
var i: u32 = 0;
|
var i: u32 = 0;
|
||||||
while (i < rd.count) : (i += 1) {
|
while (i < rd.count) : (i += 1) {
|
||||||
const item = rd.entry(i) orelse continue;
|
const item = rd.entry(i) orelse continue;
|
||||||
if (!eql(initial_ramdisk.basename(item.name), "device-manager")) continue;
|
if (!eql(initial_ramdisk.basename(item.name), "device-manager")) continue;
|
||||||
manager = process.spawnProcessSupervised(item.blob, 4, &.{ "device-manager", "test-scanout-restart" }, scheduler.currentId(), null) catch 0;
|
manager = process.spawnProcessSupervised(item.blob, 4, &.{ item.name, "test-scanout-restart" }, scheduler.currentId(), null) catch 0;
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
if (manager == 0) {
|
if (manager == 0) {
|
||||||
@@ -3267,21 +3400,23 @@ fn kernelVfsTest(boot_information: *const BootInformation) void {
|
|||||||
check("its first bytes are an ELF magic", n == 4 and header[0] == 0x7f and header[1] == 'E' and header[2] == 'L' and header[3] == 'F');
|
check("its first bytes are an ELF magic", n == 4 and header[0] == 0x7f and header[1] == 'E' and header[2] == 'L' and header[3] == 'F');
|
||||||
}
|
}
|
||||||
|
|
||||||
// Directories resolve and enumerate: /system lists services/drivers.
|
// Directories resolve and enumerate: /system lists services/drivers/
|
||||||
|
// configuration (the CSV data files ride the same initrd tree).
|
||||||
const root_directory = kernel_vfs.resolvePath("/system", false);
|
const root_directory = kernel_vfs.resolvePath("/system", false);
|
||||||
check("/system resolves to a directory node", root_directory == .kernel_node);
|
check("/system resolves to a directory node", root_directory == .kernel_node);
|
||||||
var saw_services = false;
|
var saw_services = false;
|
||||||
var saw_drivers = false;
|
var saw_drivers = false;
|
||||||
|
var saw_configuration = false;
|
||||||
var saw_stray_in_root = false;
|
var saw_stray_in_root = false;
|
||||||
var saw_files_in_services = false;
|
var saw_files_in_services = false;
|
||||||
if (root_directory == .kernel_node) {
|
if (root_directory == .kernel_node) {
|
||||||
var cursor: u64 = 0;
|
var cursor: u64 = 0;
|
||||||
var name: [64]u8 = undefined;
|
var name: [64]u8 = undefined;
|
||||||
while (kernel_vfs.nodeReaddir(root_directory.kernel_node, cursor, &name)) |entry| : (cursor += 1) {
|
while (kernel_vfs.nodeReaddir(root_directory.kernel_node, cursor, &name)) |entry| : (cursor += 1) {
|
||||||
if (eql(name[0..entry.name_len], "services")) saw_services = true else if (eql(name[0..entry.name_len], "drivers")) saw_drivers = true else saw_stray_in_root = true;
|
if (eql(name[0..entry.name_len], "services")) saw_services = true else if (eql(name[0..entry.name_len], "drivers")) saw_drivers = true else if (eql(name[0..entry.name_len], "configuration")) saw_configuration = true else saw_stray_in_root = true;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
check("readdir /system yields services and drivers", saw_services and saw_drivers);
|
check("readdir /system yields services, drivers, configuration", saw_services and saw_drivers and saw_configuration);
|
||||||
check("readdir /system yields nothing else (no /test leakage)", !saw_stray_in_root);
|
check("readdir /system yields nothing else (no /test leakage)", !saw_stray_in_root);
|
||||||
const services = kernel_vfs.resolvePath("/system/services", false);
|
const services = kernel_vfs.resolvePath("/system/services", false);
|
||||||
if (services == .kernel_node) {
|
if (services == .kernel_node) {
|
||||||
@@ -3521,6 +3656,21 @@ fn threadTestMarkerCase(boot_information: *const BootInformation, case_name: []c
|
|||||||
result();
|
result();
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Bring up the protocol namespace for a scenario that spawns its providers
|
||||||
|
/// itself. `/protocol` is served by init, PID 1 — but a scenario case wants the
|
||||||
|
/// naming layer without init's whole service list underneath it, so init is
|
||||||
|
/// started in its `registry` role: it mounts `/protocol`, reads the grants, and
|
||||||
|
/// spawns nothing (docs/os-development/protocol-namespace.md; the plan's
|
||||||
|
/// decision 9). Providers retry their bind, so racing the mount is survivable —
|
||||||
|
/// but calling this first makes the race rare.
|
||||||
|
///
|
||||||
|
/// The caller must have published the initial ramdisk already
|
||||||
|
/// (`process.setInitialRamdisk`): init reads its manifests out of it, and every
|
||||||
|
/// `/protocol` resolve goes through the same kernel VFS.
|
||||||
|
fn spawnRegistry(rd: initial_ramdisk.Reader) bool {
|
||||||
|
return spawnNamedWithArg(rd, "init", "registry");
|
||||||
|
}
|
||||||
|
|
||||||
fn spawnNamed(rd: initial_ramdisk.Reader, name: []const u8) bool {
|
fn spawnNamed(rd: initial_ramdisk.Reader, name: []const u8) bool {
|
||||||
var i: u32 = 0;
|
var i: u32 = 0;
|
||||||
while (i < rd.count) : (i += 1) {
|
while (i < rd.count) : (i += 1) {
|
||||||
@@ -3641,6 +3791,113 @@ fn childDescriptor(hid: []const u8, start: u64, len: u64) device_abi.DeviceDescr
|
|||||||
/// match `pci-bus`, and spawn it (with the bridge id as its argument) — and the spawned
|
/// match `pci-bus`, and spawn it (with the bridge id as its argument) — and the spawned
|
||||||
/// pci-bus must reach its own live marker. It uses no special privilege — the same
|
/// pci-bus must reach its own live marker. It uses no special privilege — the same
|
||||||
/// `device_enumerate` any process could call.
|
/// `device_enumerate` any process could call.
|
||||||
|
/// P2 — the registrar (docs/os-development/protocol-namespace.md). Bring up
|
||||||
|
/// `/protocol` (init in its registry role) and hand the fixture the core: it
|
||||||
|
/// asserts that an ungranted bind is refused, that the kernel's reserved prefix
|
||||||
|
/// holds, that a name a live provider holds cannot be taken, and that killing a
|
||||||
|
/// provider makes its channel fail while re-resolving the same name reaches the
|
||||||
|
/// restarted instance.
|
||||||
|
///
|
||||||
|
/// It doubles as the security case for PID 1's shared mailbox, since resolving
|
||||||
|
/// `/protocol` hands every process a sendable handle to it: a forged power
|
||||||
|
/// payload, a redirected terminate signal, a timer or exit subscription armed on
|
||||||
|
/// a foreign endpoint, and capability-carrying ping storms against both PID 1 and
|
||||||
|
/// a harness-run service. Those assertions kill the boot when they regress rather
|
||||||
|
/// than printing anything, which is the strongest form available here.
|
||||||
|
///
|
||||||
|
/// The fixture's `protocol-registry: ok` is the marker; each step also prints its
|
||||||
|
/// own line, which the harness's ordered regex reads.
|
||||||
|
fn protocolRegistryTest(boot_information: *const BootInformation) void {
|
||||||
|
log("DANOS-TEST-BEGIN: protocol-registry\n", .{});
|
||||||
|
if (boot_information.initial_ramdisk_len == 0) {
|
||||||
|
check("bootloader handed over an initial_ramdisk", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||||
|
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||||
|
check("initial_ramdisk image is valid", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
|
||||||
|
// The fixture spawns its own providers by name, so the ramdisk must be
|
||||||
|
// published; init then mounts /protocol over the same kernel VFS.
|
||||||
|
process.setInitialRamdisk(image);
|
||||||
|
check("registry (init) spawned", spawnRegistry(rd));
|
||||||
|
check("protocol-registry-test spawned", spawnNamedWithArg(rd, "protocol-registry-test", "run"));
|
||||||
|
|
||||||
|
const pass_marker = "protocol-registry: ok";
|
||||||
|
const fail_marker = "protocol-registry: FAIL";
|
||||||
|
scheduler.setPriority(1);
|
||||||
|
const deadline = architecture.millis() + 20000;
|
||||||
|
var saw_pass = false;
|
||||||
|
var saw_fail = false;
|
||||||
|
while (architecture.millis() < deadline and !saw_pass and !saw_fail) {
|
||||||
|
if (bufferHas(pass_marker)) saw_pass = true;
|
||||||
|
if (bufferHas(fail_marker)) saw_fail = true;
|
||||||
|
scheduler.yield();
|
||||||
|
}
|
||||||
|
scheduler.setPriority(4);
|
||||||
|
|
||||||
|
check("no step of the registry contract failed", !saw_fail);
|
||||||
|
check("the fixture completed every registry assertion", saw_pass);
|
||||||
|
result();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// P3 — restriction stage one (docs/os-development/protocol-namespace.md). The
|
||||||
|
/// registrar now checks `open` against `/system/configuration/protocol.csv`, and
|
||||||
|
/// a caller with no grant is told exactly what a caller asking for a name nobody
|
||||||
|
/// bound is told.
|
||||||
|
///
|
||||||
|
/// The scenario is the assertion's scaffolding: `/protocol` (init in its registry
|
||||||
|
/// role), the **input service** — which binds a real contract the fixture is
|
||||||
|
/// deliberately not granted — and the fixture. Without a live provider on the
|
||||||
|
/// forbidden name, "refused" and "not bound yet" would be the same observation
|
||||||
|
/// and the case would prove nothing; the fixture reads `/protocol`'s own listing
|
||||||
|
/// to confirm the name is there before it asks for it.
|
||||||
|
///
|
||||||
|
/// The fixture's `protocol-denied: ok` is the marker; each step prints its own
|
||||||
|
/// line, which the harness's ordered regex reads.
|
||||||
|
fn protocolDeniedTest(boot_information: *const BootInformation) void {
|
||||||
|
log("DANOS-TEST-BEGIN: protocol-denied\n", .{});
|
||||||
|
if (boot_information.initial_ramdisk_len == 0) {
|
||||||
|
check("bootloader handed over an initial_ramdisk", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||||
|
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||||
|
check("initial_ramdisk image is valid", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
|
||||||
|
process.setInitialRamdisk(image);
|
||||||
|
check("registry (init) spawned", spawnRegistry(rd));
|
||||||
|
// The provider of the contract the fixture may NOT reach. It needs no
|
||||||
|
// hardware: it binds /protocol/input and waits for subscribers.
|
||||||
|
check("input service spawned", spawnNamed(rd, "input"));
|
||||||
|
check("protocol-denied-test spawned", spawnNamedWithArg(rd, "protocol-denied-test", "run"));
|
||||||
|
|
||||||
|
const pass_marker = "protocol-denied: ok";
|
||||||
|
const fail_marker = "protocol-denied: FAIL";
|
||||||
|
scheduler.setPriority(1);
|
||||||
|
const deadline = architecture.millis() + 20000;
|
||||||
|
var saw_pass = false;
|
||||||
|
var saw_fail = false;
|
||||||
|
while (architecture.millis() < deadline and !saw_pass and !saw_fail) {
|
||||||
|
if (bufferHas(pass_marker)) saw_pass = true;
|
||||||
|
if (bufferHas(fail_marker)) saw_fail = true;
|
||||||
|
scheduler.yield();
|
||||||
|
}
|
||||||
|
scheduler.setPriority(4);
|
||||||
|
|
||||||
|
check("no step of the restriction contract failed", !saw_fail);
|
||||||
|
check("the fixture completed every restriction assertion", saw_pass);
|
||||||
|
result();
|
||||||
|
}
|
||||||
|
|
||||||
fn deviceManagerTest(boot_information: *const BootInformation) void {
|
fn deviceManagerTest(boot_information: *const BootInformation) void {
|
||||||
log("DANOS-TEST-BEGIN: device-manager\n", .{});
|
log("DANOS-TEST-BEGIN: device-manager\n", .{});
|
||||||
if (boot_information.initial_ramdisk_len == 0) {
|
if (boot_information.initial_ramdisk_len == 0) {
|
||||||
@@ -3660,6 +3917,7 @@ fn deviceManagerTest(boot_information: *const BootInformation) void {
|
|||||||
// all, it's because the manager discovered the PCI host bridge, matched, and
|
// all, it's because the manager discovered the PCI host bridge, matched, and
|
||||||
// spawned it.
|
// spawned it.
|
||||||
process.setInitialRamdisk(image);
|
process.setInitialRamdisk(image);
|
||||||
|
_ = spawnRegistry(rd); // the manager binds /protocol/device-manager
|
||||||
|
|
||||||
process.write_count = 0;
|
process.write_count = 0;
|
||||||
process.write_from_user = false;
|
process.write_from_user = false;
|
||||||
@@ -3741,9 +3999,9 @@ fn hpetDeviceId() ?u64 {
|
|||||||
/// 2. After `releaseOwner` for the binding's owner, that same entry is masked again.
|
/// 2. After `releaseOwner` for the binding's owner, that same entry is masked again.
|
||||||
///
|
///
|
||||||
/// And one property that can only be checked from kernel state: a *different* owner's
|
/// And one property that can only be checked from kernel state: a *different* owner's
|
||||||
/// binding on the same endpoint survives. Endpoints are shared (ipc_register hands out
|
/// binding on the same endpoint survives. Endpoints are shared (a capability passed in a
|
||||||
/// references), so teardown keyed on the endpoint pointer rather than the owning task
|
/// message hands out extra references), so teardown keyed on the endpoint pointer rather
|
||||||
/// would mask a live sibling driver's device line.
|
/// than the owning task would mask a live sibling driver's device line.
|
||||||
fn irqFreeTest() void {
|
fn irqFreeTest() void {
|
||||||
log("DANOS-TEST-BEGIN: irqfree\n", .{});
|
log("DANOS-TEST-BEGIN: irqfree\n", .{});
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,117 @@
|
|||||||
|
//! The single trusted door between ring 0 and a process's memory.
|
||||||
|
//!
|
||||||
|
//! Kernel code never dereferences a user virtual address. It walks that address
|
||||||
|
//! space's page tables through the physmap — kernel mappings throughout — and
|
||||||
|
//! moves the bytes there. Three properties fall out of that one decision:
|
||||||
|
//!
|
||||||
|
//! - **A bad pointer fails the system call.** danos has no fault-recovering
|
||||||
|
//! copy-in, so a raw dereference of an unmapped-but-in-range user page would
|
||||||
|
//! halt the machine. Here it is a `false` return and an `-EFAULT`.
|
||||||
|
//! - **The copy is a single fetch.** A struct pulled in once cannot be changed
|
||||||
|
//! underneath the checks that follow it — no TOCTOU against a hostile pointer.
|
||||||
|
//! - **It is SMAP-proof by construction.** No ring-0 access to a user-mapped
|
||||||
|
//! page ever happens, so the CR4.SMAP bit needs no `stac` window anywhere
|
||||||
|
//! (docs/os-development/smep-smap.md). There is no `stac` in this tree, and a
|
||||||
|
//! change that adds one is wrong by definition.
|
||||||
|
//!
|
||||||
|
//! The walk enforces the permissions ring 3 itself would face: a read needs the
|
||||||
|
//! leaf user-accessible (U/S at every level), a write needs it writable too
|
||||||
|
//! (R/W at every level). So a syscall argument cannot steer the kernel at a
|
||||||
|
//! kernel-only mapping, nor make it write a process's own read-only text — the
|
||||||
|
//! properties that matter once shared or copy-on-write mappings exist, and the
|
||||||
|
//! reason the "presence only" caveat that used to sit at the top of
|
||||||
|
//! ipc-synchronous.zig is gone.
|
||||||
|
//!
|
||||||
|
//! Scope: this module knows only about *user* address spaces. The kernel side of
|
||||||
|
//! a copy (an IPC reply staged in kernel memory, a bounce buffer) is trusted and
|
||||||
|
//! translated without permission checks — see `resolve`.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const boot_handoff = @import("boot-handoff");
|
||||||
|
const abi = @import("abi");
|
||||||
|
const architecture = @import("architecture");
|
||||||
|
|
||||||
|
const page_size = abi.page_size;
|
||||||
|
|
||||||
|
/// End of the user (low) canonical half. Every user buffer must lie below it, so
|
||||||
|
/// kernel addresses and non-canonical values are refused by the range check
|
||||||
|
/// alone, before any table is read.
|
||||||
|
pub const user_half_end: u64 = 0x0000_8000_0000_0000;
|
||||||
|
|
||||||
|
/// Whether `[virtual, virtual + len)` lies wholly inside the user half. The
|
||||||
|
/// length is compared against the remaining span rather than added to the base,
|
||||||
|
/// so a huge `len` cannot wrap the check.
|
||||||
|
pub fn userRangeOk(virtual: u64, len: usize) bool {
|
||||||
|
if (virtual >= user_half_end) return false;
|
||||||
|
return len <= user_half_end - virtual;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Resolve one address for a copy. `address_space == 0` means the kernel's own
|
||||||
|
/// tables — trusted, translated as-is. A real address space is a process's, and
|
||||||
|
/// the walk demands what ring 3 would need: user-accessible, plus writable when
|
||||||
|
/// this side of the copy is the destination.
|
||||||
|
///
|
||||||
|
/// A user frame must also be reachable *through the physmap*, because that is how
|
||||||
|
/// the copy loops touch it. The physmap covers RAM only: `paging.init` skips every
|
||||||
|
/// `.mmio` region, while `mmio_map` hands a driver its device's BAR as an ordinary
|
||||||
|
/// user-accessible mapping. Such a page satisfies the permission walk and would
|
||||||
|
/// then fault ring 0 on the physmap alias — the very #PF this layer exists to make
|
||||||
|
/// impossible — so coverage is confirmed before the address is returned, and an
|
||||||
|
/// uncovered frame is refused like any other bad buffer. (Confirming coverage,
|
||||||
|
/// rather than testing the `device_grant` bit, is what keeps physmap-backed RAM
|
||||||
|
/// that merely carries that bit — a scanout surface — usable as a buffer.)
|
||||||
|
pub fn resolve(address_space: u64, virtual: u64, for_write: bool) ?u64 {
|
||||||
|
if (address_space == 0) return architecture.translate(architecture.kernelPageTable(), virtual);
|
||||||
|
const physical = architecture.translateUser(address_space, virtual, for_write) orelse return null;
|
||||||
|
if (architecture.translate(architecture.kernelPageTable(), boot_handoff.physicalToVirtual(physical)) == null) return null;
|
||||||
|
return physical;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Copy `destination.len` bytes from `user_va` in address space `user_as` into the
|
||||||
|
/// kernel buffer `destination`. False — never a #PF — if the range escapes the
|
||||||
|
/// user half, or any source page is unmapped or not readable from ring 3.
|
||||||
|
/// Handles page-straddling buffers.
|
||||||
|
pub fn copyFromUser(user_as: u64, user_va: u64, destination: []u8) bool {
|
||||||
|
if (user_as == 0) return false; // not a user address space
|
||||||
|
if (!userRangeOk(user_va, destination.len)) return false;
|
||||||
|
var off: usize = 0;
|
||||||
|
while (off < destination.len) {
|
||||||
|
const physical = resolve(user_as, user_va + off, false) orelse return false;
|
||||||
|
const left = page_size - ((user_va + off) & (page_size - 1));
|
||||||
|
const n = @min(left, destination.len - off);
|
||||||
|
const source: [*]const u8 = @ptrFromInt(boot_handoff.physicalToVirtual(physical));
|
||||||
|
@memcpy(destination[off..][0..n], source[0..n]);
|
||||||
|
off += n;
|
||||||
|
}
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The write direction: copy the kernel buffer `source` out to `user_va` in
|
||||||
|
/// address space `user_as`. False — never a #PF, never a partial promise — if the
|
||||||
|
/// range escapes the user half, or any destination page is unmapped, kernel-only,
|
||||||
|
/// or read-only for ring 3. (A refusal mid-way may already have written earlier
|
||||||
|
/// pages; the caller fails the whole system call, so the buffer's contents are
|
||||||
|
/// meaningless either way.)
|
||||||
|
///
|
||||||
|
/// The mirror of `copyFromUser`, and the only way kernel data reaches a user
|
||||||
|
/// buffer outside the IPC path's `copyAcross`.
|
||||||
|
pub fn copyToUser(user_as: u64, user_va: u64, source: []const u8) bool {
|
||||||
|
if (user_as == 0) return false; // not a user address space
|
||||||
|
if (!userRangeOk(user_va, source.len)) return false;
|
||||||
|
var off: usize = 0;
|
||||||
|
while (off < source.len) {
|
||||||
|
const physical = resolve(user_as, user_va + off, true) orelse return false;
|
||||||
|
const left = page_size - ((user_va + off) & (page_size - 1));
|
||||||
|
const n = @min(left, source.len - off);
|
||||||
|
const destination: [*]u8 = @ptrFromInt(boot_handoff.physicalToVirtual(physical));
|
||||||
|
@memcpy(destination[0..n], source[off..][0..n]);
|
||||||
|
off += n;
|
||||||
|
}
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `copyToUser` for any fixed-layout value — the shape most write-direction
|
||||||
|
/// system calls want (a `KlogStatus`, a `FileAttributes`).
|
||||||
|
pub fn copyValueToUser(user_as: u64, user_va: u64, value: anytype) bool {
|
||||||
|
return copyToUser(user_as, user_va, std.mem.asBytes(value));
|
||||||
|
}
|
||||||
+71
-15
@@ -7,7 +7,8 @@
|
|||||||
//! mount (the initrd trees at /system and /test, the scratch ram nodes) resolves to a
|
//! mount (the initrd trees at /system and /test, the scratch ram nodes) resolves to a
|
||||||
//! stateless node TOKEN served directly by `fs_node` (read/status/readdir
|
//! stateless node TOKEN served directly by `fs_node` (read/status/readdir
|
||||||
//! with copy-out). A path under a USERSPACE mount (the fat server at
|
//! with copy-out). A path under a USERSPACE mount (the fat server at
|
||||||
//! /mnt/usb and /var) resolves to the backend's ENDPOINT: the kernel
|
//! /volumes/usb, /system/configuration, and /system/logs) resolves to the
|
||||||
|
//! backend's ENDPOINT: the kernel
|
||||||
//! installs a (deduplicated) handle in the caller's table, rewrites the
|
//! installs a (deduplicated) handle in the caller's table, rewrites the
|
||||||
//! path mount-relative, and the caller speaks the unchanged vfs-protocol
|
//! path mount-relative, and the caller speaks the unchanged vfs-protocol
|
||||||
//! to the backend over the ordinary ipc_call rendezvous. The kernel never
|
//! to the backend over the ordinary ipc_call rendezvous. The kernel never
|
||||||
@@ -21,9 +22,10 @@
|
|||||||
//! Mounting is `fs_mount(prefix, backend_handle, rewrite)`: possession of the
|
//! Mounting is `fs_mount(prefix, backend_handle, rewrite)`: possession of the
|
||||||
//! backend endpoint handle is the capability, exactly the trust of the old
|
//! backend endpoint handle is the capability, exactly the trust of the old
|
||||||
//! userspace router's op-6 cap-pass. An optional REWRITE prefix maps the mount
|
//! userspace router's op-6 cap-pass. An optional REWRITE prefix maps the mount
|
||||||
//! into the backend's namespace ("/var" -> fat's "/var" subtree while the same
|
//! into the backend's namespace ("/system/logs" -> the boot volume's
|
||||||
//! backend also serves "/mnt/usb" from its root), so FHS paths stay decoupled
|
//! identically-named subtree while the same backend also serves "/volumes/usb"
|
||||||
//! from which volume happens to carry them.
|
//! from its root), so hierarchy paths stay decoupled from which volume happens
|
||||||
|
//! to carry them.
|
||||||
|
|
||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
const abi = @import("abi");
|
const abi = @import("abi");
|
||||||
@@ -99,7 +101,7 @@ var directory_count: usize = 0;
|
|||||||
/// If `path` lies under `mount_prefix` — equal to it, or the prefix followed by
|
/// If `path` lies under `mount_prefix` — equal to it, or the prefix followed by
|
||||||
/// a path separator — return the path relative to the mount ("/" for an exact
|
/// a path separator — return the path relative to the mount ("/" for an exact
|
||||||
/// match, otherwise the tail beginning with '/'). Null when not under the
|
/// match, otherwise the tail beginning with '/'). Null when not under the
|
||||||
/// mount, so "/mnt/usb" never captures "/mnt/usbextra".
|
/// mount, so "/volumes/usb" never captures "/volumes/usbextra".
|
||||||
pub fn underMount(path: []const u8, mount_prefix: []const u8) ?[]const u8 {
|
pub fn underMount(path: []const u8, mount_prefix: []const u8) ?[]const u8 {
|
||||||
if (path.len < mount_prefix.len) return null;
|
if (path.len < mount_prefix.len) return null;
|
||||||
if (!std.mem.eql(u8, path[0..mount_prefix.len], mount_prefix)) return null;
|
if (!std.mem.eql(u8, path[0..mount_prefix.len], mount_prefix)) return null;
|
||||||
@@ -166,6 +168,11 @@ fn installMount(prefix: []const u8, kind: MountKind, backend: ?*ipc.Endpoint, re
|
|||||||
var slot: ?*Mount = null;
|
var slot: ?*Mount = null;
|
||||||
for (&mounts) |*m| {
|
for (&mounts) |*m| {
|
||||||
if (m.used and std.mem.eql(u8, m.prefixSlice(), prefix)) {
|
if (m.used and std.mem.eql(u8, m.prefixSlice(), prefix)) {
|
||||||
|
// ...except the protocol namespace. Remount-replace is how a
|
||||||
|
// restarted FAT retakes /volumes/usb; letting it retake /protocol
|
||||||
|
// would hand the whole naming layer to whoever asked second.
|
||||||
|
// First mount wins, and init (PID 1) is always first.
|
||||||
|
if (std.mem.eql(u8, prefix, protocol_root)) return;
|
||||||
if (m.backend) |old| ipc.dropRef(old);
|
if (m.backend) |old| ipc.dropRef(old);
|
||||||
slot = m;
|
slot = m;
|
||||||
break;
|
break;
|
||||||
@@ -182,11 +189,16 @@ fn installMount(prefix: []const u8, kind: MountKind, backend: ?*ipc.Endpoint, re
|
|||||||
|
|
||||||
// --- resolve -----------------------------------------------------------------
|
// --- resolve -----------------------------------------------------------------
|
||||||
|
|
||||||
|
/// Longest rewritten mount-relative path a backend resolution can carry — the
|
||||||
|
/// size `fs_resolve`'s caller has to have room for, so it is named rather than
|
||||||
|
/// spelled out at the one place that builds it.
|
||||||
|
pub const maximum_backend_path = maximum_rewrite + maximum_prefix + 160;
|
||||||
|
|
||||||
pub const Resolved = union(enum) {
|
pub const Resolved = union(enum) {
|
||||||
/// Kernel-served: a permanent node token.
|
/// Kernel-served: a permanent node token.
|
||||||
kernel_node: u64,
|
kernel_node: u64,
|
||||||
/// Backend-served: the endpoint plus the rewritten mount-relative path.
|
/// Backend-served: the endpoint plus the rewritten mount-relative path.
|
||||||
backend: struct { endpoint: *ipc.Endpoint, path: [maximum_rewrite + maximum_prefix + 160]u8, path_len: usize },
|
backend: struct { endpoint: *ipc.Endpoint, path: [maximum_backend_path]u8, path_len: usize },
|
||||||
not_found: void,
|
not_found: void,
|
||||||
};
|
};
|
||||||
|
|
||||||
@@ -326,20 +338,64 @@ pub fn nodeReaddir(node_token: u64, cursor: u64, name_out: []u8) ?struct { heade
|
|||||||
|
|
||||||
// --- mount/unmount (syscall bodies; caller resolved the handle) --------------
|
// --- mount/unmount (syscall bodies; caller resolved the handle) --------------
|
||||||
|
|
||||||
|
/// The writable subtrees a backend may mount beneath an initrd tree — exactly
|
||||||
|
/// these two, nothing else. Longest-prefix resolution then routes them to the
|
||||||
|
/// volume while every other /system and /test path stays initrd-served, so no
|
||||||
|
/// bundled binary can ever be shadowed.
|
||||||
|
const initrd_carve_outs = [_][]const u8{ "/system/configuration", "/system/logs" };
|
||||||
|
|
||||||
|
fn isInitrdCarveOut(prefix: []const u8) bool {
|
||||||
|
for (initrd_carve_outs) |allowed| {
|
||||||
|
if (std.mem.eql(u8, prefix, allowed)) return true;
|
||||||
|
}
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The protocol namespace's root — a reserved prefix, like the initrd trees.
|
||||||
|
/// Init (PID 1) mounts the registry here once at boot and the prefix then
|
||||||
|
/// refuses everything: a second mount at it, any mount *under* it (which would
|
||||||
|
/// shadow one contract), and its unmount. That is the whole kernel-side residue
|
||||||
|
/// of the naming layer — the registrar authority itself never leaves init
|
||||||
|
/// (docs/os-development/protocol-namespace.md).
|
||||||
|
const protocol_root = "/protocol";
|
||||||
|
|
||||||
|
fn protocolBound() bool {
|
||||||
|
for (&mounts) |*m| {
|
||||||
|
if (m.used and std.mem.eql(u8, m.prefixSlice(), protocol_root)) return true;
|
||||||
|
}
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether mounting at `prefix` would touch the protocol namespace. Exactly
|
||||||
|
/// `/protocol` is allowed once — while nothing holds it; anything under it,
|
||||||
|
/// ever, is refused.
|
||||||
|
fn refusesProtocolMount(prefix: []const u8) bool {
|
||||||
|
const relative = underMount(prefix, protocol_root) orelse return false;
|
||||||
|
if (relative.len != 1) return true; // strictly under /protocol: never
|
||||||
|
return protocolBound(); // /protocol itself: first mount wins
|
||||||
|
}
|
||||||
|
|
||||||
/// Mount `backend` at `prefix` with an optional backend-side `rewrite` prefix.
|
/// Mount `backend` at `prefix` with an optional backend-side `rewrite` prefix.
|
||||||
/// The endpoint reference is taken by the caller (process.zig bumps it); refuses
|
/// The endpoint reference is taken by the caller (process.zig bumps it); refuses
|
||||||
/// shadowing or replacing the initrd trees (/system, /test).
|
/// shadowing or replacing the initrd trees (/system, /test) — except the two
|
||||||
|
/// carve-outs in `initrd_carve_outs`, the writable configuration/log subtrees.
|
||||||
pub fn mountBackend(prefix: []const u8, backend: *ipc.Endpoint, rewrite: []const u8) bool {
|
pub fn mountBackend(prefix: []const u8, backend: *ipc.Endpoint, rewrite: []const u8) bool {
|
||||||
if (!isAbsolute(prefix) or prefix.len < 2 or prefix.len > maximum_prefix) return false;
|
if (!isAbsolute(prefix) or prefix.len < 2 or prefix.len > maximum_prefix) return false;
|
||||||
if (rewrite.len > maximum_rewrite) return false;
|
if (rewrite.len > maximum_rewrite) return false;
|
||||||
for (&mounts) |*m| { // the initrd trees are not shadowable
|
if (refusesProtocolMount(prefix)) return false; // the registry's prefix is claimed once
|
||||||
if (m.used and m.kind == .kernel_initrd and underMount(prefix, m.prefixSlice()) != null) return false;
|
for (&mounts) |*m| { // the initrd trees are not shadowable (carve-outs aside)
|
||||||
|
if (m.used and m.kind == .kernel_initrd and underMount(prefix, m.prefixSlice()) != null) {
|
||||||
|
if (!isInitrdCarveOut(prefix)) return false;
|
||||||
|
}
|
||||||
}
|
}
|
||||||
installMount(prefix, .backend, backend, rewrite);
|
installMount(prefix, .backend, backend, rewrite);
|
||||||
return true;
|
return true;
|
||||||
}
|
}
|
||||||
|
|
||||||
pub fn unmount(prefix: []const u8) bool {
|
pub fn unmount(prefix: []const u8) bool {
|
||||||
|
// Unmounting /protocol would delete the naming layer for everyone; nobody
|
||||||
|
// may, init included. The mount lasts the boot.
|
||||||
|
if (std.mem.eql(u8, prefix, protocol_root)) return false;
|
||||||
for (&mounts) |*m| {
|
for (&mounts) |*m| {
|
||||||
if (m.used and m.kind == .backend and std.mem.eql(u8, m.prefixSlice(), prefix)) {
|
if (m.used and m.kind == .backend and std.mem.eql(u8, m.prefixSlice(), prefix)) {
|
||||||
if (m.backend) |endpoint| ipc.dropRef(endpoint);
|
if (m.backend) |endpoint| ipc.dropRef(endpoint);
|
||||||
@@ -353,12 +409,12 @@ pub fn unmount(prefix: []const u8) bool {
|
|||||||
// --- tests (host) ------------------------------------------------------------
|
// --- tests (host) ------------------------------------------------------------
|
||||||
|
|
||||||
test "underMount matches only at path boundaries" {
|
test "underMount matches only at path boundaries" {
|
||||||
try std.testing.expectEqualStrings("/", underMount("/mnt/usb", "/mnt/usb").?);
|
try std.testing.expectEqualStrings("/", underMount("/volumes/usb", "/volumes/usb").?);
|
||||||
try std.testing.expectEqualStrings("/system/kernel", underMount("/mnt/usb/system/kernel", "/mnt/usb").?);
|
try std.testing.expectEqualStrings("/system/kernel", underMount("/volumes/usb/system/kernel", "/volumes/usb").?);
|
||||||
try std.testing.expect(underMount("/mnt/usbextra", "/mnt/usb") == null);
|
try std.testing.expect(underMount("/volumes/usbextra", "/volumes/usb") == null);
|
||||||
try std.testing.expect(underMount("/mnt", "/mnt/usb") == null);
|
try std.testing.expect(underMount("/volumes", "/volumes/usb") == null);
|
||||||
try std.testing.expect(underMount("/other", "/mnt/usb") == null);
|
try std.testing.expect(underMount("/other", "/volumes/usb") == null);
|
||||||
try std.testing.expect(underMount("greeting", "/mnt/usb") == null);
|
try std.testing.expect(underMount("greeting", "/volumes/usb") == null);
|
||||||
}
|
}
|
||||||
|
|
||||||
test "parentOf walks toward the root" {
|
test "parentOf walks toward the root" {
|
||||||
|
|||||||
@@ -12,6 +12,7 @@
|
|||||||
|
|
||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
const device = @import("driver");
|
const device = @import("driver");
|
||||||
|
const channel = @import("channel");
|
||||||
const ipc = @import("ipc");
|
const ipc = @import("ipc");
|
||||||
const process = @import("process");
|
const process = @import("process");
|
||||||
const service = @import("service");
|
const service = @import("service");
|
||||||
@@ -189,14 +190,32 @@ pub fn main(init: process.Init) void {
|
|||||||
readFadt(fadt);
|
readFadt(fadt);
|
||||||
s5_valid = readSleepS5(&persistent_namespace);
|
s5_valid = readSleepS5(&persistent_namespace);
|
||||||
|
|
||||||
|
// Every name this service needs, resolved before it becomes a provider — see
|
||||||
|
// `manager_channel`. Best-effort, as it has always been: a standalone
|
||||||
|
// bring-up with no device manager still serves power.
|
||||||
|
manager_channel = channel.openEndpoint("device-manager");
|
||||||
|
|
||||||
service.run(power_protocol.message_maximum, .{
|
service.run(power_protocol.message_maximum, .{
|
||||||
.service = .power,
|
.service = "power",
|
||||||
.init = onInit,
|
.init = onInit,
|
||||||
.on_message = onMessage,
|
.on_message = onMessage,
|
||||||
.on_notification = onNotification,
|
.on_notification = onNotification,
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// The device manager's channel, opened **before** this service binds its own
|
||||||
|
/// contract — deliberately, and load-bearing.
|
||||||
|
///
|
||||||
|
/// init is the registrar, and init is also this service's one subscriber: the
|
||||||
|
/// moment `power` is bound, init calls us to subscribe. init has a single thread,
|
||||||
|
/// so while it is blocked in that call it cannot answer anyone — including us. If
|
||||||
|
/// we opened a name after binding, the two could cross: init blocked calling us,
|
||||||
|
/// us blocked asking init to resolve a name, neither ever replying. Resolving
|
||||||
|
/// everything we need first makes that impossible, because after the bind this
|
||||||
|
/// service only ever talks to the device manager (which never calls init) and
|
||||||
|
/// then parks in the harness loop, where init's subscribe lands.
|
||||||
|
var manager_channel: ?ipc.Handle = null;
|
||||||
|
|
||||||
// Static so the harness callbacks (which run after main's stack frame is gone)
|
// Static so the harness callbacks (which run after main's stack frame is gone)
|
||||||
// can reach the namespace and interpreter.
|
// can reach the namespace and interpreter.
|
||||||
var persistent_namespace: aml.Namespace = undefined;
|
var persistent_namespace: aml.Namespace = undefined;
|
||||||
@@ -209,13 +228,13 @@ fn onInit(endpoint: ipc.Handle) bool {
|
|||||||
registered_count = 0;
|
registered_count = 0;
|
||||||
walkDevices(persistent_namespace.root, &global_interpreter);
|
walkDevices(persistent_namespace.root, &global_interpreter);
|
||||||
|
|
||||||
const manager = ipc.lookup(.device_manager);
|
const manager = manager_channel;
|
||||||
var i: usize = 0;
|
var i: usize = 0;
|
||||||
while (i < registered_count) : (i += 1) {
|
while (i < registered_count) : (i += 1) {
|
||||||
const entry = registered[i];
|
const entry = registered[i];
|
||||||
const hid = entry.hid[0..entry.hid_len];
|
const hid = entry.hid[0..entry.hid_len];
|
||||||
// The devices.csv columns (bus=acpi, hid) then the human-readable name — a
|
// The devices.csv columns (bus=acpi, hid) then the human-readable name — a
|
||||||
// would-be /etc/devices.csv row read straight off the boot log.
|
// would-be /system/configuration/devices.csv row read straight off the boot log.
|
||||||
const desc = acpi_ids.description(hid);
|
const desc = acpi_ids.description(hid);
|
||||||
if (desc.len != 0)
|
if (desc.len != 0)
|
||||||
std.log.info("device {d} bus=acpi hid={s} — {s} ({d} resources)", .{ entry.device_id, hid, desc, entry.resource_count })
|
std.log.info("device {d} bus=acpi hid={s} — {s} ({d} resources)", .{ entry.device_id, hid, desc, entry.resource_count })
|
||||||
@@ -433,15 +452,17 @@ fn onNotification(badge: u64) void {
|
|||||||
/// The `.power` protocol: subscribe (endpoint as the call's capability),
|
/// The `.power` protocol: subscribe (endpoint as the call's capability),
|
||||||
/// shutdown (PID 1 only). Device discovery uses a different endpoint (the
|
/// shutdown (PID 1 only). Device discovery uses a different endpoint (the
|
||||||
/// device manager's), so nothing here handles ChildAdded.
|
/// device manager's), so nothing here handles ChildAdded.
|
||||||
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Handle) usize {
|
fn onMessage(message: []const u8, reply: []u8, sender: u32, arrived: *ipc.Arrival) usize {
|
||||||
if (message.len < 1) return 0;
|
if (message.len < 1) return 0;
|
||||||
switch (message[0]) {
|
switch (message[0]) {
|
||||||
@intFromEnum(power_protocol.Operation.subscribe) => {
|
@intFromEnum(power_protocol.Operation.subscribe) => {
|
||||||
|
// The subscriber's endpoint is claimed only when a slot takes it;
|
||||||
|
// a full table refuses and the turn closes what arrived.
|
||||||
var status: i32 = -1;
|
var status: i32 = -1;
|
||||||
if (capability) |handle| {
|
if (arrived.peek() != null) {
|
||||||
for (&subscribers, 0..) |*slot, si| {
|
for (&subscribers, 0..) |*slot, si| {
|
||||||
if (slot.* == null) {
|
if (slot.* == null) {
|
||||||
slot.* = handle;
|
slot.* = arrived.take();
|
||||||
subscriber_tasks[si] = sender;
|
subscriber_tasks[si] = sender;
|
||||||
status = 0;
|
status = 0;
|
||||||
break;
|
break;
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
//! The acpi service as a binary package (docs/build-packages-plan.md):
|
//! The acpi service as a binary package (docs/build-packages-plan.md):
|
||||||
//! this file names the binary and EXACTLY the modules its source imports —
|
//! this file names the binary and EXACTLY the modules its source imports —
|
||||||
//! the shared recipe and the module-to-domain map live in build-support.
|
//! build-support resolves each name from the domains this zon declares.
|
||||||
//!
|
//!
|
||||||
//! The artifact is named "discovery": one swappable process per firmware
|
//! The artifact is named "discovery": one swappable process per firmware
|
||||||
//! fills the ramdisk's neutral `discovery` slot (docs/discovery.md); the
|
//! fills the ramdisk's neutral `discovery` slot (docs/discovery.md); the
|
||||||
@@ -14,8 +14,8 @@ pub fn build(b: *std.Build) void {
|
|||||||
.name = "discovery",
|
.name = "discovery",
|
||||||
.root_source_file = b.path("acpi.zig"),
|
.root_source_file = b.path("acpi.zig"),
|
||||||
.imports = &.{
|
.imports = &.{
|
||||||
"acpi-ids", "aml", "device-manager-protocol", "driver", "ipc", "logging", "memory",
|
"acpi-ids", "aml", "channel", "device-manager-protocol", "driver", "ipc", "logging",
|
||||||
"power-protocol", "process", "service", "time",
|
"memory", "power-protocol", "process", "service", "time",
|
||||||
},
|
},
|
||||||
});
|
});
|
||||||
b.installArtifact(exe);
|
b.installArtifact(exe);
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
//! The device-manager service as a binary package (docs/build-packages-plan.md):
|
//! The device-manager service as a binary package (docs/build-packages-plan.md):
|
||||||
//! this file names the binary and EXACTLY the modules its source imports —
|
//! this file names the binary and EXACTLY the modules its source imports —
|
||||||
//! the shared recipe and the module-to-domain map live in build-support.
|
//! build-support resolves each name from the domains this zon declares.
|
||||||
|
|
||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
const build_support = @import("build-support");
|
const build_support = @import("build-support");
|
||||||
|
|||||||
@@ -28,7 +28,7 @@ const registry = @import("device-registry");
|
|||||||
const fs = @import("file-system");
|
const fs = @import("file-system");
|
||||||
|
|
||||||
// --- the device registry ------------------------------------------------------
|
// --- the device registry ------------------------------------------------------
|
||||||
// Driver matching is data-driven and authoritative: /etc/devices.csv (parsed by
|
// Driver matching is data-driven and authoritative: /system/configuration/devices.csv (parsed by
|
||||||
// the device-registry module) names, per bus, which driver binds a reported
|
// the device-registry module) names, per bus, which driver binds a reported
|
||||||
// device, the most-specific match winning. There is no compiled-in fallback — a
|
// device, the most-specific match winning. There is no compiled-in fallback — a
|
||||||
// device no row matches goes unbound and is logged. This retired the hand-kept
|
// device no row matches goes unbound and is logged. This retired the hand-kept
|
||||||
@@ -41,12 +41,12 @@ var registry_source: [8192]u8 = undefined;
|
|||||||
var registry_rules: [64]registry.Rule = undefined;
|
var registry_rules: [64]registry.Rule = undefined;
|
||||||
var registry_count: usize = 0;
|
var registry_count: usize = 0;
|
||||||
|
|
||||||
/// Read and parse /etc/devices.csv once at boot. The file lives in the initial
|
/// Read and parse /system/configuration/devices.csv once at boot. The file lives in the initial
|
||||||
/// ramdisk, which the kernel serves directly — no filesystem service need be up
|
/// ramdisk, which the kernel serves directly — no filesystem service need be up
|
||||||
/// (fat is spawned after the manager), so this is a plain fs.open + read.
|
/// (fat is spawned after the manager), so this is a plain fs.open + read.
|
||||||
fn loadRegistry() void {
|
fn loadRegistry() void {
|
||||||
var file = fs.open("/etc/devices.csv", .{}) orelse {
|
var file = fs.open("/system/configuration/devices.csv", .{}) orelse {
|
||||||
_ = logging.write("/system/services/device-manager: /etc/devices.csv missing — nothing will match\n");
|
_ = logging.write("/system/services/device-manager: /system/configuration/devices.csv missing — nothing will match\n");
|
||||||
return;
|
return;
|
||||||
};
|
};
|
||||||
defer file.close();
|
defer file.close();
|
||||||
@@ -58,9 +58,9 @@ fn loadRegistry() void {
|
|||||||
}
|
}
|
||||||
const result = registry.parse(registry_source[0..used], ®istry_rules);
|
const result = registry.parse(registry_source[0..used], ®istry_rules);
|
||||||
registry_count = result.count;
|
registry_count = result.count;
|
||||||
if (result.malformed != 0) std.log.info("/etc/devices.csv: {d} malformed line(s) skipped", .{result.malformed});
|
if (result.malformed != 0) std.log.info("/system/configuration/devices.csv: {d} malformed line(s) skipped", .{result.malformed});
|
||||||
if (result.truncated) _ = logging.write("/system/services/device-manager: /etc/devices.csv has more rules than the table holds\n");
|
if (result.truncated) _ = logging.write("/system/services/device-manager: /system/configuration/devices.csv has more rules than the table holds\n");
|
||||||
std.log.info("/etc/devices.csv: {d} rule(s) loaded", .{registry_count});
|
std.log.info("/system/configuration/devices.csv: {d} rule(s) loaded", .{registry_count});
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Build a registry Identity from a bus driver's report: the bus it named, the
|
/// Build a registry Identity from a bus driver's report: the bus it named, the
|
||||||
@@ -395,13 +395,13 @@ fn initialise(endpoint: ipc.Handle) bool {
|
|||||||
return true;
|
return true;
|
||||||
}
|
}
|
||||||
|
|
||||||
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Handle) usize {
|
fn onMessage(message: []const u8, reply: []u8, sender: u32, arrived: *ipc.Arrival) usize {
|
||||||
if (message.len < 1) return 0;
|
if (message.len < 1) return 0;
|
||||||
switch (message[0]) {
|
switch (message[0]) {
|
||||||
@intFromEnum(device_manager_protocol.Operation.child_added) => return onChildAdded(message, reply, sender),
|
@intFromEnum(device_manager_protocol.Operation.child_added) => return onChildAdded(message, reply, sender),
|
||||||
@intFromEnum(device_manager_protocol.Operation.child_removed) => return onChildRemoved(message, reply, sender),
|
@intFromEnum(device_manager_protocol.Operation.child_removed) => return onChildRemoved(message, reply, sender),
|
||||||
@intFromEnum(device_manager_protocol.Operation.enumerate) => return onEnumerate(reply),
|
@intFromEnum(device_manager_protocol.Operation.enumerate) => return onEnumerate(reply),
|
||||||
@intFromEnum(device_manager_protocol.Operation.subscribe) => return onSubscribe(reply, capability),
|
@intFromEnum(device_manager_protocol.Operation.subscribe) => return onSubscribe(reply, arrived),
|
||||||
@intFromEnum(device_manager_protocol.Operation.hello) => {},
|
@intFromEnum(device_manager_protocol.Operation.hello) => {},
|
||||||
else => return 0,
|
else => return 0,
|
||||||
}
|
}
|
||||||
@@ -443,7 +443,7 @@ fn onChildAdded(message: []const u8, reply: []u8, sender: u32) usize {
|
|||||||
if (!addChild(report.parent, report.bus_address, report.identity, report.device_id, sender)) status = -1;
|
if (!addChild(report.parent, report.bus_address, report.identity, report.device_id, sender)) status = -1;
|
||||||
std.log.info("child added (device {d} port {d}, identity {d}) by {s}", .{ report.parent, report.bus_address, report.identity, driver.name() });
|
std.log.info("child added (device {d} port {d}, identity {d}) by {s}", .{ report.parent, report.bus_address, report.identity, driver.name() });
|
||||||
if (status == 0) publishEvent(message[0..device_manager_protocol.child_added_size]);
|
if (status == 0) publishEvent(message[0..device_manager_protocol.child_added_size]);
|
||||||
// Matching from reports (M19.3), now data-driven via the /etc/devices.csv
|
// Matching from reports (M19.3), now data-driven via the /system/configuration/devices.csv
|
||||||
// registry: a registered child gets the most-specific driver its identity
|
// registry: a registered child gets the most-specific driver its identity
|
||||||
// matches, once — re-reports after a bus restart dedupe on the registered
|
// matches, once — re-reports after a bus restart dedupe on the registered
|
||||||
// id, exactly like the registrations do.
|
// id, exactly like the registrations do.
|
||||||
@@ -451,7 +451,7 @@ fn onChildAdded(message: []const u8, reply: []u8, sender: u32) usize {
|
|||||||
const id = identityFromReport(report);
|
const id = identityFromReport(report);
|
||||||
if (registry.matchDriver(registry_rules[0..registry_count], id)) |match| {
|
if (registry.matchDriver(registry_rules[0..registry_count], id)) |match| {
|
||||||
if (match.ambiguous)
|
if (match.ambiguous)
|
||||||
std.log.info("/etc/devices.csv: multiple equally-specific rules match the device {s} reported; binding {s}", .{ driver.name(), match.driver });
|
std.log.info("/system/configuration/devices.csv: multiple equally-specific rules match the device {s} reported; binding {s}", .{ driver.name(), match.driver });
|
||||||
if (id.bus == .acpi) {
|
if (id.bus == .acpi) {
|
||||||
// An hid-matched driver (ps2-bus) is a singleton that finds its
|
// An hid-matched driver (ps2-bus) is a singleton that finds its
|
||||||
// own devices once spawned — spawn it once, no device assignment.
|
// own devices once spawned — spawn it once, no device assignment.
|
||||||
@@ -532,13 +532,15 @@ fn onEnumerate(reply: []u8) usize {
|
|||||||
return offset;
|
return offset;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// An application subscribed: its endpoint arrived as the call's capability.
|
/// An application subscribed: its endpoint arrived as the call's capability. The
|
||||||
fn onSubscribe(reply: []u8, capability: ?ipc.Handle) usize {
|
/// table taking a slot is what claims it (`take`); a full table refuses and lets
|
||||||
|
/// the turn close it, so a subscribe storm cannot spend the handle table too.
|
||||||
|
fn onSubscribe(reply: []u8, arrived: *ipc.Arrival) usize {
|
||||||
var status: i32 = -1;
|
var status: i32 = -1;
|
||||||
if (capability) |handle| {
|
if (arrived.peek() != null) {
|
||||||
for (&subscribers) |*slot| {
|
for (&subscribers) |*slot| {
|
||||||
if (slot.* == null) {
|
if (slot.* == null) {
|
||||||
slot.* = handle;
|
slot.* = arrived.take(); // claimed: the table holds it from here
|
||||||
status = 0;
|
status = 0;
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
@@ -566,7 +568,7 @@ pub fn main(init: process.Init) void {
|
|||||||
test_scanout_restart_mode = std.mem.eql(u8, mode, "test-scanout-restart");
|
test_scanout_restart_mode = std.mem.eql(u8, mode, "test-scanout-restart");
|
||||||
}
|
}
|
||||||
service.run(device_manager_protocol.message_maximum, .{
|
service.run(device_manager_protocol.message_maximum, .{
|
||||||
.service = .device_manager,
|
.service = "device-manager",
|
||||||
.init = initialise,
|
.init = initialise,
|
||||||
.on_message = onMessage,
|
.on_message = onMessage,
|
||||||
.on_notification = onNotification,
|
.on_notification = onNotification,
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
//! The display-demo service as a binary package (docs/build-packages-plan.md):
|
//! The display-demo service as a binary package (docs/build-packages-plan.md):
|
||||||
//! this file names the binary and EXACTLY the modules its source imports —
|
//! this file names the binary and EXACTLY the modules its source imports —
|
||||||
//! the shared recipe and the module-to-domain map live in build-support.
|
//! build-support resolves each name from the domains this zon declares.
|
||||||
|
|
||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
const build_support = @import("build-support");
|
const build_support = @import("build-support");
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
//! The display service as a binary package (docs/build-packages-plan.md):
|
//! The display service as a binary package (docs/build-packages-plan.md):
|
||||||
//! this file names the binary and EXACTLY the modules its source imports —
|
//! this file names the binary and EXACTLY the modules its source imports —
|
||||||
//! the shared recipe and the module-to-domain map live in build-support.
|
//! build-support resolves each name from the domains this zon declares.
|
||||||
|
|
||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
const build_support = @import("build-support");
|
const build_support = @import("build-support");
|
||||||
@@ -10,8 +10,8 @@ pub fn build(b: *std.Build) void {
|
|||||||
.name = "display",
|
.name = "display",
|
||||||
.root_source_file = b.path("display.zig"),
|
.root_source_file = b.path("display.zig"),
|
||||||
.imports = &.{
|
.imports = &.{
|
||||||
"display-client", "display-protocol", "driver", "input-client", "ipc", "logging",
|
"channel", "display-client", "display-protocol", "driver", "input-client", "ipc",
|
||||||
"memory", "scanout-protocol", "service", "thread", "time",
|
"logging", "memory", "scanout-protocol", "service", "thread", "time",
|
||||||
},
|
},
|
||||||
.threaded = true, // real atomics/TLS (docs/threading.md)
|
.threaded = true, // real atomics/TLS (docs/threading.md)
|
||||||
});
|
});
|
||||||
|
|||||||
@@ -16,6 +16,7 @@
|
|||||||
//! (docs/display-v2.md).
|
//! (docs/display-v2.md).
|
||||||
|
|
||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
|
const channel = @import("channel");
|
||||||
const ipc = @import("ipc");
|
const ipc = @import("ipc");
|
||||||
const input = @import("input-client");
|
const input = @import("input-client");
|
||||||
const Thread = @import("thread").Thread;
|
const Thread = @import("thread").Thread;
|
||||||
@@ -307,11 +308,17 @@ fn verifyNativePresent() void {
|
|||||||
/// present channel, switch the backend to virtio-gpu, and queue a full-screen repaint. The
|
/// present channel, switch the backend to virtio-gpu, and queue a full-screen repaint. The
|
||||||
/// present is deferred to a timer (see `service_endpoint`) so it happens after this reply
|
/// present is deferred to a timer (see `service_endpoint`) so it happens after this reply
|
||||||
/// unblocks the driver and it starts serving `.scanout`.
|
/// unblocks the driver and it starts serving `.scanout`.
|
||||||
fn attachScanout(stride: u32, width: u32, height: u32, format: u32, refresh_hz: u32, capability: ?ipc.Handle, reply: []u8) usize {
|
/// The surface arrives as the call's capability, and the harness's ownership rule
|
||||||
const cap = capability orelse return fail(reply);
|
/// applies: nothing here claims it, so the turn closes it on every path. Safe
|
||||||
|
/// because a **mapping holds its own kernel reference** (system/kernel/process.zig
|
||||||
|
/// `systemSharedMemoryMap`) — the pixels stay ours after the handle naming them
|
||||||
|
/// goes, and a driver that dies and re-announces no longer costs a handle slot
|
||||||
|
/// per restart.
|
||||||
|
fn attachScanout(stride: u32, width: u32, height: u32, format: u32, refresh_hz: u32, arrived: *ipc.Arrival, reply: []u8) usize {
|
||||||
|
const cap = arrived.peek() orelse return fail(reply);
|
||||||
if (width == 0 or height == 0 or stride < width) return fail(reply);
|
if (width == 0 or height == 0 or stride < width) return fail(reply);
|
||||||
const mapped = memory.sharedMap(cap) orelse return fail(reply);
|
const mapped = memory.sharedMap(cap) orelse return fail(reply);
|
||||||
const scanout = ipc.lookup(.scanout) orelse return fail(reply);
|
const scanout = channel.openEndpoint("scanout") orelse return fail(reply);
|
||||||
// A second announce means the driver died and was restarted (V6): re-attach to its fresh
|
// A second announce means the driver died and was restarted (V6): re-attach to its fresh
|
||||||
// scanout. (The previous shared mapping leaks — there is no shared_memory_unmap syscall yet — but the
|
// scanout. (The previous shared mapping leaks — there is no shared_memory_unmap syscall yet — but the
|
||||||
// frames are the dead driver's, reclaimed on its exit; a handful across a crash is benign.)
|
// frames are the dead driver's, reclaimed on its exit; a handful across a crash is benign.)
|
||||||
@@ -506,10 +513,13 @@ fn mouseListener(width: u32, height: u32) void {
|
|||||||
return;
|
return;
|
||||||
};
|
};
|
||||||
// Our own handle to the compositor's endpoint. IPC handles are per-thread, so we
|
// Our own handle to the compositor's endpoint. IPC handles are per-thread, so we
|
||||||
// cannot reuse the main thread's service handle — we look the service up to install a
|
// cannot reuse the main thread's — this thread resolves and opens
|
||||||
// handle in this thread's table. A poke posted here wakes the compositor loop parked
|
// `/protocol/display` exactly like any other client would, once at startup, and
|
||||||
// in replyWait (docs/threading.md: handles do not cross threads).
|
// gets its own handle. There is no special mechanism for reaching yourself: the
|
||||||
cursor_channel.poke_endpoint = ipc.lookup(.display) orelse {
|
// registry does not know or care that the provider is this process. A poke posted
|
||||||
|
// here wakes the compositor loop parked in replyWait (docs/threading.md: handles
|
||||||
|
// do not cross threads).
|
||||||
|
cursor_channel.poke_endpoint = channel.openEndpoint("display") orelse {
|
||||||
_ = logging.write("display: mouse listener could not reach the compositor endpoint\n");
|
_ = logging.write("display: mouse listener could not reach the compositor endpoint\n");
|
||||||
return;
|
return;
|
||||||
};
|
};
|
||||||
@@ -613,7 +623,7 @@ fn fail(reply: []u8) usize {
|
|||||||
return writeReply(reply, .{ .status = -1 });
|
return writeReply(reply, .{ .status = -1 });
|
||||||
}
|
}
|
||||||
|
|
||||||
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Handle) usize {
|
fn onMessage(message: []const u8, reply: []u8, sender: u32, arrived: *ipc.Arrival) usize {
|
||||||
_ = sender;
|
_ = sender;
|
||||||
if (message.len < display_protocol.request_size) return fail(reply);
|
if (message.len < display_protocol.request_size) return fail(reply);
|
||||||
const request = std.mem.bytesToValue(display_protocol.Request, message[0..display_protocol.request_size]);
|
const request = std.mem.bytesToValue(display_protocol.Request, message[0..display_protocol.request_size]);
|
||||||
@@ -657,7 +667,7 @@ fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Han
|
|||||||
return ok(reply);
|
return ok(reply);
|
||||||
},
|
},
|
||||||
@intFromEnum(display_protocol.Operation.attach_scanout) => {
|
@intFromEnum(display_protocol.Operation.attach_scanout) => {
|
||||||
return attachScanout(request.x, request.width, request.height, request.colour, request.y, capability, reply);
|
return attachScanout(request.x, request.width, request.height, request.colour, request.y, arrived, reply);
|
||||||
},
|
},
|
||||||
@intFromEnum(display_protocol.Operation.set_mode) => {
|
@intFromEnum(display_protocol.Operation.set_mode) => {
|
||||||
if (!backend.setMode(request.width, request.height)) return fail(reply);
|
if (!backend.setMode(request.width, request.height)) return fail(reply);
|
||||||
@@ -696,7 +706,7 @@ fn onNotification(badge: u64) void {
|
|||||||
|
|
||||||
pub fn main() void {
|
pub fn main() void {
|
||||||
service.run(display_protocol.message_maximum, .{
|
service.run(display_protocol.message_maximum, .{
|
||||||
.service = .display,
|
.service = "display",
|
||||||
.init = initialise,
|
.init = initialise,
|
||||||
.on_message = onMessage,
|
.on_message = onMessage,
|
||||||
.on_notification = onNotification,
|
.on_notification = onNotification,
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
//! The fat service as a binary package (docs/build-packages-plan.md):
|
//! The fat service as a binary package (docs/build-packages-plan.md):
|
||||||
//! this file names the binary and EXACTLY the modules its source imports —
|
//! this file names the binary and EXACTLY the modules its source imports —
|
||||||
//! the shared recipe and the module-to-domain map live in build-support.
|
//! build-support resolves each name from the domains this zon declares.
|
||||||
|
|
||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
const build_support = @import("build-support");
|
const build_support = @import("build-support");
|
||||||
|
|||||||
@@ -73,7 +73,7 @@ const entries_per_sector = sector_size / @sizeOf(on_disk.DirectoryEntry); // 16
|
|||||||
// A small write-through cache of single-sector (metadata) accesses: FAT sectors,
|
// A small write-through cache of single-sector (metadata) accesses: FAT sectors,
|
||||||
// directory sectors, and directory-entry writebacks. Its payoff is repeated scans
|
// directory sectors, and directory-entry writebacks. Its payoff is repeated scans
|
||||||
// — resolving many paths under the same directory (a logging burst opening dozens
|
// — resolving many paths under the same directory (a logging burst opening dozens
|
||||||
// of files under /var/log/<stamp>/) re-reads the same directory and FAT sectors,
|
// of files under /system/logs/<stamp>/) re-reads the same directory and FAT sectors,
|
||||||
// which now come from RAM instead of a USB round trip each. Bulk file data (the
|
// which now come from RAM instead of a USB round trip each. Bulk file data (the
|
||||||
// multi-sector run path) bypasses the cache — it is large and not re-read — and a
|
// multi-sector run path) bypasses the cache — it is large and not re-read — and a
|
||||||
// run write invalidates any overlapping cached sector to stay coherent.
|
// run write invalidates any overlapping cached sector to stay coherent.
|
||||||
|
|||||||
+30
-14
@@ -1,8 +1,8 @@
|
|||||||
//! system/services/fat — the FAT filesystem server. Spawned as a boot service, it
|
//! system/services/fat — the FAT filesystem server. Spawned as a boot service, it
|
||||||
//! opens the block device (a USB stick via usb-storage) under `.block`, mounts the
|
//! opens the block device (a USB stick via usb-storage) under `.block`, mounts the
|
||||||
//! FAT filesystem on it (the pure engine in engine.zig), and mounts itself into
|
//! FAT filesystem on it (the pure engine in engine.zig), and mounts itself into
|
||||||
//! the VFS at /mnt/usb. From then on the VFS forwards every open/read/write/
|
//! the VFS at /volumes/usb. From then on the VFS forwards every open/read/write/
|
||||||
//! status/readdir/close under /mnt/usb to this server, which serves the same
|
//! status/readdir/close under /volumes/usb to this server, which serves the same
|
||||||
//! vfs-protocol as a backend — turning block reads into file reads.
|
//! vfs-protocol as a backend — turning block reads into file reads.
|
||||||
//!
|
//!
|
||||||
//! The block data path never crosses IPC: a DMA bounce buffer is handed to the
|
//! The block data path never crosses IPC: a DMA bounce buffer is handed to the
|
||||||
@@ -22,7 +22,7 @@ const engine = @import("engine.zig");
|
|||||||
const on_disk = @import("on-disk.zig");
|
const on_disk = @import("on-disk.zig");
|
||||||
const vfs_protocol = @import("vfs-protocol");
|
const vfs_protocol = @import("vfs-protocol");
|
||||||
|
|
||||||
const mount_point = "/mnt/usb";
|
const mount_point = "/volumes/usb";
|
||||||
|
|
||||||
// The engine's BlockDevice, backed by the `.block` driver plus a DMA bounce
|
// The engine's BlockDevice, backed by the `.block` driver plus a DMA bounce
|
||||||
// buffer the driver reads/writes by physical address.
|
// buffer the driver reads/writes by physical address.
|
||||||
@@ -144,18 +144,25 @@ fn tryBringUp() void {
|
|||||||
};
|
};
|
||||||
std.log.info("mounted FAT ({s}, {d} clusters, partition lba {d})", .{ @tagName(filesystem.geometry.fat_type), filesystem.geometry.cluster_count, filesystem.base_lba });
|
std.log.info("mounted FAT ({s}, {d} clusters, partition lba {d})", .{ @tagName(filesystem.geometry.fat_type), filesystem.geometry.cluster_count, filesystem.base_lba });
|
||||||
|
|
||||||
// Mount ourselves into the kernel VFS at /mnt/usb — and serve /var from the
|
// Mount ourselves into the kernel VFS at /volumes/usb — and serve
|
||||||
// volume's /var subtree, so FHS paths (the logger's /var/log) stay decoupled
|
// /system/configuration and /system/logs from the volume's identically-named
|
||||||
// from which volume carries them.
|
// subtrees (the boot volume is hierarchy-shaped, so rewrite == prefix), so
|
||||||
|
// hierarchy paths (the logger's /system/logs) stay decoupled from which
|
||||||
|
// volume carries them.
|
||||||
if (file_system.mount(mount_point, endpointForMount())) {
|
if (file_system.mount(mount_point, endpointForMount())) {
|
||||||
std.log.info("mounted {s}", .{mount_point});
|
std.log.info("mounted {s}", .{mount_point});
|
||||||
} else {
|
} else {
|
||||||
_ = logging.write("/system/services/fat: could not mount /mnt/usb\n");
|
_ = logging.write("/system/services/fat: could not mount /volumes/usb\n");
|
||||||
}
|
}
|
||||||
if (file_system.mountRewritten("/var", endpointForMount(), "/var")) {
|
if (file_system.mountRewritten("/system/configuration", endpointForMount(), "/system/configuration")) {
|
||||||
std.log.info("mounted /var", .{});
|
std.log.info("mounted /system/configuration", .{});
|
||||||
} else {
|
} else {
|
||||||
_ = logging.write("/system/services/fat: could not mount /var\n");
|
_ = logging.write("/system/services/fat: could not mount /system/configuration\n");
|
||||||
|
}
|
||||||
|
if (file_system.mountRewritten("/system/logs", endpointForMount(), "/system/logs")) {
|
||||||
|
std.log.info("mounted /system/logs", .{});
|
||||||
|
} else {
|
||||||
|
_ = logging.write("/system/services/fat: could not mount /system/logs\n");
|
||||||
}
|
}
|
||||||
mounted = true;
|
mounted = true;
|
||||||
}
|
}
|
||||||
@@ -215,8 +222,14 @@ fn handleOpen(out: []u8, path: []const u8, flags: u32, sender: u32) usize {
|
|||||||
return writeReply(out, .{ .status = 0, .node = index }, &.{});
|
return writeReply(out, .{ .status = 0, .node = index }, &.{});
|
||||||
}
|
}
|
||||||
|
|
||||||
fn onMessage(message: []const u8, out: []u8, sender: u32, capability: ?ipc.Handle) usize {
|
/// The vfs protocol has no operation that takes a capability, so `arrived` is
|
||||||
_ = capability;
|
/// never claimed here — which, under the harness's ownership rule, means the
|
||||||
|
/// loop closes whatever a caller attached. That is the point of the rule: this
|
||||||
|
/// callback used to discard a `?ipc.Handle` and every request carrying one — a
|
||||||
|
/// legal thing for any client to do — spent a slot of the VFS server's
|
||||||
|
/// thirty-two until it could accept no capability at all.
|
||||||
|
fn onMessage(message: []const u8, out: []u8, sender: u32, arrived: *ipc.Arrival) usize {
|
||||||
|
_ = arrived;
|
||||||
if (!mounted) return fail(out); // storage not up (yet): fail politely, clients retry
|
if (!mounted) return fail(out); // storage not up (yet): fail politely, clients retry
|
||||||
if (message.len < vfs_protocol.request_size) return fail(out);
|
if (message.len < vfs_protocol.request_size) return fail(out);
|
||||||
const request = std.mem.bytesToValue(vfs_protocol.Request, message[0..vfs_protocol.request_size]);
|
const request = std.mem.bytesToValue(vfs_protocol.Request, message[0..vfs_protocol.request_size]);
|
||||||
@@ -298,13 +311,16 @@ fn onMessage(message: []const u8, out: []u8, sender: u32, capability: ?ipc.Handl
|
|||||||
return writeReply(out, .{ .status = 0 }, &.{});
|
return writeReply(out, .{ .status = 0 }, &.{});
|
||||||
},
|
},
|
||||||
// A backend is never itself a mount target.
|
// A backend is never itself a mount target.
|
||||||
.mount, .unmount => return fail(out),
|
// Router verbs, and the registry's claim verb: a file backend answers
|
||||||
|
// none of them (docs/os-development/protocol-namespace.md — only init
|
||||||
|
// implements `bind`).
|
||||||
|
.mount, .unmount, .bind => return fail(out),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
pub fn main() void {
|
pub fn main() void {
|
||||||
service.run(vfs_protocol.message_maximum, .{
|
service.run(vfs_protocol.message_maximum, .{
|
||||||
.service = .fat,
|
.service = "vfs",
|
||||||
.init = initialise,
|
.init = initialise,
|
||||||
.on_message = onMessage,
|
.on_message = onMessage,
|
||||||
.on_notification = onNotification,
|
.on_notification = onNotification,
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
//! The fdt service as a binary package (docs/build-packages-plan.md):
|
//! The fdt service as a binary package (docs/build-packages-plan.md):
|
||||||
//! this file names the binary and EXACTLY the modules its source imports —
|
//! this file names the binary and EXACTLY the modules its source imports —
|
||||||
//! the shared recipe and the module-to-domain map live in build-support.
|
//! build-support resolves each name from the domains this zon declares.
|
||||||
//!
|
//!
|
||||||
//! The artifact is named "discovery" like acpi's: the Raspberry Pis hand over
|
//! The artifact is named "discovery" like acpi's: the Raspberry Pis hand over
|
||||||
//! a flattened device tree, and the aarch64 target flips the root's
|
//! a flattened device tree, and the aarch64 target flips the root's
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
//! init (PID 1) as a binary package (docs/build-packages-plan.md): this file
|
//! init (PID 1) as a binary package (docs/build-packages-plan.md): this file
|
||||||
//! names the binary and EXACTLY the modules its source imports — the shared
|
//! names the binary and EXACTLY the modules its source imports — the shared
|
||||||
//! recipe and the module-to-domain map live in build-support. The root build
|
//! recipe resolves each name from the domains the zon declares. The root build
|
||||||
//! consumes the artifact for the boot image and forwards its -Dserial here.
|
//! consumes the artifact for the boot image and forwards its -Dserial here.
|
||||||
|
|
||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
@@ -11,8 +11,8 @@ pub fn build(b: *std.Build) void {
|
|||||||
.name = "init",
|
.name = "init",
|
||||||
.root_source_file = b.path("init.zig"),
|
.root_source_file = b.path("init.zig"),
|
||||||
.imports = &.{
|
.imports = &.{
|
||||||
"csv", "file-system", "ipc", "logging", "memory", "power-protocol",
|
"csv", "envelope", "file-system", "ipc", "logging", "memory", "power-protocol",
|
||||||
"process", "time",
|
"process", "time", "vfs-protocol",
|
||||||
},
|
},
|
||||||
});
|
});
|
||||||
// init reads the same `serial` flag the kernel does: its liveness heartbeat
|
// init reads the same `serial` flag the kernel does: its liveness heartbeat
|
||||||
|
|||||||
+760
-47
@@ -16,6 +16,34 @@
|
|||||||
//! power-button event runs the stop sequence over its children in reverse order
|
//! power-button event runs the stop sequence over its children in reverse order
|
||||||
//! before asking the power service to enter S5 — lifecycle (M17) and events (M21)
|
//! before asking the power service to enter S5 — lifecycle (M17) and events (M21)
|
||||||
//! composing into a clean poweroff.
|
//! composing into a clean poweroff.
|
||||||
|
//!
|
||||||
|
//! P2: init is also the **registrar** — it serves `/protocol`, the namespace where
|
||||||
|
//! a program finds everything it talks to
|
||||||
|
//! (docs/os-development/protocol-namespace.md). It is the natural home: it already
|
||||||
|
//! spawns the services and already holds the supervision link to each, so it is the
|
||||||
|
//! process that *knows* which binary is which. Registry traffic rides the same
|
||||||
|
//! endpoint as supervision, because one thread can only wait in one place — the
|
||||||
|
//! loop below answers vfs `open`/`readdir`/`bind` alongside signals, timers, power
|
||||||
|
//! events, and children's deaths.
|
||||||
|
//!
|
||||||
|
//! Which sets the security posture of this file. `fs_resolve` installs a mounted
|
||||||
|
//! backend's endpoint capability in *any* caller's handle table, so sharing the
|
||||||
|
//! mailbox means **every ring-3 process can send into PID 1**. Two rules follow,
|
||||||
|
//! and both are structural here rather than remembered per branch:
|
||||||
|
//!
|
||||||
|
//! - **Privileged action requires an attested sender.** What arrives is a
|
||||||
|
//! stranger's bytes; the only identity on it is the task id the kernel stamps.
|
||||||
|
//! Content never authorizes (`onPowerEvent`), and neither does a name — the
|
||||||
|
//! registrar attests a caller's supervision by task id (`supervisorSatisfies`).
|
||||||
|
//! - **Absence is the enforcement.** P3: `open` consults the manifest with the
|
||||||
|
//! same attested identity a `bind` does, and a caller with no grant is told
|
||||||
|
//! exactly what a caller asking for a name nobody bound is told — `-ENOENT`,
|
||||||
|
//! and no capability (`onOpen`). Restriction stage one of
|
||||||
|
//! docs/os-development/protocol-namespace.md: what a process cannot open does
|
||||||
|
//! not exist for it, so there is no "permission denied" to distinguish.
|
||||||
|
//! - **A capability that arrives is closed unless it is claimed** (`Arrival`),
|
||||||
|
//! because PID 1's thirty-two handle slots are a resource an unauthenticated
|
||||||
|
//! caller would otherwise be able to spend.
|
||||||
|
|
||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
const ipc = @import("ipc");
|
const ipc = @import("ipc");
|
||||||
@@ -24,22 +52,24 @@ const time = @import("time");
|
|||||||
const memory = @import("memory");
|
const memory = @import("memory");
|
||||||
const logging = @import("logging");
|
const logging = @import("logging");
|
||||||
const power_protocol = @import("power-protocol");
|
const power_protocol = @import("power-protocol");
|
||||||
|
const vfs_protocol = @import("vfs-protocol");
|
||||||
|
const envelope = @import("envelope");
|
||||||
const build_options = @import("build_options");
|
const build_options = @import("build_options");
|
||||||
const fs = @import("file-system");
|
const fs = @import("file-system");
|
||||||
const csv = @import("csv");
|
const csv = @import("csv");
|
||||||
|
|
||||||
/// The system services init brings up at boot are init's policy, not the kernel's —
|
/// The system services init brings up at boot are init's policy, not the kernel's —
|
||||||
/// and that policy is now data: `/etc/init.csv` (see `loadServices`), read at
|
/// and that policy is now data: `/system/configuration/init.csv` (see `loadServices`), read at
|
||||||
/// startup instead of a hardcoded list. Drivers are absent on purpose: the device
|
/// startup instead of a hardcoded list. Drivers are absent on purpose: the device
|
||||||
/// manager owns those.
|
/// manager owns those.
|
||||||
///
|
///
|
||||||
/// The most services `/etc/init.csv` can list, and the most argv entries (beyond the
|
/// The most services `/system/configuration/init.csv` can list, and the most argv entries (beyond the
|
||||||
/// path) each may carry. Fixed caps because init parses the list into static storage —
|
/// path) each may carry. Fixed caps because init parses the list into static storage —
|
||||||
/// the freestanding, no-allocator counterpart to the device manager's registry table.
|
/// the freestanding, no-allocator counterpart to the device manager's registry table.
|
||||||
const max_services = 16;
|
const max_services = 16;
|
||||||
const max_service_args = 4;
|
const max_service_args = 4;
|
||||||
|
|
||||||
/// One service init starts, parsed from a row of `/etc/init.csv`: its binary path
|
/// One service init starts, parsed from a row of `/system/configuration/init.csv`: its binary path
|
||||||
/// and argv, both slices into `init_csv` (held for the life of the process).
|
/// and argv, both slices into `init_csv` (held for the life of the process).
|
||||||
const Service = struct {
|
const Service = struct {
|
||||||
path: []const u8 = "",
|
path: []const u8 = "",
|
||||||
@@ -50,7 +80,7 @@ const Service = struct {
|
|||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
/// The `/etc/init.csv` bytes, held because the parsed services slice into them.
|
/// The `/system/configuration/init.csv` bytes, held because the parsed services slice into them.
|
||||||
var init_csv: [4096]u8 = undefined;
|
var init_csv: [4096]u8 = undefined;
|
||||||
var services: [max_services]Service = .{Service{}} ** max_services;
|
var services: [max_services]Service = .{Service{}} ** max_services;
|
||||||
var service_count: usize = 0;
|
var service_count: usize = 0;
|
||||||
@@ -65,31 +95,24 @@ var restart_counts: [max_services]u32 = .{0} ** max_services;
|
|||||||
var shutting_down = false;
|
var shutting_down = false;
|
||||||
var supervision_endpoint: ipc.Handle = 0;
|
var supervision_endpoint: ipc.Handle = 0;
|
||||||
|
|
||||||
/// Parse `/etc/init.csv` into `services`, in file order (startup order; shutdown is
|
/// Parse `/system/configuration/init.csv` into `services`, in file order (startup order; shutdown is
|
||||||
/// the reverse). Each row is a binary path followed by its argv, comma-separated;
|
/// the reverse). Each row is a binary path followed by its argv, comma-separated;
|
||||||
/// `#` comments and blank lines are ignored. The file lives in the initial ramdisk,
|
/// `#` comments and blank lines are ignored. The file lives in the initial ramdisk,
|
||||||
/// which the kernel serves directly, so init — PID 1, running before any filesystem
|
/// which the kernel serves directly, so init — PID 1, running before any filesystem
|
||||||
/// service — reads it with a plain fs.open, the same mechanism the device manager
|
/// service — reads it with a plain fs.open, the same mechanism the device manager
|
||||||
/// uses for /etc/devices.csv. A missing file means no services (the no-ramdisk
|
/// uses for /system/configuration/devices.csv. A missing file means no services (the no-ramdisk
|
||||||
/// isolation test): loud, but not fatal.
|
/// isolation test): loud, but not fatal.
|
||||||
fn loadServices() void {
|
fn loadServices() void {
|
||||||
var file = fs.open("/etc/init.csv", .{}) orelse {
|
const used = readConfiguration("/system/configuration/init.csv", &init_csv) orelse {
|
||||||
_ = logging.write("/system/services/init: /etc/init.csv missing — no services started\n");
|
_ = logging.write("/system/services/init: /system/configuration/init.csv missing — no services started\n");
|
||||||
return;
|
return;
|
||||||
};
|
};
|
||||||
defer file.close();
|
|
||||||
var used: usize = 0;
|
|
||||||
while (used < init_csv.len) {
|
|
||||||
const n = file.read(init_csv[used..]) orelse break;
|
|
||||||
if (n == 0) break;
|
|
||||||
used += n;
|
|
||||||
}
|
|
||||||
var lines = std.mem.splitScalar(u8, init_csv[0..used], '\n');
|
var lines = std.mem.splitScalar(u8, init_csv[0..used], '\n');
|
||||||
while (lines.next()) |line| {
|
while (lines.next()) |line| {
|
||||||
const body = csv.stripComment(line);
|
const body = csv.stripComment(line);
|
||||||
if (body.len == 0) continue;
|
if (body.len == 0) continue;
|
||||||
if (service_count >= services.len) {
|
if (service_count >= services.len) {
|
||||||
_ = logging.write("/system/services/init: /etc/init.csv has more services than the table holds\n");
|
_ = logging.write("/system/services/init: /system/configuration/init.csv has more services than the table holds\n");
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
var it = csv.fields(body);
|
var it = csv.fields(body);
|
||||||
@@ -107,11 +130,587 @@ fn loadServices() void {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Read a whole configuration file into `into`, returning the byte count. Both of
|
||||||
|
/// init's manifests live in the initial ramdisk the kernel serves directly, so this
|
||||||
|
/// works before any filesystem service exists. A file that fills the buffer exactly
|
||||||
|
/// is reported: a manifest silently losing its last rows is a policy change nobody
|
||||||
|
/// asked for, and the symptom (one service refused a name) points nowhere near it.
|
||||||
|
fn readConfiguration(path: []const u8, into: []u8) ?usize {
|
||||||
|
var file = fs.open(path, .{}) orelse return null;
|
||||||
|
defer file.close();
|
||||||
|
var used: usize = 0;
|
||||||
|
while (used < into.len) {
|
||||||
|
const n = file.read(into[used..]) orelse break;
|
||||||
|
if (n == 0) break;
|
||||||
|
used += n;
|
||||||
|
}
|
||||||
|
if (used == into.len) std.log.info("{s} filled the read buffer — rows past {d} bytes are lost", .{ path, used });
|
||||||
|
return used;
|
||||||
|
}
|
||||||
|
|
||||||
/// Give up restarting a service after this many crashes — a crash-loop cap, so a service
|
/// Give up restarting a service after this many crashes — a crash-loop cap, so a service
|
||||||
/// that faults immediately on every spawn doesn't respawn forever.
|
/// that faults immediately on every spawn doesn't respawn forever.
|
||||||
const maximum_restarts = 3;
|
const maximum_restarts = 3;
|
||||||
|
|
||||||
pub fn main() void {
|
// --- the registry: /protocol ------------------------------------------------
|
||||||
|
|
||||||
|
/// Longest contract name the namespace admits (`display`, `test/shared-memory`)
|
||||||
|
/// and the most that may be bound at once. Both static, like everything else
|
||||||
|
/// init holds.
|
||||||
|
const maximum_name = 64;
|
||||||
|
const maximum_bindings = 16;
|
||||||
|
const maximum_grants = 64;
|
||||||
|
|
||||||
|
/// One bound contract: the name, the provider's endpoint (a capability init
|
||||||
|
/// holds and hands to whoever opens the name), and the provenance a diagnostic
|
||||||
|
/// listing answers "who serves this?" with.
|
||||||
|
const Binding = struct {
|
||||||
|
used: bool = false,
|
||||||
|
name: [maximum_name]u8 = undefined,
|
||||||
|
name_len: usize = 0,
|
||||||
|
endpoint: ipc.Handle = 0,
|
||||||
|
task: u32 = 0,
|
||||||
|
binary: [64]u8 = undefined,
|
||||||
|
binary_len: usize = 0,
|
||||||
|
|
||||||
|
fn nameSlice(self: *const Binding) []const u8 {
|
||||||
|
return self.name[0..self.name_len];
|
||||||
|
}
|
||||||
|
fn binarySlice(self: *const Binding) []const u8 {
|
||||||
|
return self.binary[0..self.binary_len];
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
var bindings: [maximum_bindings]Binding = .{Binding{}} ** maximum_bindings;
|
||||||
|
|
||||||
|
/// What a grant row permits.
|
||||||
|
///
|
||||||
|
/// - `bind` — claim the name, i.e. provide the contract.
|
||||||
|
/// - `open` — reach the name, i.e. speak the contract to whoever provides it.
|
||||||
|
/// - `supervise` — stand in a third task's supervision chain: a task running this
|
||||||
|
/// binary, under this supervisor, may be the supervising task named by an
|
||||||
|
/// `open` row for this contract. It grants the *delegate* nothing itself.
|
||||||
|
///
|
||||||
|
/// `supervise` exists because attestation is deliberately one hop deep
|
||||||
|
/// (`supervisorSatisfies`): init vouches only for tasks it or the kernel started.
|
||||||
|
/// The driver tree is deeper than that — the device manager starts the PS/2 bus,
|
||||||
|
/// and the bus starts the keyboard and mouse drivers — so without a way to say
|
||||||
|
/// "this task is an authorized supervisor", a legitimate grandchild would be
|
||||||
|
/// indistinguishable from a laundering deputy. Naming the delegate in the
|
||||||
|
/// manifest is what tells them apart, and it is the same shape as every other
|
||||||
|
/// row: a binary, the supervisor it must have, and the contract it concerns.
|
||||||
|
/// Deliberately `open`-only — a delegate may vouch for what its children may
|
||||||
|
/// *reach*, never for what they may *claim* — so the bind path's attestation is
|
||||||
|
/// exactly what P2 shipped and every refusal it makes still holds.
|
||||||
|
const Permission = enum { bind, open, supervise };
|
||||||
|
|
||||||
|
/// One row of `/system/configuration/protocol.csv`. Every field may end in `*`,
|
||||||
|
/// which matches any tail — the subtree scoping the design doc describes, and
|
||||||
|
/// what lets one row grant the whole `/test/` family its `test/...` names.
|
||||||
|
const Grant = struct {
|
||||||
|
binary: []const u8 = "",
|
||||||
|
supervisor: []const u8 = "",
|
||||||
|
permission: Permission = .bind,
|
||||||
|
name: []const u8 = "",
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Roomier than init.csv's: this manifest carries a row per provider per spawn
|
||||||
|
/// path, a row per client per contract it reaches, and its own format
|
||||||
|
/// documentation — which is most of the bytes, and is the point of the file.
|
||||||
|
var protocol_csv: [16384]u8 = undefined;
|
||||||
|
var grants: [maximum_grants]Grant = .{Grant{}} ** maximum_grants;
|
||||||
|
var grant_count: usize = 0;
|
||||||
|
|
||||||
|
/// Parse `/system/configuration/protocol.csv` — the grant manifest. Separate from
|
||||||
|
/// init.csv because every field there after the path is argv, and overloading that
|
||||||
|
/// would be ambiguous; separate *files* also means a grant exists for binaries init
|
||||||
|
/// never spawns (the drivers, which the device manager owns).
|
||||||
|
fn loadGrants() void {
|
||||||
|
const used = readConfiguration("/system/configuration/protocol.csv", &protocol_csv) orelse {
|
||||||
|
_ = logging.write("/system/services/init: /system/configuration/protocol.csv missing — no protocol may be bound\n");
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
var lines = std.mem.splitScalar(u8, protocol_csv[0..used], '\n');
|
||||||
|
while (lines.next()) |line| {
|
||||||
|
const body = csv.stripComment(line);
|
||||||
|
if (body.len == 0) continue;
|
||||||
|
if (grant_count >= grants.len) {
|
||||||
|
_ = logging.write("/system/services/init: /system/configuration/protocol.csv has more rows than the table holds\n");
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
var it = csv.fields(body);
|
||||||
|
const binary = it.next() orelse continue;
|
||||||
|
const supervisor = it.next() orelse continue;
|
||||||
|
const permission = it.next() orelse continue;
|
||||||
|
const name = it.next() orelse continue;
|
||||||
|
if (binary.len == 0 or supervisor.len == 0 or name.len == 0) continue;
|
||||||
|
const kind: Permission = if (std.mem.eql(u8, permission, "bind"))
|
||||||
|
.bind
|
||||||
|
else if (std.mem.eql(u8, permission, "open"))
|
||||||
|
.open
|
||||||
|
else if (std.mem.eql(u8, permission, "supervise"))
|
||||||
|
.supervise
|
||||||
|
else
|
||||||
|
continue; // an unreadable row grants nothing rather than something wrong
|
||||||
|
grants[grant_count] = .{ .binary = binary, .supervisor = supervisor, .permission = kind, .name = name };
|
||||||
|
grant_count += 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Match a manifest field against a value: exact, or a trailing `*` matching any
|
||||||
|
/// tail. The wildcard is how a subtree is granted whole (`/test/*` for every test
|
||||||
|
/// fixture, `test/*` for every name they may claim).
|
||||||
|
fn matches(pattern: []const u8, value: []const u8) bool {
|
||||||
|
if (pattern.len != 0 and pattern[pattern.len - 1] == '*') {
|
||||||
|
const prefix = pattern[0 .. pattern.len - 1];
|
||||||
|
return value.len >= prefix.len and std.mem.eql(u8, value[0..prefix.len], prefix);
|
||||||
|
}
|
||||||
|
return std.mem.eql(u8, pattern, value);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A snapshot of the kernel's process records — the only identity in the system
|
||||||
|
/// that cannot be forged, because the kernel stamps it at spawn. Refreshed per
|
||||||
|
/// authorization; binds are rare, so the copy costs nothing that matters.
|
||||||
|
var process_table: [64]process.ProcessDescriptor = undefined;
|
||||||
|
var process_count: usize = 0;
|
||||||
|
var process_truncated = false;
|
||||||
|
|
||||||
|
fn refreshProcessTable() void {
|
||||||
|
const total = process.processes(&process_table);
|
||||||
|
process_count = @min(total, process_table.len);
|
||||||
|
process_truncated = total > process_table.len;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether task `id` is still alive, as the last snapshot saw it. A snapshot that
|
||||||
|
/// did not fit answers "alive" for anything it did not list: refusing a bind is
|
||||||
|
/// recoverable, stealing a live provider's name is not.
|
||||||
|
fn taskAlive(id: u32) bool {
|
||||||
|
return descriptorOf(id) != null or process_truncated;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn descriptorOf(id: u32) ?*const process.ProcessDescriptor {
|
||||||
|
for (process_table[0..process_count]) |*descriptor| {
|
||||||
|
if (descriptor.id == id) return descriptor;
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn nameOf(descriptor: *const process.ProcessDescriptor) []const u8 {
|
||||||
|
const length = @min(@as(usize, descriptor.name_length), descriptor.name.len);
|
||||||
|
return descriptor.name[0..length];
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The name a kernel task answers to in a grant row. Kernel tasks carry no
|
||||||
|
/// binary, so the manifest spells the harness's parentage `kernel`.
|
||||||
|
const kernel_supervisor = "kernel";
|
||||||
|
|
||||||
|
/// init's own task id, read once at startup. Ids are monotonic and never reused
|
||||||
|
/// (system/kernel/process.zig), so an id comparison is an *identity* test where a
|
||||||
|
/// name comparison is only a resemblance test — the whole basis of the
|
||||||
|
/// attestation below.
|
||||||
|
var own_task: u32 = 0;
|
||||||
|
|
||||||
|
/// Whether `id` is a process THIS init spawned: a lookup in its own child table,
|
||||||
|
/// which is the one record of "I started that one" nobody else can write.
|
||||||
|
fn spawnedByUs(id: u32) bool {
|
||||||
|
if (id == 0) return false;
|
||||||
|
for (child_ids[0..service_count]) |child| {
|
||||||
|
if (child == id) return true;
|
||||||
|
}
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Who is asking, attested by the kernel: the caller's binary, and the **task**
|
||||||
|
/// that spawned it — an id, not a name.
|
||||||
|
///
|
||||||
|
/// A name alone is not identity: `spawn` is ungated, so a hostile process can
|
||||||
|
/// start a granted binary itself and would inherit its grants. Neither is the
|
||||||
|
/// supervisor's *name* enough, and this is the trap the first cut fell into —
|
||||||
|
/// init and the device manager are ordinary bundled binaries, so an attacker
|
||||||
|
/// spawns its own `/system/services/init` and lets that instance spawn
|
||||||
|
/// `/system/services/input`. Both kernel-stamped names then match the grant row
|
||||||
|
/// exactly, and walking to the root of the chain does not help either: the
|
||||||
|
/// laundered chain still roots at the real PID 1. What refuses it is asking
|
||||||
|
/// *which task* the supervisor is, and only accepting one init can vouch for.
|
||||||
|
const Identity = struct {
|
||||||
|
/// The caller's binary path, exactly as the kernel stamped it at spawn.
|
||||||
|
binary: []const u8,
|
||||||
|
/// The supervising task's id. 0 means the kernel spawned the caller, which
|
||||||
|
/// no ring-3 process can arrange: every `system_spawn` stamps the caller as
|
||||||
|
/// the child's supervisor (system/kernel/process.zig `systemSpawn`).
|
||||||
|
supervisor_task: u32,
|
||||||
|
/// The supervising task's kernel-stamped binary — the grant row's supervisor
|
||||||
|
/// column is matched against this, and the refusal log prints it. `kernel`
|
||||||
|
/// when there is no supervising task.
|
||||||
|
supervisor_binary: []const u8,
|
||||||
|
/// Whether init can vouch for how the supervising task came to exist: it is
|
||||||
|
/// this init, a process this init spawned, or a process the KERNEL spawned.
|
||||||
|
/// A supervisor init cannot vouch for satisfies no row, however well its
|
||||||
|
/// name reads — that is the laundering deputy's refusal.
|
||||||
|
supervisor_vouched: bool,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// The process a task belongs to. A thread resolves to its leader: threads share
|
||||||
|
/// a binary (a thread's own record is named `thread`), and the supervision link
|
||||||
|
/// that matters is the process's.
|
||||||
|
fn leaderOf(descriptor: *const process.ProcessDescriptor) *const process.ProcessDescriptor {
|
||||||
|
if (descriptor.leader == descriptor.id) return descriptor;
|
||||||
|
return descriptorOf(descriptor.leader) orelse descriptor;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Resolve the badge on a request into an identity, one hop up the supervision
|
||||||
|
/// chain in the kernel's records — one hop is enough because the hop is attested
|
||||||
|
/// by id (see `supervisorSatisfies`), and every id in the chain init accepts is
|
||||||
|
/// one init or the kernel created.
|
||||||
|
fn identify(task: u32) ?Identity {
|
||||||
|
const caller = descriptorOf(task) orelse return null;
|
||||||
|
const leader = leaderOf(caller);
|
||||||
|
if (leader.supervisor == 0) return .{
|
||||||
|
.binary = nameOf(leader),
|
||||||
|
.supervisor_task = 0,
|
||||||
|
.supervisor_binary = kernel_supervisor,
|
||||||
|
.supervisor_vouched = true, // the kernel is the root of trust, not a claimant
|
||||||
|
};
|
||||||
|
// The supervising *task* may be a worker thread of the supervising process;
|
||||||
|
// its process is what the manifest names and what init recorded at spawn.
|
||||||
|
const supervisor = leaderOf(descriptorOf(leader.supervisor) orelse return null); // unattestable: refuse
|
||||||
|
return .{
|
||||||
|
.binary = nameOf(leader),
|
||||||
|
.supervisor_task = supervisor.id,
|
||||||
|
.supervisor_binary = nameOf(supervisor),
|
||||||
|
.supervisor_vouched = supervisor.id == own_task or
|
||||||
|
spawnedByUs(supervisor.id) or
|
||||||
|
supervisor.supervisor == 0,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether the caller's supervising task satisfies a grant row's supervisor
|
||||||
|
/// column. The column names *the authorized supervising task*, matched by
|
||||||
|
/// identity — the binary it must be, plus proof that this instance of that
|
||||||
|
/// binary is the authorized one:
|
||||||
|
///
|
||||||
|
/// - `kernel` is satisfied only by a genuinely kernel-spawned caller
|
||||||
|
/// (supervisor id 0). A ring-3 process cannot manufacture that: user
|
||||||
|
/// `system_spawn` always stamps the caller (system/kernel/process.zig).
|
||||||
|
/// - init's own binary is satisfied only when the supervising task IS this
|
||||||
|
/// init (`own_task`).
|
||||||
|
/// - any other binary — the device manager, a test fixture spawning another —
|
||||||
|
/// is satisfied only when the supervising task is one init spawned itself
|
||||||
|
/// (its own child table) or one the kernel spawned. Everything init and the
|
||||||
|
/// kernel start is therefore reachable; a chain that passes through a
|
||||||
|
/// process *neither* of them started is not.
|
||||||
|
fn supervisorSatisfies(column: []const u8, identity: Identity) bool {
|
||||||
|
if (std.mem.eql(u8, column, kernel_supervisor)) return identity.supervisor_task == 0;
|
||||||
|
if (identity.supervisor_task == 0) return false; // a kernel task answers to no binary column
|
||||||
|
if (!matches(column, identity.supervisor_binary)) return false;
|
||||||
|
return identity.supervisor_vouched;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether `identity` is granted `permission` on `name`.
|
||||||
|
fn granted(identity: Identity, permission: Permission, name: []const u8) bool {
|
||||||
|
for (grants[0..grant_count]) |grant| {
|
||||||
|
if (grant.permission != permission) continue;
|
||||||
|
if (!matches(grant.binary, identity.binary)) continue;
|
||||||
|
if (!supervisorSatisfies(grant.supervisor, identity)) continue;
|
||||||
|
if (!matches(grant.name, name)) continue;
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether `identity` may reach `name` — `granted(.open, …)`, plus the one hop
|
||||||
|
/// `open` takes that `bind` does not (`Permission.supervise`).
|
||||||
|
///
|
||||||
|
/// The hop is needed because the driver tree is three deep and attestation is
|
||||||
|
/// one: the PS/2 keyboard driver's supervising task is the PS/2 bus driver,
|
||||||
|
/// which the device manager started, which init started. Init cannot vouch for
|
||||||
|
/// the bus by acquaintance — it never met it — so the manifest says so instead,
|
||||||
|
/// and says it per contract: `ps2-bus` may be the supervisor named in an `open`
|
||||||
|
/// grant for `ps2-bus` and for `input`, and for nothing else.
|
||||||
|
fn mayOpen(identity: Identity, name: []const u8) bool {
|
||||||
|
if (granted(identity, .open, name)) return true;
|
||||||
|
return delegatedOpen(identity, name);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The delegated `open`: the row's supervisor column names the caller's actual
|
||||||
|
/// supervising task by binary, that task is one init cannot vouch for directly,
|
||||||
|
/// and a `supervise` row authorizes it for exactly this contract.
|
||||||
|
///
|
||||||
|
/// The delegate itself is attested the ordinary way (`granted` → strict
|
||||||
|
/// `supervisorSatisfies`), so the chain is still anchored one hop above it in
|
||||||
|
/// init or the kernel and the recursion stops there. Two hops of manifest, never
|
||||||
|
/// an unbounded walk — a laundering deputy is refused at the first hop nobody
|
||||||
|
/// wrote a row for.
|
||||||
|
fn delegatedOpen(identity: Identity, name: []const u8) bool {
|
||||||
|
if (identity.supervisor_task == 0) return false; // a kernel-spawned caller needs no delegate
|
||||||
|
if (identity.supervisor_vouched) return false; // already answered by `granted` above
|
||||||
|
const delegate = identify(identity.supervisor_task) orelse return false;
|
||||||
|
if (!granted(delegate, .supervise, name)) return false;
|
||||||
|
for (grants[0..grant_count]) |grant| {
|
||||||
|
if (grant.permission != .open) continue;
|
||||||
|
if (!matches(grant.binary, identity.binary)) continue;
|
||||||
|
if (!matches(grant.supervisor, identity.supervisor_binary)) continue;
|
||||||
|
if (!matches(grant.name, name)) continue;
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn findBinding(name: []const u8) ?*Binding {
|
||||||
|
for (&bindings) |*binding| {
|
||||||
|
if (binding.used and std.mem.eql(u8, binding.nameSlice(), name)) return binding;
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Release a binding: the provider's endpoint capability goes back to the handle
|
||||||
|
/// table, and the name is free for the next claimant. init's own cached power
|
||||||
|
/// channel goes with it — a closed handle number is reused by the next capability
|
||||||
|
/// that arrives, and a stale copy would quietly aim the shutdown call at a
|
||||||
|
/// stranger. So does the authorized power *task*: nothing may speak for a
|
||||||
|
/// contract nobody holds.
|
||||||
|
fn releaseBinding(binding: *Binding) void {
|
||||||
|
if (std.mem.eql(u8, binding.nameSlice(), power_contract)) {
|
||||||
|
power_endpoint = null;
|
||||||
|
power_task = null;
|
||||||
|
power_pending = false;
|
||||||
|
}
|
||||||
|
_ = ipc.close(binding.endpoint);
|
||||||
|
binding.* = .{};
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Drop every name a dead process held. Called when a supervised child dies (so
|
||||||
|
/// the restarted instance can bind again) and whenever a bind finds the current
|
||||||
|
/// owner gone — providers init does not supervise need the second path.
|
||||||
|
fn unbindTask(task: u32) void {
|
||||||
|
for (&bindings) |*binding| {
|
||||||
|
if (binding.used and binding.task == task) {
|
||||||
|
std.log.info("/protocol/{s} released ({s} is gone)", .{ binding.nameSlice(), binding.binarySlice() });
|
||||||
|
releaseBinding(binding);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A contract name as the namespace spells it: the mount-relative path a resolve
|
||||||
|
/// hands us ("/display") and the name a bind sends ("display") are the same thing
|
||||||
|
/// with and without a leading slash, so one normaliser serves both. Empty or
|
||||||
|
/// longer than the namespace admits is not a name.
|
||||||
|
fn contractName(raw: []const u8) ?[]const u8 {
|
||||||
|
const name = if (raw.len != 0 and raw[0] == '/') raw[1..] else raw;
|
||||||
|
if (name.len == 0 or name.len > maximum_name) return null;
|
||||||
|
return name;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The provider's endpoint that will ride the *next* reply, when the request was
|
||||||
|
/// an `open` that found its contract.
|
||||||
|
var pending_capability: ?ipc.Handle = null;
|
||||||
|
|
||||||
|
/// The ownership rule for a capability that arrives with a turn of the loop —
|
||||||
|
/// **the turn owns it until a handler takes it, and closes whatever is left** —
|
||||||
|
/// lives in `ipc.Arrival`, next to `replyWait`, because it is not PID 1's rule:
|
||||||
|
/// the service harness every other service runs (library/kernel/service.zig) had
|
||||||
|
/// the identical hole and now states the identical contract.
|
||||||
|
const Arrival = ipc.Arrival;
|
||||||
|
|
||||||
|
/// The one contract init is itself a client of. It never resolves the name — it
|
||||||
|
/// *is* the registry, so it reads its own table; the binding is what hands it the
|
||||||
|
/// channel.
|
||||||
|
const power_contract = "power";
|
||||||
|
|
||||||
|
/// Set when `power` is bound: init subscribes to it on the next turn of the loop,
|
||||||
|
/// never inside the bind — the provider is blocked on our reply until then, so
|
||||||
|
/// calling it here would deadlock the pair.
|
||||||
|
var power_pending = false;
|
||||||
|
var power_endpoint: ?ipc.Handle = null;
|
||||||
|
|
||||||
|
/// The one task authorized to deliver power events: whoever holds the `power`
|
||||||
|
/// binding. Recorded at the bind and cleared with the binding, so a provider that
|
||||||
|
/// dies and rebinds re-derives it with no further ceremony.
|
||||||
|
///
|
||||||
|
/// This is the *authentication* for the shutdown path. init's registry endpoint
|
||||||
|
/// is its supervision endpoint, and `fs_resolve("/protocol")` installs a sendable
|
||||||
|
/// handle to it in any caller's table — so after P2 every ring-3 process can post
|
||||||
|
/// into PID 1's mailbox. A power event may therefore never be believed on the
|
||||||
|
/// strength of its payload; it is believed because the kernel stamped the
|
||||||
|
/// sender's task id on it and that id is the provider's.
|
||||||
|
var power_task: ?u32 = null;
|
||||||
|
|
||||||
|
/// Whether the heartbeat's re-arming timer is running. A timer landing carries no
|
||||||
|
/// identity, so the loop cannot tell one timer from another — which means exactly
|
||||||
|
/// one may ever be in flight, or every landing re-arms and the beat doubles. (It
|
||||||
|
/// did: two beats a second is enough extra chatter to cut a driver's echoed line
|
||||||
|
/// in half on the shared serial stream.) So the deferred power subscribe borrows
|
||||||
|
/// the heartbeat's tick when there is one, and arms its own only when there is not.
|
||||||
|
var heartbeat_running = false;
|
||||||
|
|
||||||
|
/// Answer one registry request. Writes a vfs-protocol reply into `reply` and
|
||||||
|
/// returns its length; a capability the reply must carry lands in
|
||||||
|
/// `pending_capability`. `arrived` is the capability the *request* carried, owned
|
||||||
|
/// by the turn — nothing here has to close it, only `bind` has to claim it.
|
||||||
|
fn serveRegistry(request_bytes: []const u8, reply: []u8, sender: u32, arrived: *Arrival) usize {
|
||||||
|
if (request_bytes.len < vfs_protocol.request_size)
|
||||||
|
return answer(reply, -envelope.EPROTO, 0, 0);
|
||||||
|
// The header is read field by field rather than reinterpreted whole: the
|
||||||
|
// operation is an enum on the wire and the bytes come from anyone at all, so
|
||||||
|
// a value outside it must be a refusal, never a decoded enum.
|
||||||
|
const operation = std.mem.readInt(u32, request_bytes[0..4], .little);
|
||||||
|
const cursor = std.mem.readInt(u64, request_bytes[16..24], .little);
|
||||||
|
const declared = std.mem.readInt(u32, request_bytes[24..28], .little);
|
||||||
|
const payload_len = @min(@as(usize, declared), request_bytes.len - vfs_protocol.request_size);
|
||||||
|
const payload = request_bytes[vfs_protocol.request_size..][0..payload_len];
|
||||||
|
|
||||||
|
if (operation == @intFromEnum(vfs_protocol.Operation.bind))
|
||||||
|
return answer(reply, onBind(sender, payload, arrived), 0, 0);
|
||||||
|
// Only `bind` claims a capability; one attached to anything else is closed by
|
||||||
|
// the turn's `defer` in the loop, along with the ones sent to a request that
|
||||||
|
// was too short to name a verb at all.
|
||||||
|
if (operation == @intFromEnum(vfs_protocol.Operation.open)) return onOpen(reply, sender, payload);
|
||||||
|
if (operation == @intFromEnum(vfs_protocol.Operation.readdir)) return onReaddir(reply, cursor);
|
||||||
|
// Everything else a filesystem answers is meaningless here: `/protocol` holds
|
||||||
|
// contracts, not bytes.
|
||||||
|
return answer(reply, -envelope.ENOSYS, 0, 0);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Lay down a vfs reply header (and say how many payload bytes follow it).
|
||||||
|
fn answer(reply: []u8, status: i32, node: u64, payload_len: usize) usize {
|
||||||
|
const header = vfs_protocol.Reply{ .status = status, .node = node, .len = @intCast(payload_len) };
|
||||||
|
@memcpy(reply[0..vfs_protocol.reply_size], std.mem.asBytes(&header));
|
||||||
|
return vfs_protocol.reply_size + payload_len;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `bind(name, capability = the provider's endpoint)`. The capability is the
|
||||||
|
/// point of the call, so a bind without one is malformed. Every refusal below
|
||||||
|
/// simply returns: the endpoint stays the turn's, and the turn closes it — which
|
||||||
|
/// is why there is not one `ipc.close` on the way out of any of the six of them.
|
||||||
|
/// The success path is the only one that says anything about ownership, because
|
||||||
|
/// it is the only one that keeps the capability.
|
||||||
|
fn onBind(sender: u32, raw_name: []const u8, arrived: *Arrival) i32 {
|
||||||
|
if (arrived.peek() == null) return -envelope.EPROTO;
|
||||||
|
const name = contractName(raw_name) orelse return -envelope.ENOENT;
|
||||||
|
refreshProcessTable();
|
||||||
|
const identity = identify(sender) orelse return -envelope.EPERM;
|
||||||
|
if (!granted(identity, .bind, name)) {
|
||||||
|
std.log.info("refused bind of /protocol/{s} by {s} (pid {d}, supervisor {s} pid {d})", .{
|
||||||
|
name,
|
||||||
|
identity.binary,
|
||||||
|
sender,
|
||||||
|
identity.supervisor_binary,
|
||||||
|
identity.supervisor_task,
|
||||||
|
});
|
||||||
|
return -envelope.EPERM;
|
||||||
|
}
|
||||||
|
if (findBinding(name)) |existing| {
|
||||||
|
// Collision is an error — never last-writer-wins — unless the incumbent
|
||||||
|
// is dead, which is how a restarted provider retakes its own name.
|
||||||
|
if (taskAlive(existing.task)) {
|
||||||
|
std.log.info("refused bind of /protocol/{s}: held by {s} (pid {d})", .{ name, existing.binarySlice(), existing.task });
|
||||||
|
return -envelope.EBUSY;
|
||||||
|
}
|
||||||
|
releaseBinding(existing);
|
||||||
|
}
|
||||||
|
const slot = for (&bindings) |*binding| {
|
||||||
|
if (!binding.used) break binding;
|
||||||
|
} else return -envelope.ENOSPC;
|
||||||
|
|
||||||
|
// Claimed: the binding owns the endpoint from here, and `releaseBinding` is
|
||||||
|
// what closes it.
|
||||||
|
const endpoint = arrived.take().?;
|
||||||
|
slot.* = .{ .used = true, .endpoint = endpoint, .task = sender };
|
||||||
|
@memcpy(slot.name[0..name.len], name);
|
||||||
|
slot.name_len = name.len;
|
||||||
|
const binary_len = @min(identity.binary.len, slot.binary.len);
|
||||||
|
@memcpy(slot.binary[0..binary_len], identity.binary[0..binary_len]);
|
||||||
|
slot.binary_len = binary_len;
|
||||||
|
|
||||||
|
// Provenance, at the moment it becomes true: name -> pid -> binary path.
|
||||||
|
std.log.info("/protocol/{s} -> pid {d} {s}", .{ name, sender, slot.binarySlice() });
|
||||||
|
if (std.mem.eql(u8, name, power_contract)) {
|
||||||
|
power_endpoint = endpoint;
|
||||||
|
// The bind is also the authentication: whoever holds `power` is the one
|
||||||
|
// task whose power events init will act on (see `onPowerEvent`).
|
||||||
|
power_task = sender;
|
||||||
|
power_pending = true;
|
||||||
|
// Wake ourselves once the reply has gone out; the subscribe call cannot
|
||||||
|
// happen while the power service is still blocked on it. The heartbeat's
|
||||||
|
// tick is that wake when it is running — see `heartbeat_running`.
|
||||||
|
if (!heartbeat_running) _ = time.timerOnce(supervision_endpoint, 1);
|
||||||
|
}
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `open(name)` -> the provider's endpoint, delivered as the reply's capability.
|
||||||
|
///
|
||||||
|
/// **A refusal and an absence are the same answer, and that is the whole point.**
|
||||||
|
/// The namespace is the restriction (docs/os-development/protocol-namespace.md):
|
||||||
|
/// what a process may open is what exists for it, so "you may not have this" and
|
||||||
|
/// "there is no such thing" collapse into one reply — `-ENOENT`, no payload, no
|
||||||
|
/// capability. A caller therefore has no oracle: it cannot use `open` to learn
|
||||||
|
/// that a contract it lacks is bound, and — the reason this matters beyond
|
||||||
|
/// tidiness — stage two's supervisor can refuse, stall for a human, or substitute
|
||||||
|
/// a fake without the child being able to tell which happened.
|
||||||
|
///
|
||||||
|
/// Indistinguishable is a claim about *work done*, not only about the bytes, so
|
||||||
|
/// both questions are asked on every open whatever the first one answers: the
|
||||||
|
/// process table is refreshed, the caller identified, the grants scanned and the
|
||||||
|
/// bindings scanned, and only then is the single verdict formed. Nothing here
|
||||||
|
/// logs, either — `klog_read` is ungated (system/kernel/process.zig), so a line
|
||||||
|
/// written on one branch is a line the refused caller can read, and a serial line
|
||||||
|
/// costs milliseconds it could time. The operator's diagnosis is the pair the
|
||||||
|
/// namespace already publishes on purpose: `readdir` over `/protocol` says what is
|
||||||
|
/// bound, `/system/configuration/protocol.csv` says who may reach it, and the
|
||||||
|
/// client's own retry loop says which one it wanted.
|
||||||
|
///
|
||||||
|
/// (Not constant-time in the cryptographic sense, and not claimed to be: the two
|
||||||
|
/// scans stop at the row they match, and the optimiser is free to sink a pure
|
||||||
|
/// table walk past a branch that discards it. What is removed is the difference a
|
||||||
|
/// caller could actually measure or read — a syscall on one branch and not the
|
||||||
|
/// other, a line in a world-readable log ring, or a serial write costing
|
||||||
|
/// milliseconds.)
|
||||||
|
fn onOpen(reply: []u8, sender: u32, raw_name: []const u8) usize {
|
||||||
|
const name = contractName(raw_name) orelse return answer(reply, -envelope.ENOENT, 0, 0);
|
||||||
|
refreshProcessTable();
|
||||||
|
const identity = identify(sender);
|
||||||
|
const permitted = if (identity) |who| mayOpen(who, name) else false;
|
||||||
|
const binding = findBinding(name);
|
||||||
|
if (!permitted) return answer(reply, -envelope.ENOENT, 0, 0);
|
||||||
|
const found = binding orelse return answer(reply, -envelope.ENOENT, 0, 0);
|
||||||
|
pending_capability = found.endpoint;
|
||||||
|
return answer(reply, 0, 0, 0);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `readdir(cursor)` — the namespace, browsable. One entry per turn, as the vfs
|
||||||
|
/// protocol lists any directory: kind `protocol`, the contract's name, and the
|
||||||
|
/// provider's task id in `size`, so a plain listing answers "who serves this?".
|
||||||
|
fn onReaddir(reply: []u8, cursor: u64) usize {
|
||||||
|
var index: u64 = 0;
|
||||||
|
for (&bindings) |*binding| {
|
||||||
|
if (!binding.used) continue;
|
||||||
|
if (index != cursor) {
|
||||||
|
index += 1;
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
const name = binding.nameSlice();
|
||||||
|
const entry = vfs_protocol.DirectoryEntry{
|
||||||
|
.kind = @intFromEnum(vfs_protocol.NodeKind.protocol),
|
||||||
|
.name_len = @intCast(name.len),
|
||||||
|
.size = binding.task,
|
||||||
|
};
|
||||||
|
const total = vfs_protocol.directory_entry_size + name.len;
|
||||||
|
if (vfs_protocol.reply_size + total > reply.len) return answer(reply, -envelope.EPROTO, 0, 0);
|
||||||
|
@memcpy(reply[vfs_protocol.reply_size..][0..vfs_protocol.directory_entry_size], std.mem.asBytes(&entry));
|
||||||
|
@memcpy(reply[vfs_protocol.reply_size + vfs_protocol.directory_entry_size ..][0..name.len], name);
|
||||||
|
return answer(reply, 0, 0, total);
|
||||||
|
}
|
||||||
|
return answer(reply, 0, 0, 0); // end of directory
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn main(startup: process.Init) void {
|
||||||
|
// `registry` is the scenario mode: serve /protocol and nothing else. The
|
||||||
|
// kernel test harness spawns its own providers directly, so it wants the
|
||||||
|
// naming layer up without init's whole service list underneath it
|
||||||
|
// (docs/security-track-plan.md, decision 9).
|
||||||
|
const registry_only = if (startup.arguments.get(1)) |role| std.mem.eql(u8, role, "registry") else false;
|
||||||
|
|
||||||
// Prove the heap end to end: allocate through the runtime allocator (which
|
// Prove the heap end to end: allocate through the runtime allocator (which
|
||||||
// mmaps pages from the kernel and carves them with the free list), write into
|
// mmaps pages from the kernel and carves them with the free list), write into
|
||||||
// that heap buffer (exercising the widened debug_write bounds check), and
|
// that heap buffer (exercising the widened debug_write bounds check), and
|
||||||
@@ -128,25 +727,38 @@ pub fn main() void {
|
|||||||
|
|
||||||
// One endpoint carries everything init waits on: children's exit
|
// One endpoint carries everything init waits on: children's exit
|
||||||
// notifications (they are spawned supervised against it), init's own
|
// notifications (they are spawned supervised against it), init's own
|
||||||
// signals, and power events it subscribes to. All arrive in the loop below.
|
// signals, power events it subscribes to, and — since PID 1 is the registrar
|
||||||
|
// — every /protocol request. One thread can wait in one place, so they share
|
||||||
|
// a mailbox and the loop below tells them apart.
|
||||||
supervision_endpoint = ipc.createIpcEndpoint() orelse {
|
supervision_endpoint = ipc.createIpcEndpoint() orelse {
|
||||||
_ = logging.write("/system/services/init: no endpoint\n");
|
_ = logging.write("/system/services/init: no endpoint\n");
|
||||||
return;
|
return;
|
||||||
};
|
};
|
||||||
_ = process.bindSignals(supervision_endpoint);
|
_ = process.bindSignals(supervision_endpoint);
|
||||||
|
|
||||||
|
// Our own id, before anything can ask us a question. It is half of the
|
||||||
|
// registrar's authority: a grant row naming init as the supervisor is
|
||||||
|
// satisfied by *this* task and no other instance of this binary
|
||||||
|
// (`supervisorSatisfies`).
|
||||||
|
own_task = process.taskId();
|
||||||
|
|
||||||
|
// The namespace goes up BEFORE anything is spawned, so a service's first
|
||||||
|
// bind lands rather than retrying. The kernel reserves the prefix: this
|
||||||
|
// mount is the only one it will ever hold.
|
||||||
|
loadGrants();
|
||||||
|
if (!fs.mount("/protocol", supervision_endpoint)) {
|
||||||
|
_ = logging.write("/system/services/init: /protocol already mounted — not the registrar\n");
|
||||||
|
}
|
||||||
|
|
||||||
// Load the service list, then bring each up supervised so init can stop them
|
// Load the service list, then bring each up supervised so init can stop them
|
||||||
// cleanly. Best-effort and silent: each service announces its own readiness,
|
// cleanly. Best-effort and silent: each service announces its own readiness,
|
||||||
// and with no /etc/init.csv (an isolation test) the loop starts nothing.
|
// and with no /system/configuration/init.csv (an isolation test) the loop starts nothing.
|
||||||
|
if (!registry_only) {
|
||||||
loadServices();
|
loadServices();
|
||||||
for (services[0..service_count], 0..) |*service, i| {
|
for (services[0..service_count], 0..) |*service, i| {
|
||||||
if (process.spawnSupervised(service.path, service.arguments(), supervision_endpoint)) |id| child_ids[i] = id;
|
if (process.spawnSupervised(service.path, service.arguments(), supervision_endpoint)) |id| child_ids[i] = id;
|
||||||
}
|
}
|
||||||
|
}
|
||||||
// Subscribe to power events (retry: the power service registers well after
|
|
||||||
// init starts). Best-effort — without it, a `terminate` signal still
|
|
||||||
// triggers the same shutdown path.
|
|
||||||
subscribePower();
|
|
||||||
|
|
||||||
// A re-arming timer drives the liveness heartbeat — proof PID 1 is alive (the
|
// A re-arming timer drives the liveness heartbeat — proof PID 1 is alive (the
|
||||||
// init test's marker) and a -Dserial diagnostic. It is a serial/test-build-only
|
// init test's marker) and a -Dserial diagnostic. It is a serial/test-build-only
|
||||||
@@ -154,32 +766,125 @@ pub fn main() void {
|
|||||||
// wakes only for real work (signals, power events, children's exits), never for a
|
// wakes only for real work (signals, power events, children's exits), never for a
|
||||||
// periodic beat. `build_options.serial` is comptime, so the heartbeat — its timer
|
// periodic beat. `build_options.serial` is comptime, so the heartbeat — its timer
|
||||||
// and the handler below — folds away entirely when serial is off.
|
// and the handler below — folds away entirely when serial is off.
|
||||||
if (build_options.serial) _ = time.timerOnce(supervision_endpoint, 1000);
|
heartbeat_running = build_options.serial and !registry_only;
|
||||||
|
if (heartbeat_running) _ = time.timerOnce(supervision_endpoint, 1000);
|
||||||
|
|
||||||
var receive: [power_protocol.message_maximum]u8 = undefined;
|
var receive: [vfs_protocol.message_maximum]u8 = undefined;
|
||||||
|
var reply_buffer: [vfs_protocol.message_maximum]u8 = undefined;
|
||||||
|
var reply_len: usize = 0;
|
||||||
|
var reply_capability: ?ipc.Handle = null;
|
||||||
while (true) {
|
while (true) {
|
||||||
const got = ipc.replyWait(supervision_endpoint, &.{}, &receive, null);
|
const got = ipc.replyWait(supervision_endpoint, reply_buffer[0..reply_len], &receive, reply_capability);
|
||||||
|
reply_len = 0; // nothing owed until this turn's request says otherwise
|
||||||
|
reply_capability = null;
|
||||||
|
|
||||||
|
// Whatever capability came with this turn is the turn's, and the turn
|
||||||
|
// closes it unless a handler claims it. Structural on purpose — see
|
||||||
|
// `Arrival`; it is what keeps a zero-length call from spending a handle
|
||||||
|
// slot of PID 1's per call.
|
||||||
|
var arrived: Arrival = .{ .handle = got.cap };
|
||||||
|
defer arrived.release();
|
||||||
|
|
||||||
|
if (got.isNotification()) {
|
||||||
|
// Every badge on this branch is stamped by the KERNEL, and a stranger
|
||||||
|
// cannot stamp one: `ipc_send` — the only way a ring-3 process puts
|
||||||
|
// something in this mailbox with no reply owed — sets exactly
|
||||||
|
// `notify_badge_bit | notify_message_bit` and fills the low bits with
|
||||||
|
// the sender's own task id (system/kernel/ipc-synchronous.zig,
|
||||||
|
// `sendLocked`).
|
||||||
|
//
|
||||||
|
// That is a statement about `ipc_send`, and on its own it proved far
|
||||||
|
// too little: an attacker does not use `ipc_send` to forge a signal,
|
||||||
|
// it asks the kernel to deliver a real one *here*. `fs_resolve`
|
||||||
|
// hands any process a sendable handle to this endpoint, and
|
||||||
|
// `signal_bind`/`timer_bind`/`process_subscribe`/spawn's exit
|
||||||
|
// endpoint all used to accept any handle the caller held — so a
|
||||||
|
// stranger could point its own signal delivery at PID 1 and signal
|
||||||
|
// itself, and the terminate badge landing here was genuine in every
|
||||||
|
// bit. What makes these branches trustworthy is therefore in the
|
||||||
|
// KERNEL, not in this comment: binding a kernel notification to an
|
||||||
|
// endpoint now requires *owning* that endpoint (`ipc.ownedBy`), so a
|
||||||
|
// signal here comes only from our supervisor or our own group, a
|
||||||
|
// timer landing only from a timer we armed, and a child-exit notice
|
||||||
|
// only from a child we spawned. The buffered-message branch below is
|
||||||
|
// the one still carrying a stranger's bytes, and it is the one that
|
||||||
|
// authenticates its sender.
|
||||||
if (process.signalsFrom(got.badge)) |signals| {
|
if (process.signalsFrom(got.badge)) |signals| {
|
||||||
if (signals.has(.terminate)) shutDown();
|
if (signals.has(.terminate)) shutDown();
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
if (build_options.serial and got.isTimer()) {
|
if (got.isTimer()) {
|
||||||
|
// The pending power subscription rides any timer landing: by the
|
||||||
|
// time one arrives, the bind's reply has left and the power
|
||||||
|
// service is serving again.
|
||||||
|
if (power_pending) {
|
||||||
|
power_pending = false;
|
||||||
|
subscribePower();
|
||||||
|
}
|
||||||
|
if (heartbeat_running) {
|
||||||
_ = logging.write("/system/services/init: heartbeat\n");
|
_ = logging.write("/system/services/init: heartbeat\n");
|
||||||
_ = time.timerOnce(supervision_endpoint, 1000);
|
_ = time.timerOnce(supervision_endpoint, 1000);
|
||||||
|
}
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
if (got.isMessage() and got.len >= 2 and receive[0] == @intFromEnum(power_protocol.Operation.event)) {
|
if (got.isMessage()) {
|
||||||
// A power event (the only buffered messages init receives).
|
// A buffered message: the only thing here an anonymous stranger
|
||||||
if (receive[1] == @intFromEnum(power_protocol.Event.power_button)) shutDown();
|
// can put in front of PID 1. Authenticated by sender, never by
|
||||||
|
// payload — see `onPowerEvent`.
|
||||||
|
onPowerEvent(got.senderTaskId(), receive[0..got.len]);
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
if (got.isChildExit()) {
|
if (got.isChildExit()) {
|
||||||
restartChild(got.childProcessId());
|
restartChild(got.childProcessId());
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
// Anything else: keep waiting.
|
continue; // anything else: keep waiting
|
||||||
if (got.isNotification()) continue;
|
|
||||||
}
|
}
|
||||||
|
// The universal ping, answered by the empty reply. A ping may still carry
|
||||||
|
// a capability — the kernel installs one regardless of length — and this
|
||||||
|
// `continue` disposes of it through the turn's `defer`, which is exactly
|
||||||
|
// what it failed to do when the close lived in the branches.
|
||||||
|
if (got.len == 0) continue;
|
||||||
|
reply_len = serveRegistry(receive[0..got.len], &reply_buffer, got.senderTaskId(), &arrived);
|
||||||
|
reply_capability = pending_capability;
|
||||||
|
pending_capability = null;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A buffered message claiming to be a power event.
|
||||||
|
///
|
||||||
|
/// **Privileged control traffic is authenticated by sender, never by content.**
|
||||||
|
/// init's registry endpoint is its supervision endpoint, and `fs_resolve` installs
|
||||||
|
/// a sendable handle to any mount's backend in *any* caller's table
|
||||||
|
/// (system/kernel/process.zig), so after P2 every ring-3 process holds a handle it
|
||||||
|
/// can `ipc_send` into. Two payload bytes were once enough to reach `shutDown()`
|
||||||
|
/// from here — which stops every service and parks PID 1 in its final sleep,
|
||||||
|
/// destroying the registry for the rest of the boot, and does it for any process
|
||||||
|
/// that cares to ask.
|
||||||
|
///
|
||||||
|
/// The sender's task id is the fix, because it is not the sender's to choose: the
|
||||||
|
/// kernel stamps it into the badge's low bits as it copies the message into the
|
||||||
|
/// ring. Init is the registry, so it knows exactly which task holds `power`, and
|
||||||
|
/// that task alone is believed. A provider that dies and rebinds moves the
|
||||||
|
/// authorization with the binding; a name nothing holds authorizes nobody. Task
|
||||||
|
/// ids are never reused, so even a dead provider's id cannot be inherited.
|
||||||
|
///
|
||||||
|
/// (One task, not one process: the ACPI service is single-threaded and publishes
|
||||||
|
/// from the same task that bound the name. A threaded provider would want its
|
||||||
|
/// leader compared instead — which is a change to make when one appears, not a
|
||||||
|
/// looser rule to leave lying around for it.)
|
||||||
|
fn onPowerEvent(sender: u32, payload: []const u8) void {
|
||||||
|
const authorized = power_task orelse {
|
||||||
|
std.log.info("ignored a power event from pid {d}: nothing holds /protocol/power", .{sender});
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
if (sender != authorized) {
|
||||||
|
std.log.info("ignored a power event from pid {d}: /protocol/power is pid {d}", .{ sender, authorized });
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
if (payload.len < 2) return;
|
||||||
|
if (payload[0] != @intFromEnum(power_protocol.Operation.event)) return;
|
||||||
|
if (payload[1] == @intFromEnum(power_protocol.Event.power_button)) shutDown();
|
||||||
}
|
}
|
||||||
|
|
||||||
/// A supervised boot service died. Find which one and restart it — unless it exited
|
/// A supervised boot service died. Find which one and restart it — unless it exited
|
||||||
@@ -187,6 +892,10 @@ pub fn main() void {
|
|||||||
/// cap. Reclaiming the dead process is already the kernel's job (docs/process-lifecycle.md
|
/// cap. Reclaiming the dead process is already the kernel's job (docs/process-lifecycle.md
|
||||||
/// iron rule 1); init only decides whether to bring it back.
|
/// iron rule 1); init only decides whether to bring it back.
|
||||||
fn restartChild(id: u32) void {
|
fn restartChild(id: u32) void {
|
||||||
|
// Whatever it served, it serves no longer: the name goes back before the
|
||||||
|
// replacement asks for it, so the restarted instance binds rather than
|
||||||
|
// colliding with its own corpse.
|
||||||
|
unbindTask(id);
|
||||||
if (shutting_down) return; // deaths during the stop sequence are expected, not crashes
|
if (shutting_down) return; // deaths during the stop sequence are expected, not crashes
|
||||||
for (services[0..service_count], 0..) |*service, i| {
|
for (services[0..service_count], 0..) |*service, i| {
|
||||||
if (child_ids[i] != id) continue;
|
if (child_ids[i] != id) continue;
|
||||||
@@ -209,22 +918,26 @@ fn restartChild(id: u32) void {
|
|||||||
// An untracked child (e.g. the log-flush one-shot): nothing to restart.
|
// An untracked child (e.g. the log-flush one-shot): nothing to restart.
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Look up the power service and subscribe our endpoint (handed over as the
|
/// Subscribe our endpoint (handed over as the call's capability) to the power
|
||||||
/// call's capability) so events arrive as buffered messages here.
|
/// service, so events arrive as buffered messages here. init is the registry, so
|
||||||
|
/// it never resolves `/protocol/power` — it reads its own table, which is also
|
||||||
|
/// what makes this reachable at all: the subscription is armed by the bind that
|
||||||
|
/// put the endpoint there.
|
||||||
|
///
|
||||||
|
/// **This is the only place PID 1 blocks on another process, and it is the one
|
||||||
|
/// hazard the registrar has.** One thread serves both the namespace and this
|
||||||
|
/// call, so while it is outstanding init answers nobody: if the callee were
|
||||||
|
/// itself blocked asking init to resolve a name, the pair would never move. Two
|
||||||
|
/// things keep that from happening — the call is deferred to the next turn of
|
||||||
|
/// the loop (so the provider has its bind reply and is on its way to
|
||||||
|
/// `replyWait`), and the power provider resolves every name it needs *before* it
|
||||||
|
/// binds (system/services/acpi/acpi.zig, `manager_channel`). Any future service
|
||||||
|
/// init calls owes the same discipline.
|
||||||
fn subscribePower() void {
|
fn subscribePower() void {
|
||||||
var handle: ?ipc.Handle = null;
|
const handle = power_endpoint orelse return;
|
||||||
var tries: u32 = 0;
|
|
||||||
while (handle == null and tries < 200) : (tries += 1) {
|
|
||||||
handle = ipc.lookup(.power);
|
|
||||||
if (handle == null) time.sleepMillis(20);
|
|
||||||
}
|
|
||||||
// A missing power service is not fatal — init proceeds to its heartbeat and
|
|
||||||
// a `terminate` signal still drives shutdown. Silent so the no-ramdisk init
|
|
||||||
// test's heartbeat marker is the next line written.
|
|
||||||
const h = handle orelse return;
|
|
||||||
const request = power_protocol.Subscribe{};
|
const request = power_protocol.Subscribe{};
|
||||||
var reply: [power_protocol.message_maximum]u8 = undefined;
|
var reply: [power_protocol.message_maximum]u8 = undefined;
|
||||||
_ = ipc.callCap(h, std.mem.asBytes(&request), &reply, supervision_endpoint) catch {};
|
_ = ipc.callCap(handle, std.mem.asBytes(&request), &reply, supervision_endpoint) catch {};
|
||||||
}
|
}
|
||||||
|
|
||||||
/// The stop sequence: persist the log while storage is still up, then terminate
|
/// The stop sequence: persist the log while storage is still up, then terminate
|
||||||
@@ -242,7 +955,7 @@ fn shutDown() void {
|
|||||||
i -= 1;
|
i -= 1;
|
||||||
if (child_ids[i] != 0) process.stop(child_ids[i], 2000, supervision_endpoint);
|
if (child_ids[i] != 0) process.stop(child_ids[i], 2000, supervision_endpoint);
|
||||||
}
|
}
|
||||||
if (ipc.lookup(.power)) |h| {
|
if (power_endpoint) |h| {
|
||||||
const request = power_protocol.Shutdown{};
|
const request = power_protocol.Shutdown{};
|
||||||
var reply: [power_protocol.message_maximum]u8 = undefined;
|
var reply: [power_protocol.message_maximum]u8 = undefined;
|
||||||
_ = ipc.call(h, std.mem.asBytes(&request), &reply) catch {};
|
_ = ipc.call(h, std.mem.asBytes(&request), &reply) catch {};
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
//! The input service as a binary package (docs/build-packages-plan.md):
|
//! The input service as a binary package (docs/build-packages-plan.md):
|
||||||
//! this file names the binary and EXACTLY the modules its source imports —
|
//! this file names the binary and EXACTLY the modules its source imports —
|
||||||
//! the shared recipe and the module-to-domain map live in build-support.
|
//! build-support resolves each name from the domains this zon declares.
|
||||||
|
|
||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
const build_support = @import("build-support");
|
const build_support = @import("build-support");
|
||||||
@@ -9,7 +9,7 @@ pub fn build(b: *std.Build) void {
|
|||||||
const exe = build_support.userBinary(b, .{
|
const exe = build_support.userBinary(b, .{
|
||||||
.name = "input",
|
.name = "input",
|
||||||
.root_source_file = b.path("input.zig"),
|
.root_source_file = b.path("input.zig"),
|
||||||
.imports = &.{ "input-client", "input-protocol", "ipc", "logging", "process", "service" },
|
.imports = &.{ "channel", "input-client", "input-protocol", "ipc", "logging", "process", "service" },
|
||||||
});
|
});
|
||||||
b.installArtifact(exe);
|
b.installArtifact(exe);
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1,5 +1,5 @@
|
|||||||
//! system/services/input — the user-space input service. Shipped in the initial_ramdisk,
|
//! system/services/input — the user-space input service. Shipped in the initial_ramdisk,
|
||||||
//! spawned as a ring-3 process, and published under the well-known `input` service id. It
|
//! spawned as a ring-3 process, and bound at `/protocol/input`. It
|
||||||
//! is the fan-out point between **sources** (keyboard, mouse, and joystick/gamepad drivers)
|
//! is the fan-out point between **sources** (keyboard, mouse, and joystick/gamepad drivers)
|
||||||
//! and **subscribers** (any program that wants input): a source `publish`es an
|
//! and **subscribers** (any program that wants input): a source `publish`es an
|
||||||
//! `InputEvent`, and the service pushes it to every subscriber whose interest mask includes
|
//! `InputEvent`, and the service pushes it to every subscriber whose interest mask includes
|
||||||
@@ -20,6 +20,7 @@
|
|||||||
//! handle and `ipc.send`s each event to it.
|
//! handle and `ipc.send`s each event to it.
|
||||||
|
|
||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
|
const channel = @import("channel");
|
||||||
const ipc = @import("ipc");
|
const ipc = @import("ipc");
|
||||||
const process = @import("process");
|
const process = @import("process");
|
||||||
const service = @import("service");
|
const service = @import("service");
|
||||||
@@ -58,7 +59,13 @@ fn pruneDeadSubscribers() void {
|
|||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if (!alive) sub.* = .{};
|
// The slot owns the endpoint capability it was handed, so reclaiming the
|
||||||
|
// slot closes it — otherwise a process that subscribes and dies costs a
|
||||||
|
// handle-table slot that never comes back.
|
||||||
|
if (!alive) {
|
||||||
|
_ = ipc.close(sub.endpoint);
|
||||||
|
sub.* = .{};
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -84,10 +91,13 @@ fn broadcast(event: input_protocol.InputEvent) void {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Handle one request. `got` carries the sender badge (a task id) and, for subscribe, the
|
/// Handle one request. `got` carries the sender badge (a task id); `arrived` carries the
|
||||||
/// subscriber's endpoint capability in `got.cap`. Writes a `Reply` into `out` and returns
|
/// capability the request came with, under the same ownership rule the service harness
|
||||||
/// its length.
|
/// states (`ipc.Arrival`): **it belongs to the turn, and only a handler that means to keep
|
||||||
fn handle(message: []const u8, got: ipc.Received, out: []u8) usize {
|
/// it says `take`.** Everything else here — a short message, a `publish`, a subscribe that
|
||||||
|
/// finds the table full — simply returns, and the loop closes what arrived. Writes a
|
||||||
|
/// `Reply` into `out` and returns its length.
|
||||||
|
fn handle(message: []const u8, got: ipc.Received, out: []u8, arrived: *ipc.Arrival) usize {
|
||||||
const reply = struct {
|
const reply = struct {
|
||||||
fn write(buffer: []u8, status: i32) usize {
|
fn write(buffer: []u8, status: i32) usize {
|
||||||
const header = input_protocol.Reply{ .status = status };
|
const header = input_protocol.Reply{ .status = status };
|
||||||
@@ -101,11 +111,12 @@ fn handle(message: []const u8, got: ipc.Received, out: []u8) usize {
|
|||||||
|
|
||||||
switch (@as(input_protocol.Operation, @enumFromInt(request.operation))) {
|
switch (@as(input_protocol.Operation, @enumFromInt(request.operation))) {
|
||||||
.subscribe => {
|
.subscribe => {
|
||||||
const endpoint = got.cap orelse return reply.write(out, -1); // no endpoint passed
|
const endpoint = arrived.peek() orelse return reply.write(out, -1); // no endpoint passed
|
||||||
// A zero mask means "everything" (a subscriber that named no class still wants input).
|
// A zero mask means "everything" (a subscriber that named no class still wants input).
|
||||||
const mask = if (request.device_mask == 0) input_protocol.device_all else request.device_mask;
|
const mask = if (request.device_mask == 0) input_protocol.device_all else request.device_mask;
|
||||||
pruneDeadSubscribers();
|
pruneDeadSubscribers();
|
||||||
if (!addSubscriber(endpoint, @intCast(got.badge), mask)) return reply.write(out, -1); // table full
|
if (!addSubscriber(endpoint, @intCast(got.badge), mask)) return reply.write(out, -1); // table full
|
||||||
|
_ = arrived.take(); // claimed: the subscriber table holds it until that task dies
|
||||||
return reply.write(out, 0);
|
return reply.write(out, 0);
|
||||||
},
|
},
|
||||||
.publish => {
|
.publish => {
|
||||||
@@ -120,8 +131,8 @@ pub fn main() void {
|
|||||||
_ = logging.write("/system/services/input: no endpoint\n");
|
_ = logging.write("/system/services/input: no endpoint\n");
|
||||||
return;
|
return;
|
||||||
};
|
};
|
||||||
if (!ipc.register(.input, endpoint)) {
|
if (!channel.bindPatiently("input", endpoint)) {
|
||||||
_ = logging.write("/system/services/input: register failed\n");
|
_ = logging.write("/system/services/input: could not bind /protocol/input\n");
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
_ = logging.write("/system/services/input: ready\n");
|
_ = logging.write("/system/services/input: ready\n");
|
||||||
@@ -131,12 +142,21 @@ pub fn main() void {
|
|||||||
var receive: [input_protocol.request_size]u8 = undefined;
|
var receive: [input_protocol.request_size]u8 = undefined;
|
||||||
while (true) {
|
while (true) {
|
||||||
const got = ipc.replyWait(endpoint, reply_buffer[0..reply_len], &receive, null);
|
const got = ipc.replyWait(endpoint, reply_buffer[0..reply_len], &receive, null);
|
||||||
|
// Whatever capability came with this turn is the turn's, and the turn closes it
|
||||||
|
// unless `handle` claims it (`ipc.Arrival`). The kernel installs a sent capability
|
||||||
|
// whatever the message's length or kind, so this covers the notification
|
||||||
|
// `continue` and every refusal inside `handle` — otherwise about thirty-two
|
||||||
|
// capability-carrying calls, which need no authorization at all, exhaust this
|
||||||
|
// service's handle table and no further subscribe can ever land.
|
||||||
|
var arrived: ipc.Arrival = .{ .handle = got.cap };
|
||||||
|
defer arrived.release();
|
||||||
|
|
||||||
// Only synchronous client requests (subscribe/publish) arrive here; nothing sends
|
// Only synchronous client requests (subscribe/publish) arrive here; nothing sends
|
||||||
// this service asynchronous messages, so a notification wake would be spurious.
|
// this service asynchronous messages, so a notification wake would be spurious.
|
||||||
if (got.isNotification()) {
|
if (got.isNotification()) {
|
||||||
reply_len = 0;
|
reply_len = 0;
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
reply_len = handle(receive[0..got.len], got, &reply_buffer);
|
reply_len = handle(receive[0..got.len], got, &reply_buffer, &arrived);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -4,7 +4,7 @@
|
|||||||
//! demultiplexes it into **one file per process** on the flash volume:
|
//! demultiplexes it into **one file per process** on the flash volume:
|
||||||
//!
|
//!
|
||||||
//! <base>/<boot-stamp>/<binary-path>.log
|
//! <base>/<boot-stamp>/<binary-path>.log
|
||||||
//! e.g. /mnt/usb/var/log/2026-07-21T101530Z/system/services/fat.log
|
//! e.g. /volumes/usb/system/logs/2026-07-21T101530Z/system/services/fat.log
|
||||||
//!
|
//!
|
||||||
//! The boot stamp is the wall-clock time of boot (from klog_status), so one
|
//! The boot stamp is the wall-clock time of boot (from klog_status), so one
|
||||||
//! boot session is one self-contained directory; the kernel's own records go to
|
//! boot session is one self-contained directory; the kernel's own records go to
|
||||||
@@ -37,11 +37,11 @@ const time = @import("time");
|
|||||||
const logging = @import("logging");
|
const logging = @import("logging");
|
||||||
|
|
||||||
|
|
||||||
/// Where log trees live: the FHS path. The kernel VFS routes /var to whatever
|
/// Where log trees live: the hierarchy path. The kernel VFS routes /system/logs
|
||||||
/// volume the fat server mounted there (today: the /var subtree of the USB
|
/// to whatever volume the fat server mounted there (today: the /system/logs
|
||||||
/// flash volume) — swapping the persistent medium later touches fat's two
|
/// subtree of the USB flash volume) — swapping the persistent medium later
|
||||||
/// mount calls, never this constant.
|
/// touches fat's mount calls, never this constant.
|
||||||
const base = "/var/log";
|
const base = "/system/logs";
|
||||||
|
|
||||||
/// Drain cadence and the quiet period after which files are closed (flushed).
|
/// Drain cadence and the quiet period after which files are closed (flushed).
|
||||||
const tick_ms = 250;
|
const tick_ms = 250;
|
||||||
@@ -99,11 +99,11 @@ fn initialise(harness_endpoint: ipc.Handle) bool {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// The logger serves no protocol; the ping is answered by the harness.
|
/// The logger serves no protocol; the ping is answered by the harness.
|
||||||
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Handle) usize {
|
fn onMessage(message: []const u8, reply: []u8, sender: u32, arrived: *ipc.Arrival) usize {
|
||||||
_ = message;
|
_ = message;
|
||||||
_ = reply;
|
_ = reply;
|
||||||
_ = sender;
|
_ = sender;
|
||||||
_ = capability;
|
_ = arrived; // nothing here takes a capability: the harness closes what arrives
|
||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -128,7 +128,7 @@ fn onTerminate() void {
|
|||||||
|
|
||||||
fn tick() void {
|
fn tick() void {
|
||||||
if (!storage_ready) {
|
if (!storage_ready) {
|
||||||
// makePath doubles as the readiness probe: while /var is unmounted the
|
// makePath doubles as the readiness probe: while /system/logs is unmounted the
|
||||||
// resolve fails fast (no storage round trip) and the ring buffers; the
|
// resolve fails fast (no storage round trip) and the ring buffers; the
|
||||||
// first success creates the whole per-boot tree.
|
// first success creates the whole per-boot tree.
|
||||||
if (!fs.makePath(boot_directory[0..boot_directory_len])) return;
|
if (!fs.makePath(boot_directory[0..boot_directory_len])) return;
|
||||||
|
|||||||
+59
-6
@@ -172,7 +172,7 @@ CASES = [
|
|||||||
"smp": 4,
|
"smp": 4,
|
||||||
"timeout": 150,
|
"timeout": 150,
|
||||||
"qemu_extra": ["-device", "intel-iommu,intremap=off"],
|
"qemu_extra": ["-device", "intel-iommu,intremap=off"],
|
||||||
"expect": r"(?s)(?=.*/system/kernel: iommu online)(?=.*fat: mounted /mnt/usb)(?=.*fat-test: ok)",
|
"expect": r"(?s)(?=.*/system/kernel: iommu online)(?=.*fat: mounted /volumes/usb)(?=.*fat-test: ok)",
|
||||||
"fail": r"DANOS-TEST-RESULT: FAIL|DANOS-IOMMU-FAULT"},
|
"fail": r"DANOS-TEST-RESULT: FAIL|DANOS-IOMMU-FAULT"},
|
||||||
# DMA + MSI under translation: interrupt-IN reports arrive through translated DMA and
|
# DMA + MSI under translation: interrupt-IN reports arrive through translated DMA and
|
||||||
# the xHC's MSI/MSI-X still delivers (the 0xFEE00000 interrupt window bypasses second-
|
# the xHC's MSI/MSI-X still delivers (the 0xFEE00000 interrupt window bypasses second-
|
||||||
@@ -208,7 +208,7 @@ CASES = [
|
|||||||
"smp": 4,
|
"smp": 4,
|
||||||
"timeout": 150,
|
"timeout": 150,
|
||||||
"qemu_extra": ["-device", "amd-iommu,dma-remap=on,intremap=off"],
|
"qemu_extra": ["-device", "amd-iommu,dma-remap=on,intremap=off"],
|
||||||
"expect": r"(?s)(?=.*iommu online \(AMD-Vi\))(?=.*fat: mounted /mnt/usb)(?=.*fat-test: ok)",
|
"expect": r"(?s)(?=.*iommu online \(AMD-Vi\))(?=.*fat: mounted /volumes/usb)(?=.*fat-test: ok)",
|
||||||
"fail": r"DANOS-TEST-RESULT: FAIL|DANOS-IOMMU-FAULT"},
|
"fail": r"DANOS-TEST-RESULT: FAIL|DANOS-IOMMU-FAULT"},
|
||||||
# Port I/O grants: a claimed device's io_port resource lets a driver read/write its
|
# Port I/O grants: a claimed device's io_port resource lets a driver read/write its
|
||||||
# ports (PS/2 status 0x64), gated by the claim; out-of-range/unclaimed is refused.
|
# ports (PS/2 status 0x64), gated by the claim; out-of-range/unclaimed is refused.
|
||||||
@@ -343,6 +343,16 @@ CASES = [
|
|||||||
{"name": "usermem",
|
{"name": "usermem",
|
||||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
|
# The checked copy layer (system/kernel/user-memory.zig): kernel-side unit
|
||||||
|
# checks for the bound, presence, and leaf U/S + writable refusals, then a
|
||||||
|
# fixture aiming unmapped-but-in-range pointers at klog_read/klog_status/
|
||||||
|
# process_enumerate/device_enumerate/fs_resolve/debug_write. Each must come
|
||||||
|
# back as a wrapped -errno with the machine still running — before H1 every
|
||||||
|
# one of them dereferenced the bad page in ring 0 and halted it.
|
||||||
|
{"name": "user-memory",
|
||||||
|
"timeout": 60,
|
||||||
|
"expect": r"(?s)(?=.*DANOS-TEST-RESULT: PASS)(?=.*user-memory-test: ok)",
|
||||||
|
"fail": r"user-memory-test: FAIL|DANOS-TEST-RESULT: FAIL|CPU EXCEPTION|KERNEL PANIC"},
|
||||||
# Isolation: a ring-3 read of a kernel-only page must #PF with error code
|
# Isolation: a ring-3 read of a kernel-only page must #PF with error code
|
||||||
# 0x5 (present|user) at the user IP. ([\s\S] spans lines; `.` doesn't.)
|
# 0x5 (present|user) at the user IP. ([\s\S] spans lines; `.` doesn't.)
|
||||||
{"name": "user-pf",
|
{"name": "user-pf",
|
||||||
@@ -623,13 +633,13 @@ CASES = [
|
|||||||
"expect": r"usb-storage: ready[\s\S]*usb-storage: block 0 signature 0x55aa",
|
"expect": r"usb-storage: ready[\s\S]*usb-storage: block 0 signature 0x55aa",
|
||||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
# FAT mount end to end: the fat server mounts the boot usb-storage device (the
|
# FAT mount end to end: the fat server mounts the boot usb-storage device (the
|
||||||
# FAT32 image) into the VFS at /mnt/usb. A fat-test client then lists and reads
|
# FAT32 image) into the VFS at /volumes/usb. A fat-test client then lists and reads
|
||||||
# through the mount — proof of the whole stack: block device -> FAT parse ->
|
# through the mount — proof of the whole stack: block device -> FAT parse ->
|
||||||
# VFS routing -> file read.
|
# VFS routing -> file read.
|
||||||
{"name": "fat-mount",
|
{"name": "fat-mount",
|
||||||
"smp": 4,
|
"smp": 4,
|
||||||
"timeout": 150,
|
"timeout": 150,
|
||||||
"expect": r"fat: mounted /mnt/usb[\s\S]*fat-test: ok",
|
"expect": r"fat: mounted /volumes/usb[\s\S]*fat-test: ok",
|
||||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
# Phase 2b: mkdir/unlink through the mount. Reuses the fat-mount build — the
|
# Phase 2b: mkdir/unlink through the mount. Reuses the fat-mount build — the
|
||||||
# fat-test client, after listing, makes a directory, writes+reads a file inside
|
# fat-test client, after listing, makes a directory, writes+reads a file inside
|
||||||
@@ -714,7 +724,7 @@ CASES = [
|
|||||||
"smp": 4,
|
"smp": 4,
|
||||||
"timeout": 150,
|
"timeout": 150,
|
||||||
"qmp_after": {"delay": 8, "command": "system_powerdown"},
|
"qmp_after": {"delay": 8, "command": "system_powerdown"},
|
||||||
"expect": r"logger: logging to /var/log/\d{4}-\d{2}-\d{2}T\d{6}Z[\s\S]*"
|
"expect": r"logger: logging to /system/logs/\d{4}-\d{2}-\d{2}T\d{6}Z[\s\S]*"
|
||||||
r"init: shutting down[\s\S]*"
|
r"init: shutting down[\s\S]*"
|
||||||
r"logger: flushed through sequence \d+[\s\S]*"
|
r"logger: flushed through sequence \d+[\s\S]*"
|
||||||
r"power: entering S5",
|
r"power: entering S5",
|
||||||
@@ -836,6 +846,49 @@ CASES = [
|
|||||||
{"name": "input",
|
{"name": "input",
|
||||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
|
# The protocol registry (docs/os-development/protocol-namespace.md): init serves
|
||||||
|
# /protocol, and the fixture drives the registrar's whole contract — an ungranted
|
||||||
|
# bind refused (-EPERM), the kernel's reserved prefix holding against mount and
|
||||||
|
# unmount, a live owner's name refused (-EBUSY), and a killed provider's channel
|
||||||
|
# failing while a re-resolve reaches the restarted instance. Each step prints its
|
||||||
|
# own line, so a failure says which rule broke, not merely that one did.
|
||||||
|
# Three of the steps are the registrar's security contract, and each fails
|
||||||
|
# catastrophically rather than quietly if it regresses: a forged power event
|
||||||
|
# posted to PID 1's mailbox must not shut the machine down (a regression ends
|
||||||
|
# the boot), capability-carrying zero-length pings must not consume PID 1's
|
||||||
|
# handle table (a regression makes every later bind impossible), and a granted
|
||||||
|
# binary spawned by an unauthorized task must be refused while the same binary
|
||||||
|
# spawned by an authorized one is not (the laundering deputy).
|
||||||
|
{"name": "protocol-registry",
|
||||||
|
"expect": r"(?s)(?=.*protocol-registry: ungranted bind refused)"
|
||||||
|
r"(?=.*protocol-registry: /protocol reserved)"
|
||||||
|
r"(?=.*protocol-registry: forged power event ignored)"
|
||||||
|
r"(?=.*protocol-registry: capability-carrying pings did not exhaust the registrar)"
|
||||||
|
r"(?=.*protocol-registry: collision refused)"
|
||||||
|
r"(?=.*protocol-registry: dead channel refused)"
|
||||||
|
r"(?=.*protocol-registry: restarted provider reached)"
|
||||||
|
r"(?=.*protocol-registry: laundering deputy refused)"
|
||||||
|
r"(?=.*protocol-registry: foreign signal binding refused)"
|
||||||
|
r"(?=.*protocol-registry: foreign timer and exit binding refused)(?=.*protocol-registry: foreign receive refused)"
|
||||||
|
r"(?=.*protocol-registry: capability-carrying pings did not exhaust the harness)"
|
||||||
|
r"(?=.*DANOS-TEST-RESULT: PASS)",
|
||||||
|
"fail": r"DANOS-TEST-RESULT: FAIL|protocol-registry: FAIL"},
|
||||||
|
# Restriction stage one (docs/os-development/protocol-namespace.md): the
|
||||||
|
# registrar checks `open` against /system/configuration/protocol.csv, and a
|
||||||
|
# caller with no grant gets the same answer as a caller naming a contract
|
||||||
|
# nobody bound. The scenario boots /protocol plus the input service, so the
|
||||||
|
# forbidden name is genuinely BOUND — the fixture reads the namespace listing
|
||||||
|
# to prove it — and then compares the refusal with an unbound name field by
|
||||||
|
# field: status, node, payload length, the whole reply packet, and the
|
||||||
|
# presence of a capability. All three failure shapes (refused-and-bound,
|
||||||
|
# granted-and-unbound, neither) must collapse into one answer.
|
||||||
|
{"name": "protocol-denied",
|
||||||
|
"expect": r"(?s)(?=.*protocol-denied: granted open succeeded)"
|
||||||
|
r"(?=.*protocol-denied: ungranted open refused as absent)"
|
||||||
|
r"(?=.*protocol-denied: refusal is indistinguishable from absence)"
|
||||||
|
r"(?=.*protocol-denied: ok)"
|
||||||
|
r"(?=.*DANOS-TEST-RESULT: PASS)",
|
||||||
|
"fail": r"DANOS-TEST-RESULT: FAIL|protocol-denied: FAIL"},
|
||||||
# Device manager: a ring-3 service enumerates /system/devices, matches the PCI host
|
# Device manager: a ring-3 service enumerates /system/devices, matches the PCI host
|
||||||
# bridge to pci-bus, and spawns it — end-to-end proof of discover -> match -> spawn
|
# bridge to pci-bus, and spawns it — end-to-end proof of discover -> match -> spawn
|
||||||
# -> driver-up (the spawned pci-bus logs "<N> functions found").
|
# -> driver-up (the spawned pci-bus logs "<N> functions found").
|
||||||
@@ -935,7 +988,7 @@ def run_case(arch, case):
|
|||||||
# The bootable FAT32 USB image the build produced (tools/make-fat-image.py),
|
# The bootable FAT32 USB image the build produced (tools/make-fat-image.py),
|
||||||
# presented to the guest as a usb-storage device (see qemu_args).
|
# presented to the guest as a usb-storage device (see qemu_args).
|
||||||
# Boot a per-run COPY of the image: the guest MUTATES its boot volume (the
|
# Boot a per-run COPY of the image: the guest MUTATES its boot volume (the
|
||||||
# fat tests create/delete files; the logger writes /var/log), and QEMU is
|
# fat tests create/delete files; the logger writes /system/logs), and QEMU is
|
||||||
# hard-killed after a match — booting the build artifact in place let one
|
# hard-killed after a match — booting the build artifact in place let one
|
||||||
# run's leftovers fail the next (a stale TESTDIR trips the mkdir-duplicate
|
# run's leftovers fail the next (a stale TESTDIR trips the mkdir-duplicate
|
||||||
# refusal) and dirtied the build cache's own output.
|
# refusal) and dirtied the build cache's own output.
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
//! The args-echo test fixture as a binary package (docs/build-packages-plan.md):
|
//! The args-echo test fixture as a binary package (docs/build-packages-plan.md):
|
||||||
//! this file names the binary and EXACTLY the modules its source imports —
|
//! this file names the binary and EXACTLY the modules its source imports —
|
||||||
//! the shared recipe and the module-to-domain map live in build-support.
|
//! build-support resolves each name from the domains this zon declares.
|
||||||
|
|
||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
const build_support = @import("build-support");
|
const build_support = @import("build-support");
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
//! The crash-test test fixture as a binary package (docs/build-packages-plan.md):
|
//! The crash-test test fixture as a binary package (docs/build-packages-plan.md):
|
||||||
//! this file names the binary and EXACTLY the modules its source imports —
|
//! this file names the binary and EXACTLY the modules its source imports —
|
||||||
//! the shared recipe and the module-to-domain map live in build-support.
|
//! build-support resolves each name from the domains this zon declares.
|
||||||
|
|
||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
const build_support = @import("build-support");
|
const build_support = @import("build-support");
|
||||||
@@ -9,7 +9,7 @@ pub fn build(b: *std.Build) void {
|
|||||||
const exe = build_support.userBinary(b, .{
|
const exe = build_support.userBinary(b, .{
|
||||||
.name = "crash-test",
|
.name = "crash-test",
|
||||||
.root_source_file = b.path("crash-test.zig"),
|
.root_source_file = b.path("crash-test.zig"),
|
||||||
.imports = &.{ "device-manager-protocol", "driver", "ipc", "logging", "process", "time" },
|
.imports = &.{ "channel", "device-manager-protocol", "driver", "ipc", "logging", "process", "time" },
|
||||||
});
|
});
|
||||||
b.installArtifact(exe);
|
b.installArtifact(exe);
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -7,6 +7,7 @@
|
|||||||
//! binary), it exits silently so it cannot derange other tests.
|
//! binary), it exits silently so it cannot derange other tests.
|
||||||
|
|
||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
|
const channel = @import("channel");
|
||||||
const ipc = @import("ipc");
|
const ipc = @import("ipc");
|
||||||
const process = @import("process");
|
const process = @import("process");
|
||||||
const time = @import("time");
|
const time = @import("time");
|
||||||
@@ -29,7 +30,7 @@ pub fn main(init: process.Init) void {
|
|||||||
var manager: ?ipc.Handle = null;
|
var manager: ?ipc.Handle = null;
|
||||||
var tries: u32 = 0;
|
var tries: u32 = 0;
|
||||||
while (manager == null and tries < 100) : (tries += 1) {
|
while (manager == null and tries < 100) : (tries += 1) {
|
||||||
manager = ipc.lookup(.device_manager);
|
manager = channel.openEndpoint("device-manager");
|
||||||
if (manager == null) time.sleepMillis(20);
|
if (manager == null) time.sleepMillis(20);
|
||||||
}
|
}
|
||||||
const h = manager orelse return;
|
const h = manager orelse return;
|
||||||
|
|||||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user