Compare commits
33
Commits
3fc2d5b083
...
threading
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
def34e71fc | ||
|
|
1b33f48acd | ||
|
|
b3a8147bd7 | ||
|
|
0730e77530 | ||
|
|
73df864fd2 | ||
|
|
11e363896f | ||
|
|
6e8b02d771 | ||
|
|
d26515706e | ||
|
|
acf8ff2c33 | ||
|
|
e5dcc9790b | ||
|
|
f8ad4ac971 | ||
|
|
914af52b94 | ||
|
|
bdd8a48476 | ||
|
|
f309ce04f4 | ||
|
|
9989ebbec7 | ||
|
|
626e3c5e9b | ||
|
|
2c63c76288 | ||
|
|
05fc1764de | ||
|
|
65bb04d890 | ||
|
|
24c49f56e1 | ||
|
|
0217662808 | ||
|
|
4231301896 | ||
|
|
58927ed7e5 | ||
|
|
6e0e0a62c6 | ||
|
|
88ad432758 | ||
|
|
9333d0572f | ||
|
|
f3342118f5 | ||
|
|
2723b6f778 | ||
|
|
a01a4f3b3d | ||
|
|
e1605e3235 | ||
|
|
1e80c57484 | ||
|
|
c3e9c59086 | ||
|
|
88ed3c5417 |
@@ -1 +0,0 @@
|
||||
0.16.0
|
||||
@@ -1,8 +1,8 @@
|
||||
# DanOS
|
||||
Codename: Shodan
|
||||
Version: 1
|
||||
|
||||
A small resilient operating system, written from scratch in Zig.
|
||||
**Codename: Shodan**
|
||||
|
||||
A very small resilient operating system.
|
||||
|
||||
## Zen of DanOS:
|
||||
|
||||
@@ -18,12 +18,12 @@ A small resilient operating system, written from scratch in Zig.
|
||||
- Useful during driver development.
|
||||
- Drivers can claim MMIO / ports
|
||||
- Driver resources (e.g. IRQ/Port/MMIO) claims are automatically cleaned up if the driver dies or is killed
|
||||
- Drivers can also hook into the process lifecyle to clean up or reset hardware
|
||||
- Drivers can also hook into the process lifecycle to clean up or reset hardware
|
||||
- No legacy to deal with
|
||||
- Zig code uses a clean coding style (Zen of Zig)
|
||||
- Favor reading code over writing code.
|
||||
- No magic numbers.
|
||||
- No shortend names unless its for ABI compatibility or acronyms
|
||||
- No shortened names unless its for ABI compatibility or acronyms
|
||||
- Inter-Process Communication (IPC)
|
||||
- Publish and subscribe to Asynchronous Messages
|
||||
- Talk to services and processes synchronously
|
||||
@@ -54,6 +54,17 @@ the UEFI bootloader at `zig-out/EFI/BOOT/BOOTX64.efi`, the kernel at
|
||||
`zig-out/system/kernel`, init at `zig-out/system/services/init`, drivers under
|
||||
`zig-out/system/drivers/`, and the initial-ramdisk at `zig-out/boot/`.
|
||||
|
||||
## Release media
|
||||
|
||||
```sh
|
||||
zig build release-x86-64
|
||||
```
|
||||
|
||||
Produces `zig-out/danos-x86-64.iso`, a hybrid ISO that boots flashed raw to a
|
||||
USB stick (balenaEtcher, dd) or burned to optical media — see
|
||||
[docs/release-iso.md](docs/release-iso.md). `zig build check-iso-image`
|
||||
validates it without booting.
|
||||
|
||||
## Run
|
||||
|
||||
Boot it in QEMU with OVMF (opens a display window):
|
||||
|
||||
@@ -54,6 +54,12 @@ fn timestamp(b: *std.Build) []const u8 {
|
||||
/// can't reach), linked against the `runtime` runtime library with the shared user
|
||||
/// link script. Pinned to LLVM + LLD so the script's PHDRS (segment permissions)
|
||||
/// are authoritative — the kernel's W^X user-ELF loader requires exact perms.
|
||||
///
|
||||
/// The compilation root is not the program's own file but the shared shim
|
||||
/// library/runtime/root.zig, which supplies the root declarations (`main`
|
||||
/// re-export, panic handler, `_start` pull) so a program only defines
|
||||
/// `pub fn main`. The program's file becomes the `program` module the shim
|
||||
/// imports; reach it through `programModule` to add per-binary imports.
|
||||
fn addUserBinary(
|
||||
b: *std.Build,
|
||||
target: std.Build.ResolvedTarget,
|
||||
@@ -64,27 +70,66 @@ fn addUserBinary(
|
||||
name: []const u8,
|
||||
root: []const u8,
|
||||
) *std.Build.Step.Compile {
|
||||
return addUserBinaryImpl(b, target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, name, root, false);
|
||||
}
|
||||
|
||||
/// As `addUserBinary`, but built multi-threaded (`single_threaded = false`) so real
|
||||
/// atomics/TLS work — required before a binary may call `runtime.Thread.spawn`
|
||||
/// (docs/threading.md). Threads are a deliberate per-binary opt-in.
|
||||
fn addThreadedUserBinary(
|
||||
b: *std.Build,
|
||||
target: std.Build.ResolvedTarget,
|
||||
runtime_module: *std.Build.Module,
|
||||
mmio_module: *std.Build.Module,
|
||||
xkeyboard_config_module: *std.Build.Module,
|
||||
acpi_ids_module: *std.Build.Module,
|
||||
name: []const u8,
|
||||
root: []const u8,
|
||||
) *std.Build.Step.Compile {
|
||||
return addUserBinaryImpl(b, target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, name, root, true);
|
||||
}
|
||||
|
||||
fn addUserBinaryImpl(
|
||||
b: *std.Build,
|
||||
target: std.Build.ResolvedTarget,
|
||||
runtime_module: *std.Build.Module,
|
||||
mmio_module: *std.Build.Module,
|
||||
xkeyboard_config_module: *std.Build.Module,
|
||||
acpi_ids_module: *std.Build.Module,
|
||||
name: []const u8,
|
||||
root: []const u8,
|
||||
threaded: bool,
|
||||
) *std.Build.Step.Compile {
|
||||
// Settings (target, optimize, code model, ...) live on the root module only;
|
||||
// the program and runtime modules leave theirs null and inherit them.
|
||||
const program_module = b.createModule(.{
|
||||
.root_source_file = b.path(root),
|
||||
.imports = &.{
|
||||
.{ .name = "runtime", .module = runtime_module },
|
||||
// Typed volatile MMIO + memory barriers, for drivers. See library/mmio/.
|
||||
.{ .name = "mmio", .module = mmio_module },
|
||||
// Keyboard layouts (keycode + modifiers -> keysym/character), available
|
||||
// to any program that wants it. See library/xkeyboard-config/.
|
||||
.{ .name = "xkeyboard-config", .module = xkeyboard_config_module },
|
||||
// ACPI/PnP hardware-ID registry, so drivers name devices
|
||||
// (HardwareId.ps2_keyboard) instead of magic "_HID" strings.
|
||||
.{ .name = "acpi-ids", .module = acpi_ids_module },
|
||||
},
|
||||
});
|
||||
const exe = b.addExecutable(.{
|
||||
.name = name,
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path(root),
|
||||
.root_source_file = b.path("library/runtime/root.zig"),
|
||||
.target = target,
|
||||
.optimize = .ReleaseSmall,
|
||||
.code_model = .large,
|
||||
.single_threaded = true,
|
||||
.single_threaded = !threaded, // a threaded binary needs real atomics/TLS
|
||||
.sanitize_c = .off,
|
||||
.stack_check = false,
|
||||
.stack_protector = false,
|
||||
.imports = &.{
|
||||
.{ .name = "runtime", .module = runtime_module },
|
||||
// Typed volatile MMIO + memory barriers, for drivers. See library/mmio/.
|
||||
.{ .name = "mmio", .module = mmio_module },
|
||||
// Keyboard layouts (keycode + modifiers -> keysym/character), available
|
||||
// to any program that wants it. See library/xkeyboard-config/.
|
||||
.{ .name = "xkeyboard-config", .module = xkeyboard_config_module },
|
||||
// ACPI/PnP hardware-ID registry, so drivers name devices
|
||||
// (HardwareId.ps2_keyboard) instead of magic "_HID" strings.
|
||||
.{ .name = "acpi-ids", .module = acpi_ids_module },
|
||||
.{ .name = "program", .module = program_module },
|
||||
},
|
||||
}),
|
||||
});
|
||||
@@ -96,6 +141,14 @@ fn addUserBinary(
|
||||
return exe;
|
||||
}
|
||||
|
||||
/// The `program` module of a binary built by `addUserBinary` — the module rooted
|
||||
/// at the program's own source file. Per-binary imports (protocol modules, bus
|
||||
/// ABIs) go here, not on the root shim: module imports are not transitive, so an
|
||||
/// import added to the root would be invisible to the program's code.
|
||||
fn programModule(exe: *std.Build.Step.Compile) *std.Build.Module {
|
||||
return exe.root_module.import_table.get("program").?;
|
||||
}
|
||||
|
||||
/// The modules the kernel imports, gathered once so both kernel variants (the
|
||||
/// installed one and the serial-enabled one `run-x86-64` boots) are built from
|
||||
/// the same set. `build_options` is *not* here — it carries `serial`/`test_case`,
|
||||
@@ -171,8 +224,9 @@ fn addKernel(
|
||||
/// Assemble the bootable FAT32 image (the in-repo Python builder) holding what
|
||||
/// the firmware and loader need off the ESP: the EFI stub, `kernel`, `init`, and
|
||||
/// the initial-ramdisk. Factored so the serial-enabled `run-x86-64` variant can
|
||||
/// bundle its own kernel while sharing the (serial-independent) loader, init, and
|
||||
/// ramdisk. Returns the image's LazyPath.
|
||||
/// bundle its own serial kernel while sharing the loader, init, and ramdisk — all
|
||||
/// built once per invocation (the loader's boot breadcrumbs and init's heartbeat
|
||||
/// both follow the top-level -Dserial). Returns the image's LazyPath.
|
||||
fn addBootImage(
|
||||
b: *std.Build,
|
||||
kernel_bin: std.Build.LazyPath,
|
||||
@@ -355,6 +409,13 @@ pub fn build(b: *std.Build) void {
|
||||
});
|
||||
runtime_module.addImport("display-protocol", display_protocol_module);
|
||||
|
||||
// The scanout protocol: the compositor's outbound present channel to a native scanout
|
||||
// driver (virtio-gpu), separate from the client-facing display protocol (docs/display-v2.md).
|
||||
const scanout_protocol_module = b.addModule("scanout-protocol", .{
|
||||
.root_source_file = b.path("system/services/display/scanout-protocol.zig"),
|
||||
});
|
||||
runtime_module.addImport("scanout-protocol", scanout_protocol_module);
|
||||
|
||||
// The power protocol: system power's domain-named surface (docs/power.md).
|
||||
const power_protocol_module = b.addModule("power-protocol", .{
|
||||
.root_source_file = b.path("system/services/power/protocol.zig"),
|
||||
@@ -433,6 +494,14 @@ pub fn build(b: *std.Build) void {
|
||||
// linked into the kernel's user region against the `runtime` runtime library, and
|
||||
// started in ring 3 by the kernel's user-ELF loader.
|
||||
const init_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "init", "system/services/init/init.zig");
|
||||
// init reads the same `serial` flag the kernel does: its liveness heartbeat is a
|
||||
// serial/test-build diagnostic (the QEMU harness's init tests assert on it, and
|
||||
// -Dserial images emit it), so a flashable image runs a purely event-driven PID 1
|
||||
// that wakes only for real work. The test harness builds with -Dserial=true, so
|
||||
// the heartbeat stays present under test.
|
||||
const init_options = b.addOptions();
|
||||
init_options.addOption(bool, "serial", serial);
|
||||
programModule(init_exe).addImport("build_options", init_options.createModule());
|
||||
const init_install = b.addInstallArtifact(init_exe, .{ .dest_dir = .{ .override = .{ .custom = "system/services" } } });
|
||||
b.getInstallStep().dependOn(&init_install.step);
|
||||
|
||||
@@ -449,30 +518,33 @@ pub fn build(b: *std.Build) void {
|
||||
// The xHCI bus driver builds chapter-9 requests and decodes descriptors from
|
||||
// usb-abi, and reports each interface's (class,subclass,protocol) identity via
|
||||
// usb-ids.packTriple.
|
||||
usb_xhci_bus_exe.root_module.addImport("usb-abi", usb_abi_module);
|
||||
usb_xhci_bus_exe.root_module.addImport("usb-ids", usb_ids_module);
|
||||
usb_xhci_bus_exe.root_module.addImport("usb-transfer-protocol", usb_transfer_protocol_module);
|
||||
programModule(usb_xhci_bus_exe).addImport("usb-abi", usb_abi_module);
|
||||
programModule(usb_xhci_bus_exe).addImport("usb-ids", usb_ids_module);
|
||||
programModule(usb_xhci_bus_exe).addImport("usb-transfer-protocol", usb_transfer_protocol_module);
|
||||
// The USB HID class drivers: keyboard and mouse. They own no hardware — each
|
||||
// opens its device through runtime.usb (the transfer protocol) and publishes to
|
||||
// the input service. They build chapter-9 class requests from usb-abi.
|
||||
const usb_hid_keyboard_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "usb-hid-keyboard", "system/drivers/usb-hid/keyboard.zig");
|
||||
usb_hid_keyboard_exe.root_module.addImport("usb-abi", usb_abi_module);
|
||||
programModule(usb_hid_keyboard_exe).addImport("usb-abi", usb_abi_module);
|
||||
const usb_hid_mouse_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "usb-hid-mouse", "system/drivers/usb-hid/mouse.zig");
|
||||
usb_hid_mouse_exe.root_module.addImport("usb-abi", usb_abi_module);
|
||||
programModule(usb_hid_mouse_exe).addImport("usb-abi", usb_abi_module);
|
||||
// The USB mass-storage class driver: opens its device via runtime.usb, drives it
|
||||
// with Bulk-Only Transport + SCSI, and serves the block protocol under `.block`.
|
||||
const usb_storage_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "usb-storage", "system/drivers/usb-storage/usb-storage.zig");
|
||||
usb_storage_exe.root_module.addImport("block-protocol", block_protocol_module);
|
||||
programModule(usb_storage_exe).addImport("block-protocol", block_protocol_module);
|
||||
// The FAT filesystem server: mounts the block device and serves it into the VFS
|
||||
// at /mnt/usb. Its engine (engine.zig / on-disk.zig) is imported relatively.
|
||||
const fat_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "fat", "system/services/fat/fat.zig");
|
||||
const display_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "display", "system/services/display/display.zig");
|
||||
const display_demo_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "display-demo", "system/services/display-demo/display-demo.zig");
|
||||
const virtio_gpu_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "virtio-gpu", "system/drivers/virtio-gpu/virtio-gpu.zig");
|
||||
const shm_server_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "shm-server", "system/services/shm-server/shm-server.zig");
|
||||
const shm_client_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "shm-client", "system/services/shm-client/shm-client.zig");
|
||||
const fat_test_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "fat-test", "system/services/fat/fat-test.zig");
|
||||
const pci_bus_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "pci-bus", "system/drivers/pci-bus/pci-bus.zig");
|
||||
// The PCI bus driver decodes each function's class triple to human names in its
|
||||
// boot log (class/subclass/prog-IF), so pull in the shared pci-class reference.
|
||||
pci_bus_exe.root_module.addImport("pci-class", pci_class_module);
|
||||
programModule(pci_bus_exe).addImport("pci-class", pci_class_module);
|
||||
// A test fixture, not a real driver: hellos to the device manager, then faults —
|
||||
// what the driver-restart scenario drives the crash-loop cap with.
|
||||
const crash_test_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "crash-test", "system/services/crash-test/crash-test.zig");
|
||||
@@ -491,13 +563,13 @@ pub fn build(b: *std.Build) void {
|
||||
.fdt => "system/services/fdt/fdt.zig",
|
||||
};
|
||||
const discovery_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "discovery", discovery_source);
|
||||
if (discovery == .acpi) discovery_exe.root_module.addImport("aml", aml_module);
|
||||
if (discovery == .acpi) programModule(discovery_exe).addImport("aml", aml_module);
|
||||
const device_manager_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "device-manager", "system/services/device-manager/device-manager.zig");
|
||||
// Names the xHCI PCI class triple from the shared taxonomy instead of a bare 0x0C0330.
|
||||
device_manager_exe.root_module.addImport("pci-class", pci_class_module);
|
||||
programModule(device_manager_exe).addImport("pci-class", pci_class_module);
|
||||
// The manager matches reported USB interfaces by their (class,subclass,protocol)
|
||||
// triple (usbDriverForIdentity), built from the named usb-ids codes.
|
||||
device_manager_exe.root_module.addImport("usb-ids", usb_ids_module);
|
||||
programModule(device_manager_exe).addImport("usb-ids", usb_ids_module);
|
||||
// The input service and its exercisers: the fan-out server, a hardware-free synthetic
|
||||
// source, and a subscriber that doubles as the `input` test's oracle. See docs/input.md.
|
||||
const input_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "input", "system/services/input/input.zig");
|
||||
@@ -506,6 +578,9 @@ pub fn build(b: *std.Build) void {
|
||||
const args_echo_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "args-echo", "system/services/args-echo/args-echo.zig");
|
||||
const process_test_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "process-test", "system/services/process-test/process-test.zig");
|
||||
const log_flush_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "log-flush", "system/services/log-flush/log-flush.zig");
|
||||
// The first multi-threaded binary: exercises runtime.Thread over the thread ABI
|
||||
// (docs/threading.md). Built threaded so its shared-memory poll is real.
|
||||
const thread_test_exe = addThreadedUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "thread-test", "system/services/thread-test/thread-test.zig");
|
||||
|
||||
// Pack the user binaries into the initial_ramdisk image with the host-side Python tool
|
||||
// (the container format is trivial, and Python sidesteps std API churn). Args:
|
||||
@@ -539,10 +614,18 @@ pub fn build(b: *std.Build) void {
|
||||
mk_run.addFileArg(display_exe.getEmittedBin());
|
||||
mk_run.addArg("display-demo");
|
||||
mk_run.addFileArg(display_demo_exe.getEmittedBin());
|
||||
mk_run.addArg("virtio-gpu");
|
||||
mk_run.addFileArg(virtio_gpu_exe.getEmittedBin());
|
||||
mk_run.addArg("shm-server");
|
||||
mk_run.addFileArg(shm_server_exe.getEmittedBin());
|
||||
mk_run.addArg("shm-client");
|
||||
mk_run.addFileArg(shm_client_exe.getEmittedBin());
|
||||
mk_run.addArg("pci-bus");
|
||||
mk_run.addFileArg(pci_bus_exe.getEmittedBin());
|
||||
mk_run.addArg("crash-test");
|
||||
mk_run.addFileArg(crash_test_exe.getEmittedBin());
|
||||
mk_run.addArg("thread-test");
|
||||
mk_run.addFileArg(thread_test_exe.getEmittedBin());
|
||||
mk_run.addArg("device-list");
|
||||
mk_run.addFileArg(device_list_exe.getEmittedBin());
|
||||
mk_run.addArg("discovery");
|
||||
@@ -645,6 +728,31 @@ pub fn build(b: *std.Build) void {
|
||||
const check_fat_step = b.step("check-fat-image", "Verify the FAT32 USB image is valid and bootable");
|
||||
check_fat_step.dependOn(&check_fat.step);
|
||||
|
||||
// --- release-x86-64: danos-x86-64.iso, the flashable release image ---
|
||||
// Wrap the FAT32 boot volume in a hybrid ISO (the in-repo Python builder
|
||||
// again, no xorriso/isohybrid): an ISO9660 whose El Torito EFI boot entry
|
||||
// and MBR ESP partition entry both point at the embedded FAT image. One
|
||||
// file then boots every way release media is consumed — flashed raw to a
|
||||
// USB stick with Etcher or dd, or burned to optical media — while
|
||||
// danos-usb.img stays the raw superfloppy QEMU and the test harness boot.
|
||||
const mk_iso = b.addSystemCommand(&.{"python3"});
|
||||
mk_iso.addFileArg(b.path("tools/make-iso-image.py"));
|
||||
const iso_image = mk_iso.addOutputFileArg("danos-x86-64.iso");
|
||||
mk_iso.addFileArg(fat_image);
|
||||
const iso_install = b.addInstallFile(iso_image, "danos-x86-64.iso");
|
||||
const release_step = b.step("release-x86-64", "Build the flashable x86-64 release ISO (zig-out/danos-x86-64.iso; flash with Etcher or dd)");
|
||||
release_step.dependOn(&iso_install.step);
|
||||
|
||||
// `zig build check-iso-image` — the ISO builder's own --verify (mirroring
|
||||
// check-fat-image): the MBR partition, the El Torito catalog, and the
|
||||
// embedded FAT32 image must all agree.
|
||||
const check_iso = b.addSystemCommand(&.{"python3"});
|
||||
check_iso.addFileArg(b.path("tools/make-iso-image.py"));
|
||||
check_iso.addArg("--verify");
|
||||
check_iso.addFileArg(iso_image);
|
||||
const check_iso_step = b.step("check-iso-image", "Verify the release ISO is a valid hybrid (MBR ESP partition + El Torito EFI entry)");
|
||||
check_iso_step.dependOn(&check_iso.step);
|
||||
|
||||
// --- run-x86-64: boot the x86-64 kernel in QEMU via UEFI/OVMF ---
|
||||
// Firmware lives in different places per OS/distro, so probe the known
|
||||
// layouts (Architecture, Debian/Ubuntu, Fedora, macOS Homebrew) and use the first
|
||||
@@ -775,6 +883,8 @@ pub fn build(b: *std.Build) void {
|
||||
"system/services/fat/engine.zig", // FAT read/write over a RAM-backed image
|
||||
"system/services/display/compositor.zig", // Rect math + fill/composite/blit-tile
|
||||
"system/services/display/protocol.zig", // pack(): native pixel encoding per format
|
||||
"system/drivers/virtio-gpu/virtio-gpu-protocol.zig", // virtio-gpu command struct sizes
|
||||
"system/drivers/virtio-gpu/virtio-pci.zig", // virtio 1.0 PCI transport struct sizes
|
||||
}) |root| {
|
||||
const mod_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
|
||||
+40
-11
@@ -39,42 +39,53 @@ rather than restate it. Roughly in the order things happen at runtime:
|
||||
endpoints — the backbone the microkernel's isolated servers talk over.
|
||||
12. **[syscall.md](syscall.md) — system calls.** How ring 3 asks the kernel for
|
||||
something: the `syscall`/`sysret` fast path, the trap frame, and why the table is
|
||||
deliberately tiny.
|
||||
13. **[drivers.md](drivers.md) — writing a driver.** The payoff: a driver is an
|
||||
deliberately tiny. The numbers are a **private** ABI — [vdso.md](vdso.md) designs
|
||||
the public boundary that will hide them.
|
||||
13. **[vfs-protocol.md](vfs-protocol.md) — the VFS wire protocol.** The language-neutral
|
||||
byte-level spec of the file protocol spoken over IPC: request/reply headers,
|
||||
the operation table, mount routing, and the append-only evolution rules — the
|
||||
first IPC protocol documented as public ABI.
|
||||
14. **[drivers.md](drivers.md) — writing a driver.** The payoff: a driver is an
|
||||
ordinary ring-3 process that claims a device, maps its registers, and **sleeps
|
||||
until its hardware interrupts it**. The claim is the capability; `irq_ack` is the
|
||||
unmask.
|
||||
14. **[driver-model.md](driver-model.md) — buses, classes and host controllers.** How
|
||||
15. **[driver-model.md](driver-model.md) — buses, classes and host controllers.** How
|
||||
real driver stacks factor into three shapes and how families share code. The
|
||||
three primitives it proposed are long since built (M13 capability passing,
|
||||
M14 DMA + barriers, M15 MSI), and the driver *contract* on top of them —
|
||||
hello, supervision, restart — is built too (device-manager.md, M18).
|
||||
15. **[process-management.md](process-management.md) — process management.** The
|
||||
16. **[process-management.md](process-management.md) — process management.** The
|
||||
microkernel's `ps`/`kill`/SIGCHLD: enumerate as a table snapshot, the
|
||||
supervision link as the kill authority, and child-exit notifications over the
|
||||
same endpoints IRQs arrive on.
|
||||
16. **[process-lifecycle.md](process-lifecycle.md) — the process lifecycle.** Built
|
||||
17. **[process-lifecycle.md](process-lifecycle.md) — the process lifecycle.** Built
|
||||
(M17): signals over IPC as the one lifecycle vocabulary every process speaks — the
|
||||
POSIX.1-1990 words with message delivery instead of stack hijack, the stable
|
||||
`runtime.process` interface, exit reasons, published exit events any stateful
|
||||
service can subscribe to (the VFS releasing dead clients' handles), and the two
|
||||
iron rules (cleanup is the kernel's job; kill is not a signal).
|
||||
17. **[device-manager.md](device-manager.md) — the device manager.** Built (M18,
|
||||
18. **[device-manager.md](device-manager.md) — the device manager.** Built (M18,
|
||||
through the app surface): the
|
||||
tree, the matcher, and the supervisor. Tree structure lives in the manager,
|
||||
authority stays in the kernel; bus drivers report what they see; drivers are
|
||||
restarted through the lifecycle vocabulary — the plan that turns
|
||||
[resilience.md](resilience.md)'s restart goal into increments.
|
||||
18. **[input.md](input.md) — the input module.** Broadcasting input events (keyboard,
|
||||
19. **[input.md](input.md) — the input module.** Broadcasting input events (keyboard,
|
||||
mouse, joystick): why a synchronous rendezvous can't fan out to many listeners, the
|
||||
asynchronous `ipc_send` primitive built to fix it, and the per-device subscribe/publish
|
||||
service layered on top.
|
||||
19. **[display.md](display.md) — the display service.** The display half of the GUI
|
||||
20. **[display.md](display.md) — the display service.** The display half of the GUI
|
||||
track: a user-space compositor that owns the framebuffer, composes a layer stack into
|
||||
a double buffer, and presents it. Why GOP and the PCI display device are two views of
|
||||
one controller, the device-node + write-combining handoff, and what flicker-free buys
|
||||
that tear-free doesn't. Plan: [display-plan.md](display-plan.md).
|
||||
20. **[halting.md](halting.md) — halting.** Why a kernel can't just "exit", and
|
||||
that tear-free doesn't. Plan: [display-plan.md](display-plan.md). **v2** (complete) makes
|
||||
scanout a pluggable backend — GOP floor + a native virtio-gpu driver, hot-attached, with
|
||||
runtime mode-set, EDID, fenced vsync presents, and restart re-attach:
|
||||
[display-v2.md](display-v2.md), plan [display-v2-plan.md](display-v2-plan.md). Looking
|
||||
further out, two research snapshots survey what a *native* driver for real GPU silicon
|
||||
would take as another `.scanout` backend: [nvidia-gpus.md](nvidia-gpus.md) (RTX 3060 /
|
||||
Ampere) and [intel-igpu.md](intel-igpu.md) (Intel iGPU).
|
||||
21. **[halting.md](halting.md) — halting.** Why a kernel can't just "exit", and
|
||||
how `while (true) hlt` parks the CPU safely once there's nothing left to do.
|
||||
|
||||
Start with the north star:
|
||||
@@ -93,6 +104,19 @@ Start with the north star:
|
||||
port to **one seam** (`std.os.danos`), so we build `runtime.os` (→ that seam) plus a
|
||||
thin `runtime.fs`, retire the `posix` shim, and follow a phased path to
|
||||
`zig build-exe hello.zig` running on danos — **not** Linux-ABI emulation.
|
||||
- **[threading.md](threading.md) — threads, the std-shaped way.** **Built** (M1–M6):
|
||||
`runtime.Thread` mirrors `std.Thread`'s API (spawn/join/detach, Mutex/Condition/
|
||||
Semaphore) over a **private** thread ABI — several tasks sharing one address space via
|
||||
a `thread_spawn` syscall, futex-backed blocking, aspace refcounting. Why it's the
|
||||
native type and not literal `std.Thread` (the [private ABI](syscall.md)), and why
|
||||
threads stay a narrow opt-in against the [resilience](resilience.md) default. Build
|
||||
plan + gates: [threading-plan.md](threading-plan.md).
|
||||
- **[vdso.md](vdso.md) — the vDSO, the public system-call boundary.** A design note
|
||||
(not built yet) on keeping `abi.zig` genuinely private: a kernel-supplied, C-ABI
|
||||
entry blob mapped into every process as the *only* way into the kernel — so the
|
||||
syscall numbers can be renumbered or randomised at will, and Rust/C binaries get a
|
||||
stable boundary without danos growing a dynamic linker. danos's public ABI = the
|
||||
vDSO + the documented IPC wire protocols ([vfs-protocol.md](vfs-protocol.md) first).
|
||||
|
||||
Cutting across all of these:
|
||||
|
||||
@@ -100,6 +124,11 @@ Cutting across all of these:
|
||||
hardware needed to run danos: minimum specs (UEFI x86-64, ACPI, PCIe ECAM,
|
||||
xHCI, ~128 MiB RAM) grounded in what the boot path actually assumes, plus a
|
||||
plain-language guide matching Intel/AMD CPU generations by name.
|
||||
- **[release-iso.md](release-iso.md) — the release ISO.** The flashable boot
|
||||
media: `zig build release-x86-64` wraps the FAT32 boot volume in a hybrid ISO
|
||||
(MBR ESP partition + El Torito EFI entry, one embedded image) that Etcher/dd
|
||||
flash to USB or a burner writes to disc — built by an in-repo pure-Python
|
||||
tool, like the FAT image itself.
|
||||
- **[arch.md](arch.md) — the architecture split.** How CPU-specific code is kept
|
||||
behind a build-time `arch` module so the generic kernel never names x86_64,
|
||||
leaving room for other systems (e.g. an AArch64 Raspberry Pi) later.
|
||||
@@ -244,5 +273,5 @@ exception in [coding-standards.md](coding-standards.md) applies to that seam.
|
||||
| danos-native runtime (`runtime`): syscall wrappers, heap, IPC, device access, the file API (`fs`) — the stable application ABI | `library/runtime/` |
|
||||
| System services (init, the VFS server + `protocol`, the device-manager) | `system/services/` |
|
||||
| Device drivers, one sub-project each (`pci-bus`, `ps2-bus`, `usb-xhci-bus` bus drivers) | `system/drivers/` |
|
||||
| Build + `run-x86-64` (QEMU/OVMF) | `build.zig` |
|
||||
| Build + `run-x86-64` (QEMU/OVMF) + `release-x86-64` (the flashable ISO) | `build.zig` |
|
||||
| QEMU integration test harness | `test/qemu_test.py` |
|
||||
|
||||
@@ -0,0 +1,173 @@
|
||||
# Display v2 — build plan (pluggable scanout: GOP floor + virtio-gpu native)
|
||||
|
||||
The ordered, checkpointable build-out for [display-v2.md](display-v2.md). Each milestone
|
||||
lands on its own and ends in a **verifiable gate** — shaped for a `/loop` run, like
|
||||
[display-plan.md](display-plan.md). Read display-v2.md first for the *why*.
|
||||
|
||||
## Locked decisions (do not relitigate)
|
||||
|
||||
- **First native backend = virtio-gpu** (VM standard: mode-set + present/flush + vsync).
|
||||
- **Dynamic hot-attach**: boot on GOP, upgrade to native when the driver **announces**
|
||||
(push, not polling); re-attach across driver restarts; GOP is the floor for "no driver
|
||||
ever," not a live fall-back after a reprogram.
|
||||
- **v2 builds the `shm` capability** (endpoints → memory objects), shared with the future
|
||||
client-surface path.
|
||||
- The compositor's layers/back-buffer/damage are **unchanged**; only scanout is pluggable.
|
||||
|
||||
## Conventions
|
||||
|
||||
Follow [coding-standards.md](coding-standards.md): spell out non-acronym abbreviations,
|
||||
kebab-case file names, no `Co-Authored-By` trailers. New user binaries go through
|
||||
`addUserBinary` and get packed into the initial-ramdisk; protocols are
|
||||
`b.addModule("…-protocol", …)` imported into `runtime`; new syscalls extend
|
||||
[abi.zig](../system/abi.zig) `SystemCall` + a `library/runtime` wrapper.
|
||||
|
||||
## How to verify along the way
|
||||
|
||||
**Every gate is serial-checkable — no screenshots** (this plan is built to run unattended).
|
||||
Where "does it actually display" would otherwise need a human eyeball, the code **reads its
|
||||
own pixels back**: the scanout resource is CPU-visible RAM (shm-backed) and the back buffer
|
||||
is cacheable, so a driver/compositor can write a known value, read it back, and log a
|
||||
pass/fail — and a virtio `resource_flush` is confirmed by the device **acking it on the
|
||||
used ring**. Those two together (pixel-readback + flush-ack) are the automated stand-in for
|
||||
"it's on screen."
|
||||
|
||||
- `zig build test` — host unit tests (backend selection, virtio struct sizes/encodings,
|
||||
pixel-check helpers).
|
||||
- `python3 test/qemu_test.py <case>` — boots the kernel in QEMU; asserts on serial markers.
|
||||
The virtio cases boot with `-device virtio-gpu` (a per-case `qemu_extra`).
|
||||
- `run-x86-64` renders to a window — for the human's own satisfaction, **not** a gate.
|
||||
|
||||
---
|
||||
|
||||
## V1 — The scanout backend seam (refactor, no behaviour change) ✅
|
||||
|
||||
Extract scanout from the compositor so today's path becomes one backend among future ones.
|
||||
|
||||
- [x] `system/services/display/backend.zig`: a `Backend` tagged union with `info()`,
|
||||
`surface()` (the cacheable compose target), `present(damage)`, and capability flags
|
||||
(`canModeSet`/`hasVsync`, both false for GOP).
|
||||
- [x] The v1 GOP path is now `backend.Gop` (claims the `display` node, WC-maps the LFB,
|
||||
keeps the cacheable back buffer, `present` = the damage-rect WC copy). display.zig
|
||||
composes into `backend.surface()` and calls `backend.present(damage)` — no LFB or
|
||||
framebuffer geometry left in the compositor core.
|
||||
- [x] The selection decision is the pure `chooseKind(native_available)` (gop unless a
|
||||
native driver announced), split from the syscall-bound `select()`/`Gop.init()`.
|
||||
|
||||
**Gate (met):** `display-service` + `display-demo` pass **unchanged** (pure refactor; GOP
|
||||
is the only backend), and `zig build test` stays green.
|
||||
|
||||
## V2 — The `shm` cross-process memory capability (kernel) ✅
|
||||
|
||||
- [x] [abi.zig](../system/abi.zig): `shm_create` (34) / `shm_map` (35) syscalls + a
|
||||
`shm_test` service id. Handlers in process.zig: `shm_create(len)` allocates contiguous,
|
||||
zeroed, **cacheable** frames, wraps them in a refcounted object, installs a capability
|
||||
handle, maps them into the caller's shm arena → returns vaddr + handle; `shm_map(cap)`
|
||||
maps the same physical pages into the receiver. Reclaimed on death (see below).
|
||||
- [x] The capability core (ipc-synchronous.zig) is now **kind-tagged**: `scheduler.Task`'s
|
||||
handle table holds `HandleObject{kind, ptr}`; `closeHandles` and `shareCapability`
|
||||
dispatch by kind, so an `ShmObject` rides an `ipc_call` `send_cap` exactly like an
|
||||
endpoint and frees only when its last capability drops. `mapUserSharedInto` (paging)
|
||||
maps WB-cacheable + `device_grant`, so a sharer's teardown never frees the shared
|
||||
frames — the object owns them.
|
||||
- [x] `library/runtime/shm.zig` (+ barrel export): `create(len) -> Region{ptr, handle, len}`,
|
||||
`map(handle) -> ptr`.
|
||||
|
||||
**Gate (met):** `python3 test/qemu_test.py shm` — `shm-client` creates a region, writes a
|
||||
pattern, and passes its capability to `shm-server` as an `ipc_call` send_cap; the server
|
||||
`shm_map`s it and reads the **same bytes** back → `shm: shared 4096 bytes ok`. Guardrail:
|
||||
`ipc`/`ipc-call`/`ipc-cap`, `supervision`, `dma`, `usermem`, `display-service`, and host
|
||||
tests all still pass — the handle-table change broke no existing IPC.
|
||||
|
||||
## V3 — The virtio-gpu driver: bring-up + a frame on screen ✅
|
||||
|
||||
- [x] `system/drivers/virtio-gpu/`: claim the virtio-gpu PCI function (device-manager
|
||||
match on the display/other class triple, driver self-confirms vendor 0x1AF4/device
|
||||
0x1050 from config space), enable memory-space + bus-master, walk the vendor
|
||||
capabilities in config space to find common-config + notify, map the BAR, negotiate
|
||||
VERSION_1, and stand up the control virtqueue in coherent DMA. `virtio-gpu-protocol.zig`
|
||||
+ `virtio-pci.zig` for the control/transport structs (host-tested sizes).
|
||||
- [x] Create a 2D scanout resource backed by a coherent DMA region (V4 swaps this for the
|
||||
shm-shared surface), `attach_backing`, `set_scanout` to scanout 0, `transfer_to_host_2d`
|
||||
+ `resource_flush` of a test pattern, and wait on the used ring.
|
||||
- [x] Register a `scanout` service (`ServiceId.scanout` = 11).
|
||||
|
||||
**Gate (met):** the `virtio-gpu` case (QEMU `-device virtio-gpu-pci`) boots the
|
||||
device-manager stack, which discovers the function and spawns the driver; the driver writes
|
||||
a known test pattern into the scanout backing, `transfer_to_host_2d` + `resource_flush`es
|
||||
it, and **waits for the device's used-ring ack**, then reads the backing back and checks the
|
||||
pattern — logging `virtio-gpu: scanout 640x480 online` and `virtio-gpu: flush acked, pixel
|
||||
check ok`. That proves virtqueue + resource + attach + set_scanout + transfer + flush end to
|
||||
end without a screenshot (the used-ring ack is the device confirming it consumed the frame).
|
||||
|
||||
## V4 — The native backend + hot-attach ✅
|
||||
|
||||
- [x] `backend.VirtioGpu` in the compositor: `surface()` = the shared `shm` scanout surface
|
||||
(the compositor composes straight into the device's resource backing; x86 DMA is
|
||||
coherent, so the cacheable shared pages need no flush), `present(damage)` = a `present`
|
||||
request over the driver's `.scanout` endpoint (→ transfer-to-host + resource flush).
|
||||
- [x] The driver **announces** to `.display` after bring-up (looks it up with a bounded retry,
|
||||
sends `attach_scanout` with the geometry + the shared surface as an `ipc_call` send_cap).
|
||||
The compositor maps it, looks up `.scanout` itself (no need to pass the endpoint — the
|
||||
driver registered it), switches backend, and re-composites the current frame full-screen.
|
||||
The present is deferred to a one-shot timer so it runs *after* the reply unblocks the
|
||||
driver and it serves `.scanout` — presenting inline would deadlock.
|
||||
- [x] Boot still starts on `backend.Gop`; the upgrade happens on announce. `shm_physical` (a
|
||||
new syscall) gives the driver the guest-physical of the shared surface for `attach_backing`.
|
||||
|
||||
**Gate (met):** the `display-native` case (QEMU `-device virtio-gpu-pci`, `mem` bumped since it
|
||||
boots the whole system) starts the compositor + `display-demo` + device-manager; the driver
|
||||
announces, the compositor logs `display: scanout upgraded to virtio-gpu`, drives frames through
|
||||
the native backend, and **reads a pixel back** from the shared surface after a present to
|
||||
confirm the composited frame landed (`display: native present verified`), while `display-demo:
|
||||
ok` still fires — checked order-independently. Without `-device virtio-gpu-pci` nothing is
|
||||
announced and it stays on GOP: the v1 `display-service`/`display-demo` gates pass unchanged.
|
||||
|
||||
## V5 — Mode-setting, EDID, and vsync ✅
|
||||
|
||||
- [x] The driver negotiates `VIRTIO_GPU_F_EDID` (when offered) and reads the monitor's EDID,
|
||||
logging its preferred mode; it offers a small mode list over `.scanout` `get_modes`. The
|
||||
resource + shared surface are sized to the largest mode, so `set_mode` just re-points the
|
||||
scanout rectangle (no resource/surface churn) — a runtime resolution change. `runtime.display`
|
||||
gains `modes()` / `setMode()` (display-protocol `get_modes`/`set_mode`, forwarded to the backend).
|
||||
- [x] Every `resource_flush` is issued fenced (`VIRTIO_GPU_FLAG_FENCE`); the device signals the
|
||||
fence when the frame is on screen, which the used-ring ack the synchronous present waits on
|
||||
already gates — a tear-free present.
|
||||
- [x] `backend.VirtioGpu` reports `canModeSet` / `hasVsync` = true.
|
||||
|
||||
**Gate (met):** the `display-modeset` case (reusing the display-native boot) upgrades to
|
||||
virtio-gpu, queries the driver's modes, `setMode`s to a different resolution, and confirms the
|
||||
change by reading the backend's geometry back (`display: mode set to {w}x{h}, verified`); the
|
||||
fenced present path is exercised and confirmed (`display: vsync present ok`) — both from serial,
|
||||
passing 3/3. The driver also logs the EDID preferred mode (`virtio-gpu: EDID preferred mode …`).
|
||||
|
||||
## V6 — Resilience (restart + re-attach) + tests + docs ✅
|
||||
|
||||
- [x] The virtio-gpu driver now **hellos** the device manager (role: bus) so it is properly
|
||||
supervised — no longer stopped at the hello deadline — and is restarted on death. On
|
||||
driver loss the compositor keeps the last frame (its `.scanout` calls now return
|
||||
`-EPEER` instead of hanging — a kernel fix: an endpoint is marked dead when its owner
|
||||
dies) and **re-attaches** when the restarted driver re-announces. A permanent give-up
|
||||
(crash-loop cap) leaves the frozen frame; GOP is not re-taken.
|
||||
- [x] `test/qemu_test.py`: the `virtio-gpu`, `display-native` (hot-attach), `display-modeset`,
|
||||
and `display-reattach` (driver-kill/re-attach) cases. display-v2.md status updated.
|
||||
|
||||
**Gate (met):** the `display-reattach` case — device-manager (in `test-scanout-restart` mode)
|
||||
kills the virtio-gpu driver once after it hellos; the restart policy respawns it, it
|
||||
re-announces, and the compositor logs `display: scanout re-attached` after the initial
|
||||
`display: scanout upgraded to virtio-gpu`, with no CPU exception / panic (the compositor
|
||||
survives) — passing 3/3. All v1 + v2 cases (host tests, `ipc`/`ipc-call`/`ipc-cap`,
|
||||
`supervision`, `shm`, `display-service`, `display-demo`, `virtio-gpu`, `display-native`,
|
||||
`display-modeset`) pass; default `zig build` is clean.
|
||||
|
||||
---
|
||||
|
||||
## Deferred (explicitly not in this plan)
|
||||
|
||||
- **Client-rendered surfaces** — now unblocked by the `shm` capability (V2): an app renders
|
||||
its own bitmap and hands the compositor a reference. A natural follow-on.
|
||||
- **Bochs DISPI backend** — a simpler second native backend (mode-set only, dumb scanout);
|
||||
slots behind the same interface if wanted.
|
||||
- **Real-GPU (NVIDIA/AMD/Intel) drivers** — out of scope; those devices stay on the GOP
|
||||
floor by design.
|
||||
- **Hardware-accelerated compositing / multiple heads** — future.
|
||||
@@ -0,0 +1,131 @@
|
||||
# The display service v2: a pluggable scanout backend
|
||||
|
||||
**Status: complete (V1–V6).** The compositor boots on the GOP framebuffer and, when a
|
||||
virtio-gpu driver announces itself, hot-attaches a native backend over the shared `shm`
|
||||
scanout surface — with runtime mode-setting, EDID, and fenced (vsync) presents, and it
|
||||
re-attaches across driver restarts. All serial-gated (see [display-v2-plan.md](display-v2-plan.md)).
|
||||
|
||||
v1 ([display.md](display.md)) is a compositor that owns the **GOP framebuffer** — it
|
||||
composites a layer stack into a cacheable back buffer and streams damage to the linear
|
||||
framebuffer the firmware handed over. That path is portable and good: it drives any GPU,
|
||||
including a real NVIDIA card at an ultrawide's native resolution, with zero GPU-specific
|
||||
code. v2 keeps it as the **floor** and makes *scanout* — how a finished frame reaches the
|
||||
panel — a **pluggable backend**, so the compositor can **upgrade to a real GPU driver when
|
||||
one is present** and fall back to the framebuffer when it isn't.
|
||||
|
||||
The compositor itself (layers, back buffer, damage) does not change. Only the last step —
|
||||
"put this frame on screen" — becomes swappable.
|
||||
|
||||
## The shape
|
||||
|
||||
```
|
||||
compositor (display service) ── layer stack + back buffer + damage (unchanged)
|
||||
│ composites a frame, then: backend.present(damage)
|
||||
▼
|
||||
scanout backend (selected at runtime — GOP by default, native when it appears)
|
||||
│
|
||||
├─ GopBackend the v1 path: WC copy back→front to the firmware LFB.
|
||||
│ Always available. No mode-set, no vsync. THE FLOOR.
|
||||
│
|
||||
└─ VirtioGpuBackend talks to a virtio-gpu driver process over a `scanout`
|
||||
service: present via a shared resource + flush (real vsync),
|
||||
EDID mode list, runtime mode-set.
|
||||
```
|
||||
|
||||
A **backend** is a small interface the compositor calls:
|
||||
|
||||
- `surface()` → the pixels to compose into and their geometry `{ptr, pitch, format, w, h}`
|
||||
(the LFB for GOP; a shared scanout resource for virtio-gpu),
|
||||
- `present(damage: Rect)` → make the damaged region visible (a no-op-ish WC copy for GOP;
|
||||
a virtio flush, optionally vsync-fenced, for the native path),
|
||||
- capability queries — `canModeSet`, `hasVsync` — and, when supported, `modes()` /
|
||||
`setMode(m)`.
|
||||
|
||||
The compositor composes into `surface()` and calls `present(damage)` exactly as it does
|
||||
today; everything device-specific lives behind the interface.
|
||||
|
||||
## Selection and hot-attach
|
||||
|
||||
The choice is **dynamic**, because a GPU driver is spawned asynchronously (the device
|
||||
manager brings it up after boot), and because danos is meant to be resilient:
|
||||
|
||||
1. **Boot on GOP.** The compositor starts on `GopBackend` immediately, so there is never a
|
||||
blank screen while drivers load — the exact v1 behaviour.
|
||||
2. **Upgrade on announce.** When the virtio-gpu driver has claimed its device and set up a
|
||||
scanout, it **announces itself to the display service** (a `push`: the driver looks up
|
||||
`.display` and sends an *attach-scanout* message carrying its `scanout` endpoint as a
|
||||
capability). The compositor switches to `VirtioGpuBackend` and re-presents the current
|
||||
frame full-screen. Push beats polling — the compositor doesn't know a priori which
|
||||
driver, if any, exists, and danos has no service-registration pub/sub.
|
||||
3. **Native is restartable, not fallback-on-crash.** Once a native driver has reprogrammed
|
||||
the device, the firmware's GOP framebuffer is **stale** — "native → GOP" is not a clean
|
||||
fall-back. So a native driver that **crashes** is *restarted* by its supervisor (the
|
||||
resilience work already merged), re-announces, and the compositor **re-attaches**
|
||||
(native → native). The screen freezes on the last frame during the gap — acceptable.
|
||||
4. **GOP is the floor for "no driver was ever there."** On a real GPU (NVIDIA/AMD/Intel)
|
||||
the class-0x03 device matches nothing in the driver table, no `scanout` is ever
|
||||
announced, and the compositor stays on GOP forever — no special-casing. Only if a
|
||||
native driver *permanently* gives up (crash-loop cap) does the compositor attempt GOP
|
||||
again, and even then only if the LFB is still mappable.
|
||||
|
||||
## The shared-memory primitive this needs
|
||||
|
||||
virtio-gpu's scanout resource is **guest RAM** — the driver allocates it and attaches it
|
||||
to a virtio resource, and the compositor composes into it. That means the compositor
|
||||
writing into the driver's buffer is **cross-process memory sharing**, the primitive v1
|
||||
deferred (docs/display.md, "What v1 does not do"). v2 builds it: the natural generalization
|
||||
of M13 capability-passing from *endpoints* to *memory objects* —
|
||||
|
||||
```
|
||||
shm_create(len) -> {handle, vaddr} // a shareable, page-aligned RAM region
|
||||
… pass `handle` as the send_cap on an ipc_call …
|
||||
shm_map(cap) -> vaddr // the receiver maps the same physical pages
|
||||
```
|
||||
|
||||
The payoff is leverage: the **same** primitive unlocks **both** native GPU drivers *and*
|
||||
client-rendered surfaces (an app composing its own bitmap and handing the compositor a
|
||||
reference instead of drawing by command). One piece of kernel work, two features.
|
||||
|
||||
## The virtio-gpu driver
|
||||
|
||||
A new ring-3 driver process (the topology v1 anticipated — "split the driver from the
|
||||
compositor when a second backend arrives"). It claims the virtio-gpu PCI function, and:
|
||||
|
||||
- sets up the **virtqueues** (control + cursor) and the device's config space,
|
||||
- creates a **2D scanout resource** backed by an `shm` region, `attach_backing`s it,
|
||||
`set_scanout`s it to a CRTC, and `resource_flush`es damaged rectangles,
|
||||
- reads **EDID** (the `GET_EDID` control command) for the mode list, and `set_scanout`
|
||||
at a chosen mode for **runtime mode-setting**,
|
||||
- registers a `scanout` service and announces to the display service.
|
||||
|
||||
Its `resource_flush` is the real **present** — and gives a genuine **vsync/tear-free**
|
||||
path a dumb GOP framebuffer can't.
|
||||
|
||||
## What v2 unlocks — and its honest scope
|
||||
|
||||
Behind the abstraction, a native backend gives runtime **mode-setting** (resolution /
|
||||
refresh / bpp), **EDID** enumeration, and **vsync**. But only on devices we have a driver
|
||||
for — realistically **VMs** (virtio-gpu, and later maybe Bochs DISPI). Real discrete GPUs
|
||||
need per-vendor KMS-class drivers that aren't getting written, so they **stay on GOP** —
|
||||
which is genuinely fine (v1 on the NVIDIA box is smooth). So v2's real value is twofold:
|
||||
the **pluggable architecture** (a driver slots in when one exists) and a **rich, vsync'd
|
||||
path in VMs**, where danos development happens. The framebuffer floor never goes away.
|
||||
|
||||
## Locked decisions
|
||||
|
||||
- **First native backend: virtio-gpu** — the VM standard; gives mode-set + a real
|
||||
present/flush (and vsync), and exercises the whole pluggable design. Tested with QEMU
|
||||
`-device virtio-gpu`.
|
||||
- **Dynamic hot-attach** — boot on GOP, upgrade to native on the driver's announce,
|
||||
re-attach across driver restarts; GOP is the floor for "no driver ever," not a live
|
||||
fall-back after a reprogram.
|
||||
- **Detection = push** (the driver announces to `.display`), not compositor polling.
|
||||
- **v2 builds the `shm` capability** (endpoints → memory objects), shared with the future
|
||||
client-surface path.
|
||||
|
||||
## See also
|
||||
|
||||
- [display.md](display.md) — v1: the compositor, the GOP-vs-device split, the WC discipline.
|
||||
- [display-v2-plan.md](display-v2-plan.md) — the ordered build-out.
|
||||
- [driver-model.md](driver-model.md) — claim / `mmio_map` / MSI / capability passing (M13).
|
||||
- [resilience.md](resilience.md) — the restart machinery the hot-attach leans on.
|
||||
@@ -0,0 +1,556 @@
|
||||
# Native Intel iGPU display support — feasibility and roadmap
|
||||
|
||||
**Status: research snapshot, not implemented.** This records what a *minimal, display-only*
|
||||
native driver for an **Intel integrated GPU** — EDID read + mode-set + framebuffer scanout, with
|
||||
**no** 3D/media/compute — would take, and how it slots into danos's pluggable scanout
|
||||
architecture. It is a survey of primary sources (Intel's open-source
|
||||
[Programmer's Reference Manuals](https://www.intel.com/content/www/us/en/docs/graphics-for-linux/developer-reference/1-0/overview.html),
|
||||
coreboot's [libgfxinit](https://doc.coreboot.org/gfx/libgfxinit.html), the Linux
|
||||
[i915 display](https://github.com/torvalds/linux/tree/master/drivers/gpu/drm/i915/display) driver,
|
||||
and Haiku's [intel_extreme](https://github.com/haiku/haiku/tree/master/src/add-ons/kernel/drivers/graphics/intel_extreme/)),
|
||||
not an implementation. It is the companion to [nvidia-gpus.md](nvidia-gpus.md) and should be read
|
||||
against it — the two answer the same question for opposite silicon.
|
||||
|
||||
Read [display.md](display.md) and [display-v2.md](display-v2.md) first — this doc assumes the v2
|
||||
model where scanout is a **pluggable backend** and a native driver is just another `.scanout`
|
||||
service (like the virtio-gpu one), announcing to the compositor over `attach_scanout`.
|
||||
|
||||
## TL;DR
|
||||
|
||||
- **Intel is a materially easier, lower-tier target than the NVIDIA RTX 3060 — and the reason is
|
||||
documentation, not silicon.** Intel publishes official, register-level, per-platform **Display
|
||||
Engine** PRMs with named registers, bitfields, and numbered enable sequences; NVIDIA publishes
|
||||
no display PRM and forces reverse-engineering against GPL nouveau. A minimal Intel display-only
|
||||
driver is roughly **tier 2 to low-tier 3** for well-covered generations (Skylake / Kaby Lake /
|
||||
Coffee Lake), versus NVIDIA's **tier 4** for GA106. This is the load-bearing conclusion.
|
||||
- **The display block is a genuinely separable register domain.** Mode-set + scanout touch only
|
||||
display registers (pipes, planes, transcoders, DDI buffers, PLLs, power wells, GMBUS/AUX) — **no
|
||||
render engine, no command streamer, no GEM/3D, no signed microcode.** Two small carve-outs, both
|
||||
trivial pokes that do *not* pull in the render engine: a real CDCLK frequency change writes the
|
||||
shared GT PCODE mailbox, and the plane's surface register is a GGTT (memory-interface) address.
|
||||
- **There is no firmware wall on the display path.** The only display microcontroller (DMC / "CSR",
|
||||
Skylake+) is **optional** — its sole job is saving/restoring display state across DC5/DC6
|
||||
low-power idle. Without it, i915 prints "Disabling runtime power management" and mode-sets and
|
||||
scans out normally. GuC/HuC are render/media coprocessors, never touched by a display driver.
|
||||
Pre-Skylake parts have no display microcontroller at all yet mode-set fine. There is **nothing
|
||||
analogous to NVIDIA's GSP**.
|
||||
- **The scanout memory model is dramatically simpler than a discrete GPU.** Intel iGPUs have **no
|
||||
VRAM**: the display scans out of ordinary system RAM addressed through the Global GTT (GGTT), a
|
||||
flat single-level page table. Linear (untiled) framebuffers are first-class. You need **no
|
||||
GEM/TTM, no VMM, no VRAM allocator, no BAR1 aperture juggling** — the exact machinery the NVIDIA
|
||||
path forces on you.
|
||||
- **coreboot libgfxinit is a compact, complete, display-only reference** doing precisely this scope
|
||||
(EDID + PLL/mode-set + scanout, zero 3D) in ~22k lines of formally-analysed SPARK/Ada — versus
|
||||
i915's ~400k lines. It is a *read-and-reimplement* reference, not drop-in code (GPL-2.0-or-later,
|
||||
and Ada, not Zig).
|
||||
- **The clean-room, permissively-licensed path is real** — you can implement from the PRM without
|
||||
reading GPL code, and Haiku's MIT `intel_extreme` is a permissive precedent. This is the decisive
|
||||
contrast with NVIDIA, where no vendor register spec exists.
|
||||
- **The practical catch is hardware, not software.** On a desktop with an RTX 3060, the monitor is
|
||||
almost certainly cabled to the *card*, so an iGPU driver would light a dark motherboard port; the
|
||||
CPU may be an **F-SKU with the iGPU fused off entirely**; and every clean-room reference targets
|
||||
*older* Intel. Intel is the right target to **learn** display bring-up — "run it on my machine"
|
||||
is a separate, machine-dependent question that may not resolve in the reader's favour.
|
||||
- **Recommendation:** as with the NVIDIA doc, GOP already gives native-resolution scanout with zero
|
||||
GPU code. A native Intel driver buys runtime mode changes, hardware vsync, and multihead — and it
|
||||
reaches "first pixel" far faster than the NVIDIA path *if* the target machine actually has a
|
||||
usable, cable-attached iGPU of a documented generation.
|
||||
|
||||
## Display engine architecture, and why it's separable
|
||||
|
||||
For the common single-display path (SST DisplayPort / HDMI / eDP), the Intel display data flow is a
|
||||
small, fully documented, essentially fixed sequence:
|
||||
|
||||
```
|
||||
memory surface → PLANE(s) → PIPE → TRANSCODER → DDI (drives IO/PHY) → connector
|
||||
```
|
||||
|
||||
The Tiger Lake PRM Vol 12 states it verbatim: *"The front end of the display contains the pipes.
|
||||
The pipes connect to the transcoders. The transcoders, except for wireless, connect to the DDIs to
|
||||
drive the IO/PHY."* A **pipe** blends planes (primary/sprite/cursor) into one raster stream; the
|
||||
**transcoder** wraps it in port-protocol timing (DP/HDMI/eDP/DSI); the **DDI** is the physical port
|
||||
and PHY. Pipe, Planes, Transcoder, and Digital Display Interface are each first-class PRM chapters
|
||||
with per-object files in libgfxinit
|
||||
([TGL PRM Vol 12](https://cdrdv2-public.intel.com/705833/intel-gfx-prm-osrc-tgl-vol-12-display-engine.pdf)).
|
||||
|
||||
**Two honest qualifications** the raw research overstated (per verification):
|
||||
|
||||
- The pipeline is *not* strictly linear in all cases — the same PRM pages document optional branches
|
||||
a minimal driver simply ignores (wireless writeback to memory, MIPI DSI, DisplayPort multistream
|
||||
many-to-one, DSC/tiled pipe-joining). Ignoring them does not weaken feasibility.
|
||||
- The four-object model *as named* is **Haswell-onward** (DDI introduced ~2013), not "every gen."
|
||||
Pre-Haswell used FDI + PCH transcoders + port-specific encoders. Within the modern iGPU range
|
||||
danos would realistically target (Skylake → Meteor/Lunar Lake) the model is stable.
|
||||
|
||||
**The DPLL/clock block is a separate, per-port programmable clock source** and is one of the harder,
|
||||
most gen-specific pieces: pick/enable a PLL, route its output to the DDI, then bring up the port.
|
||||
The register layout and divider math change substantially per generation — pre-SKL SPLL/WRPLL/LCPLL,
|
||||
Skylake+ shared DPLL0–3, Gen11+ combo-PHY plus Type-C MG/DKL PLLs. Pixel-clock computation is a
|
||||
classic per-gen rewrite.
|
||||
|
||||
### Separable from render — the single most important enabler
|
||||
|
||||
The display is a distinct register domain from render/media, and this is confirmed at the primary
|
||||
level: the TGL PRM ships display as its own volume (Vol 12), separate from Render Engine (Vol 9) and
|
||||
Media (Vol 11); Linux's KMS "is provided by Intel Display Driver, and **shared with drm/xe**"
|
||||
([kernel.org i915](https://docs.kernel.org/gpu/i915.html)) — i.e. the display module is
|
||||
reused across two different GPU drivers. A full mode-set lights a display end-to-end using only power
|
||||
wells, PLL/port-clock, DDI-buffer/PHY, transcoder and pipe registers — **zero render commands, zero
|
||||
GEM objects, zero command-streamer.** libgfxinit is decisive proof: complete EDID + modeset +
|
||||
framebuffer with no render/3D code at all.
|
||||
|
||||
Two carve-outs the "touches ONLY display registers" phrasing needs (per verification), **neither of
|
||||
which drags in the render engine**:
|
||||
|
||||
1. A mode-set that changes the **Core Display Clock (CDCLK)** frequency/voltage pokes the shared **GT
|
||||
Driver Mailbox** (PCODE/PCU power-controller interface), per Vol 12's own "Display Voltage
|
||||
Frequency Switching" step. A trivial register handshake, documented alongside the display sequence.
|
||||
2. The primary plane's surface register (`PLANE_SURF`) holds a **GGTT graphics address** (a
|
||||
memory-interface concept, not covered in Vol 12). Using pre-mapped stolen memory — as libgfxinit
|
||||
does — sidesteps any active GGTT programming. See [Memory and scanout](#memory-and-scanout).
|
||||
|
||||
### Per-gen churn: what's stable, what you rewrite
|
||||
|
||||
The **object model** (pipes/planes/transcoders/DDIs, GMBUS-for-EDID, double-buffered plane registers
|
||||
armed atomically) is conceptually stable from Ironlake/Haswell through Tiger Lake. What you rewrite
|
||||
per generation is:
|
||||
|
||||
1. the **CPU-vs-PCH split and interconnect**,
|
||||
2. the **port/PHY + DPLL** programming,
|
||||
3. **register offsets + power-well / CDCLK topology**, and
|
||||
4. the **mode-set enable sequence itself** (power-well ordering, PLL lock, DDI-buffer enable,
|
||||
transcoder clock-select) — an effective fourth axis the raw research folded into (1)/(2).
|
||||
|
||||
Interconnect eras, with the timeline **corrected** (the cited Haiku doc was chronologically loose):
|
||||
|
||||
- **Gen5 Ironlake (2010) → Ivy Bridge:** FDI (Flexible Display Interface) links the CPU display
|
||||
engine to PCH-resident ports. The FDI/PCH-split era begins at **Ironlake**, not Gen7.
|
||||
- **Haswell (Gen7.5):** the main digital outputs come **back onto the CPU die as DDIs** (DDI A = eDP)
|
||||
— the *opposite* of "moving output to the PCH," and it collapses the FDI/PCH dance **for the
|
||||
digital ports only**. FDI is **retained** for the legacy VGA/CRT path (DDI E → PCH CRT DAC), so a
|
||||
driver gets the single DDI code path only by omitting analog VGA (which a minimal driver does).
|
||||
- **Skylake (Gen9):** reworks clock/PLL, CDCLK, and the power-well model; introduces the optional DMC.
|
||||
- **Gen11 Ice Lake / Gen12 Tiger Lake:** add combo-PHY + USB-Type-C/Thunderbolt MG/DKL PHYs — the
|
||||
single biggest cost increase, and the reason "newest silicon" is *not* the easiest target. (DSC is
|
||||
documented per-**pipe**; MSO is an eDP feature — not "per-transcoder" as the raw research said.)
|
||||
|
||||
### The tractable sweet spot
|
||||
|
||||
The documented, tractable sweet spot for a from-scratch display-only driver is the
|
||||
**Haswell (Gen7.5) / Broadwell (Gen8) DDI family, with Skylake (Gen9) as the modern-hardware pick**
|
||||
since it shares the same DDI object model. Rationale:
|
||||
|
||||
- Broadwell has a complete, freely downloadable
|
||||
[PRM Vol 11 Display](https://cdrdv2-public.intel.com/690828/intel-gfx-prm-osrc-bdw-vol-11-display.pdf);
|
||||
its engine (3 pipes A/B/C, 4 transcoders incl. transcoder-EDP that floats onto any pipe, DDI A–E,
|
||||
WRPLL/SPLL/LCPLL) is the classic "DDI + transcoder + WRPLL" model.
|
||||
- It predates the combo-PHY / Type-C / MG-DKL complexity of Ice Lake / Tiger Lake.
|
||||
- libgfxinit's DDI **connector/EDID/DP layer is uniform from Haswell through Coffee Lake**, so the
|
||||
hardest-to-get-right port logic generalises widely.
|
||||
|
||||
Two supporting claims from the raw research are **wrong and corrected here (verification):**
|
||||
|
||||
- **The BDW and SKL PRMs are NOT 0BSD-licensed.** Both carry a Creative Commons
|
||||
**Attribution-NoDerivatives** notice. Only the *newer* OSRC PRMs (Tiger Lake 2021 onward) put their
|
||||
embedded code samples under **Zero-Clause BSD**. So for the recommended Haswell/Broadwell/Skylake
|
||||
generations there are no "copy-pasteable 0BSD code samples" — the legal basis is *reimplementation
|
||||
from a CC-BY-ND spec* (register facts are not copyrightable), not copying.
|
||||
- **FDI+PCH is not fully eliminated on Haswell/Broadwell.** The BDW PRM keeps FDI for the DDI E → PCH
|
||||
CRT DAC. The "one DDI code path" holds only for the digital outputs a minimal driver targets.
|
||||
|
||||
Sandy/Ivy Bridge (Gen6/7) is where the hobby-doc walkthroughs concentrate (the OSDev GMBUS/EDID
|
||||
material) but carries the FDI+PCH split cost. *(Low confidence on the OSDev specifics — the wiki
|
||||
returns 403 to automated fetches and its "guaranteed to work" phrasing is a hobby assertion, not a
|
||||
silicon guarantee.)*
|
||||
|
||||
## Documentation — and the clean-room question
|
||||
|
||||
This is the crux of the whole comparison. **Intel hands you the register spec that NVIDIA withholds.**
|
||||
|
||||
- The Tiger Lake **"Vol 12: Display Engine"** PRM is a real, first-party, open-source document —
|
||||
**433 pages, verified by direct download** — with named registers + addresses + bitfield tables
|
||||
(`TRANS_DDI_FUNC_CTL`, `DDI_BUF_CTL`, `DP_TP_CTL`, `PLANE_STRIDE`, `DPLL_CFGCR0/1`, `CDCLK_CTL`,
|
||||
`PWR_WELL_CTL_DDI`, …) and **numbered, step-by-step enable sequences** with explicit writes, wait
|
||||
conditions, and microsecond timeouts. It even includes the "magic value" tables older PRMs deferred
|
||||
to the driver (DisplayPort PLL DCO/divider values; voltage-swing/de-emphasis in mV). *"A spec you
|
||||
could write a driver from directly"* is well-supported, not hyperbole
|
||||
([TGL Vol 12](https://cdrdv2-public.intel.com/705833/intel-gfx-prm-osrc-tgl-vol-12-display-engine.pdf)).
|
||||
- **Clean-room, permissively-licensed implementation is legally and practically feasible from the
|
||||
PRM alone.** CC-BY-ND governs redistribution of the *document*; register addresses and bit
|
||||
definitions are functional facts, and original code implementing a described hardware interface is
|
||||
not a derivative of the PDF. *(This is standard copyright reasoning, not adjudicated case law —
|
||||
treat it as well-grounded, not settled.)* Two independent implementations already exist built
|
||||
essentially from these docs (libgfxinit, Haiku), so the spec is demonstrably sufficient.
|
||||
|
||||
**The documentation ceiling — corrected.** The raw research said public PRMs stop "roughly at Ice
|
||||
Lake / Tiger Lake." Verification refuted this: full public **"Vol 12 Display Engine"** PRMs exist for
|
||||
Ice Lake, Lakefield, Tiger Lake, Rocket Lake, DG1, **and DG2/Arc "Alchemist" (Gen12.5, 2022)** —
|
||||
[the ACM display PRM is public](https://www.x.org/docs/intel/ACM/intel-gfx-prm-osrc-acm-vol12-displayengine.pdf).
|
||||
The genuine cliff is **Meteor Lake (2023) and newer**: those have only a high-level architecture
|
||||
overview, no register-level display PRM, and i915 references their display registers by opaque
|
||||
internal **Bspec numeric IDs**. Alder Lake and Raptor Lake iGPUs are Gen12 Xe-LP display — the same
|
||||
IP as Tiger Lake — so despite lacking a dedicated PRM they are effectively covered by the TGL PRM.
|
||||
|
||||
Net: a from-docs driver can confidently target **Skylake through DG2/Arc**, which is essentially the
|
||||
entire current laptop/NUC installed base; only Meteor Lake and later slide back toward the NVIDIA
|
||||
situation (reverse-engineering or reading GPL i915). The PRMs also survived 01.org's shutdown and are
|
||||
mirrored in several stable places (Intel's cdrdv2 host, the
|
||||
[Igalia CC-BY-ND archive](https://github.com/Igalia/intel-osrc-gfx-prm) for Gen4–Gen9.5,
|
||||
[kiwitree](https://kiwitree.net/~lina/intel-gfx-docs/prm/), x.org) — not a single point of failure.
|
||||
*(Note: the Igalia archive stops at Kaby Lake and contains no Display Engine volume; the TGL/DG2
|
||||
display PRMs are separate Intel/x.org downloads.)*
|
||||
|
||||
## coreboot libgfxinit — the native reference
|
||||
|
||||
[libgfxinit](https://doc.coreboot.org/gfx/libgfxinit.html) is the closest thing to a template danos
|
||||
could ask for: a self-contained **native modeset library** (no VBIOS/int10, no firmware blobs) that
|
||||
probes displays via EDID over DDC/I²C and DP AUX, and drives LVDS, eDP, DP1–3, HDMI1–3, analog VGA,
|
||||
plus USB-C DP/HDMI alt-mode on Tiger Lake. It sets up pipes (Primary/Secondary/Tertiary), planes,
|
||||
transcoders, PLLs, panel power/backlight, the GTT, and framebuffer scanout — **display-only, zero
|
||||
3D/media/compute**, which is exactly danos's scope. Its public entry is essentially
|
||||
`Initialize()` then `Update_Outputs(Pipe_Configs)`, where each `Pipe_Config` carries
|
||||
`{Port, Framebuffer, Cursor, Mode}` — a near-perfect fit for a pluggable scanout backend.
|
||||
|
||||
Why it beats i915 as a reference (**verified by measurement**): **131 Ada source files, ~818 KB,
|
||||
~22k code lines** across *all* generations, factored precisely along the axes you care about (`edid`,
|
||||
`dp_aux`, `dp_training`, `pipe_setup`, `transcoder`, `plls`, `connectors`, `port_detect`), with
|
||||
**none** of the DRM/KMS/GEM/TTM, GT/3D, RC6/RPS, or GuC/HuC machinery that makes
|
||||
`drivers/gpu/drm/i915` **~419k lines / 900 files / 12 MB**. (A grep confirms *zero* gem/ttm/guc/huc/
|
||||
execbuf identifiers in the tree.) It depends only on a small HW-access shim, `libhwbase`
|
||||
(`HW.PCI`, `HW.Port_IO`, `HW.MMIO`, `HW.Time`), which maps naturally onto danos's MMIO-grant + IPC
|
||||
primitives — you provide Zig equivalents and the modeset logic sits on top. *(Correction to the raw
|
||||
research: the widely-quoted "~13–14k LOC" is only the generic `common/` layer; the eight
|
||||
per-generation subdirs roughly double it.)*
|
||||
|
||||
**It is a read-and-reimplement reference, not drop-in code.** Two hard constraints:
|
||||
|
||||
- **License is GPL-2.0-or-later** (the COPYING file is GPLv2; per-file headers add "or any later
|
||||
version"). The CC-BY-4.0 on the docs *site* is a footer, not the source license. Copyleft applies
|
||||
to ported code.
|
||||
- **It is SPARK/Ada, and designed to run as coreboot boot-firmware**, not a runtime OS driver. A
|
||||
danos port means either an Ada/GNAT toolchain in the build or hand-transliteration into Zig; the
|
||||
SPARK "absence of runtime errors" proof does **not** carry over to your reimplementation (and note
|
||||
it proves absence of runtime errors, **not** functional modeset correctness).
|
||||
|
||||
Two more caveats worth knowing: its **error handling is limited** — "only the case that no display
|
||||
could be found counts as failure"; a later DP link-training failure is *not* propagated. And its
|
||||
**verified-in-coreboot** hardware list stops at **Coffee Lake + Apollo Lake**, even though the tree
|
||||
contains a `tigerlake/` directory (Ice Lake has no directory at all, and Alder Lake support is only
|
||||
"begun"). So treat Haswell..Coffee Lake as the trustworthy transliteration window and TGL as
|
||||
present-but-less-proven.
|
||||
|
||||
The orchestration reads as a clean state machine (`hw-gfx-gma.adb` `Enable_Output`):
|
||||
`Fill_Port_Config → Preferred_Link_Setting → PLLs.Alloc → [retry] Connectors.Pre_On →
|
||||
Display_Controller.On → Connectors.Post_On`, with a literal *"try each DP-lane configuration twice"*
|
||||
inner retry and an outer link-setting step-down. `hw-gfx-dp_training.adb` (398 lines) is a complete,
|
||||
generic DP link-training implementation (TP1/TP2/TP3, CR + EQ loops, swing/pre-emphasis adjust from
|
||||
sink status). Per-generation buffer translations plug in underneath via
|
||||
`Program_Buffer_Translations`, gated on `Config.Has_DDI_Buffer_Trans`. All of this was confirmed
|
||||
against the source line-by-line.
|
||||
|
||||
## The EDID + mode-set path (Haswell/Broadwell target)
|
||||
|
||||
The whole path is memory-mapped register programming with polled status bits — no command ring, no
|
||||
microcode, no DMA channel.
|
||||
|
||||
**EDID over DDC (GMBUS).** Pure MMIO poking of the GMBUS I²C controller (`GMBUS0`–`GMBUS5`): `GMBUS0`
|
||||
selects pin-pair/port + clock; `GMBUS1` carries slave address (`0x50` for EDID), byte count,
|
||||
direction, SW-ready; `GMBUS2` exposes HW-ready/NAK/ACTIVE to poll; `GMBUS3` is a 4-byte data FIFO;
|
||||
`GMBUS5` gives the 2-byte segment index for E-DDC. A read is: write `GMBUS0`, write `GMBUS1`
|
||||
(`CYCLE_WAIT | count | SLAVE_READ | SW_RDY | slave<<addr`), loop {poll `HW_RDY`, read 4 bytes}, then
|
||||
STOP ([i915 intel_gmbus.c](https://github.com/torvalds/linux/blob/master/drivers/gpu/drm/i915/display/intel_gmbus.c)).
|
||||
|
||||
**EDID + DPCD over DP AUX.** For DisplayPort/eDP, EDID (as I²C-over-AUX to `0x50`) and all DPCD
|
||||
capability/link-status registers are read over the AUX channel: per-DDI `DDI_AUX_CTL` + 5×
|
||||
`DDI_AUX_DATA`. Build a 3–5 byte header + payload, set SEND_BUSY, poll it clear, read
|
||||
DONE/TIMEOUT/RECEIVE_ERROR. Message size 1–20 bytes; spec requires ≥3 retries. On Haswell/BDW the AUX
|
||||
clock divider is programmed explicitly; SKL+ derive it automatically
|
||||
([i915 intel_dp_aux.c](https://github.com/torvalds/linux/blob/master/drivers/gpu/drm/i915/display/intel_dp_aux.c)).
|
||||
Both GMBUS and DP-AUX live in libgfxinit's shared `common/` — cheap and nearly gen-invariant.
|
||||
|
||||
**The mode-set is a fixed, documented register sequence.** The Broadwell DisplayPort enable order
|
||||
(verbatim from BDW PRM Vol 11, pp.98–99): (1) DDI lane capability; (2) panel power sequencing if
|
||||
needed; (3) enable the CPU display PLL (WRPLL/SPLL) and wait ~20 µs; (4) Port Clock Select → DDI,
|
||||
enable `DP_TP_CTL` with training pattern 1, configure `DDI_BUF_TRANS`, enable `DDI_BUF_CTL`, wait
|
||||
>518 µs, run link training, set `DP_TP_CTL` to Normal (Idle first for eDP); (5) Transcoder Clock
|
||||
Select, enable the plane, panel fitter if needed, program transcoder timings + M/N/TU, enable
|
||||
`TRANS_DDI_FUNC_CTL`, enable `TRANS_CONF`, then backlight. Disable is the exact reverse — a bounded
|
||||
checklist.
|
||||
|
||||
**DisplayPort/eDP link training is driver-driven in software over AUX** — the CPU runs the
|
||||
clock-recovery and channel-equalization state machines by hand; it is **not** offloaded to a hardware
|
||||
sequencer or firmware. The source side exposes only primitives: `DP_TP_CTL` selects the training
|
||||
pattern the port emits; `DDI_BUF_CTL`/`DDI_BUF_TRANS` set voltage-swing/pre-emphasis. The driver
|
||||
loops: emit pattern + set source levels → write `TRAINING_PATTERN_SET` (DPCD 0x102) + `TRAINING_LANEx_SET`
|
||||
(0x103) over AUX → delay (100 µs CR / 400 µs EQ) → read `LANE_STATUS` → on failure adjust to the
|
||||
sink's `ADJUST_REQUEST` values and retry. A few hundred lines of ordinary CPU/AUX code (libgfxinit
|
||||
`Train_DP`: CR loop 1..32, EQ loop 1..6). **This is the single fiddliest, most fragile piece** — a
|
||||
TMDS/HDMI panel avoids it entirely, and targeting an already-lit eDP panel avoids most of it.
|
||||
|
||||
**The clock (WRPLL) is documented divider math, not a magic table.** On Haswell/BDW the WRPLL derives
|
||||
the symbol clock from a 2700 MHz LCPLL reference through R2/N2/P dividers with VCO 2400–4800 MHz —
|
||||
small integer arithmetic. DP is *easier* than HDMI because it runs at a few fixed link rates (1.62 /
|
||||
2.7 / 5.4 GHz), so a DP/eDP-only minimal driver can often use fixed rates and skip most of the search.
|
||||
|
||||
**Plane/scanout programming is trivial for a compositor.** The primary plane is `PRI_CTL`
|
||||
(enable + pixel format), `PRI_STRIDE`, `PRI_SURF` (surface base — writing it triggers the atomic
|
||||
update), `PRI_OFFSET`; formats include 32-bit BGRX 8:8:8 and 16-bit BGRX 5:6:5 — a direct match for a
|
||||
linear XRGB compositor buffer. Plane registers are double-buffered and latch at vblank via an
|
||||
**arming** write — so a page-flip is "write base + stride + size, then the arming write." This is
|
||||
*exactly* the primitive danos's damage-driven compositor already expresses over GOP/virtio-gpu; the
|
||||
incremental work is "program these display-domain registers," not a new scanout model. The panel
|
||||
fitter (`PF_WIN_POS`/`PF_WIN_SZ`/`PF_CTRL`) can be left disabled for native-resolution scanout;
|
||||
Skylake+ replaces it with a shared pipe-scaler (`PS_CTRL`).
|
||||
|
||||
**Smallest useful target:** eDP (DDI A / transcoder-EDP) or a single DP output at native resolution,
|
||||
panel fitter off, plane in 32bpp XRGB. That is: GMBUS + I²C-over-AUX EDID/DPCD, one fixed-rate or
|
||||
WRPLL config, the ~20-step enable sequence, the software CR/EQ loop, and `PRI_*` plane setup with
|
||||
`PRI_SURF`-write flips. Out of scope: 3D, media, tiling, RC6/power-gating, PSR, audio.
|
||||
|
||||
## Memory and scanout
|
||||
|
||||
This is where Intel's *architecture* — not just its docs — makes the job smaller, and it is the
|
||||
biggest single simplification versus a discrete GPU.
|
||||
|
||||
- **No VRAM.** Intel iGPUs have a unified memory architecture; the display scans out of ordinary
|
||||
**system RAM** addressed through the **Global GTT (GGTT)**. The only way to give the GPU memory is
|
||||
to bind system pages into the GGTT
|
||||
([i915/GEM crashcourse](https://blog.ffwll.ch/2012/10/i915gem-crashcourse.html)).
|
||||
- **The plane surface register is a GGTT offset**, not a raw physical address — the display walks the
|
||||
GGTT to fetch pixels, so a scanout buffer must be GGTT-mapped (global, not per-process). libgfxinit
|
||||
writes the framebuffer offset straight into `DSPSURF`/`PLANE_SURF` masked to 4 KB.
|
||||
- **Linear (untiled) scanout is a first-class supported mode** — the plane's tiling field value 0 is
|
||||
Linear. No X/Y/Yf tiling engine is needed for a display-only driver. (UEFI GOP itself hands off a
|
||||
linear framebuffer the plane is already scanning.)
|
||||
- **No memory manager.** You need only (1) some contiguous-ish system pages and (2) GGTT PTEs
|
||||
pointing at them (`physical_addr | valid_bit` — the GGTT is a flat single-level array of PTEs in
|
||||
the `GTTMMADR` MMIO BAR), then program the plane. **No GEM/TTM/PPGTT/GuC.** coreboot's native-init
|
||||
literally does `for(i…) WRITE32(base + i*inc | 1, (i*4) | 1)`.
|
||||
- **"Stolen memory"** (GSM/DSM) is firmware-reserved system RAM where the firmware places the GGTT
|
||||
itself and the boot framebuffer. A driver is not obligated to keep scanout there — it can rebind
|
||||
GGTT entries to its own pages. Stolen memory matters mainly for *inheriting* the GOP framebuffer at
|
||||
handoff.
|
||||
|
||||
**The contrast with NVIDIA is stark.** On a discrete GPU the scanout surface must live in **VRAM**
|
||||
(nouveau always pins scanout to VRAM), CPU access goes through the **BAR1** aperture (which on
|
||||
consumer cards can be far smaller than total VRAM unless Resizable BAR is on), and you need a
|
||||
contiguous aligned VRAM allocator plus a BAR1 mapping. The Intel iGPU path **eliminates all of that**
|
||||
— scanout is plain system RAM, and a userspace compositor can write the framebuffer pages directly
|
||||
(as danos already does with the GOP WC framebuffer).
|
||||
|
||||
Because danos boots via GOP, an Intel driver attaches to a display whose **GGTT is already populated
|
||||
and whose plane is already scanning a linear framebuffer at native resolution.** A minimal driver can
|
||||
reuse that live mapping and reprogram the running plane rather than come up from cold — the same
|
||||
"attach to a live display" advantage the NVIDIA doc identifies, but with a far smaller register
|
||||
surface and no firmware wall. *(Low-confidence, per-target details to pin from the specific gen's
|
||||
PRM: GGTT PTE size — 4-byte pre-gen8 vs 8-byte gen8+ — the `GTTMMADR`/aperture BAR layout, surface
|
||||
alignment — 4 KB floor but some gens/tilings want 256 KB — and whether the display's GGTT-mediated
|
||||
DMA sits before or after danos's M16 IOMMU on the target platform.)*
|
||||
|
||||
## Firmware
|
||||
|
||||
A minimal display-only Intel driver is **effectively firmware-free — more so than NVIDIA.**
|
||||
|
||||
- **DMC (Display Microcontroller, "CSR", Skylake+) is NOT required for mode-set or scanout.** Its
|
||||
sole job is saving/restoring display-engine registers across DC5/DC6 low-power idle. Absent, i915
|
||||
prints *"Failed to load DMC firmware … Disabling runtime power management"* and the display
|
||||
mode-sets and scans out normally — you lose only the deep display idle states, not output
|
||||
([intel_dmc.c](https://github.com/torvalds/linux/blob/master/drivers/gpu/drm/i915/display/intel_dmc.c);
|
||||
corroborated by multiple distro bug threads). *(A source-level `HAS_DMC` early-return citation would
|
||||
strengthen this beyond distro testimony, but the conclusion is well-supported.)*
|
||||
- **Pre-Skylake parts have no display microcontroller at all** yet perform full mode-set (and even
|
||||
Panel Self Refresh). This confirms the display engine is fundamentally CPU/MMIO-driven; the
|
||||
microcontroller is an add-on for autonomous idling, not a prerequisite for lighting a panel.
|
||||
Targeting a pre-Skylake or DMC-optional generation sidesteps the question entirely.
|
||||
- **GuC and HuC are render/media microcontrollers on the GT side** — GuC schedules the render engines,
|
||||
HuC assists HEVC/H.265 codec (plus later HDCP/PXP/GSC). Neither is in the scanout path; a
|
||||
display-only driver never loads them
|
||||
([kernel.org microcontrollers](https://docs.kernel.org/gpu/i915.html)).
|
||||
- **PSR firmware lives on the panel**, not in the OS — a minimal driver simply doesn't enable PSR.
|
||||
- **Type-C/TCSS (Ice Lake+) firmware** (PMC/IOM/PHY) is part of platform BIOS/coreboot init and the
|
||||
hardware, *not* a signed blob the display driver loads at runtime. A driver attaching to an
|
||||
already-lit GOP connector, or targeting classic DDI ports, avoids it. *(Cold DP-alt-mode changes
|
||||
from a userspace driver on modern TCSS platforms were not traced to primary source — flagged.)*
|
||||
|
||||
There is **no signed-firmware wall over the Intel GPU at all** on the display path. This is the
|
||||
architectural opposite of NVIDIA's mandatory, unsignable, ABI-unstable GSP — which even on the
|
||||
near-side "direct" display path is a permanent maintenance liability for anything beyond scanout.
|
||||
|
||||
## Licensing
|
||||
|
||||
The situation is *better* than NVIDIA's but still nuanced.
|
||||
|
||||
- **The two best code references are both GPL** — Linux i915 (GPL-2.0) and coreboot libgfxinit
|
||||
(GPL-2.0-or-later). You cannot copy either into a permissively-licensed danos. libgfxinit's WRPLL
|
||||
divider math is itself copied from i915, so it carries the same encumbrance.
|
||||
- **But you don't need to copy code.** The Intel PRM is a *specification*, and a clean-room Zig
|
||||
implementation written from the PRM (using libgfxinit/i915 only to understand behaviour, never to
|
||||
copy) is legitimate — register numbers and bit definitions are functional facts, not copyrightable
|
||||
expression. This is the exact inverse of the NVIDIA case, where no such spec exists and the only
|
||||
guide is the GPL/RE'd code itself.
|
||||
- **A permissive precedent exists: Haiku's `intel_extreme` is MIT-licensed** and was built from
|
||||
Intel's public docs. So if danos wants a permissive license, the model is: implement from the PRM,
|
||||
optionally read MIT Haiku for structure, treat GPL libgfxinit/i915 as documentation-of-last-resort.
|
||||
- **A licensing nuance on the recommended generations:** the "copy the 0BSD PRM code samples" shortcut
|
||||
only applies to Tiger-Lake-era (2021+) PRMs. The Haswell/Broadwell/Skylake PRMs are CC-BY-ND, so
|
||||
their register *facts* are free to implement but there are no code samples to lift.
|
||||
|
||||
As with the NVIDIA doc: danos's userspace-driver-over-IPC model (a driver is a separate process behind
|
||||
a defined protocol) is the cleanest possible license boundary if the project ever chooses to ship a
|
||||
GPL display-driver binary and keep the rest of danos permissive — but that is a boundary judgement
|
||||
wanting real diligence, not a settled fact. The clean-room-from-PRM route avoids the question.
|
||||
|
||||
## Prior art outside Linux
|
||||
|
||||
This is a **real contrast with NVIDIA**, where no one has built a from-scratch native driver outside
|
||||
Linux. For Intel there are **multiple independent, non-Linux, clean-room native modeset
|
||||
implementations** to learn from:
|
||||
|
||||
- **coreboot libgfxinit** — SPARK/Ada, G45/GM45 and Arrandale → Coffee Lake + Apollo Lake (TGL
|
||||
in-tree), the strongest structural reference.
|
||||
- **Haiku `intel_extreme`** — modeset-only (no 2D/3D accel), **MIT-licensed**, i845 through Sandy
|
||||
Bridge solid, newer Gemini/Ice/Tiger Lake in progress but "hit or miss, as the driver lags behind
|
||||
the specs" ([Haiku generations](https://www.haiku-os.org/docs/develop/drivers/intel_extreme/generations.html),
|
||||
[Phoronix Sept 2024](https://www.phoronix.com/news/Haiku-OS-September-2024)).
|
||||
- **SerenityOS** — added basic native Intel graphics ([PR #6277](https://github.com/SerenityOS/serenity/pull/6277)),
|
||||
though only for very old ICH7-class hardware.
|
||||
- **managarm** — native Intel G45 support.
|
||||
|
||||
The catch: **every clean-room non-Linux implementation targets old hardware.** A modern Gen12 "Xe"
|
||||
desktop iGPU is beyond all of them; for the very newest parts only GPL i915 covers the registers. So
|
||||
the wealth of prior art is real but concentrated below Tiger Lake.
|
||||
|
||||
## The practical desktop caveat
|
||||
|
||||
Before any effort estimate is trusted, three hardware realities — the honest reason "Intel is easier"
|
||||
does **not** automatically mean "it'll light up the reader's monitor":
|
||||
|
||||
1. **Muxing / cabling.** On a desktop with a discrete RTX 3060, the monitor is almost certainly
|
||||
plugged into the *card's* outputs, not the motherboard's. An iGPU driver would light a
|
||||
**different, currently-dark** output. To see danos on Intel the reader would have to physically
|
||||
move the cable to a motherboard video port **and** likely enable the iGPU / "IGD Multi-Monitor" in
|
||||
BIOS. Intel-first probably does **not** light the current display without re-cabling.
|
||||
2. **No iGPU at all.** Intel **F-SKU** desktop chips (i5-9400F, i5-12400F, i5-13400F, i7-13700KF, …)
|
||||
ship the graphics **fused off** and cannot be re-enabled. These are extremely common in
|
||||
budget/mid gaming builds paired with an RTX 3060. On an F-SKU (or an X-series HEDT part) the
|
||||
Intel-iGPU path is a **non-starter** regardless of cabling.
|
||||
3. **Generation coverage.** If the CPU *is* a recent non-F part, its iGPU may be Gen12 Xe (Alder/
|
||||
Raptor Lake), beyond libgfxinit's verified set and beyond most non-Linux prior art — leaving GPL
|
||||
i915 (or the TGL-class PRM, which covers Alder/Raptor display IP) as the only reference.
|
||||
|
||||
A cleaner path for *learning* without the hardware lottery: an older bare-metal Intel box (Haswell/
|
||||
Skylake NUC or laptop) whose panel is natively on the iGPU. Note QEMU does **not** emulate an Intel
|
||||
iGPU display engine, so a VM cannot exercise a real Intel modeset path — virtio-gpu (already working)
|
||||
is the VM answer.
|
||||
|
||||
## Alternatives, and the honest Intel-vs-NVIDIA verdict
|
||||
|
||||
| Option | What you get | The tradeoff |
|
||||
|---|---|---|
|
||||
| **Stay on GOP** (working today) | Native-res scanout, zero GPU code/firmware/maintenance | Resolution frozen at ExitBootServices; no runtime mode change, no hardware vsync, no multihead |
|
||||
| **Intel iGPU, reuse-GOP** | EDID read + plane page-flips on the GOP-set mode | Still bounded to GOP's resolution; but real driver-owned scanout |
|
||||
| **Intel iGPU, full modeset** (this doc) | Runtime modeset, vsync, multihead, from public docs | Tier 2–3 effort; DP link training; per-gen churn; **needs a cable-attached, documented iGPU** |
|
||||
| **Native NVIDIA GA106 direct** ([nvidia-gpus.md](nvidia-gpus.md)) | Same, on the RTX 3060 the monitor is actually plugged into | **Tier 4**; GPL-only reference; DMA channel modeset; de-emphasised legacy path |
|
||||
| **GA106 via GSP/OGKM** | Also unlocks 3D later | Tier 5; unstable version-pinned firmware ABI |
|
||||
|
||||
**The verdict for *this reader* (RTX 3060 box):** For pure "see danos on my screen," **NVIDIA-direct
|
||||
is paradoxically the more relevant path**, because the monitor is already cabled to the 3060 and GOP
|
||||
already drives it — a native NVIDIA driver reprograms *that* live display. An Intel driver, however
|
||||
much easier to *write*, likely lights a dark motherboard port the reader isn't looking at, or hits an
|
||||
F-SKU with no iGPU.
|
||||
|
||||
**The verdict for *learning display bring-up*:** **Intel wins decisively.** Public register PRMs, four
|
||||
independent open reference drivers, an MIT precedent (Haiku), a compact formally-analysed blueprint
|
||||
(libgfxinit), no signed-firmware wall, no VRAM/BAR memory manager, and a legitimate permissive
|
||||
clean-room path. It reaches "first pixel" far faster than the NVIDIA native path — *on hardware that
|
||||
actually has a cable-attached, documented Intel iGPU.* Those two goals — "run on my machine" and
|
||||
"learn the craft" — point at different silicon, and that is the honest bottom line.
|
||||
|
||||
## "First light" milestones — a danos `.scanout` service
|
||||
|
||||
Framed as a danos `.scanout` service (like the virtio-gpu and proposed NVIDIA ones), inheriting the
|
||||
GOP-initialized display — no firmware, no cold POST:
|
||||
|
||||
1. **PCI/BAR bring-up** — enumerate the iGPU, map its MMIO BAR (`GTTMMADR` + register block) and the
|
||||
aperture BAR via danos MMIO grants; confirm the display engine is GOP-live.
|
||||
2. **EDID** — implement GMBUS DDC (`0x50`) and DP AUX; read + parse the panel EDID and DPCD caps.
|
||||
*(Smallest self-contained, gen-invariant milestone — a good first commit.)*
|
||||
3. **First pixel = reprogram, don't re-modeset** — with GOP's mode and GGTT mapping inherited,
|
||||
reprogram the running plane (`PRI_CTL`/`PRI_STRIDE`/`PRI_SURF`, linear, 32bpp XRGB) to point at a
|
||||
danos-owned system-RAM buffer; prove a page-flip via the `PRI_SURF` arming write on the *current*
|
||||
mode before changing timings. This defers the entire DPLL/DDI/transcoder/link-training surface —
|
||||
the hardest, most gen-specific ~70% of the work.
|
||||
4. **GGTT ownership** — write your own GGTT PTEs (via an MMIO grant to `GTTMMADR`) pointing at
|
||||
compositor-owned pages, for double-buffered damage-driven present.
|
||||
5. **Wire into the compositor `.scanout` backend** (`attach_scanout`); add vsync via the display
|
||||
vblank interrupt (IRQ-as-IPC).
|
||||
6. **Full mode-set** (the hard, gen-specific step) — for one chosen generation (Haswell/Broadwell or
|
||||
Skylake): WRPLL/DPLL programming, the ~20-step DDI/transcoder/pipe enable sequence, panel power
|
||||
sequencing for eDP (`PP_CONTROL`/`PP_ON_DELAYS`/`PP_OFF_DELAYS` — a common black-screen pitfall).
|
||||
7. **DisplayPort link training** — only if the panel is DP and GOP's link can't be reused; the
|
||||
software CR/EQ state machine over AUX. TMDS/HDMI avoids it; a live eDP panel avoids most of it.
|
||||
8. **Multihead**, then optionally a second generation once one is solid.
|
||||
|
||||
Keep the GOP backend as the fallback the whole way — a stall at any step still leaves danos with a
|
||||
working display, exactly the resilience v2 already provides via re-attach.
|
||||
|
||||
## Reading list
|
||||
|
||||
**Native reference — coreboot libgfxinit (GPL-2.0-or-later, SPARK/Ada):**
|
||||
- `common/hw-gfx-gma.adb` — `Enable_Output`, the end-to-end modeset state machine.
|
||||
- `common/hw-gfx-dp_training.adb` — the complete generic DP link-training CR/EQ loops.
|
||||
- `common/hw-gfx-gma-pipe_setup.adb` — plane/pipe/scaler + `DSPSURF`/`DSPSTRIDE`/`DSPCNTR` scanout.
|
||||
- `common/hw-gfx-gma-transcoder.adb` — timing generator; `common/hw-gfx-edid.adb`,
|
||||
`hw-gfx-gma-i2c.adb`, `hw-gfx-dp_aux_ch.adb` — EDID/DDC/AUX; `hw-gfx-gma-registers.ads` — offsets.
|
||||
- `common/haswell*/`, `skylake/`, `tigerlake/` — the per-gen PLL/PHY/buffer-translation backends.
|
||||
|
||||
**Vendor register specs — Intel OSRC PRMs:**
|
||||
- [Broadwell Vol 11: Display](https://cdrdv2-public.intel.com/690828/intel-gfx-prm-osrc-bdw-vol-11-display.pdf)
|
||||
(CC-BY-ND) — the recommended Haswell/Broadwell-class enable sequences, plane, panel fitter.
|
||||
- [Tiger Lake Vol 12: Display Engine](https://cdrdv2-public.intel.com/705833/intel-gfx-prm-osrc-tgl-vol-12-display-engine.pdf)
|
||||
(code samples 0BSD) — the most complete modern reference incl. PLL/voltage-swing value tables.
|
||||
- [DG2/Arc Vol 12: Display Engine](https://www.x.org/docs/intel/ACM/intel-gfx-prm-osrc-acm-vol12-displayengine.pdf)
|
||||
— the newest public display PRM (Gen12.5, 2022).
|
||||
- [Igalia CC-BY-ND archive](https://github.com/Igalia/intel-osrc-gfx-prm) (Gen4–Gen9.5) and the
|
||||
[kiwitree mirror](https://kiwitree.net/~lina/intel-gfx-docs/prm/) — stable mirrors.
|
||||
|
||||
**GPL reference-of-last-resort — Linux i915 display:**
|
||||
- `intel_gmbus.c`, `intel_dp_aux.c` — the concrete EDID/DDC and DP-AUX register sequences.
|
||||
- `intel_ddi.c` / `intel_ddi_buf_trans.c`, `intel_cdclk.c`, `intel_dpll_mgr.c` — DDI/CDCLK/PLL;
|
||||
`i9xx_plane.c`, `intel_crtc.c` — plane/pipe; `intel_dp.c` — link training. Huge and modular; a
|
||||
reference to confirm undocumented quirks, not a template.
|
||||
|
||||
**Permissive prior art — Haiku `intel_extreme` (MIT):**
|
||||
- [`src/add-ons/kernel/drivers/graphics/intel_extreme/`](https://github.com/haiku/haiku/tree/master/src/add-ons/kernel/drivers/graphics/intel_extreme/)
|
||||
— a second independent modeset-only driver; MIT, so structurally readable for a permissive danos.
|
||||
- [generations.html](https://www.haiku-os.org/docs/develop/drivers/intel_extreme/generations.html)
|
||||
— the best plain-English per-generation fault-line map.
|
||||
|
||||
## Open questions (unresolved by the survey)
|
||||
|
||||
- **Does the target machine have a usable, cable-attached iGPU at all?** F-SKU check, CPU generation,
|
||||
and monitor cabling must be resolved before any effort estimate is trusted (see
|
||||
[practical caveat](#the-practical-desktop-caveat)).
|
||||
- **Does danos even need native mode-*setting*, or only plane/scanout control on the GOP-set mode?**
|
||||
If runtime mode changes aren't required, the driver collapses to EDID + plane page-flips, dropping
|
||||
the DPLL/DDI/link-training ~70% of the work.
|
||||
- **GGTT vs raw physical:** confirm from the exact target-gen PRM that `PLANE_SURF` is interpreted as
|
||||
a GGTT graphics address (well-established, but per-gen confirmation advisable), and the PTE size /
|
||||
`GTTMMADR` / aperture layout for writing GGTT entries.
|
||||
- **Reuse the firmware/GOP GGTT + framebuffer, or install your own GGTT entries?** The latter (needed
|
||||
for double-buffering) means writing GGTT PTEs from the userspace driver via an MMIO grant.
|
||||
- **eDP panel power sequencing** (`PP_*`, T1–T12 delays) — not covered in this pass and a common
|
||||
black-screen source.
|
||||
- **IOMMU interaction** — whether the display's GGTT-mediated DMA needs IOMMU passthrough for the
|
||||
framebuffer pages under danos's M16 IOMMU, or sits before the IOMMU on the target platform.
|
||||
- **DP link-training / AUX robustness and per-generation register drift** are the dominant *risks* —
|
||||
not documentation scarcity.
|
||||
- **Exact Haswell/BDW MMIO offsets** (commonly cited: GMBUS ~`0xC5100`, `DDI_AUX_CTL_A` ~`0x64010`,
|
||||
`DDI_BUF_CTL_A` ~`0x64000`, `DP_TP_CTL_A` ~`0x64040`) were not extracted verbatim from the PRM —
|
||||
confirm against `i915_reg.h` before coding.
|
||||
|
||||
---
|
||||
|
||||
*Research snapshot; verify against current libgfxinit / i915 source and the specific target
|
||||
generation's PRM before building. Intel's public-PRM coverage and the muxing/F-SKU realities of a
|
||||
given machine both change what is actually achievable.*
|
||||
@@ -0,0 +1,246 @@
|
||||
# Native NVIDIA GPU support — feasibility and roadmap
|
||||
|
||||
**Status: research snapshot, not implemented.** This records what a *native* display driver for a
|
||||
real discrete NVIDIA GPU — specifically an **RTX 3060 (Ampere GA106)** — would take, and how it
|
||||
would slot into danos's pluggable scanout architecture. It is a survey of primary sources
|
||||
(NVIDIA's [open-gpu-kernel-modules](https://github.com/NVIDIA/open-gpu-kernel-modules), the Linux
|
||||
[nouveau/nvkm](https://github.com/torvalds/linux/tree/master/drivers/gpu/drm/nouveau) driver,
|
||||
NVIDIA's [open-gpu-doc](https://nvidia.github.io/open-gpu-doc/), and
|
||||
[linux-firmware](https://github.com/NVIDIA/linux-firmware)), not an implementation. The NVIDIA
|
||||
driver landscape moves quickly (GSP defaults, firmware ABIs); treat specifics as a mid-decade
|
||||
snapshot and re-verify against current source before building.
|
||||
|
||||
Read [display.md](display.md) and [display-v2.md](display-v2.md) first — this doc assumes the
|
||||
v2 model where scanout is a **pluggable backend** and a native driver is just another `.scanout`
|
||||
service (like the virtio-gpu one), announcing to the compositor over `attach_scanout`.
|
||||
|
||||
## TL;DR
|
||||
|
||||
- A **minimal display-only driver** (EDID + mode-set + framebuffer scanout, **no** 3D/compute)
|
||||
for the RTX 3060 **can and should avoid the GSP entirely**. nouveau has a register-level,
|
||||
CPU-driven display path for Ampere (`nvkm/engine/disp/ga102.c`) that lights up GA106 with no
|
||||
external firmware; the signed-firmware wall gates the **compute/graphics** engines (PGRAPH),
|
||||
**not** the display controller. "GSP is mandatory on Ampere" is true only for NVIDIA's own
|
||||
RM-object route.
|
||||
- **danos's UEFI GOP boot is the single biggest thing in its favour.** The VBIOS/GOP has already
|
||||
run devinit and brought up the display PLLs, so a driver attaches to a **live, initialized**
|
||||
GA106 — no firmware load, no cold-boot POST, no devinit interpreter. You reprogram a running
|
||||
display rather than bring one up from cold.
|
||||
- It is still a **hard, multi-week-to-months expert effort** (effort tier ≈ 4/5) dominated by
|
||||
NVDisplay channel-DMA programming, SOR/head routing, DisplayPort AUX + link training, and the
|
||||
display supervisor handshake. The GSP/RM route is tier 5 (near-infeasible solo).
|
||||
- The **licensing tension is counterintuitive**: the permissively-licensed reference (NVIDIA
|
||||
open-gpu-kernel-modules, MIT/GPLv2) is the **hard GSP path**; the register-level display code
|
||||
you actually want lives in **GPL nouveau**. See [Licensing](#licensing).
|
||||
- The **window is closing**: GA10x (Ampere) is the *last* NVIDIA family with a register-level
|
||||
display path — Ada (RTX 40) deleted its non-GSP display HAL. Targeting Ampere specifically
|
||||
matters.
|
||||
- **Recommendation:** for *this card*, GOP already gives native-resolution scanout with zero GPU
|
||||
code and zero maintenance. A native driver buys only runtime mode changes, hardware
|
||||
vsync/vblank, and multihead. It is justified if that runtime control is a danos goal, or to
|
||||
*learn the craft* — for which an Intel iGPU or a pre-Turing NVIDIA card reaches "first pixel"
|
||||
far faster.
|
||||
|
||||
## The GSP wall, and why display sits on the near side of it
|
||||
|
||||
On Turing and later, NVIDIA split its driver's Resource Manager into a host **CPU-RM** and a
|
||||
**GSP-RM** running on an on-die RISC-V core ("Peregrine"), talking over RPC
|
||||
([LWN 953144](https://lwn.net/Articles/953144/)). The GSP is a *full resource manager*, not a
|
||||
display coprocessor — there is no "display-only" GSP image and no small display RPC subset. Its
|
||||
boot chain is entirely signed and mandatory: a VBIOS-resident **FWSEC-FRTS** app carves a
|
||||
write-protected region (WPR2), a signed **Booter** on the SEC2 falcon loads the GSP bootloader,
|
||||
and that loads **GSP-RM** inside WPR. The firmware ships pre-computed signatures and the driver
|
||||
picks one by an on-chip fuse-version register — **you cannot self-sign**, and there is **no stable
|
||||
firmware ABI** (it is revised every driver release; nouveau and the Rust nova-core driver each pin
|
||||
exactly one version). A GSP driver is a permanent maintenance liability, not a one-time build
|
||||
([LWN 1037379](https://lwn.net/Articles/1037379/),
|
||||
[nova-core cover letter](https://lore.freedesktop.org/nouveau/20250826-nova_firmware-v2-7-93566252fe3a@nvidia.com/T/)).
|
||||
|
||||
**But display doesn't need any of that on Ampere.** `nvkm/engine/disp/ga102.c` dual-dispatches:
|
||||
|
||||
```
|
||||
if (nvkm_gsp_rm(device->gsp)) return r535_disp_new(&ga102_disp, ...); // GSP RPC path
|
||||
return nvkm_disp_new_(&ga102_disp, ...); // direct register path
|
||||
```
|
||||
|
||||
Both branches use the same `ga102_disp` HAL and the same `GA102_DISP_*` class IDs; GSP merely
|
||||
swaps register programming for RPC. GA106 (chipset `0x176`) is wired to `ga102_disp_new` in the
|
||||
device table, identical to GA102/103/104/107. Ampere lit up displays via the **direct** path in
|
||||
Linux 5.11/5.17 — two years before GSP-RM landed (6.7, 2023)
|
||||
([ga102.c](https://raw.githubusercontent.com/torvalds/linux/master/drivers/gpu/drm/nouveau/nvkm/engine/disp/ga102.c),
|
||||
[Phoronix GA106](https://www.phoronix.com/news/Nouveau-NVIDIA-GA106)).
|
||||
|
||||
**Caveat — this is now the legacy path.** As of Linux 6.18, nouveau defaults to GSP on
|
||||
Turing/Ampere; the direct path is a retained, forceable fallback (`nouveau.config=NvGspRm=0`, and
|
||||
automatic when GSP firmware is absent). It is stable and proven, but NVIDIA and nova-core are
|
||||
moving to GSP-only, and **Ada already deleted its non-GSP display HAL**. GA10x is the last family
|
||||
that keeps a register-level display path.
|
||||
|
||||
## What "direct" actually entails
|
||||
|
||||
"Direct" is not "plain register pokes." Only SOR / PLL / DP-link / clock setup is bare MMIO. The
|
||||
**mode-set and scanout themselves flow through the NVDisplay channels — a DMA pushbuffer**:
|
||||
|
||||
- Display classes for Ampere (the C670 family): core `GA102_DISP_CORE_CHANNEL_DMA` (`0xc67d`),
|
||||
window `0xc67e`, window-immediate `0xc67b`, cursor `0xc67a` (headers `clc67d.h` / `clc67e.h` /
|
||||
`clc67a.h` in [open-gpu-doc `classes/display/`](https://github.com/NVIDIA/open-gpu-doc/tree/master/classes/display)).
|
||||
- The core channel needs **instance memory, a RAMHT, DMA objects, and a channel user-MMIO
|
||||
region** ([disp/chan.c](https://raw.githubusercontent.com/torvalds/linux/master/drivers/gpu/drm/nouveau/nvkm/engine/disp/chan.c)).
|
||||
The register-level "plumbing" to allocate/kick a channel is in NVIDIA's GA102 display register
|
||||
manual: `NV_PDISP_FE_CHNCTL_CORE/WIN/CURS`, `NV_PDISP_FE_PBBASE/PBBASEHI`
|
||||
([dev_display_withoffset.ref.txt](https://raw.githubusercontent.com/NVIDIA/open-gpu-doc/master/manuals/ampere/ga102/dev_display_withoffset.ref.txt)).
|
||||
- **Mode-set is a method stream** on the core channel: `HEAD_SET_RASTER_*`,
|
||||
`HEAD_SET_PIXEL_CLOCK_FREQUENCY`, `HEAD_SET_CONTROL_OUTPUT_RESOURCE`, `SOR_SET_CONTROL`
|
||||
(protocol select), viewport/scaler, then `UPDATE`. The window channel points at the scanout
|
||||
surface (`SET_CONTEXT_DMA_ISO`, `SET_STORAGE`, `SET_OFFSET`).
|
||||
- After `UPDATE` you must complete the display **supervisor** interrupt handshake (SV1/SV2/SV3).
|
||||
|
||||
**EDID and DisplayPort are a separate subdev you must port.** open-gpu-doc documents *none* of
|
||||
EDID/DDC/AUX. On the direct path you read EDID in-driver via nouveau's `nvkm/subdev/i2c`: bit-bang
|
||||
**DDC/I²C at address `0x50`** (E-DDC `0x30`) for TMDS/HDMI, or native **DP AUX** in `i2c/aux.c`
|
||||
for DisplayPort. DisplayPort **link training** (the `dp.c` `train_cr` / `train_eq` state machine
|
||||
over AUX — clock recovery, lane/rate, voltage-swing/pre-emphasis) is the single hardest and most
|
||||
fragile piece; a DVI/HDMI (TMDS) panel avoids it entirely.
|
||||
|
||||
## The memory floor (smaller than you'd fear)
|
||||
|
||||
Neither route hands you a framebuffer allocator — even GSP-RM does not manage the scanout
|
||||
framebuffer; the driver owns VRAM and merely tells GSP where its page directory is. But
|
||||
display-only is a small fraction of a full GEM/TTM stack:
|
||||
|
||||
- **Pitch-linear (untiled) scanout is allowed** on nv50→Ampere — the window's storage method has a
|
||||
`PITCH` layout mode, so you skip block-linear tiling math
|
||||
([wndwc37e.c](https://raw.githubusercontent.com/torvalds/linux/master/drivers/gpu/drm/nouveau/dispnv50/wndwc37e.c)).
|
||||
- The window references its surface through a simple **display context-DMA**
|
||||
(`SET_CONTEXT_DMA_ISO` + a 256-byte-granular `SET_OFFSET = addr>>8`) — a base/limit descriptor,
|
||||
**not** the GPU's 5-level compute page tables. **No full GPU VMM is needed** for scanout.
|
||||
- The surface must live in **VRAM** in practice (nouveau always pins scanout to VRAM). *Open
|
||||
question:* whether GA10x can scan out from a system-memory (GART) surface via a sysmem-target
|
||||
ctxdma — which would let danos skip a VRAM allocator. No source forbids it; nouveau never does
|
||||
it (confidence: medium).
|
||||
- **CPU access** to the framebuffer for compositing goes through **BAR1** (a VRAM aperture); BAR0
|
||||
is the 16 MB register window. BAR1 can be smaller than 12 GB of VRAM unless Resizable BAR maps
|
||||
it all.
|
||||
|
||||
**Net:** you need (1) a contiguous aligned VRAM allocator (256-byte base, pitch a multiple of
|
||||
64 bytes — confirm against the Ampere display refs), (2) a little instmem for the channel
|
||||
pushbuffers + iso ctxdma, (3) a BAR1 CPU mapping. You do **not** need the 5-level VMM, GEM/TTM
|
||||
eviction, or tiling.
|
||||
|
||||
## Licensing
|
||||
|
||||
The tension is the opposite of convenient:
|
||||
|
||||
- **NVIDIA open-gpu-kernel-modules is dual MIT/GPLv2** — usable under MIT, no copyleft on your
|
||||
other code — **but its display logic is the GSP/RM-object route.** Its class headers
|
||||
(`cl0073.h`, `cl2080.h`, `ctrl0073*.h`) are useful, permissive references.
|
||||
- **nouveau is GPLv2**, and the **register-level display sequences you actually want live in
|
||||
nouveau**, not in the MIT code. So the *easy technical path is the GPL-licensed one.* Reading
|
||||
GPL nouveau and reimplementing it in Zig is a derivative-work risk proportional to how closely
|
||||
your code tracks its structure/constants.
|
||||
|
||||
Options: **(a)** accept that the danos NVIDIA display driver is a **GPL component**. danos's
|
||||
userspace-driver-over-IPC model (a driver is a separate process behind a defined protocol, not
|
||||
linked into the kernel) is about the cleanest possible GPL boundary, so the GPL would be contained
|
||||
to that one binary and the rest of danos could keep its own license — but this is a
|
||||
licensing-boundary judgement that wants real diligence, not a settled fact. **(b)** clean-room
|
||||
from *specification* rather than *code*: [envytools](https://envytools.readthedocs.io) + NVIDIA's
|
||||
open-gpu-doc register manuals + the MIT OGKM class headers, treating nouveau as
|
||||
documentation-of-last-resort.
|
||||
|
||||
**Firmware licensing is moot for the direct path** (no firmware is loaded). For completeness: the
|
||||
GSP blobs are marked redistributable under `LICENCE.nvidia`, which permits use by **any
|
||||
OSI-approved open-source OS** (not just Linux), on NVIDIA GPUs, **unmodified**, with **no
|
||||
reverse-engineering of the firmware binary**. The one gate — is danos released under an OSI
|
||||
license? — is only reached on the GSP route, which this doc recommends against for this card.
|
||||
|
||||
## Prior art
|
||||
|
||||
**No one has built a from-scratch native NVIDIA driver outside Linux.** FreeBSD ships
|
||||
`nvidia-drm-kmod`, a *port of NVIDIA's own closed `nvidia-drm.ko`* loading the GSP blob (its old
|
||||
nouveau port was removed). Haiku's NVIDIA support is likewise a *port of OGKM* (GSP, Turing+, very
|
||||
alpha). OpenBSD / DragonFly have neither. Every non-Linux OS that supports modern NVIDIA chose to
|
||||
**wrap NVIDIA's GSP stack** rather than write a native driver. A danos direct-register driver
|
||||
would have exactly one reference implementation — GPL nouveau — and no non-Linux precedent.
|
||||
|
||||
## Alternatives
|
||||
|
||||
| Option | What you get | The tradeoff |
|
||||
|---|---|---|
|
||||
| **Stay on GOP** (working today) | Native-res scanout, zero GPU code/firmware/maintenance | Resolution frozen at ExitBootServices; **no runtime mode change, no hardware vsync, no multihead** |
|
||||
| **Pre-Turing NVIDIA** (Kepler / early Maxwell) | Direct EVO/disp-core + CRTC/PLL modeset, **no signed firmware, no coprocessor**; mature nouveau reference | Older display class; not this card; only reclocking is firmware-gated |
|
||||
| **Intel iGPU** | **Publicly documented** register interfaces (Intel PRMs); no coprocessor mediating modeset | i915 is huge + generation-specific; write one generation from the PRM |
|
||||
| **Native GA106 direct** (this doc) | Runtime modeset, vsync, multihead on the actual card | Tier-4 effort; GPL reference; DP link training; legacy/de-emphasized path |
|
||||
| **GA106 via GSP/OGKM** | Also unlocks 3D / reclocking later | Tier-5; ~14k-line ante; unstable version-pinned ABI; unprecedented outside Linux |
|
||||
|
||||
## "First light" milestones (direct path, inheriting GOP state)
|
||||
|
||||
Framed as a danos `.scanout` service (like the virtio-gpu driver), taking the direct register path
|
||||
and inheriting the GOP-initialized display — no signed firmware, no devinit, no GSP:
|
||||
|
||||
1. **PCI/BAR bring-up** — enumerate GA106 (`0x176`), map **BAR0** (registers) and **BAR1** (VRAM
|
||||
aperture) via danos MMIO grants; confirm the display engine is GOP-live.
|
||||
2. **VRAM + instmem allocator** — contiguous aligned VRAM for the scanout surface (256-byte base)
|
||||
+ small instmem for pushbuffers / RAMHT / iso ctxdma. No VMM, no TTM.
|
||||
3. **EDID** — port `nvkm/subdev/i2c` DDC (`0x50`) + DP-AUX (`aux.c`); read + parse the panel EDID.
|
||||
4. **Core channel up** — allocate the `0xc67d` core channel as a DMA pushbuffer; stand up the
|
||||
SV1/SV2/SV3 supervisor-interrupt handshake.
|
||||
5. **First pixel = reprogram, don't re-POST** — bind a window (`0xc67e`) at the existing WC
|
||||
framebuffer via `SET_CONTEXT_DMA_ISO` + `SET_OFFSET`, pitch-linear, `UPDATE`; prove you can
|
||||
drive the *current* GOP mode from your own channel before changing anything.
|
||||
6. **Modeset** — push raster timings on a head, route head→SOR→connector, program the pixel-clock
|
||||
PLL, switch to an EDID mode (needs the `clc67d/e` method opcodes from the OGKM headers + the
|
||||
supervisor timing from nouveau `head.c`).
|
||||
7. **DisplayPort link training** — only if the panel is DP and GOP's link can't be reused; the
|
||||
`dp.c` `train_cr`/`train_eq` state machine. TMDS/HDMI is far simpler.
|
||||
8. **Wire into the compositor `.scanout` backend** (`attach_scanout`), add vsync via the display
|
||||
interrupt, then multihead.
|
||||
|
||||
Keep the GOP backend as the fallback the whole way — a stall at any step still leaves danos with a
|
||||
working display (exactly the resilience v2 already provides via re-attach).
|
||||
|
||||
## Reading list
|
||||
|
||||
**Direct path — nouveau (GPLv2):**
|
||||
- `nvkm/engine/disp/ga102.c` — the GA10x display HAL + the GSP/non-GSP dispatch.
|
||||
- `nvkm/engine/disp/{head.c, ior.c, dp.c, hdmi.c, chan.c}` — head/SOR routing, DP AUX + link
|
||||
training, channel-DMA plumbing.
|
||||
- `dispnv50/{corec37d.c, corec57d.c, wndwc37e.c, wndwc57e.c, wndwc67e.c, headc37d.c, cursc37a.c}`.
|
||||
- `nvkm/subdev/i2c` (DDC + `aux.c`) for EDID; `nvkm/subdev/bios/init.c` + `devinit/` **only** if
|
||||
you ever have to re-POST (danos's GOP handoff means you shouldn't).
|
||||
|
||||
**Object model / GSP path — NVIDIA OGKM (MIT/GPLv2):** class headers `cl0073.h`, `cl2080.h`,
|
||||
`ctrl0073system.h`, `ctrl0073specific.h`; `src/nvidia/` for RM control sequences.
|
||||
`nvidia-modeset.ko` (NVKMS) is a *policy* layer over RM and can be bypassed entirely.
|
||||
[nova-core](https://lore.freedesktop.org/nouveau/) (Rust) is the forward-looking reference for GSP
|
||||
boot mechanics (falcon signing, queue rings, RPC).
|
||||
|
||||
**Register / method specs — NVIDIA open-gpu-doc:**
|
||||
- [`classes/display/README.txt`](https://raw.githubusercontent.com/NVIDIA/open-gpu-doc/master/classes/display/README.txt)
|
||||
— the channel model + class-to-GPU map (read first).
|
||||
- [`classes/display/clc67d.h`](https://raw.githubusercontent.com/NVIDIA/open-gpu-doc/master/classes/display/clc67d.h)
|
||||
+ `clc67e.h` / `clc67a.h` — the Ampere core/window/cursor mode-set method vocabulary.
|
||||
- [`manuals/ampere/ga102/dev_display_withoffset.ref.txt`](https://raw.githubusercontent.com/NVIDIA/open-gpu-doc/master/manuals/ampere/ga102/dev_display_withoffset.ref.txt)
|
||||
— `NV_PDISP_FE_*` channel/pushbuffer registers + SOR.
|
||||
- [`DCB`](https://github.com/NVIDIA/open-gpu-doc/tree/master/DCB) — connector→output-resource
|
||||
routing; [`Devinit`](https://github.com/NVIDIA/open-gpu-doc/tree/master/Devinit) +
|
||||
[`BIOS-Information-Table`](https://github.com/NVIDIA/open-gpu-doc/tree/master/BIOS-Information-Table)
|
||||
— VBIOS parsing (bring-up reference; not needed if inheriting GOP).
|
||||
- The 632 KB Volta [`dev_display.ref`](https://download.nvidia.com/open-gpu-doc/Display-Ref-Manuals/1/gv100/dev_display.ref)
|
||||
is the best shot at SOR-DP/AUX register detail the smaller Ampere file omits.
|
||||
|
||||
## Open questions (unresolved by the survey)
|
||||
|
||||
Each needs a direct read of the named nouveau file or experimentation on the actual card:
|
||||
|
||||
- Exact GA106 register/method offsets and PADLINK→SOR→connector wiring (can vary by board vendor).
|
||||
- Whether *any* PLL/devinit re-run is unavoidable vs. fully inherited from GOP.
|
||||
- Whether DisplayPort needs full retraining on takeover, or the GOP-established link can be reused.
|
||||
- The precise SV1/SV2/SV3 supervisor sequence.
|
||||
- Whether a system-memory-target scanout ctxdma could eliminate the VRAM allocator.
|
||||
- The exact `clc67d.h`/`clc67e.h` method opcode numbers (not captured verbatim in the survey).
|
||||
|
||||
---
|
||||
|
||||
*Research snapshot; verify against current nouveau / open-gpu-kernel-modules source before
|
||||
building — NVIDIA's GSP defaults and firmware ABIs change per release.*
|
||||
@@ -0,0 +1,9 @@
|
||||
# OS Developer Guide
|
||||
|
||||
This document is for those who need to understand the architectural decisions behind the OS.
|
||||
|
||||
## Written in Zig?
|
||||
|
||||
The os was initially written in zig because it has excellent support for EFI. With zig, we could forgo using a third party bootloader, reducing the time to boot up the kernel. Following the "Zen of Zig", helped to produce the most readable codebase for an operating system ever created. So those, new to OS development could quickly get up to speed.
|
||||
|
||||
|
||||
@@ -0,0 +1,70 @@
|
||||
# The release ISO — flashable boot media
|
||||
|
||||
`zig build release-x86-64` produces **`zig-out/danos-x86-64.iso`**, the file you
|
||||
hand to someone who wants to try danos on a real machine: point
|
||||
[balenaEtcher](https://etcher.balena.io) (or Raspberry Pi Imager, or plain `dd`)
|
||||
at it, flash a USB stick, and boot the stick. The same file also burns to
|
||||
optical media. `zig build check-iso-image` validates it without booting.
|
||||
|
||||
```
|
||||
zig build release-x86-64
|
||||
# Etcher: select danos-x86-64.iso → select the stick → Flash
|
||||
# or: sudo dd if=zig-out/danos-x86-64.iso of=/dev/rdiskN bs=4m (macOS; triple-check N)
|
||||
```
|
||||
|
||||
## Why an ISO when danos-usb.img already boots
|
||||
|
||||
`danos-usb.img` is a raw FAT32 **superfloppy** — a filesystem starting at
|
||||
sector 0, no partition table. UEFI firmware accepts that from a USB stick (it
|
||||
probes whole-disk FAT before giving up), which is why `dd`-ing the .img works
|
||||
and why QEMU and the test harness boot it directly. But it is a
|
||||
developer-shaped artifact: flashing apps expect an ISO, and a superfloppy
|
||||
can't be burned to a CD/DVD or carry a partition table for pickier firmware.
|
||||
|
||||
The ISO wraps that same FAT image — bit-identical, built by the same
|
||||
`tools/make-fat-image.py` — in a container that boots everywhere release media
|
||||
gets consumed. One payload, two images: the .img stays the raw volume the QEMU
|
||||
harness mounts and boots, the .iso is what leaves the building.
|
||||
|
||||
## How a hybrid ISO boots twice
|
||||
|
||||
The trick (the same one Linux distribution ISOs use, usually via `xorriso
|
||||
-isohybrid…`) is that ISO9660 reserves its first 32 KiB as a **system area** it
|
||||
never touches — exactly where an MBR lives on a disk. So one file can carry two
|
||||
tables of contents, both pointing at the same embedded FAT image:
|
||||
|
||||
* **Flashed to USB (Etcher, dd):** firmware sees a disk whose sector 0 is an
|
||||
MBR with one partition of type `0xEF` (EFI System Partition) covering the
|
||||
embedded FAT image. It mounts that ESP and runs `\EFI\BOOT\BOOTX64.efi` —
|
||||
the standard removable-media path ([efi.md](efi.md)).
|
||||
* **Burned to optical media:** firmware reads the ISO9660 volume descriptors
|
||||
at sector 16 and finds an **El Torito** boot record. Its catalog has one
|
||||
entry, platform ID `0xEF` (EFI), whose start LBA is — again — the embedded
|
||||
FAT image. The firmware exposes that image as a virtual disk and runs the
|
||||
same `BOOTX64.efi` off it.
|
||||
|
||||
Neither path involves the legacy BIOS boot-sector machinery: danos is
|
||||
UEFI-only ([system-requirements.md](system-requirements.md)), so the MBR holds
|
||||
no boot code, just the partition entry, and the El Torito entry is EFI-class,
|
||||
not floppy emulation.
|
||||
|
||||
One El Torito wrinkle: the catalog's sector-count field is 16-bit (units of
|
||||
512 bytes), so it can name at most 32 MiB — less than the 64 MiB FAT image.
|
||||
That is fine in practice: firmware sizes the FAT filesystem from its own BPB,
|
||||
and the boot files sit in the first few MiB of the image (clusters are
|
||||
allocated from the front) either way. The USB path has no such cap.
|
||||
|
||||
## The builder
|
||||
|
||||
`tools/make-iso-image.py` follows the house rule of
|
||||
[make-fat-image.py](../tools/make-fat-image.py): pure Python 3 standard
|
||||
library, no external tools (no xorriso, mkisofs, or isohybrid), with a
|
||||
`--verify` mode the `check-iso-image` step runs — it checks that the MBR
|
||||
partition and the El Torito catalog agree on where the FAT image lives and
|
||||
that a FAT32 boot sector is actually there. Every timestamp field in the ISO
|
||||
is zeroed, so the build is reproducible byte-for-byte.
|
||||
|
||||
The ISO9660 filesystem around the boot machinery is minimal but real: a root
|
||||
directory listing `BOOT.CAT` (the catalog) and `EFI.IMG` (the FAT image), so
|
||||
`file`, mount tools, and archive browsers can open the ISO and see what's in
|
||||
it.
|
||||
@@ -0,0 +1,290 @@
|
||||
# Threading — build plan (`runtime.Thread` over a private thread ABI)
|
||||
|
||||
The ordered, checkpointable build-out for [threading.md](threading.md). Each milestone
|
||||
lands on its own and ends in a **verifiable gate** — shaped for a `/loop` run, like
|
||||
[display-v2-plan.md](display-v2-plan.md). Read threading.md first for the *why*.
|
||||
|
||||
## Locked decisions (do not relitigate)
|
||||
|
||||
- **`runtime.Thread` mirrors `std.Thread`'s API; the implementation is danos-native.**
|
||||
Not literal `std.Thread` — that would break the [private ABI](syscall.md).
|
||||
- **Threads are a narrow, per-binary opt-in.** Default concurrency stays process + IPC
|
||||
([resilience.md](resilience.md)); only a service that asks is built
|
||||
`single_threaded = false`.
|
||||
- **Blocking is futex-backed, never spin-backed** — waiters park in the kernel so an
|
||||
idle core still halts ([halting.md](halting.md)).
|
||||
- **New syscalls are private**: extend [abi.zig](../system/abi.zig) `SystemCall` after
|
||||
`shm_physical = 36` (`thread_spawn = 37`, `thread_exit = 38`, `current_core = 39`,
|
||||
`futex_wait = 40`, `futex_wake = 41`) + a `library/runtime` wrapper; user code never names a number.
|
||||
- **Restart granularity stays the process** — a faulting thread kills its process; the
|
||||
supervisor restarts the process, which respawns its threads.
|
||||
|
||||
## Conventions
|
||||
|
||||
Follow [coding-standards.md](coding-standards.md): spell out non-acronym abbreviations,
|
||||
kebab-case file names, no `Co-Authored-By` trailers. New user binaries go through
|
||||
`addUserBinary` (with the new `threaded` flag where a binary spawns threads) and get
|
||||
packed into the initial-ramdisk; new syscalls extend [abi.zig](../system/abi.zig)
|
||||
`SystemCall` + a `library/runtime` wrapper; test services live beside the code they
|
||||
exercise and register a `ServiceId` if they must be looked up.
|
||||
|
||||
## How to verify along the way
|
||||
|
||||
**Every gate is serial-checkable — no screenshots** (this plan runs unattended). A
|
||||
thread proves it ran by writing to **shared memory** the parent reads back, and proves
|
||||
parallelism by stamping the **core index** it ran on (like the `smp`/`affinity` cases).
|
||||
|
||||
- `zig build test` — host unit tests (closure packing, mutex state machine, futex
|
||||
wrapper encodings).
|
||||
- `python3 test/qemu_test.py <case>` — boots the kernel in QEMU; asserts on serial
|
||||
markers. Thread cases set `smp: true` (real parallelism) and bump `mem` (they boot
|
||||
the process/scheduler stack); each milestone **adds its case to `CASES`** so its gate
|
||||
is runnable.
|
||||
- **Guardrail every milestone:** the concurrency-sensitive existing cases stay green —
|
||||
`smoke`, `sched`, `priority`, `smp`, `affinity`, `process`, `process-kill`,
|
||||
`supervision`, `fault-recovery`, `vfs-client-death`, `ipc`/`ipc-cap`,
|
||||
`display-service`. A threading change that regresses those is rejected.
|
||||
|
||||
## Unattended execution (the loop contract)
|
||||
|
||||
This plan runs to completion **without human input**. Every design choice is already
|
||||
fixed in *Locked decisions*; the checkboxes are the only state. A loop iteration must:
|
||||
|
||||
1. **Resume** at the first milestone that still has an unchecked `- [ ]`. (All earlier
|
||||
milestones are done — do not revisit them.)
|
||||
2. **Work on a branch.** On the first iteration, branch off `main` (e.g. `threading`);
|
||||
never commit threading work to `main`. All work stays local — **do not push**.
|
||||
3. **Implement** every unchecked item in that milestone, including adding its
|
||||
`-Dtest-case` to `CASES` in [test/qemu_test.py](../test/qemu_test.py) (with
|
||||
`smp: true` / a `mem` bump where noted) so the gate is runnable.
|
||||
4. **Run the gate**: `python3 test/qemu_test.py <case>`, then the full **guardrail
|
||||
set**, then `zig build` (clean) and `zig build test` (green).
|
||||
5. **Decide, do not ask:**
|
||||
- **Green** = the milestone's case prints its stated marker(s) and reports `PASS`,
|
||||
the whole guardrail set passes, `zig build` is clean, and host tests are green.
|
||||
→ tick this milestone's boxes **and** its `**Gate:**`-referenced case, `git commit`
|
||||
(`threads(M<n>): <summary>`, no `Co-Authored-By` trailer per
|
||||
[coding-standards.md](coding-standards.md)), and continue to the next milestone in
|
||||
the same iteration if budget remains; otherwise let the loop re-fire.
|
||||
- **Red** = anything above fails. Diagnose from the captured serial log
|
||||
(`zig-out/qemu-test/<case>-failed-serial.log`) and fix in place, then re-run — up to
|
||||
**3 fix attempts** for that gate. A concurrency case that fails then passes on a
|
||||
bare re-run is **flaky, not green**: re-run it **twice more** and treat green only
|
||||
if it passes all; otherwise fix the race (a real threading bug), don't paper over
|
||||
it.
|
||||
6. **A genuinely ambiguous fork is not a stop.** Pick the option most consistent with
|
||||
[threading.md](threading.md)'s *Locked decisions*, note the choice in the commit
|
||||
message, and continue. Do not pause for confirmation on in-scope, reversible work —
|
||||
this plan is that authorization.
|
||||
|
||||
**The only stop conditions:**
|
||||
|
||||
- **Done** — every milestone box is checked (M1–M6), `zig build` clean, whole
|
||||
`thread-*` suite + guardrail green. Update threading.md's status line to "built" (that
|
||||
is M6's own task) and stop.
|
||||
- **Blocked** — a gate is still red after 3 fix attempts, or a step needs something
|
||||
outside the repo (a toolchain change, new hardware, a decision no locked decision
|
||||
covers). Append `> **BLOCKED (M<n>):** <what failed, what was tried, the serial
|
||||
marker missing>` under that milestone, commit the WIP on the branch, and stop. Do not
|
||||
thrash further and do not silently skip the milestone.
|
||||
|
||||
Nothing else warrants stopping — not "should I proceed?", not "is this right?". The
|
||||
checkboxes + git history are the resumable record; the next iteration picks up from the
|
||||
first unchecked box.
|
||||
|
||||
---
|
||||
|
||||
## M1 — Address-space refcount (kernel foundation, no API, no behaviour change) ✅
|
||||
|
||||
The one invariant change threads require, landed and proven **before** anything shares
|
||||
an address space. Today aspace is 1:1 with a task and teardown destroys it on any user
|
||||
task's exit; make destruction happen on the **last** exit.
|
||||
|
||||
- [x] A refcount keyed by the address-space root, held in `scheduler.zig`
|
||||
(`aspace_refs`): `retainAspace` takes a reference in `spawnUserLocked` (on the
|
||||
success path, after the slot + stack are secured), all under the big kernel lock.
|
||||
- [x] Both task-teardown paths ([scheduler.zig](../system/kernel/scheduler.zig):
|
||||
`exitUserLocked` and `destroyTaskLocked`) call `releaseAspace`, which decrements
|
||||
and only `destroyAddressSpace`s at **zero**; an unretained space (hand-built test
|
||||
spaces) is destroyed directly, preserving prior behaviour.
|
||||
- [x] `-Dtest-case=aspace-refcount`: spawn and reap several ring-3 processes in sequence
|
||||
and assert (via test-observable `liveAspaceCount`/`aspaceDestroyCount`) that the
|
||||
live-space count returns to **baseline** and destructions advance by exactly that
|
||||
many — each space destroyed exactly once, no leak, no double-free. (Refcount
|
||||
observables, not raw frame counts, since kernel stacks are still leaked on exit.)
|
||||
|
||||
**Gate (met):** `python3 test/qemu_test.py aspace-refcount` passes
|
||||
(`aspace-refcount: spaces released to baseline ok` → `DANOS-TEST-RESULT: PASS`), and the
|
||||
full guardrail set passes unchanged — 13/13 (`smoke`, `sched`, `priority`, `smp`,
|
||||
`affinity`, `process`, `process-kill`, `supervision`, `fault-recovery`,
|
||||
`vfs-client-death`, `ipc`, `ipc-cap`, `display-service`); default `zig build` clean,
|
||||
`zig build test` green. The reframing is invisible until an aspace is actually shared.
|
||||
|
||||
## M2 — `thread_spawn` + `thread_exit`: a thread runs in the shared address space ✅
|
||||
|
||||
Spawn only — no join yet. Prove a second task executes in the **caller's** address
|
||||
space and exits cleanly.
|
||||
|
||||
- [x] [abi.zig](../system/abi.zig): `thread_spawn = 37`, `thread_exit = 38`. Handlers in
|
||||
process.zig; `thread_spawn` calls `scheduler.spawnThread` (shares the caller's
|
||||
aspace, `retainAspace`); `thread_exit` ends the task like a process `exit(0)`
|
||||
(`terminateCurrent` → `releaseAspace`). The closure pointer is delivered in the new
|
||||
thread's **rdi** via a new `jump_to_user_arg` asm path (`t.user_arg`, 0 for a
|
||||
process) — no naked runtime asm.
|
||||
- [x] `library/runtime/thread.zig` (barrel-exported as `runtime.Thread`): `spawn` maps a
|
||||
stack (`mmap`), heap-allocates the `{args}` closure, and calls
|
||||
`thread_spawn(&Closure.entry, stack_top, closure)`; `Closure.entry` (a plain C-ABI
|
||||
Zig fn, closure in rdi) runs the function and calls `thread_exit`. Stack top is
|
||||
16-aligned-minus-8 for the C entry.
|
||||
- [x] A `threaded` flag on the user-binary recipe (`addThreadedUserBinary` →
|
||||
`single_threaded = false`); `thread-test` is the first opt-in binary.
|
||||
- [x] `-Dtest-case=thread-spawn`: `thread-test` spawns a worker that writes a sentinel to
|
||||
a **shared** global and release-stores `done`; the main thread acquire-polls `done`
|
||||
and asserts the shared global holds the sentinel — proof the worker ran in the same
|
||||
address space.
|
||||
|
||||
**Gate (met):** `python3 test/qemu_test.py thread-spawn` passes
|
||||
(`thread-test: child ran in shared aspace ok` → `DANOS-TEST-RESULT: PASS`); guardrail set
|
||||
16/16 green (incl. `args`/`init`/`process`, which exercise the new `jump_to_user_arg`
|
||||
process path with arg 0) plus `aspace-refcount`; `zig build` clean, `zig build test`
|
||||
green.
|
||||
|
||||
> **Note (deferred to M3+):** the mmap arena is per-*task* (`heap_next`), so two threads
|
||||
> in one aspace that both `mmap` would collide. Fine for M2 (only the parent maps, for the
|
||||
> child's stack); make the arena per-aspace and the runtime heap thread-safe alongside the
|
||||
> `Mutex` work (M5).
|
||||
|
||||
## M3 — `join` + `detach` + real parallelism ✅
|
||||
|
||||
- [x] `join` over the existing exit-notification path
|
||||
([process-lifecycle.md](process-lifecycle.md)): `thread_spawn` gained a 4th arg, an
|
||||
`exit_endpoint` handle (resolved + refcounted like `spawnProcessSupervised`, via
|
||||
`spawnThreadSupervised`); `join` blocks in `ipc_reply_wait` on that endpoint until
|
||||
the child-exit notice for its `tid`, then `munmap`s the stack. `detach` relinquishes
|
||||
the join right (its stack is reclaimed at process exit — kernel-reaper reclaim for
|
||||
detached threads is deferred; see note).
|
||||
- [x] `runtime.Thread.join` / `detach`, plus `Thread.currentCore()` (a new `current_core`
|
||||
= 39 syscall) for the parallelism proof. `getCurrentId` deferred to M6 (TLS), where
|
||||
a lighter self-id fits. The closure now rides the **thread's own stack** (not the
|
||||
heap) — private per thread, so spawn/join touch no shared heap.
|
||||
- [x] `-Dtest-case=thread-join` (`smp: 4`): `thread-test` join mode spawns N=4 workers
|
||||
that each do K=100k `@atomicRmw`-increments on a shared counter and stamp the core
|
||||
they ran on; the main thread joins all N and asserts `counter == N*K` **and**
|
||||
`@popCount(cores_seen) > 1` (genuine cross-core parallelism), then a detached worker
|
||||
proves `detach` runs without a join.
|
||||
|
||||
**Gate (met):** `python3 test/qemu_test.py thread-join` passes (`thread-test: join ok` →
|
||||
`DANOS-TEST-RESULT: PASS`), robust across 4 runs; guardrail 17/17 green (incl. `smp`,
|
||||
`affinity`, `process-kill`, and `args`/`init`/`process` on the exit-endpoint spawn path)
|
||||
plus `aspace-refcount`/`thread-spawn`; `zig build` clean, `zig build test` green.
|
||||
|
||||
> **Note (deferred):** a detached thread's stack is freed only at process exit (not by the
|
||||
> reaper on thread exit) — kernel user-stack tracking + reclaim is a later refinement. And
|
||||
> the runtime heap is still not thread-safe: threads that both allocate concurrently would
|
||||
> race (the thread *machinery* avoids the heap, but worker code sharing an allocator does
|
||||
> not). Both fold into the M5 `Mutex`/allocator work.
|
||||
|
||||
## M4 — Futex: the one blocking primitive ✅
|
||||
|
||||
- [x] [abi.zig](../system/abi.zig): `futex_wait = 40`, `futex_wake = 41`. A waiter is a
|
||||
`.blocked` task tagged with `Task.futex_addr` (no queue linkage);
|
||||
`futex_wait(addr, expected, timeout_ns)` reads the user word under the big lock,
|
||||
parks iff `*addr == expected`, and returns on wake or timeout; `futex_wake(addr,
|
||||
count)` scans the task table and readies up to `count` matching waiters (same
|
||||
address space). No spinning — a parked waiter leaves its core free to `hlt`. A
|
||||
timed wait also sets `wake_at`, so the timer's `wakeExpired` wakes it; `futex_addr`
|
||||
staying non-zero (only `futex_wake` clears it) is how the waiter tells timeout from
|
||||
a real wake.
|
||||
- [x] `runtime.Thread.Futex` (`wait` / `timedWait` / `wake`) over the syscall wrappers.
|
||||
- [x] `-Dtest-case=thread-futex` (`smp: 4`): a waiter thread prints `waiting` and
|
||||
`futex_wait`s on a word; the main thread publishes it, prints `waking`, and
|
||||
`futex_wake`s; the waiter prints `woke`. Then a `timedWait` on an unwoken word
|
||||
reports `error.Timeout`.
|
||||
|
||||
**Gate (met):** `python3 test/qemu_test.py thread-futex` passes, robust across 3 runs —
|
||||
the case's **ordered** regex asserts `waiting → waking → woke → PASS` on the serial
|
||||
stream (the handoff proof), and `thread-futex: timeout ok` confirms the timeout.
|
||||
Guardrail 18/18 green (incl. `sleep`/`event`/`ipc` blocking paths) + `aspace-refcount`,
|
||||
`thread-spawn`, `thread-join`; `zig build` clean, `zig build test` green.
|
||||
|
||||
> **Note:** the kernel test checks only the freshest verdict marker via `bufferHas` (the
|
||||
> in-memory log ring buffer evicts older lines); ordering is asserted against the full
|
||||
> serial stream by the qemu regex instead.
|
||||
|
||||
## M5 — `Mutex` + `Condition` + `Semaphore` ✅
|
||||
|
||||
- [x] `runtime.Thread.Mutex` (three-state futex mutex: CAS fast path, `futex_wait`/`wake`
|
||||
slow path), `Condition` (`wait`/`timedWait`/`signal`/`broadcast`, a futex sequence
|
||||
counter), `Semaphore` (permits over `Mutex`+`Condition`) — the same state machines
|
||||
`std.Thread` uses, ported onto our `Futex`.
|
||||
- [x] `-Dtest-case=thread-mutex` (`smp: 4`): a bounded producer/consumer — 2 producers +
|
||||
2 consumers over one `Mutex` and two `Condition`s move N=2000 unique items through
|
||||
an 8-slot ring; the consumed checksum and tally match exactly (no lost/duplicated
|
||||
item, no overrun) under real cross-core contention. The small ring forces producers
|
||||
to block on full and consumers on empty, exercising `Condition.wait`.
|
||||
|
||||
**Gate (met):** `python3 test/qemu_test.py thread-mutex` passes (`thread-mutex: ok` →
|
||||
`DANOS-TEST-RESULT: PASS`), robust across 3 runs; guardrail 17/17 green (incl.
|
||||
`sleep`/`event`/`ipc`) + all M1–M4 thread cases; `zig build` clean, `zig build test`
|
||||
green.
|
||||
|
||||
> **Deferred (with rationale):**
|
||||
> - **`join` → futex completion word** — the exit-endpoint join (M3) is correct and
|
||||
> tested. A futex-completion join needs the *kernel* to clear+wake a word after the
|
||||
> thread is fully off its stack (a CLONE_CHILD_CLEARTID-style mechanism); doing it in
|
||||
> the thread's own trampoline would let `join` `munmap` the stack while the thread still
|
||||
> runs on it (use-after-free). Left on the exit-endpoint path; the kernel clear-on-exit
|
||||
> is a later, separate refinement.
|
||||
> - **Host unit tests for the state machines** — `Mutex`/`Condition` bottom out in the
|
||||
> `futex_*` syscalls, unavailable on the host without a mockable `Futex` seam. The QEMU
|
||||
> `thread-mutex` gate exercises them under real concurrency instead; a host-side mock is
|
||||
> future work.
|
||||
|
||||
## M6 — `getCurrentId`, docs, and CI wiring ✅
|
||||
|
||||
- [x] `getCurrentId` via a small `thread_self = 42` syscall (`runtime.Thread.getCurrentId`
|
||||
returns the kernel task id). **Per-thread `threadlocal` TLS is deferred** — no
|
||||
consumer needs it, and it would require context-switching `fs.base` per task (real
|
||||
kernel + per-switch cost) for an unused feature; threaded binaries have run fine
|
||||
without it through M2–M5. threading.md's TLS reasoning already scoped it as
|
||||
deferred-unless-needed. When a consumer appears, the shape is: `thread_spawn`
|
||||
allocates a per-thread TLS block, sets `fs.base`, and the context switch saves/
|
||||
restores it.
|
||||
- [x] `RwLock` / `WaitGroup` deferred (no consumer yet); they slot onto the same
|
||||
`Futex`/`Mutex`/`Condition` when wanted.
|
||||
- [x] All `thread-*` cases wired into [test/qemu_test.py](../test/qemu_test.py)
|
||||
(`thread-spawn`/`-join`/`-futex`/`-mutex`/`-id`); threading.md + docs/README.md
|
||||
status updated to **built**; the worked example is threading.md's win-condition.
|
||||
- [x] `-Dtest-case=thread-id` (`smp: 4`): two workers read `getCurrentId`; the main
|
||||
thread confirms all three ids are non-zero and distinct — each thread has its own
|
||||
kernel identity. (Renamed from `thread-tls`, which implied `threadlocal`.)
|
||||
|
||||
**Gate (met):** `python3 test/qemu_test.py thread-id` passes; the whole `thread-*` suite
|
||||
(`thread-spawn`/`-join`/`-futex`/`-mutex`/`-id`) plus the full guardrail set pass; default
|
||||
`zig build` clean, `zig build test` green.
|
||||
|
||||
---
|
||||
|
||||
## Status: built
|
||||
|
||||
M1–M6 complete. danos has `runtime.Thread` — `spawn`/`join`/`detach`, cross-core
|
||||
parallelism, futex, and `Mutex`/`Condition`/`Semaphore`, all over a private thread ABI
|
||||
behind the runtime. Deferred (with rationale, no consumer yet): `threadlocal` TLS,
|
||||
`RwLock`/`WaitGroup`, kernel clear-on-exit for a futex-completion `join`, a per-aspace
|
||||
mmap arena / thread-safe runtime heap, and host-side unit tests via a mockable `Futex`.
|
||||
|
||||
---
|
||||
|
||||
## Deferred (explicitly not in this plan)
|
||||
|
||||
- **Cross-process shared-memory futex** — the `(aspace, vaddr)` key can become a
|
||||
physical-address key so two processes share a futex through an [shm](display-v2.md)
|
||||
region. Not needed for intra-process threads.
|
||||
- **Per-thread priorities / affinity distinct from the process** — threads inherit the
|
||||
process priority ([scheduling.md](scheduling.md)); revisit only if it earns its keep.
|
||||
- **Per-thread signal delivery** — signals stay process-scoped
|
||||
([process-lifecycle.md](process-lifecycle.md)).
|
||||
- **A `pthread`/POSIX surface** — the API is `std.Thread`-shaped Zig, nothing more.
|
||||
- **A real `std.Thread` backend** — arrives with self-hosting
|
||||
([zig-self-hosting.md](zig-self-hosting.md)); it sits on these same primitives, so it
|
||||
swaps the impl under `runtime.Thread`, not the call sites.
|
||||
@@ -0,0 +1,316 @@
|
||||
# Threading: `runtime.Thread`, a std-shaped API over a private thread ABI
|
||||
|
||||
A note on danos **threads** — several tasks sharing one address space — provided by a
|
||||
`runtime.Thread` type that mirrors the shape of Zig's `std.Thread` while keeping every
|
||||
kernel entry behind the [runtime](../library/runtime). **Built** (M1–M6, see
|
||||
[threading-plan.md](threading-plan.md)): `spawn`/`join`/`detach`, cross-core
|
||||
parallelism, a futex (`futex_wait`/`futex_wake`), and a futex-backed
|
||||
`Mutex`/`Condition`/`Semaphore`, plus `getCurrentId`/`currentCore`. Deferred by design
|
||||
(no consumer yet): per-thread `threadlocal` TLS, `RwLock`/`WaitGroup`, and migrating
|
||||
`join` to a futex completion word — see the plan's M5/M6 notes. The analysis is against
|
||||
**Zig 0.16** (the pinned toolchain); `std.Thread`'s internals move between releases, so
|
||||
treat upstream shapes as "0.16.x."
|
||||
|
||||
## The win condition
|
||||
|
||||
A danos service can write
|
||||
|
||||
```zig
|
||||
const t = try runtime.Thread.spawn(.{}, worker, .{ctx});
|
||||
// ... do other work concurrently ...
|
||||
t.join();
|
||||
```
|
||||
|
||||
and get real parallelism across cores — with `runtime.Thread.Mutex`,
|
||||
`runtime.Thread.Condition`, and `runtime.Thread.Semaphore` available for
|
||||
coordination — **without any code path reaching the kernel except through the
|
||||
runtime**. The call sites read exactly like `std.Thread`, so the day danos becomes a
|
||||
real Zig target (see [self-hosting](#the-self-hosting-endgame)) we swap the
|
||||
implementation underneath, not the API above.
|
||||
|
||||
## Locked decisions (do not relitigate)
|
||||
|
||||
- **We build `runtime.Thread`, not literal `std.Thread`.** It mirrors std's *API and
|
||||
features*; the implementation underneath is danos-native. See
|
||||
[Why not literal std.Thread](#why-not-literal-stdthread).
|
||||
- **Threads are a narrow, opt-in capability — not the default concurrency tool.** The
|
||||
default for resilience stays **process + IPC** ([resilience.md](resilience.md),
|
||||
[ipc.md](ipc.md)). See [Where threads fit](#where-threads-fit-the-resilience-tension).
|
||||
- **Blocking synchronization is futex-backed, never spin-backed.** Waiters sleep in
|
||||
the kernel so an idle core still halts ([halting.md](halting.md)).
|
||||
- **Per-binary opt-in to multi-threaded codegen.** Only a service that asks for
|
||||
threads is built `single_threaded = false`; the rest stay lean and single-threaded.
|
||||
- **The thread ABI is private.** New syscalls extend [abi.zig](../system/abi.zig)
|
||||
`SystemCall` and are reached only through `library/runtime` wrappers, exactly like
|
||||
every other danos syscall ([syscall.md](syscall.md)) — numbers stay renumberable.
|
||||
|
||||
## Why not literal `std.Thread`
|
||||
|
||||
danos's ABI invariant is that the **runtime is the sole holder of the syscall ABI**,
|
||||
and that ABI is private and renumberable ([syscall.md](syscall.md) — "unstable
|
||||
private ABI"). That is a security and evolvability asset: no compiled binary can
|
||||
hardcode a syscall number, and the kernel can renumber freely because only the
|
||||
runtime — rebuilt in lockstep — knows the mapping.
|
||||
|
||||
`std.Thread` is incompatible with that invariant on two counts:
|
||||
|
||||
1. **It selects its backend from `builtin.os.tag`, and issues syscalls directly.**
|
||||
danos targets `.os_tag = .freestanding` ([build.zig](../build.zig)), for which
|
||||
`std.Thread` resolves to an unsupported stub that `@compileError`s. Adding a real
|
||||
backend would either bake danos syscall numbers into std (breaking ABI privacy and
|
||||
renumbering) or fork std to route back through the runtime — a permanent rebase
|
||||
cost that buys nothing the native type doesn't.
|
||||
2. **Our user binaries are built `single_threaded = true`** ([build.zig](../build.zig)
|
||||
`addUserBinary`), which compiles threading out entirely and makes atomics and TLS
|
||||
single-threaded. Threads need this flipped per binary regardless.
|
||||
|
||||
So we take the *shape* of `std.Thread`, not the *type*. The cost of replicating the
|
||||
surface (spawn/join/Mutex/Condition) is small; the cost of the std type is the ABI
|
||||
invariant.
|
||||
|
||||
## Where threads fit: the resilience tension
|
||||
|
||||
Threads are in genuine tension with a resilience-first microkernel, and it is worth
|
||||
being explicit so we do not reach for them by reflex.
|
||||
|
||||
The reason danos pays for a microkernel is **fault isolation**
|
||||
([resilience.md](resilience.md)): a component corrupts its own address space, faults,
|
||||
and is **restarted** without touching anyone else — because the boundary *is* the
|
||||
address space. Threads deliberately remove that boundary *within* a process:
|
||||
|
||||
- Threads share one address space, so one thread's stray write corrupts them all —
|
||||
there is no isolation **between** threads.
|
||||
- Threads share fate: a fault in any thread, or a "kill the process" decision, takes
|
||||
down **all** of them. Restartability lives at the process level, not the thread
|
||||
level.
|
||||
- Shared mutable state reintroduces data races — the failure class the
|
||||
isolate-and-message model was chosen to avoid.
|
||||
|
||||
**Therefore:** the default answer to "make X concurrent" stays *another process over
|
||||
IPC* (isolated, independently restartable) or a single event loop with several
|
||||
message sources. Reach for a thread only inside **one** service that needs genuine
|
||||
**shared-memory, low-latency parallelism** and can accept intra-service fate-sharing —
|
||||
e.g. a compositor splitting tile compositing across cores, where per-tile IPC would be
|
||||
too chatty. "Input on one thread, display on another" is *not* that case; it wants two
|
||||
processes. The isolation boundary stays at process granularity.
|
||||
|
||||
## The API surface (mirrors `std.Thread`)
|
||||
|
||||
Lives in `library/runtime/thread.zig`, re-exported as `runtime.Thread`.
|
||||
|
||||
```zig
|
||||
pub const Thread = struct {
|
||||
pub const Id = u32; // the kernel task id
|
||||
pub const SpawnConfig = struct {
|
||||
stack_size: usize = default_stack_size,
|
||||
allocator: ?std.mem.Allocator = null, // for the closure + stack bookkeeping
|
||||
};
|
||||
pub const SpawnError = error{ OutOfMemory, ThreadQuotaExceeded, SystemResources };
|
||||
|
||||
pub fn spawn(config: SpawnConfig, comptime function: anytype, args: anytype) SpawnError!Thread;
|
||||
pub fn join(self: Thread) void; // block until the thread ends, reclaim its stack
|
||||
pub fn detach(self: Thread) void; // give up the right to join; kernel reclaims on exit
|
||||
pub fn getCurrentId() Id;
|
||||
pub fn yield() void; // -> existing `yield` syscall
|
||||
|
||||
pub const Mutex = struct { pub fn lock(*Mutex) void; pub fn tryLock(*Mutex) bool; pub fn unlock(*Mutex) void; };
|
||||
pub const Condition = struct { pub fn wait(*Condition, *Mutex) void; pub fn timedWait(*Condition, *Mutex, u64) error{Timeout}!void; pub fn signal(*Condition) void; pub fn broadcast(*Condition) void; };
|
||||
pub const Semaphore = struct { pub fn wait(*Semaphore) void; pub fn post(*Semaphore) void; };
|
||||
pub const Futex = struct { pub fn wait(*const atomic.Value(u32), u32) void; pub fn timedWait(...) error{Timeout}!void; pub fn wake(*const atomic.Value(u32), u32) void; };
|
||||
// RwLock / ResetEvent / WaitGroup follow the same pattern, added as needed.
|
||||
};
|
||||
```
|
||||
|
||||
Deviations from `std.Thread`, called out honestly:
|
||||
|
||||
- **The thread function's return value is discarded** (as `std.Thread.join` returns
|
||||
`void`). Return data through shared state or a `Semaphore`/`Condition`, not the
|
||||
return.
|
||||
- `getCpuCount()` maps to the existing SMP core count ([smp.md](smp.md)); a service
|
||||
rarely needs it.
|
||||
|
||||
## Kernel primitives (new private syscalls)
|
||||
|
||||
Four new entries extend [abi.zig](../system/abi.zig) `SystemCall` after
|
||||
`shm_physical = 36`, each with a `library/runtime` wrapper:
|
||||
|
||||
| Syscall | Signature | Purpose |
|
||||
|---|---|---|
|
||||
| `thread_spawn` | `(entry, stack_top, arg) -> tid` | create a task sharing the **caller's** address space |
|
||||
| `thread_exit` | `(stack_base, stack_len)` | end the calling thread; hand back its stack range for reclaim |
|
||||
| `futex_wait` | `(addr, expected, timeout_ns) -> status` | block if `*addr == expected`, until woken or timeout |
|
||||
| `futex_wake` | `(addr, count) -> woken` | wake up to `count` waiters on `addr` |
|
||||
|
||||
Plus one invariant change with no new syscall: **address-space reference counting**.
|
||||
|
||||
## Mechanics
|
||||
|
||||
### Address-space reference counting
|
||||
|
||||
Today an address space is 1:1 with a task: `spawnUserLocked` records `aspace` on the
|
||||
Task, and teardown does `destroyAddressSpace(t.aspace)` when **any** user task exits
|
||||
([scheduler.zig](../system/kernel/scheduler.zig)). With threads, several tasks share
|
||||
one `aspace`, so the first to exit would rip the address space out from under its
|
||||
siblings.
|
||||
|
||||
Fix: a small refcount keyed by the address-space root (`createAddressSpace` in
|
||||
[process.zig](../system/kernel/process.zig) sets it to 1). `thread_spawn` increments
|
||||
it; task teardown decrements and only calls `destroyAddressSpace` at **zero**. All of
|
||||
this is already under the big kernel lock, so no new locking. This is the one piece
|
||||
that must land and be proven before anything shares an address space.
|
||||
|
||||
### `thread_spawn` and the trampoline
|
||||
|
||||
The scheduler already accepts an arbitrary `aspace` and does **not** smuggle values
|
||||
through registers — `startUserTask` reads the entry/stack from the Task and
|
||||
`jumpToUser`s ([scheduler.zig](../system/kernel/scheduler.zig)). That makes the thread
|
||||
path clean:
|
||||
|
||||
1. The runtime's `spawn` `mmap`s a stack (syscall `4`), heap-allocates a closure —
|
||||
`{ fn_ptr, args_tuple, completion }`, the std "Instance" pattern — and writes the
|
||||
closure pointer to the **top word of the new stack**.
|
||||
2. It calls `thread_spawn(entry = &threadTrampoline, stack_top, arg = closure_ptr)`.
|
||||
The kernel calls the same `spawnUserLocked` path with the **caller's aspace**
|
||||
(refcount++), `entry`, and `user_sp = stack_top`.
|
||||
3. `threadTrampoline` (a small runtime shim) reads the closure off its stack, calls
|
||||
the user function, then calls `thread_exit`. No new register ABI — the closure
|
||||
pointer rides the stack the runtime set up, mirroring how `startUserTask` avoids
|
||||
register smuggling.
|
||||
|
||||
Unlike a process start, there is **no** System V argc/argv/auxv block
|
||||
([sysv.md](sysv.md)) — a thread stack carries only the closure pointer.
|
||||
|
||||
### Lifetime: exit, join, detach, stack reclaim
|
||||
|
||||
- **`thread_exit`** marks the task dead and hands the kernel the thread's user-stack
|
||||
range. The kernel reaps the task on the scheduler (already running on a *kernel*
|
||||
stack, so it can safely unmap the user stack), decrements the aspace refcount, and
|
||||
frees the task slot.
|
||||
- **`join` — Stage 1** reuses the existing exit-notification machinery
|
||||
([process-lifecycle.md](process-lifecycle.md)): `spawn` passes a per-thread
|
||||
`exit_endpoint`, and `join` blocks in `ipc_reply_wait` until the child-exit
|
||||
notification for that `tid` arrives, then `munmap`s the stack. No futex needed to
|
||||
land spawn/join.
|
||||
- **`join` — Stage 2 refinement** migrates to the std shape: a `completion` word in
|
||||
the closure that `thread_exit`'s trampoline `futex_wake`s and `join` `futex_wait`s
|
||||
on — dropping the per-thread endpoint. Kept as a refinement so Stage 1 ships first.
|
||||
- **`detach`** relinquishes the join right; the kernel reclaims the stack and slot on
|
||||
`thread_exit` (a detached thread's stack range is unmapped by the reaper, since no
|
||||
joiner will).
|
||||
|
||||
### Futex, and the sync primitives on top
|
||||
|
||||
`futex_wait`/`futex_wake` are the one blocking primitive; `Mutex`, `Condition`, and
|
||||
`Semaphore` are ordinary user-space state machines over an `atomic.Value(u32)` that
|
||||
call the futex wrappers on the slow path — the same construction `std.Thread` uses,
|
||||
so the algorithms port directly.
|
||||
|
||||
Keying: threads share an address space, so a **virtual address within that aspace**
|
||||
identifies a futex uniquely; the kernel keys its wait queue by `(aspace_root, vaddr)`.
|
||||
Keying by the **physical** address instead (translate `vaddr -> paddr` on entry) is a
|
||||
deliberate forward door: it lets two *processes* share a futex through an
|
||||
[shm](display-v2.md) region later, without changing the API. We start with the
|
||||
private-per-aspace key and note the physical-key upgrade.
|
||||
|
||||
No spinning: a contended lock parks the task in the kernel and the core is free to run
|
||||
other work or `hlt` ([halting.md](halting.md)). This is why futex is a locked
|
||||
decision, not a "maybe later."
|
||||
|
||||
### TLS and `getCurrentId`
|
||||
|
||||
danos sets up no `fs.base` TLS today (fine under `single_threaded`). Two scoped needs:
|
||||
|
||||
- **`getCurrentId`** returns the kernel task id — either a trivial syscall or, better,
|
||||
a value the runtime stashes in a per-thread control block.
|
||||
- **`threadlocal` variables** need a real per-thread TLS block and `fs.base` set per
|
||||
thread. `thread_spawn` sets `fs.base` to a runtime-allocated per-thread block; full
|
||||
`threadlocal` support is Stage 3, only if a consumer needs it. Nothing in the core
|
||||
spawn/join/mutex path requires `threadlocal`.
|
||||
|
||||
### Build: multi-threaded codegen, opt-in
|
||||
|
||||
`addUserBinary` gains a `threaded: bool = false` parameter; when set it builds that
|
||||
binary `single_threaded = false` so atomics and (later) TLS are real. Threads and
|
||||
atomics are unsound in a `single_threaded` image, so a binary must opt in **before**
|
||||
it may call `runtime.Thread.spawn`. Everyone else stays single-threaded and lean.
|
||||
|
||||
## Interaction with the rest of the kernel
|
||||
|
||||
- **Scheduler / SMP** ([scheduling.md](scheduling.md), [smp.md](smp.md)): a thread is
|
||||
just another `Task` with an `aspace` shared with its siblings; the existing
|
||||
per-core ready queues, priorities, and affinity apply unchanged. Threads of one
|
||||
process can run on different cores simultaneously — that is the point.
|
||||
- **Halting** ([halting.md](halting.md)): futex-parked waiters keep the "idle core
|
||||
halts" property intact under lock contention — no busy-wait.
|
||||
- **Lifecycle** ([process-lifecycle.md](process-lifecycle.md)): killing a process
|
||||
must kill *all* its threads and only then drop the last aspace ref. The kill path
|
||||
already targets a process; it fans out to every task on that aspace.
|
||||
- **Resilience** ([resilience.md](resilience.md)): a faulting thread kills its whole
|
||||
process (shared fate). The supervisor restarts the **process**, which respawns its
|
||||
threads from a known-good state — restart granularity stays the process.
|
||||
|
||||
## Build-out plan (staged, each gate serial-checkable)
|
||||
|
||||
The ordered, `/loop`-runnable milestones live in
|
||||
**[threading-plan.md](threading-plan.md)** (shaped like
|
||||
[display-v2-plan.md](display-v2-plan.md)): every milestone lands on its own and ends in
|
||||
a verifiable gate (`python3 test/qemu_test.py <case>`, asserting serial markers;
|
||||
`zig build test` for host unit tests). The stages below are the shape it expands.
|
||||
|
||||
- **Stage 0 — address-space refcount.** Refcount on the aspace root; teardown destroys
|
||||
at zero. No API yet; nothing shares an aspace, so refcount is 1 everywhere.
|
||||
*Gate:* the full QEMU suite stays green (no regression) — proves the reframing is
|
||||
invisible until used.
|
||||
- **Stage 1 — spawn / join / detach.** `thread_spawn` + `thread_exit`, the trampoline,
|
||||
stacks via `mmap`, join over the exit-endpoint, the `threaded` build flag.
|
||||
*Gate:* `-Dtest-case=thread-spawn` — a threaded test service spawns N threads that
|
||||
each `@atomicRmw`-increment a shared counter, the parent joins all N, and asserts
|
||||
the total is exactly N × iterations. Runs `smp` (multi-core) to prove real
|
||||
parallelism.
|
||||
- **Stage 2 — blocking synchronization.** `futex_wait`/`futex_wake` + `Futex`,
|
||||
`Mutex`, `Condition`, `Semaphore`; optionally migrate join to a futex completion
|
||||
word. *Gate:* `-Dtest-case=thread-mutex` — a bounded producer/consumer over a
|
||||
`Mutex` + `Condition` moves K items with no lost wakeups and no busy-wait (assert
|
||||
the consumer blocked, e.g. via a low idle tick count).
|
||||
- **Stage 3 — polish.** Per-thread TLS / `fs.base` and `threadlocal` (only if a
|
||||
consumer needs it), `RwLock`/`WaitGroup` as demanded, and this doc's cases wired
|
||||
into [test/qemu_test.py](../test/qemu_test.py).
|
||||
|
||||
## Conventions
|
||||
|
||||
Follow [coding-standards.md](coding-standards.md): spell out non-acronym
|
||||
abbreviations, kebab-case file names, no `Co-Authored-By` trailers. New syscalls
|
||||
extend [abi.zig](../system/abi.zig) `SystemCall` + a `library/runtime` wrapper
|
||||
([syscall.md](syscall.md)). `runtime.Thread` is a first-class runtime module, the same
|
||||
way `runtime.process` ([process-lifecycle.md](process-lifecycle.md)) and `runtime.ipc`
|
||||
are — user code never names a syscall.
|
||||
|
||||
## Non-goals
|
||||
|
||||
- **No preemptive user-space signals delivered to a specific thread.** Signals stay
|
||||
process-scoped ([process-lifecycle.md](process-lifecycle.md)).
|
||||
- **No thread priorities distinct from the process.** Threads inherit the process
|
||||
priority; per-thread priority is a later question if it ever earns its keep.
|
||||
- **No cross-process shared-memory futex yet** — the physical-address key leaves the
|
||||
door open, but the first cut is private-per-aspace.
|
||||
- **No `pthread`/POSIX surface.** The API is `std.Thread`-shaped Zig, nothing more.
|
||||
|
||||
## The self-hosting endgame
|
||||
|
||||
When danos becomes a real Zig target and we (eventually) add a danos backend to std
|
||||
([zig-self-hosting.md](zig-self-hosting.md)), `std.Thread` can sit *on top of* these
|
||||
same kernel primitives — the danos `std.Thread.Impl` would call the very
|
||||
`thread_spawn`/`futex_*` wrappers `runtime.Thread` already uses. Because
|
||||
`runtime.Thread` was built API-compatible from day one, that transition swaps the
|
||||
implementation, not a single call site. Designing to the std shape now is what makes
|
||||
the later self-hosting lift cheap.
|
||||
|
||||
## Further reading
|
||||
|
||||
- [scheduling.md](scheduling.md), [smp.md](smp.md) — the task model these threads join.
|
||||
- [resilience.md](resilience.md), [vision.md](vision.md) — why isolation is the default
|
||||
and threads are the exception.
|
||||
- [syscall.md](syscall.md), [ipc.md](ipc.md) — the private ABI and the messaging model
|
||||
threads sit beside.
|
||||
- [halting.md](halting.md) — the idle/halt property futex-backed blocking preserves.
|
||||
- [zig-self-hosting.md](zig-self-hosting.md) — the target this bends toward.
|
||||
+206
@@ -0,0 +1,206 @@
|
||||
# The vDSO — the public system-call boundary
|
||||
|
||||
> **Status:** design note, not built. The runtime today issues raw `syscall`
|
||||
> instructions from `library/runtime/system-call.zig` using the numbers in
|
||||
> `system/abi.zig`. This note designs the layer that replaces that arrangement:
|
||||
> a **kernel-supplied, C-ABI entry library** mapped into every process — the
|
||||
> only supported way into the kernel — so the raw numbers can stay private,
|
||||
> be renumbered at will, and eventually be randomised per boot.
|
||||
|
||||
## Why: the ABI danos promises, and the one it doesn't
|
||||
|
||||
`system/abi.zig` is the **private** kernel ↔ runtime contract. Its header says
|
||||
so: the numbers are an implementation detail the runtime hides and may
|
||||
renumber, the same split as libSystem over the XNU syscalls on macOS or win32
|
||||
over the NT syscalls on Windows. Linux — with its world-visible, frozen
|
||||
syscall table — is the outlier, not the norm.
|
||||
|
||||
That stance has consequences the moment binaries exist that we don't rebuild
|
||||
ourselves:
|
||||
|
||||
1. **Third-party binaries** (docs/zig-self-hosting.md) must keep working across
|
||||
kernel updates. If they contain raw `syscall` instructions with today's
|
||||
numbers baked in, every renumbering breaks the world — the ABI would be
|
||||
*de facto* public no matter what the header says. Go on macOS made exactly
|
||||
this mistake: it issued XNU syscalls directly instead of going through
|
||||
libSystem, and macOS updates repeatedly broke every Go binary until Go
|
||||
switched to the library like everyone else.
|
||||
2. **Not everything is Zig.** A Rust or C program can't import the `runtime`
|
||||
module. The public boundary has to be expressible in the one calling
|
||||
convention every language speaks: the C ABI.
|
||||
3. **Randomised syscall numbers** — a hardening option we want open — only
|
||||
work if no user binary anywhere knows a number at build time. The binding
|
||||
must happen at *load time*, from something the kernel controls.
|
||||
|
||||
All three point at the same well-known shape: a **vDSO** (virtual dynamic
|
||||
shared object). The kernel carries a small blob of user-mode code, maps it
|
||||
into every process at spawn, and that blob — not the application — contains
|
||||
the `syscall` instructions. Fuchsia works exactly this way: its vDSO is the
|
||||
*only* kernel entry, version-matched by construction because the kernel itself
|
||||
injects it. Because the kernel and the blob ship as one artifact, there is
|
||||
**no version skew, no loader, no search path, and no shared file on disk** —
|
||||
which is what makes this the resilient way to have a private ABI
|
||||
(docs/resilience.md), where a conventional `ld.so` + `/lib/libdanos.so`
|
||||
arrangement would add a loader to every spawn and a single shared point of
|
||||
failure.
|
||||
|
||||
The public danos ABI then has exactly two layers, neither of which is
|
||||
`abi.zig`:
|
||||
|
||||
| Layer | Contract | Spoken by |
|
||||
|-------|----------|-----------|
|
||||
| **vDSO** | C-ABI functions, this note | every language's thin shim (`runtime.system` for Zig, a `-sys` crate for Rust, a header for C) |
|
||||
| **IPC wire protocols** | byte layouts over `ipc_call` ([vfs-protocol.md](vfs-protocol.md) is the first one documented) | any client that can lay out bytes |
|
||||
|
||||
Everything above those — the heap, `runtime.fs`, the service harness — is
|
||||
per-language convenience, compiled into each binary from source, exactly as
|
||||
today. Nothing about the Zig runtime's shape changes; it just stops being the
|
||||
*only* door.
|
||||
|
||||
## The blob
|
||||
|
||||
A single copy of the vDSO code lives in the kernel image (built by
|
||||
`build.zig` as a tiny freestanding object, embedded like the AP trampoline).
|
||||
At boot the kernel finalises it once — this is where randomised numbers would
|
||||
be patched in — and thereafter maps the **same physical pages** read-execute
|
||||
into every process's address space. The blob is:
|
||||
|
||||
- **Position-independent.** It is mapped at a per-process randomised base, so
|
||||
it must be PIC (rip-relative addressing only — no relocations to process).
|
||||
- **Stateless and re-entrant.** No writable data. Anything stateful belongs to
|
||||
the process, not the vDSO.
|
||||
- **Architecture-specific.** The x86-64 blob wraps `syscall`; an aarch64 blob
|
||||
wraps `svc #0`. It lives beside the other per-architecture kernel sources
|
||||
(`system/kernel/architecture/<arch>/`), selected the same way the
|
||||
`architecture` module is (docs/arch.md).
|
||||
|
||||
### Shape: a function table, not an ELF
|
||||
|
||||
A real `.so` with a dynamic symbol table is the conventional vDSO shape, but
|
||||
linking against one at load time needs a dynamic linker in every binary —
|
||||
machinery danos deliberately doesn't have. Instead the v1 shape is the
|
||||
simplest thing that is still a stable contract — a **function-pointer table**
|
||||
at the vDSO base:
|
||||
|
||||
```
|
||||
offset 0 u64 magic 'danosVDS' — a mapped-the-wrong-thing guard
|
||||
offset 8 u64 api_level incremented when the table grows
|
||||
offset 16 u64 count number of table entries that follow
|
||||
offset 24 u64 table[count] function pointers into the vDSO's own code
|
||||
```
|
||||
|
||||
Table *indices* are the public constants (published in a C header,
|
||||
`danos.h`), assigned once and append-only — the same discipline the IPC
|
||||
protocols use for operation values. The pointers point at stubs inside the
|
||||
blob; what those stubs put in `rax` is nobody's business but the kernel's.
|
||||
A language shim binds in one step: read the base from the init block, check
|
||||
the magic, keep the table pointer. Feature detection for a binary built
|
||||
against older headers is `count`/`api_level` — a kernel never removes or
|
||||
reorders entries.
|
||||
|
||||
(If danos ever grows a real dynamic linker, the same blob can additionally
|
||||
present an ELF `dynsym` without breaking the table — Fuchsia's vDSO is
|
||||
likewise both a mappable blob and a linkable `.so`. That is a later
|
||||
convenience, not a requirement.)
|
||||
|
||||
### Delivery: the auxiliary vector
|
||||
|
||||
The kernel already builds a System V entry block — argc, argv, envp
|
||||
terminator, **auxiliary vector** — on every new process's stack
|
||||
(`buildEntryStack`, read by `runtime.start`). The vDSO base rides in a new
|
||||
auxv entry, exactly Linux's `AT_SYSINFO_EHDR` move. No new syscall, no magic
|
||||
address, and a language shim finds it the same portable way on every
|
||||
architecture.
|
||||
|
||||
## The function surface
|
||||
|
||||
One table entry per kernel call, C ABI (System V AMD64), names prefixed
|
||||
`danos_`. The current `SystemCall` set maps directly; integer arguments and
|
||||
returns are `u64`, errors return as negative values exactly as today.
|
||||
|
||||
The calls that return two values in `rax:rdx` today — `dma_alloc`
|
||||
(vaddr + paddr), `msi_bind` (address + data), `shm_create` (vaddr + handle) —
|
||||
become functions returning a two-`u64` struct. The System V ABI returns a
|
||||
16-byte struct in `rax:rdx`, so the stub is a plain `syscall; ret` — the
|
||||
C-ABI spelling of the existing convention, at zero cost.
|
||||
|
||||
Grouped as `abi.zig` groups them:
|
||||
|
||||
| Group | Functions |
|
||||
|-------|-----------|
|
||||
| process | `danos_exit`, `danos_yield`, `danos_sleep`, `danos_spawn`, `danos_process_enumerate`, `danos_process_kill`, `danos_process_exit_reason`, `danos_process_subscribe`, `danos_process_signal`, `danos_signal_bind` |
|
||||
| memory | `danos_mmap`, `danos_munmap`, `danos_dma_alloc`, `danos_dma_free`, `danos_shm_create`, `danos_shm_map`, `danos_shm_physical` |
|
||||
| ipc | `danos_endpoint_create`, `danos_ipc_register`, `danos_ipc_lookup`, `danos_ipc_call`, `danos_ipc_reply_wait`, `danos_ipc_send` |
|
||||
| devices | `danos_device_enumerate`, `danos_device_claim`, `danos_device_register`, `danos_mmio_map`, `danos_irq_bind`, `danos_irq_ack`, `danos_msi_bind`, `danos_io_read`, `danos_io_write` |
|
||||
| time | `danos_clock`, `danos_wall_clock`, `danos_timer_bind` |
|
||||
| diagnostics | `danos_debug_write`, `danos_klog_read` |
|
||||
|
||||
The constants that ride alongside the calls — mmap protection bits, DMA
|
||||
flags, notification badge bits, `ExitReason`, `Signal`, well-known service
|
||||
ids, `page_size`, the IPC message maximum — move to the public header too:
|
||||
they are wire values a Rust program needs verbatim. What stays private in
|
||||
`abi.zig` is exactly the thing the vDSO exists to hide: the `SystemCall`
|
||||
numbers and the trap convention.
|
||||
|
||||
## Enforcement, and an honest threat model
|
||||
|
||||
Renumbering only has teeth if the kernel **refuses syscalls that don't come
|
||||
from the vDSO**. The check is cheap: on kernel entry, the saved user `rip`
|
||||
must lie inside the calling process's vDSO mapping; otherwise the process is
|
||||
killed with a fault-class exit reason (its supervisor restarts or gives up,
|
||||
docs/process-lifecycle.md — a foreign-syscall attempt is a bug or an attack,
|
||||
never something to limp past). Fuchsia enforces exactly this.
|
||||
|
||||
What this buys, precisely:
|
||||
|
||||
- **ABI freedom** — the real prize. The numbers can change per release or per
|
||||
boot and nothing outside the kernel image cares. The private ABI stays
|
||||
actually private, permanently.
|
||||
- **A single audited chokepoint** for kernel entry, per process, at a
|
||||
randomised address.
|
||||
- **Raised bar for exploits**: shellcode can't issue a hard-coded `syscall`;
|
||||
it must first discover the per-process vDSO base (ASLR) and call through
|
||||
it.
|
||||
|
||||
What it does *not* buy: an attacker with arbitrary code execution in a
|
||||
process can still *call* the vDSO functions — they are mapped executable in
|
||||
that process, and return-oriented chains reach them. Syscall randomisation is
|
||||
hardening, not a security boundary; the security boundary remains the
|
||||
capability model (what the process's endpoints and device claims let it do).
|
||||
It is worth building anyway — for the ABI freedom first and the hardening
|
||||
second — but the design should never be sold as more than that.
|
||||
|
||||
## Migration
|
||||
|
||||
Phased so every step ships alone (the M-milestone discipline):
|
||||
|
||||
1. **The blob + the table.** Build the vDSO, map it at spawn, deliver the
|
||||
base via auxv. `runtime.system-call.zig` binds through the table when the
|
||||
auxv entry is present, falls back to raw `syscall` when absent — the whole
|
||||
tree keeps booting during the transition.
|
||||
2. **Cut the runtime over.** Delete the raw stubs; `runtime` no longer
|
||||
imports the `SystemCall` numbers at all (`abi.zig`'s enum becomes
|
||||
kernel-internal). The QEMU suite passing proves the table carries the
|
||||
whole system.
|
||||
3. **Enforce + randomise.** Add the `rip`-range check, then per-boot number
|
||||
randomisation patched into the blob at kernel init. A test boots with
|
||||
randomisation on and runs the full suite.
|
||||
4. **The other languages.** Publish `danos.h`; a Rust `danos-sys` crate wraps
|
||||
the table. This is also the seam `std.os.danos` calls through when the Zig
|
||||
self-hosting fork lands (docs/zig-self-hosting.md) — the vDSO is what
|
||||
makes that seam stable across kernel versions.
|
||||
|
||||
## What deliberately stays out
|
||||
|
||||
- **No dynamic linker, no `/lib/*.so`.** The vDSO is kernel-injected precisely
|
||||
so danos binaries can stay fully static above it. Sharing *library code*
|
||||
across processes stays what it is today: a service behind IPC, or source
|
||||
compiled into each binary.
|
||||
- **No file/device I/O in the vDSO.** The microkernel line doesn't move: the
|
||||
vDSO wraps the same deliberately tiny table (docs/syscall.md); files are
|
||||
still the VFS server's business over IPC.
|
||||
- **No fast-path user-mode implementations yet.** Linux's vDSO exists mostly
|
||||
to answer `gettimeofday` without a kernel entry. `danos_clock` could one
|
||||
day read the calibrated TSC in user mode the same way — the blob is where
|
||||
such an optimisation would live — but that is an optimisation, not part of
|
||||
this design's contract.
|
||||
@@ -0,0 +1,170 @@
|
||||
# The VFS wire protocol
|
||||
|
||||
> **Status:** built and spoken today between `runtime.fs` (the client) and the
|
||||
> VFS server (`system/services/vfs`), with mounted backends (the FAT server)
|
||||
> speaking the same protocol behind the router. The Zig source of truth is
|
||||
> `system/services/vfs/protocol.zig` (the `vfs-protocol` module), whose unit
|
||||
> tests pin the sizes and values below. This page is the **language-neutral
|
||||
> wire specification** of that contract — what a Rust or C client implements
|
||||
> ([vdso.md](vdso.md) explains why the IPC protocols, not the syscall
|
||||
> numbers, are danos's public ABI).
|
||||
|
||||
## Transport
|
||||
|
||||
A VFS exchange is one synchronous IPC rendezvous (`ipc_call`,
|
||||
docs/ipc.md): the client sends one message and blocks; the server replies
|
||||
with one message. The endpoint is found by well-known service id
|
||||
(`ipc_lookup`, service id **1** = vfs).
|
||||
|
||||
- A message is at most **256 bytes** (`message_maximum`).
|
||||
- A request is a fixed 32-byte **Request** header followed by an inline
|
||||
payload of at most **224 bytes** (`maximum_payload`) — a path, or write
|
||||
bytes. There is no multi-message request: paths and single reads/writes
|
||||
must fit, and larger transfers loop (see *read* / *write*).
|
||||
- A reply is a fixed 24-byte **Reply** header followed by an inline payload —
|
||||
read bytes, a `FileStatus`, or a `DirectoryEntry`.
|
||||
- All integers are **little-endian**; layouts are C layout for x86-64
|
||||
(`extern struct`), offsets given below so nothing need be inferred.
|
||||
|
||||
The kernel never parses any of this — it only moves the bytes
|
||||
(docs/syscall.md); files are entirely a user-space affair.
|
||||
|
||||
## Request header — 32 bytes
|
||||
|
||||
| offset | size | field | meaning |
|
||||
|-------:|-----:|-------|---------|
|
||||
| 0 | 4 | `operation` | an **Operation** value (below) |
|
||||
| 4 | 4 | — | padding |
|
||||
| 8 | 8 | `node` | the server-side open-node id from a prior `open`; 0 for path-based operations |
|
||||
| 16 | 8 | `offset` | byte position for read/write; entry index (cursor) for readdir; else 0 |
|
||||
| 24 | 4 | `len` | payload length for path/write operations; requested byte count for read |
|
||||
| 28 | 4 | `flags` | open flags (below); else 0 |
|
||||
|
||||
## Reply header — 24 bytes
|
||||
|
||||
| offset | size | field | meaning |
|
||||
|-------:|-----:|-------|---------|
|
||||
| 0 | 4 | `status` | **0 = success**, negative = failure (signed) |
|
||||
| 4 | 4 | — | padding |
|
||||
| 8 | 8 | `node` | the new open-node id (for `open`); else 0 |
|
||||
| 16 | 4 | `len` | reply payload length in bytes |
|
||||
| 20 | 4 | — | padding |
|
||||
|
||||
On failure the router replies `status = -1`; a mounted backend's negative
|
||||
status is forwarded to the client verbatim. A richer errno vocabulary is
|
||||
future work — clients must treat *any* negative status as failure, not match
|
||||
on -1.
|
||||
|
||||
## Operations
|
||||
|
||||
Values are append-only and never renumbered (the same evolution rule every
|
||||
danos protocol follows); an unrecognised operation gets a `status = -1`
|
||||
reply.
|
||||
|
||||
| value | operation | request payload | reply |
|
||||
|------:|-----------|-----------------|-------|
|
||||
| 0 | `open` | the path (`len` = its length), `flags` as below | `node` = open-node id |
|
||||
| 1 | `close` | — (`node` set) | status only |
|
||||
| 2 | `read` | — (`node`, `offset`, `len` = wanted count) | `len` bytes read, payload = the bytes; `len` 0 at end of file |
|
||||
| 3 | `write` | the bytes (`node`, `offset`, `len` = count) | `len` = bytes accepted (may be short — loop) |
|
||||
| 4 | `status` | — (`node` set) | payload = **FileStatus** (24 bytes) |
|
||||
| 5 | `readdir` | — (`node` = a directory, `offset` = cursor) | payload = one **DirectoryEntry** + name; `len` 0 at end |
|
||||
| 6 | `mount` | the mount-point path; the backend endpoint rides as the call's **capability** | status only |
|
||||
| 7 | `unmount` | the mount-point path | status only |
|
||||
| 8 | `mkdir` | the path | status only |
|
||||
| 9 | `unlink` | the path | status only |
|
||||
| 10 | `rename` | old path, one `0x00`, new path (`len` = total) | status only |
|
||||
|
||||
Notes per operation:
|
||||
|
||||
- **open** — paths are absolute (`/mnt/usb/notes.txt`) or bare names
|
||||
(`greeting`); bare names resolve in the VFS's flat ramfs, absolute paths
|
||||
route through the mount table (below). The returned `node` is an id in the
|
||||
*router's* open table; clients never see a backend's own ids.
|
||||
- **read / write** — a single exchange moves at most 224 bytes
|
||||
(`maximum_payload`); the client loops, advancing `offset` by the returned
|
||||
`len`, until done (read) or the slice is written (write). A `write` reply
|
||||
shorter than requested is progress, not an error; a `len` of 0 means no
|
||||
forward progress — stop rather than spin.
|
||||
- **readdir** — `offset` is a **cursor: the entry index**, not a byte
|
||||
position. Each call returns exactly one entry; the client increments the
|
||||
cursor by 1. A reply with `len` 0 is end-of-directory. The directory must
|
||||
have been opened with the `directory` flag.
|
||||
- **mount** — the one operation that passes a **capability**: the caller
|
||||
(a filesystem server, e.g. FAT) sends its own request endpoint as the
|
||||
`ipc_call` capability argument, and the router forwards everything under
|
||||
the mount point to it — speaking this same protocol, with paths rewritten
|
||||
relative to the mount. Prefixes match at path boundaries only
|
||||
(`/mnt/usb` never captures `/mnt/usbextra`); the longest matching prefix
|
||||
wins.
|
||||
- **rename** — same-directory rename only (the router requires old and new to
|
||||
resolve under one mount).
|
||||
|
||||
## Open flags
|
||||
|
||||
Bitwise OR in `Request.flags`, meaningful for `open` only:
|
||||
|
||||
| bit | name | meaning |
|
||||
|----:|------|---------|
|
||||
| 1 | `create` | create the file if it does not exist |
|
||||
| 2 | `directory` | open a directory node for `readdir` rather than a file |
|
||||
| 4 | `truncate` | truncate an existing file to zero length on open (replace, don't overwrite in place) |
|
||||
|
||||
## FileStatus — 24 bytes (the `status` reply payload)
|
||||
|
||||
| offset | size | field | meaning |
|
||||
|-------:|-----:|-------|---------|
|
||||
| 0 | 8 | `size` | file size in bytes |
|
||||
| 8 | 4 | `kind` | a **NodeKind** value |
|
||||
| 12 | 4 | — | padding |
|
||||
| 16 | 8 | `mtime` | modification time, Unix epoch seconds UTC; 0 if the backend keeps none |
|
||||
|
||||
## DirectoryEntry — 16 bytes + name (the `readdir` reply payload)
|
||||
|
||||
| offset | size | field | meaning |
|
||||
|-------:|-----:|-------|---------|
|
||||
| 0 | 4 | `kind` | a **NodeKind** value |
|
||||
| 4 | 4 | `name_len` | length of the name that follows |
|
||||
| 8 | 8 | `size` | the entry's size in bytes |
|
||||
| 16 | `name_len` | name | the entry's name, not NUL-terminated |
|
||||
|
||||
## NodeKind
|
||||
|
||||
Aligned to the FSH file-type table
|
||||
(docs/danos-file-system-hierarchy-FSH.md):
|
||||
|
||||
| value | kind |
|
||||
|------:|------|
|
||||
| 0 | regular file |
|
||||
| 1 | directory |
|
||||
| 2 | character device |
|
||||
| 3 | block device |
|
||||
| 4 | symbolic link |
|
||||
| 5 | fifo |
|
||||
| 6 | socket |
|
||||
|
||||
Clients should map unknown values to *regular* rather than reject — the
|
||||
table can grow.
|
||||
|
||||
## Lifetimes and trust
|
||||
|
||||
Open-node ids live in the server. A client that dies without closing leaks
|
||||
nothing permanently: the VFS subscribes to the kernel's published process-exit
|
||||
events (docs/process-lifecycle.md) and releases a dead client's handles,
|
||||
closing forwarded backend nodes best-effort. Ids are plain integers, not
|
||||
capabilities — the VFS trusts its callers with each other's ids today, which
|
||||
is acceptable while every client is part of the system image and worth
|
||||
revisiting (per-client id namespaces) before third-party binaries arrive.
|
||||
|
||||
## Evolution rules
|
||||
|
||||
What a non-Zig implementation may rely on, and what it must not:
|
||||
|
||||
- Operation values, flag bits, `NodeKind` values, and struct layouts are
|
||||
**append-only and frozen once shipped** — the unit tests in `protocol.zig`
|
||||
pin them exactly so a refactor can't silently move them.
|
||||
- The 256-byte message ceiling is a property of the current IPC transport,
|
||||
not a promise; clients should read `maximum_payload`-shaped limits from the
|
||||
reply lengths they actually get (loop-until-done), not hard-code 224.
|
||||
- Negative statuses beyond -1 will appear (an errno vocabulary); success is
|
||||
exactly 0.
|
||||
@@ -59,6 +59,33 @@ pub fn present() bool {
|
||||
return transact(.{ .operation = @intFromEnum(protocol.Operation.present) }, &reply);
|
||||
}
|
||||
|
||||
/// One selectable display mode.
|
||||
pub const Mode = protocol.Mode;
|
||||
|
||||
/// Fill `out` with the resolutions the display can switch to; returns how many were written
|
||||
/// (zero on the GOP floor, or if the service never came up).
|
||||
pub fn modes(out: []Mode) usize {
|
||||
const h = service() orelse return 0;
|
||||
var request = protocol.Request{ .operation = @intFromEnum(protocol.Operation.get_modes) };
|
||||
var reply: [protocol.modes_reply_size]u8 = undefined;
|
||||
const len = ipc.call(h, std.mem.asBytes(&request), &reply) catch return 0;
|
||||
if (len < protocol.modes_reply_size) return 0;
|
||||
const answer = std.mem.bytesToValue(protocol.ModesReply, reply[0..protocol.modes_reply_size]);
|
||||
if (answer.status != 0) return 0;
|
||||
const count = @min(@min(answer.count, protocol.max_modes), out.len);
|
||||
for (0..count) |i| out[i] = answer.modes[i];
|
||||
return count;
|
||||
}
|
||||
|
||||
/// Change the display resolution. Only a native backend that supports mode-setting honours it
|
||||
/// (on the GOP floor it returns false); on success the display's `info()` reports the new mode.
|
||||
pub fn setMode(width: u32, height: u32) bool {
|
||||
var reply: protocol.Reply = undefined;
|
||||
const changed = transact(.{ .operation = @intFromEnum(protocol.Operation.set_mode), .width = width, .height = height }, &reply);
|
||||
if (changed) mode = null; // the cached mode is stale now
|
||||
return changed;
|
||||
}
|
||||
|
||||
/// The mode, cached after the first `info()` so `color()` doesn't round-trip per pixel.
|
||||
var mode: ?Info = null;
|
||||
|
||||
|
||||
@@ -0,0 +1,19 @@
|
||||
//! The root module every user binary is compiled through (build.zig,
|
||||
//! `addUserBinary`). The program's own file is imported as `program`, and this
|
||||
//! shim contributes the declarations Zig resolves from the compilation root —
|
||||
//! `main` (dispatched by runtime.start) and the panic handler — and pulls in the
|
||||
//! `_start` entry shim. A program therefore only defines `pub fn main`; nothing
|
||||
//! else is required in its source file.
|
||||
|
||||
const runtime = @import("runtime");
|
||||
const program = @import("program");
|
||||
|
||||
/// Resolved as `@import("root").main` by runtime.start's comptime dispatch.
|
||||
pub const main = program.main;
|
||||
|
||||
/// The panic handler for every safety check in the image (runtime.start.panic).
|
||||
pub const panic = runtime.panic;
|
||||
|
||||
comptime {
|
||||
_ = &runtime.start._start; // pull the runtime entry shim into the image
|
||||
}
|
||||
@@ -4,12 +4,11 @@
|
||||
//! shim. It is compiled into each binary (inheriting its `.large` code model and
|
||||
//! freestanding target), so all user programs share one implementation.
|
||||
//!
|
||||
//! A user binary needs three lines:
|
||||
//! const runtime = @import("runtime");
|
||||
//! pub const panic = runtime.panic;
|
||||
//! comptime { _ = &runtime.start._start; } // pull the entry shim in
|
||||
//! and a `pub fn main() void` or `pub fn main(init: runtime.process.Init) void`
|
||||
//! (arguments arrive via `init`).
|
||||
//! A user binary only defines a `pub fn main() void` or
|
||||
//! `pub fn main(init: runtime.process.Init) void` (arguments arrive via `init`).
|
||||
//! The panic handler and the `_start` entry pull live in the shared compilation
|
||||
//! root, library/runtime/root.zig, which build.zig wires around every program —
|
||||
//! nothing to declare per source file.
|
||||
|
||||
pub const system = @import("system.zig");
|
||||
/// Monotonic time, delays, and deadlines over the kernel clock/sleep/timer syscalls
|
||||
@@ -38,6 +37,11 @@ pub const device = @import("device.zig");
|
||||
/// DMA-capable memory for drivers: contiguous, pinned, uncacheable buffers.
|
||||
pub const dma = @import("dma.zig");
|
||||
|
||||
/// Shared cacheable memory: create a region + capability, pass the capability to another
|
||||
/// process (an `ipc_call` send_cap), map the same pages there. See library/runtime/shm.zig
|
||||
/// and docs/display-v2.md.
|
||||
pub const shm = @import("shm.zig");
|
||||
|
||||
/// USB class-driver client: open a device on the xHCI bus and drive it
|
||||
/// (control / interrupt / bulk transfers). See library/runtime/usb.zig.
|
||||
pub const usb = @import("usb.zig");
|
||||
@@ -51,18 +55,25 @@ pub const block = @import("block.zig");
|
||||
pub const display = @import("display.zig");
|
||||
/// The display wire protocol (shared with the display service and its clients).
|
||||
pub const display_protocol = @import("display-protocol");
|
||||
/// The scanout wire protocol: the compositor's present channel to a native scanout driver
|
||||
/// (virtio-gpu). See system/services/display/scanout-protocol.zig and docs/display-v2.md.
|
||||
pub const scanout_protocol = @import("scanout-protocol");
|
||||
|
||||
/// The danos-native file API (open/read/write/list over the user-space VFS) — the
|
||||
/// layer danos programs use directly, and where the operations that later become
|
||||
/// `std.os.danos` are staged. See docs/zig-self-hosting.md.
|
||||
pub const fs = @import("fs.zig");
|
||||
|
||||
/// Re-exported so a user binary can `pub const panic = runtime.panic;`.
|
||||
/// Re-exported so the root shim (root.zig) can install it as the panic handler.
|
||||
pub const panic = start.panic;
|
||||
|
||||
/// Process entry types: the `Init` handed to `main`, and its `Arguments`.
|
||||
pub const process = @import("process.zig");
|
||||
|
||||
/// Threads: `runtime.Thread`, std.Thread-shaped, over the private thread ABI
|
||||
/// (docs/threading.md). A binary must be built multi-threaded to spawn.
|
||||
pub const Thread = @import("thread.zig").Thread;
|
||||
|
||||
/// The service harness: one replyWait loop folding requests, signals, and
|
||||
/// notifications into callbacks (docs/process-lifecycle.md).
|
||||
pub const service = @import("service.zig");
|
||||
|
||||
@@ -0,0 +1,57 @@
|
||||
//! User-space shared memory: `shm_create` / `shm_map`. A process creates a shareable,
|
||||
//! zeroed, cacheable RAM region and gets back a pointer plus a **capability handle**; it
|
||||
//! passes that handle to another process as an `ipc_call` send_cap, and the receiver
|
||||
//! `shm_map`s it to map the same physical pages. The kernel primitive under the display
|
||||
//! compositor↔native-driver and app↔compositor surface paths (docs/display-v2.md). The
|
||||
//! generalization of capability passing from endpoints to memory objects.
|
||||
|
||||
const abi = @import("abi");
|
||||
const sc = @import("system-call.zig");
|
||||
const ipc = @import("ipc.zig");
|
||||
|
||||
inline fn failed(r: usize) bool {
|
||||
return r > ~@as(usize, 0) - 4095; // a wrapped -errno lands in the top page
|
||||
}
|
||||
|
||||
/// A shared region: the `ptr` the CPU touches, and the `handle` (a capability) to hand to
|
||||
/// another process as an `ipc_call` send_cap.
|
||||
pub const Region = struct {
|
||||
ptr: [*]u8,
|
||||
handle: ipc.Handle,
|
||||
len: usize,
|
||||
};
|
||||
|
||||
/// Grant `len` bytes (rounded up to whole pages) of shareable, zeroed, cacheable RAM.
|
||||
/// Returns the region or null on failure. Two return values — vaddr in rax, handle in rdx —
|
||||
/// so this is a hand-written stub like `dma.alloc`.
|
||||
pub fn create(len: usize) ?Region {
|
||||
var rax: usize = undefined;
|
||||
var rdx: usize = undefined; // out: the capability handle
|
||||
asm volatile ("syscall"
|
||||
: [rax] "={rax}" (rax),
|
||||
[rdx] "={rdx}" (rdx),
|
||||
: [n] "{rax}" (@intFromEnum(abi.SystemCall.shm_create)),
|
||||
[a0] "{rdi}" (len),
|
||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||
if (failed(rax)) return null;
|
||||
return .{ .ptr = @ptrFromInt(rax), .handle = rdx, .len = len };
|
||||
}
|
||||
|
||||
/// Map the shared region named by a capability `handle` this process received (via an
|
||||
/// `ipc_call` send_cap) into its address space — the same physical pages the creator sees.
|
||||
/// Returns the pointer, or null on failure.
|
||||
pub fn map(handle: ipc.Handle) ?[*]u8 {
|
||||
const r = sc.systemCall1(.shm_map, handle);
|
||||
if (failed(r)) return null;
|
||||
return @ptrFromInt(r);
|
||||
}
|
||||
|
||||
/// The guest-physical base of the shared region named by `handle` (which this process must
|
||||
/// hold a capability for). The region's frames are contiguous, so this single address plus
|
||||
/// the region length is all a device needs — e.g. a virtio-gpu driver programming an
|
||||
/// `attach_backing`. Returns null on failure.
|
||||
pub fn physical(handle: ipc.Handle) ?usize {
|
||||
const r = sc.systemCall1(.shm_physical, handle);
|
||||
if (failed(r)) return null;
|
||||
return r;
|
||||
}
|
||||
@@ -1,6 +1,7 @@
|
||||
//! The user-space process entry shim. Every user binary roots `_start` here (via
|
||||
//! `entry = _start` in build.zig) and forces this file to be analysed with
|
||||
//! `comptime { _ = &runtime.start._start; }`, so the whole runtime is linked in.
|
||||
//! `entry = _start` in build.zig); the shared compilation root, root.zig, forces
|
||||
//! this file to be analysed with `comptime { _ = &runtime.start._start; }`, so
|
||||
//! the whole runtime is linked in.
|
||||
|
||||
const std = @import("std");
|
||||
const system = @import("system.zig");
|
||||
@@ -36,7 +37,7 @@ export fn rt_start(stack: [*]const u64) callconv(.c) noreturn {
|
||||
/// Comptime-dispatch on root.main's signature, in the spirit of std's start.zig:
|
||||
/// zero parameters or one `process.Init`; returns void, noreturn, u8, !void, or !u8.
|
||||
fn callMain(init: process.Init) u8 {
|
||||
const root = @import("root"); // the user binary's root source file
|
||||
const root = @import("root"); // root.zig, re-exporting the program's main
|
||||
const main_information = @typeInfo(@TypeOf(root.main)).@"fn";
|
||||
|
||||
const call_arguments = switch (main_information.params.len) {
|
||||
|
||||
@@ -0,0 +1,260 @@
|
||||
//! `runtime.Thread` — threads for danos, shaped like Zig's `std.Thread` but built on
|
||||
//! danos's private thread ABI (docs/threading.md). Several tasks share one address
|
||||
//! space; `spawn` starts one, the kernel delivers the closure pointer in the new
|
||||
//! thread's rdi, a plain Zig trampoline runs the user function and calls `thread_exit`,
|
||||
//! and `join` blocks on the thread's exit notification. See docs/threading.md for why
|
||||
//! this mirrors `std.Thread`'s API rather than being the literal type.
|
||||
//!
|
||||
//! The closure (the function's captured args) lives at the **top of the thread's own
|
||||
//! stack**, not the heap — each thread's stack is private, so there is no shared-heap
|
||||
//! concurrency in the spawn/join machinery (the runtime heap is not yet thread-safe).
|
||||
//! A binary must be built multi-threaded (`addThreadedUserBinary`) before it may spawn.
|
||||
|
||||
const std = @import("std");
|
||||
const abi = @import("abi");
|
||||
const sc = @import("system-call.zig");
|
||||
const system = @import("system.zig");
|
||||
const ipc = @import("ipc.zig");
|
||||
|
||||
/// A thread stack, if the caller does not override it. 64 KiB of mmap'd, zeroed pages.
|
||||
pub const default_stack_size: usize = 64 * 1024;
|
||||
|
||||
pub const Thread = struct {
|
||||
/// The kernel task id of the spawned thread.
|
||||
tid: u32,
|
||||
/// The endpoint the kernel notifies when this thread ends — what `join` blocks on.
|
||||
exit_endpoint: ipc.Handle,
|
||||
/// The mmap'd stack, reclaimed by `join` (or at process exit after `detach`).
|
||||
stack_base: usize,
|
||||
stack_size: usize,
|
||||
|
||||
pub const Id = u32;
|
||||
|
||||
pub const SpawnConfig = struct {
|
||||
/// Bytes of stack, rounded up to whole pages by the kernel's mmap.
|
||||
stack_size: usize = default_stack_size,
|
||||
};
|
||||
|
||||
pub const SpawnError = error{
|
||||
/// The kernel refused the thread, the stack mmap failed, or no endpoint was free.
|
||||
SystemResources,
|
||||
};
|
||||
|
||||
/// Start `function(args...)` on a new thread sharing this address space. Mirrors
|
||||
/// `std.Thread.spawn`. The thread's return value is discarded (as in `std.Thread`);
|
||||
/// return data through shared state.
|
||||
pub fn spawn(config: SpawnConfig, comptime function: anytype, args: anytype) SpawnError!Thread {
|
||||
const Args = @TypeOf(args);
|
||||
const Closure = struct {
|
||||
args: Args,
|
||||
/// Entered directly by the kernel with `self` in rdi (C ABI). Runs the user
|
||||
/// function, then ends the thread — never returns.
|
||||
fn entry(self_addr: usize) callconv(.c) noreturn {
|
||||
const self: *@This() = @ptrFromInt(self_addr);
|
||||
@call(.auto, function, self.args);
|
||||
exitThread();
|
||||
}
|
||||
};
|
||||
|
||||
// The endpoint the kernel posts this thread's exit notification to.
|
||||
const endpoint = ipc.createIpcEndpoint() orelse return error.SystemResources;
|
||||
|
||||
const base = system.mmap(config.stack_size, system.PROT_READ | system.PROT_WRITE);
|
||||
if (system.mmapFailed(base)) return error.SystemResources;
|
||||
|
||||
// Lay the closure at the very top of the thread's own stack, then start the
|
||||
// thread's rsp just below it (16-aligned minus 8, the alignment a `call` leaves
|
||||
// for a C-ABI entry) so the growing stack never overwrites the args.
|
||||
var closure_addr = (base + config.stack_size) - @sizeOf(Closure);
|
||||
closure_addr &= ~@as(usize, @alignOf(Closure) - 1); // align the closure down
|
||||
const closure: *Closure = @ptrFromInt(closure_addr);
|
||||
closure.* = .{ .args = args };
|
||||
|
||||
var stack_top = closure_addr & ~@as(usize, 15); // 16-align below the closure
|
||||
stack_top -= 8; // ...then rsp % 16 == 8 at the C entry
|
||||
|
||||
const tid = threadSpawn(@intFromPtr(&Closure.entry), stack_top, closure_addr, endpoint);
|
||||
if (threadSpawnFailed(tid)) {
|
||||
_ = system.munmap(base, config.stack_size);
|
||||
return error.SystemResources;
|
||||
}
|
||||
return .{ .tid = @intCast(tid), .exit_endpoint = endpoint, .stack_base = base, .stack_size = config.stack_size };
|
||||
}
|
||||
|
||||
/// Block until this thread finishes, then reclaim its stack. Mirrors
|
||||
/// `std.Thread.join`. The exit endpoint is private to this thread, so the first
|
||||
/// child-exit notification on it is this thread's.
|
||||
pub fn join(self: Thread) void {
|
||||
var receive: [0]u8 = undefined;
|
||||
while (true) {
|
||||
const got = ipc.replyWait(self.exit_endpoint, &.{}, &receive, null);
|
||||
if (got.isChildExit() and got.childProcessId() == self.tid) break;
|
||||
}
|
||||
_ = system.munmap(self.stack_base, self.stack_size);
|
||||
}
|
||||
|
||||
/// Relinquish the right to join: never wait for or reclaim this thread. Its stack is
|
||||
/// reclaimed at process exit (docs/threading-plan.md M3 — kernel-reaper stack reclaim
|
||||
/// for detached threads is a later refinement). Mirrors `std.Thread.detach`.
|
||||
pub fn detach(self: Thread) void {
|
||||
_ = self;
|
||||
}
|
||||
|
||||
/// The calling thread's id (its kernel task id). Mirrors `std.Thread.getCurrentId`.
|
||||
pub fn getCurrentId() Id {
|
||||
return @intCast(sc.systemCall0(.thread_self));
|
||||
}
|
||||
|
||||
/// The dense 0-based index of the core the calling thread is running on. A danos
|
||||
/// extension beyond `std.Thread`, used to observe genuine cross-core parallelism.
|
||||
pub fn currentCore() Id {
|
||||
return @intCast(sc.systemCall0(.current_core));
|
||||
}
|
||||
|
||||
/// `std.Thread.Futex`-shaped block/wake on a `u32` atomic — the primitive the
|
||||
/// blocking `Mutex`/`Condition`/`Semaphore` are built on. Waiters park in the
|
||||
/// kernel (no busy-wait), so an idle core still halts (docs/halting.md).
|
||||
pub const Futex = struct {
|
||||
/// Block while `ptr.* == expect`. Returns when woken by `wake`, or promptly if
|
||||
/// the value already differs (safe against spurious returns, as in std): the
|
||||
/// caller re-checks its condition in a loop.
|
||||
pub fn wait(ptr: *const std.atomic.Value(u32), expect: u32) void {
|
||||
_ = futexWait(@intFromPtr(ptr), expect, 0);
|
||||
}
|
||||
|
||||
/// As `wait`, but returns `error.Timeout` if `timeout_ns` elapses first.
|
||||
pub fn timedWait(ptr: *const std.atomic.Value(u32), expect: u32, timeout_ns: u64) error{Timeout}!void {
|
||||
if (futexWait(@intFromPtr(ptr), expect, timeout_ns) == abi.futex_timed_out) return error.Timeout;
|
||||
}
|
||||
|
||||
/// Wake up to `max_waiters` threads blocked on `ptr`.
|
||||
pub fn wake(ptr: *const std.atomic.Value(u32), max_waiters: u32) void {
|
||||
_ = futexWake(@intFromPtr(ptr), max_waiters);
|
||||
}
|
||||
};
|
||||
|
||||
/// A mutual-exclusion lock, `std.Thread.Mutex`-shaped. The classic three-state
|
||||
/// futex mutex (unlocked / locked / contended): the fast path is a single CAS, and
|
||||
/// only a contended lock ever enters the kernel.
|
||||
pub const Mutex = struct {
|
||||
state: std.atomic.Value(u32) = std.atomic.Value(u32).init(unlocked),
|
||||
|
||||
const unlocked: u32 = 0;
|
||||
const locked: u32 = 1;
|
||||
const contended: u32 = 2;
|
||||
|
||||
/// Try to take the lock without blocking; returns whether it was acquired.
|
||||
pub fn tryLock(m: *Mutex) bool {
|
||||
return m.state.cmpxchgStrong(unlocked, locked, .acquire, .monotonic) == null;
|
||||
}
|
||||
|
||||
/// Acquire the lock, blocking in the kernel while it is contended.
|
||||
pub fn lock(m: *Mutex) void {
|
||||
if (m.state.cmpxchgStrong(unlocked, locked, .acquire, .monotonic) != null) m.lockSlow();
|
||||
}
|
||||
|
||||
fn lockSlow(m: *Mutex) void {
|
||||
@branchHint(.cold);
|
||||
// Mark the lock contended and take it as soon as it falls unlocked; park on
|
||||
// the futex while it stays contended. Marking contended may cause a spurious
|
||||
// wake on unlock (harmless), never a missed one.
|
||||
while (m.state.swap(contended, .acquire) != unlocked) {
|
||||
Futex.wait(&m.state, contended);
|
||||
}
|
||||
}
|
||||
|
||||
/// Release the lock; wake one waiter if the lock was contended.
|
||||
pub fn unlock(m: *Mutex) void {
|
||||
if (m.state.swap(unlocked, .release) == contended) Futex.wake(&m.state, 1);
|
||||
}
|
||||
};
|
||||
|
||||
/// A condition variable, `std.Thread.Condition`-shaped. Spurious wakeups are
|
||||
/// allowed — always wait in a predicate loop with the mutex held. Built on a futex
|
||||
/// sequence counter: a waiter samples the seq, drops the mutex, and parks until the
|
||||
/// seq changes (a signal that races the unlock bumps the seq, so it is not missed).
|
||||
pub const Condition = struct {
|
||||
seq: std.atomic.Value(u32) = std.atomic.Value(u32).init(0),
|
||||
|
||||
/// Atomically release `mutex` and block until signalled, then re-acquire it.
|
||||
pub fn wait(c: *Condition, mutex: *Mutex) void {
|
||||
const seq = c.seq.load(.acquire);
|
||||
mutex.unlock();
|
||||
Futex.wait(&c.seq, seq);
|
||||
mutex.lock();
|
||||
}
|
||||
|
||||
/// As `wait`, but returns `error.Timeout` if `timeout_ns` elapses first. The
|
||||
/// mutex is re-acquired either way.
|
||||
pub fn timedWait(c: *Condition, mutex: *Mutex, timeout_ns: u64) error{Timeout}!void {
|
||||
const seq = c.seq.load(.acquire);
|
||||
mutex.unlock();
|
||||
const timed_out = if (Futex.timedWait(&c.seq, seq, timeout_ns)) |_| false else |_| true;
|
||||
mutex.lock();
|
||||
if (timed_out) return error.Timeout;
|
||||
}
|
||||
|
||||
/// Wake one waiter.
|
||||
pub fn signal(c: *Condition) void {
|
||||
_ = c.seq.fetchAdd(1, .release);
|
||||
Futex.wake(&c.seq, 1);
|
||||
}
|
||||
|
||||
/// Wake all waiters.
|
||||
pub fn broadcast(c: *Condition) void {
|
||||
_ = c.seq.fetchAdd(1, .release);
|
||||
Futex.wake(&c.seq, std.math.maxInt(u32));
|
||||
}
|
||||
};
|
||||
|
||||
/// A counting semaphore, `std.Thread.Semaphore`-shaped: a permit count guarded by a
|
||||
/// `Mutex` + `Condition`.
|
||||
pub const Semaphore = struct {
|
||||
mutex: Mutex = .{},
|
||||
cond: Condition = .{},
|
||||
permits: usize = 0,
|
||||
|
||||
/// Take a permit, blocking until one is available.
|
||||
pub fn wait(s: *Semaphore) void {
|
||||
s.mutex.lock();
|
||||
defer s.mutex.unlock();
|
||||
while (s.permits == 0) s.cond.wait(&s.mutex);
|
||||
s.permits -= 1;
|
||||
}
|
||||
|
||||
/// Return a permit and wake a waiter.
|
||||
pub fn post(s: *Semaphore) void {
|
||||
s.mutex.lock();
|
||||
defer s.mutex.unlock();
|
||||
s.permits += 1;
|
||||
s.cond.signal();
|
||||
}
|
||||
};
|
||||
};
|
||||
|
||||
/// thread_spawn(entry, stack_top, arg, exit_endpoint) -> tid, or a wrapped error.
|
||||
fn threadSpawn(entry: usize, stack_top: usize, arg: usize, exit_endpoint: ipc.Handle) usize {
|
||||
return sc.systemCall4(.thread_spawn, entry, stack_top, arg, exit_endpoint);
|
||||
}
|
||||
|
||||
/// The kernel returns a real (small) task id on success and a wrapped `-1` on failure;
|
||||
/// no valid task id ever exceeds a u32.
|
||||
inline fn threadSpawnFailed(ret: usize) bool {
|
||||
return ret > std.math.maxInt(u32);
|
||||
}
|
||||
|
||||
/// End the calling thread. Never returns.
|
||||
fn exitThread() noreturn {
|
||||
_ = sc.systemCall0(.thread_exit);
|
||||
unreachable;
|
||||
}
|
||||
|
||||
/// futex_wait(addr, expect, timeout_ns) -> status (abi.futex_*).
|
||||
fn futexWait(addr: usize, expect: u32, timeout_ns: u64) usize {
|
||||
return sc.systemCall3(.futex_wait, addr, expect, timeout_ns);
|
||||
}
|
||||
|
||||
/// futex_wake(addr, count) -> number woken.
|
||||
fn futexWake(addr: usize, count: u32) usize {
|
||||
return sc.systemCall2(.futex_wake, addr, count);
|
||||
}
|
||||
@@ -60,9 +60,23 @@ pub const SystemCall = enum(u64) {
|
||||
timer_bind = 31, // timer_bind(endpoint, ms) -> 0/-errno: one-shot timer — posts a notification when ms elapse
|
||||
klog_read = 32, // klog_read(offset, ptr, len) -> bytes copied: copy the kernel RAM log buffer out to a user buffer (for persisting the boot log to disk)
|
||||
wall_clock = 33, // wall_clock() -> Unix epoch seconds (UTC): the RTC wall-clock time, for filesystem timestamps (mtime). Monotonic time is `clock`.
|
||||
shm_create = 34, // shm_create(len) -> vaddr (rax), handle (rdx): a shareable, zeroed, cacheable RAM region mapped into this AS; the handle is a capability passed to another process as an ipc_call send_cap (docs/display-v2.md)
|
||||
shm_map = 35, // shm_map(cap) -> vaddr: map the shared region named by a received capability into this AS (the same physical pages the creator sees)
|
||||
shm_physical = 36, // shm_physical(cap) -> paddr: the guest-physical base of a shared region held by capability, so a driver can program it into a device (e.g. virtio-gpu attach_backing); the pages are contiguous (docs/display-v2.md)
|
||||
thread_spawn = 37, // thread_spawn(entry, stack_top, arg, exit_endpoint) -> tid: start a task sharing the caller's address space at `entry` on `stack_top`, `arg` in rdi; exit_endpoint (a handle, or no_cap) is notified when it ends — how join waits (docs/threading.md)
|
||||
thread_exit = 38, // thread_exit(): end the calling thread, dropping one reference to its address space (destroyed on the last)
|
||||
current_core = 39, // current_core() -> index: the dense 0-based index of the core the caller is running on (for parallelism/affinity introspection)
|
||||
futex_wait = 40, // futex_wait(addr, expected, timeout_ns) -> status: if *addr == expected, block until woken or the timeout; returns futex_woken/mismatch/timed_out (docs/threading.md)
|
||||
futex_wake = 41, // futex_wake(addr, count) -> woken: wake up to `count` tasks blocked in futex_wait on `addr` in this address space
|
||||
thread_self = 42, // thread_self() -> tid: the calling thread's kernel task id (runtime.Thread.getCurrentId)
|
||||
_,
|
||||
};
|
||||
|
||||
/// `futex_wait` return codes (in rax).
|
||||
pub const futex_woken: u64 = 0; // woken by a futex_wake
|
||||
pub const futex_mismatch: u64 = 1; // *addr != expected on entry; the caller did not block
|
||||
pub const futex_timed_out: u64 = 2; // the timeout elapsed before a wake
|
||||
|
||||
/// How a process ended — recorded by the kernel at death, queried by the
|
||||
/// supervisor with `process_exit_reason`, and the input to its restart decision
|
||||
/// (docs/process-lifecycle.md): a clean exit meant to stop, a fault wants a
|
||||
@@ -184,6 +198,8 @@ pub const ServiceId = enum(u32) {
|
||||
block = 7, // a block-device driver (USB mass storage today): read/write of fixed-size blocks, the storage a filesystem sits on
|
||||
fat = 8, // the FAT filesystem server; the VFS mounts it and forwards paths under its mount point (/mnt/usb) to it
|
||||
display = 9, // the display service: owns the framebuffer, composites a layer stack, presents frames (docs/display.md)
|
||||
shm_test = 10, // the shm test server (V2): a client passes it a shared-memory capability, it maps + verifies (docs/display-v2.md)
|
||||
scanout = 11, // a native scanout driver (virtio-gpu): the compositor finds it here to upgrade off the GOP framebuffer (docs/display-v2.md)
|
||||
_,
|
||||
};
|
||||
|
||||
|
||||
@@ -263,8 +263,3 @@ pub fn main(init: runtime.process.Init) void {
|
||||
.on_message = onMessage,
|
||||
});
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start; // pull the runtime entry shim into the image
|
||||
}
|
||||
|
||||
@@ -177,8 +177,3 @@ pub fn main(init: runtime.process.Init) void {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start;
|
||||
}
|
||||
|
||||
@@ -60,7 +60,7 @@ pub fn main(init: runtime.process.Init) void {
|
||||
_ = runtime.system.write("/system/drivers/ps2-bus/mouse: out of memory\n");
|
||||
return;
|
||||
};
|
||||
if (device.findDeviceDescriptorByHid(buffer, hid) == null) {
|
||||
if (ps2.findMouseDescriptor(buffer) == null) {
|
||||
writeLine("/system/drivers/ps2-bus/mouse: no device for hid {s}\n", .{hid});
|
||||
return;
|
||||
}
|
||||
@@ -137,8 +137,3 @@ pub fn main(init: runtime.process.Init) void {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start;
|
||||
}
|
||||
|
||||
@@ -244,7 +244,7 @@ pub fn main() void {
|
||||
// to the *port*, whatever device identify found on it.
|
||||
var maybe_auxiliary_interrupt: ?struct { device_id: u64, interrupt_index: u64, gsi: u64 } = null;
|
||||
if (port_device_types[@intFromEnum(ps2.Port.two)] != null) {
|
||||
if (device.findDeviceDescriptorByHid(buffer, acpi_ids.HardwareId.ps2_mouse.hid())) |descriptor| {
|
||||
if (ps2.findMouseDescriptor(buffer)) |descriptor| {
|
||||
if (findInterruptResourceIndex(descriptor)) |auxiliary_index| {
|
||||
if (device.claim(descriptor.id) and device.irqBind(descriptor.id, auxiliary_index, endpoint)) {
|
||||
maybe_auxiliary_interrupt = .{
|
||||
@@ -308,8 +308,3 @@ pub fn main() void {
|
||||
reply_len = handleAttach(receive[0..got.len], got, &reply_buffer);
|
||||
}
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start;
|
||||
}
|
||||
|
||||
@@ -270,6 +270,22 @@ pub const DeviceType = enum(u32) {
|
||||
}
|
||||
};
|
||||
|
||||
/// The `_HID`s a PS/2 pointing device (the controller's aux channel) can enumerate
|
||||
/// under. It is the same 8042 mouse channel whichever id the firmware chose:
|
||||
/// QEMU/OVMF report the generic `.ps2_mouse` (PNP0F13), VirtualBox reports
|
||||
/// `.microsoft_ps2_mouse` (PNP0F03). Both mean "the mouse on port two".
|
||||
pub const mouse_hardware_ids = [_]acpi_ids.HardwareId{ .ps2_mouse, .microsoft_ps2_mouse };
|
||||
|
||||
/// Find the aux (mouse) device's ACPI node, whichever of the PS/2-mouse `_HID`s the
|
||||
/// firmware used — the bus needs it to bind IRQ12, and the mouse driver to confirm
|
||||
/// its device is present. Returns the first match, or null if none is reported.
|
||||
pub fn findMouseDescriptor(buffer: []device.DeviceDescriptor) ?device.DeviceDescriptor {
|
||||
for (mouse_hardware_ids) |id| {
|
||||
if (device.findDeviceDescriptorByHid(buffer, id.hid())) |descriptor| return descriptor;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
// --- the bus <-> child-driver forwarding protocol -----------------------------
|
||||
//
|
||||
// The 8042's ports and IRQ1 live on the PNP0303 node that only the ps2-bus driver
|
||||
|
||||
@@ -167,8 +167,3 @@ pub fn main(init: runtime.process.Init) void {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start;
|
||||
}
|
||||
|
||||
@@ -133,8 +133,3 @@ pub fn main(init: runtime.process.Init) void {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start;
|
||||
}
|
||||
|
||||
@@ -180,8 +180,3 @@ pub fn main(init: runtime.process.Init) void {
|
||||
.on_message = onMessage,
|
||||
});
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start;
|
||||
}
|
||||
|
||||
@@ -417,8 +417,3 @@ pub fn main(init: runtime.process.Init) void {
|
||||
.on_notification = onNotification,
|
||||
});
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start; // pull the runtime entry shim into the image
|
||||
}
|
||||
|
||||
@@ -0,0 +1,139 @@
|
||||
//! The virtio-gpu control protocol — the command/response structs the driver exchanges with
|
||||
//! the device over its control virtqueue (virtio spec, "GPU Device"). `extern` structs, so
|
||||
//! the layout matches the little-endian wire format exactly. Host-tested for size. See
|
||||
//! docs/display-v2.md.
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
/// Control command / response types (virtio_gpu_ctrl_type). Commands are 0x01xx, responses
|
||||
/// 0x11xx (ok) / 0x12xx (error).
|
||||
pub const CmdType = enum(u32) {
|
||||
get_display_info = 0x0100,
|
||||
resource_create_2d = 0x0101,
|
||||
resource_unref = 0x0102,
|
||||
set_scanout = 0x0103,
|
||||
resource_flush = 0x0104,
|
||||
transfer_to_host_2d = 0x0105,
|
||||
resource_attach_backing = 0x0106,
|
||||
resource_detach_backing = 0x0107,
|
||||
get_edid = 0x010a,
|
||||
|
||||
resp_ok_nodata = 0x1100,
|
||||
resp_ok_display_info = 0x1101,
|
||||
resp_ok_edid = 0x1104,
|
||||
resp_err_unspec = 0x1200,
|
||||
_,
|
||||
};
|
||||
|
||||
/// Set in a command's `flags` to request a fence; the device echoes `fence_id` in the
|
||||
/// response and does not report completion until the command's effects are visible.
|
||||
pub const flag_fence: u32 = 1 << 0;
|
||||
|
||||
/// VIRTIO_GPU_F_EDID — device feature bit 1 (the low feature word): the device answers the
|
||||
/// `get_edid` command. Negotiate it only when the device offers it.
|
||||
pub const feature_edid: u32 = 1 << 1;
|
||||
|
||||
/// virtio_gpu_ctrl_hdr — the header on every command and response.
|
||||
pub const CtrlHdr = extern struct {
|
||||
type: u32,
|
||||
flags: u32 = 0,
|
||||
fence_id: u64 = 0,
|
||||
ctx_id: u32 = 0,
|
||||
ring_idx: u8 = 0,
|
||||
padding: [3]u8 = .{ 0, 0, 0 },
|
||||
};
|
||||
|
||||
pub const Rect = extern struct {
|
||||
x: u32,
|
||||
y: u32,
|
||||
width: u32,
|
||||
height: u32,
|
||||
};
|
||||
|
||||
/// 2D pixel formats. QEMU's virtio-gpu host default is B8G8R8X8 (matches our bgrx).
|
||||
pub const format_b8g8r8x8_unorm: u32 = 2;
|
||||
pub const format_r8g8b8x8_unorm: u32 = 134;
|
||||
|
||||
pub const ResourceCreate2d = extern struct {
|
||||
hdr: CtrlHdr,
|
||||
resource_id: u32,
|
||||
format: u32,
|
||||
width: u32,
|
||||
height: u32,
|
||||
};
|
||||
|
||||
/// One scatter-gather entry of a resource's guest backing (a physical span).
|
||||
pub const MemEntry = extern struct {
|
||||
addr: u64,
|
||||
length: u32,
|
||||
padding: u32 = 0,
|
||||
};
|
||||
|
||||
/// Header for RESOURCE_ATTACH_BACKING; `nr_entries` `MemEntry` follow it inline.
|
||||
pub const ResourceAttachBacking = extern struct {
|
||||
hdr: CtrlHdr,
|
||||
resource_id: u32,
|
||||
nr_entries: u32,
|
||||
};
|
||||
|
||||
pub const SetScanout = extern struct {
|
||||
hdr: CtrlHdr,
|
||||
rect: Rect,
|
||||
scanout_id: u32,
|
||||
resource_id: u32,
|
||||
};
|
||||
|
||||
pub const ResourceFlush = extern struct {
|
||||
hdr: CtrlHdr,
|
||||
rect: Rect,
|
||||
resource_id: u32,
|
||||
padding: u32 = 0,
|
||||
};
|
||||
|
||||
/// Copy the guest backing into the host resource for `rect` (2D resources must transfer
|
||||
/// before a flush shows the update).
|
||||
pub const TransferToHost2d = extern struct {
|
||||
hdr: CtrlHdr,
|
||||
rect: Rect,
|
||||
offset: u64,
|
||||
resource_id: u32,
|
||||
padding: u32 = 0,
|
||||
};
|
||||
|
||||
pub const max_scanouts = 16;
|
||||
|
||||
pub const DisplayOne = extern struct {
|
||||
rect: Rect,
|
||||
enabled: u32,
|
||||
flags: u32,
|
||||
};
|
||||
|
||||
pub const RespDisplayInfo = extern struct {
|
||||
hdr: CtrlHdr,
|
||||
pmodes: [max_scanouts]DisplayOne,
|
||||
};
|
||||
|
||||
pub const GetEdid = extern struct {
|
||||
hdr: CtrlHdr,
|
||||
scanout: u32,
|
||||
padding: u32 = 0,
|
||||
};
|
||||
|
||||
pub const RespEdid = extern struct {
|
||||
hdr: CtrlHdr,
|
||||
size: u32,
|
||||
padding: u32 = 0,
|
||||
edid: [1024]u8,
|
||||
};
|
||||
|
||||
test "virtio-gpu struct sizes match the wire layout" {
|
||||
try std.testing.expectEqual(@as(usize, 24), @sizeOf(CtrlHdr));
|
||||
try std.testing.expectEqual(@as(usize, 16), @sizeOf(Rect));
|
||||
try std.testing.expectEqual(@as(usize, 40), @sizeOf(ResourceCreate2d));
|
||||
try std.testing.expectEqual(@as(usize, 16), @sizeOf(MemEntry));
|
||||
try std.testing.expectEqual(@as(usize, 32), @sizeOf(ResourceAttachBacking));
|
||||
try std.testing.expectEqual(@as(usize, 48), @sizeOf(SetScanout));
|
||||
try std.testing.expectEqual(@as(usize, 48), @sizeOf(ResourceFlush));
|
||||
try std.testing.expectEqual(@as(usize, 56), @sizeOf(TransferToHost2d));
|
||||
try std.testing.expectEqual(@as(usize, 24 + 4 + 4 + 1024), @sizeOf(RespEdid));
|
||||
}
|
||||
@@ -0,0 +1,617 @@
|
||||
//! /system/drivers/virtio-gpu — the virtio-gpu (virtio 1.0, modern PCI) display driver.
|
||||
//! The device manager spawns it for the display/other PCI function (class 0x0380) whose
|
||||
//! config space says vendor 0x1AF4 / device 0x1050; this instance claims that device and
|
||||
//! brings up a single 2D scanout.
|
||||
//!
|
||||
//! V3 (this increment): the whole path end to end, proven from serial without a screenshot.
|
||||
//! Claim the function, map its config space (resource 0) and the BAR that carries the
|
||||
//! virtio structures, walk the vendor capabilities to find common-config / notify, reset
|
||||
//! and negotiate VERSION_1, stand up the control virtqueue in DMA memory, then drive the
|
||||
//! GPU: RESOURCE_CREATE_2D → ATTACH_BACKING (a coherent DMA buffer) → SET_SCANOUT, paint a
|
||||
//! known test pattern, TRANSFER_TO_HOST_2D → RESOURCE_FLUSH, and **wait for the device's
|
||||
//! used-ring ack**. Reading the backing back confirms it is CPU-visible; the ack confirms
|
||||
//! the device consumed the frame. The compositor backend, hot-attach, mode-set/EDID, and
|
||||
//! restart/re-attach are V4–V6. See docs/display-v2.md.
|
||||
|
||||
const std = @import("std");
|
||||
const runtime = @import("runtime");
|
||||
const mmio = @import("mmio");
|
||||
const device = runtime.device;
|
||||
const dma = runtime.dma;
|
||||
const shm = runtime.shm;
|
||||
const system = runtime.system;
|
||||
const ipc = runtime.ipc;
|
||||
const dp = runtime.display_protocol;
|
||||
const sp = runtime.scanout_protocol;
|
||||
const dm = runtime.device_manager_protocol;
|
||||
const vp = @import("virtio-pci.zig");
|
||||
const vg = @import("virtio-gpu-protocol.zig");
|
||||
|
||||
/// The DisplayFormat (device-abi) our B8G8R8X8 scanout resource presents: bgrx = 1. Handed to
|
||||
/// the compositor in the announce so it packs colours in the surface's byte order.
|
||||
const display_format_bgrx: u32 = 1;
|
||||
|
||||
/// The PCI vendor/device ids of a modern virtio-gpu (Red Hat / virtio; GPU is a
|
||||
/// virtio-1.0-only device, so the id is always the modern 0x1050 — no legacy variant).
|
||||
const virtio_vendor: u16 = 0x1AF4;
|
||||
const virtio_gpu_device: u16 = 0x1050;
|
||||
|
||||
/// The scanout resource + shared surface are sized to the *largest* mode we offer; a mode
|
||||
/// change (V5) re-points the scanout rectangle within it, so the resource, its backing, and
|
||||
/// the shared surface never churn — and the surface's row stride is always `max_width`, which
|
||||
/// the compositor is told in the announce. Kept modest so the backing is an easy contiguous run.
|
||||
const max_width: u32 = 800;
|
||||
const max_height: u32 = 600;
|
||||
const scanout_bytes: usize = @as(usize, max_width) * max_height * 4;
|
||||
const resource_id: u32 = 1;
|
||||
|
||||
/// The modes this scanout offers (all ≤ max). The first is the mode it comes up in.
|
||||
const Mode = struct { width: u32, height: u32 };
|
||||
const offered_modes = [_]Mode{ .{ .width = 640, .height = 480 }, .{ .width = 800, .height = 600 } };
|
||||
|
||||
/// The active mode — the scanout rectangle within the max-sized surface. Changed by `set_mode`.
|
||||
var current_width: u32 = offered_modes[0].width;
|
||||
var current_height: u32 = offered_modes[0].height;
|
||||
|
||||
/// Monotonic fence id for fenced (vsync) flushes; the device signals the fence when the flush
|
||||
/// is complete, which its used-ring ack already gates our synchronous present on.
|
||||
var fence_next: u64 = 1;
|
||||
|
||||
/// Whether the device offered VIRTIO_GPU_F_EDID, so `get_edid` is worth issuing.
|
||||
var edid_available = false;
|
||||
|
||||
/// The control virtqueue. We drive it synchronously — one command, notify, poll the used
|
||||
/// ring — so a depth of 16 is ample; we ask the device to shrink to it (virtio 1.0 lets the
|
||||
/// driver reduce queue_size), keeping the whole ring inside one page.
|
||||
const queue_size: u16 = 16;
|
||||
const desc_offset: usize = 0; // 16 * 16 = 256 bytes
|
||||
const avail_offset: usize = 256; // flags + idx + ring[16] + used_event = 38 bytes
|
||||
const used_offset: usize = 1024; // flags + idx + ring[16] + avail_event = 134 bytes
|
||||
|
||||
/// The command scratch: the request the device reads, then its response, in one DMA page.
|
||||
const request_offset: usize = 0;
|
||||
const response_offset: usize = 2048;
|
||||
|
||||
var device_id: u64 = 0;
|
||||
|
||||
// Mapped virtio structures (virtual addresses into the device's BAR).
|
||||
var common_base: usize = 0;
|
||||
var notify_base: usize = 0;
|
||||
var notify_multiplier: u32 = 0;
|
||||
var notify_addr: usize = 0;
|
||||
|
||||
// Per-BAR mapping cache: several capabilities usually share one BAR, and mmio_map must not
|
||||
// be asked to map the same resource twice.
|
||||
var bar_virtual: [6]usize = .{ 0, 0, 0, 0, 0, 0 };
|
||||
|
||||
// DMA memory: the virtqueue rings and the command scratch.
|
||||
var ring: dma.Region = undefined;
|
||||
var command: dma.Region = undefined;
|
||||
|
||||
// The scanout backing is a **shared** (shm) region, not DMA: cacheable so the compositor
|
||||
// composites into it cheaply (x86 DMA is coherent, so the device still sees the writes), and
|
||||
// shareable so the same physical pages the device scans out of are the ones the compositor
|
||||
// paints. The driver keeps the capability to hand to the compositor in the announce.
|
||||
var surface: shm.Region = undefined;
|
||||
|
||||
// Split-virtqueue producer/consumer shadows.
|
||||
var avail_shadow: u16 = 0;
|
||||
var used_shadow: u16 = 0;
|
||||
|
||||
/// Format one whole log line and emit it in a single `write`, so this driver's output can
|
||||
/// never interleave mid-line with the other drivers the manager runs concurrently.
|
||||
fn log(comptime fmt: []const u8, arguments: anytype) void {
|
||||
var line: [160]u8 = undefined;
|
||||
_ = system.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
||||
}
|
||||
|
||||
// --- common-config register access (little-endian MMIO at `common_base`) ---------------
|
||||
|
||||
fn cfgRead(comptime T: type, comptime field: []const u8) T {
|
||||
return mmio.read(T, common_base + @offsetOf(vp.CommonCfg, field));
|
||||
}
|
||||
fn cfgWrite(comptime T: type, comptime field: []const u8, value: T) void {
|
||||
mmio.write(T, common_base + @offsetOf(vp.CommonCfg, field), value);
|
||||
}
|
||||
/// Write a 64-bit common-config register as two 32-bit halves (low then high) — the widest
|
||||
/// access every virtio-pci host is required to accept for the queue-address registers.
|
||||
fn cfgWrite64(comptime field: []const u8, value: u64) void {
|
||||
const at = common_base + @offsetOf(vp.CommonCfg, field);
|
||||
mmio.write(u32, at, @truncate(value));
|
||||
mmio.write(u32, at + 4, @truncate(value >> 32));
|
||||
}
|
||||
fn orStatus(bit: u8) void {
|
||||
cfgWrite(u8, "device_status", cfgRead(u8, "device_status") | bit);
|
||||
}
|
||||
|
||||
// --- PCI config-space capability walk (config space is resource 0) ---------------------
|
||||
|
||||
/// Map the BAR numbered `bar` (0..5) and return its virtual base, correlating the BAR's
|
||||
/// physical address (read from config space) with one of our device resources — because a
|
||||
/// virtio capability names a BAR *number*, while `mmio_map` takes a *resource index* (and
|
||||
/// resource 0 is config space, so BAR resources are re-numbered and gaps skipped).
|
||||
fn mapBar(config: usize, descriptor: *const device.DeviceDescriptor, bar: u8) ?usize {
|
||||
if (bar >= 6) return null;
|
||||
if (bar_virtual[bar] != 0) return bar_virtual[bar];
|
||||
|
||||
const low = mmio.read(u32, config + 0x10 + @as(usize, bar) * 4);
|
||||
if (low & 0x1 != 0) return null; // an I/O-space BAR — virtio structures are in memory BARs
|
||||
var base: u64 = low & 0xFFFF_FFF0;
|
||||
if ((low & 0x6) == 0x4) { // 64-bit memory BAR: the high half is the next dword
|
||||
const high = mmio.read(u32, config + 0x10 + (@as(usize, bar) + 1) * 4);
|
||||
base |= @as(u64, high) << 32;
|
||||
}
|
||||
|
||||
for (descriptor.resources[0..@intCast(descriptor.resource_count)], 0..) |resource, index| {
|
||||
if (resource.kind == @intFromEnum(device.ResourceKind.memory) and resource.start == base) {
|
||||
const v = device.mmioMap(device_id, index) orelse return null;
|
||||
bar_virtual[bar] = v;
|
||||
return v;
|
||||
}
|
||||
}
|
||||
log("virtio-gpu: BAR {d} (physical 0x{x}) is not a mapped resource\n", .{ bar, base });
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Walk the PCI capability list from mapped config space, recording the common-config and
|
||||
/// notify structures (the only two V3 needs). Returns false if either is missing.
|
||||
fn walkCapabilities(config: usize, descriptor: *const device.DeviceDescriptor) bool {
|
||||
if (mmio.read(u16, config + 0x06) & 0x10 == 0) { // Status bit 4: capabilities list present
|
||||
log("virtio-gpu: device has no PCI capability list\n", .{});
|
||||
return false;
|
||||
}
|
||||
var cap: u8 = @as(u8, @truncate(mmio.read(u8, config + 0x34))) & 0xFC;
|
||||
var guard: u32 = 0;
|
||||
while (cap != 0 and guard < 48) : (guard += 1) {
|
||||
const at = config + cap;
|
||||
const id = mmio.read(u8, at + 0);
|
||||
const next = mmio.read(u8, at + 1) & 0xFC;
|
||||
// Only map BARs for the structures V3 uses (common + notify). The other virtio
|
||||
// capabilities (isr, device, and especially the cfg_pci back-door, which carries a
|
||||
// placeholder bar=0/offset=0) reference BARs we never touch, so mapping them would
|
||||
// just log spurious "not a mapped resource" noise.
|
||||
if (id == vp.pci_cap_vendor) {
|
||||
const cfg_type = mmio.read(u8, at + 3);
|
||||
if (cfg_type == vp.cfg_common or cfg_type == vp.cfg_notify) {
|
||||
const bar = mmio.read(u8, at + 4);
|
||||
const offset = mmio.read(u32, at + 8);
|
||||
if (mapBar(config, descriptor, bar)) |bar_base| {
|
||||
if (cfg_type == vp.cfg_common) {
|
||||
common_base = bar_base + offset;
|
||||
} else {
|
||||
notify_base = bar_base + offset;
|
||||
notify_multiplier = mmio.read(u32, at + 16); // virtio_pci_notify_cap tail
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
cap = next;
|
||||
}
|
||||
if (common_base == 0 or notify_base == 0) {
|
||||
log("virtio-gpu: missing common-config or notify capability\n", .{});
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
// --- the control virtqueue -------------------------------------------------------------
|
||||
|
||||
/// Publish the two-descriptor chain (request read by the device, response written by it),
|
||||
/// notify the control queue, and wait for the device to return the buffer on the used ring.
|
||||
fn submit(request_len: usize, response_len: usize) bool {
|
||||
const desc: [*]vp.Desc = @ptrFromInt(ring.virtual + desc_offset);
|
||||
desc[0] = .{
|
||||
.addr = command.physical + request_offset,
|
||||
.len = @intCast(request_len),
|
||||
.flags = vp.desc_flag_next,
|
||||
.next = 1,
|
||||
};
|
||||
desc[1] = .{
|
||||
.addr = command.physical + response_offset,
|
||||
.len = @intCast(response_len),
|
||||
.flags = vp.desc_flag_write,
|
||||
.next = 0,
|
||||
};
|
||||
|
||||
const avail_ring: [*]u16 = @ptrFromInt(ring.virtual + avail_offset + 4);
|
||||
avail_ring[avail_shadow % queue_size] = 0; // head of the chain is descriptor 0
|
||||
mmio.wmb();
|
||||
avail_shadow +%= 1;
|
||||
mmio.write(u16, ring.virtual + avail_offset + 2, avail_shadow); // avail.idx
|
||||
mmio.wmb();
|
||||
|
||||
mmio.write(u16, notify_addr, 0); // ring the control queue's doorbell
|
||||
return waitUsed();
|
||||
}
|
||||
|
||||
/// Spin, then sleep-poll, on the used-ring index until the device advances it. QEMU
|
||||
/// processes the notify on its own thread, so the ack usually lands immediately; the sleep
|
||||
/// fallback covers a device that defers it without burning the CPU.
|
||||
fn waitUsed() bool {
|
||||
var tries: u32 = 0;
|
||||
while (tries < 2000) : (tries += 1) {
|
||||
mmio.rmb();
|
||||
const idx = mmio.read(u16, ring.virtual + used_offset + 2); // used.idx
|
||||
if (idx != used_shadow) {
|
||||
used_shadow = idx;
|
||||
return true;
|
||||
}
|
||||
if (tries > 8) system.sleep(1);
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/// The type field of the response the device wrote — `resp_ok_nodata` on success.
|
||||
fn responseType() u32 {
|
||||
const response: *vg.CtrlHdr = @ptrFromInt(command.virtual + response_offset);
|
||||
return response.type;
|
||||
}
|
||||
|
||||
/// Submit a command whose response is a bare header, returning its response type (0 if the
|
||||
/// device never acked).
|
||||
fn command_nodata(request_len: usize) u32 {
|
||||
if (!submit(request_len, @sizeOf(vg.CtrlHdr))) return 0;
|
||||
return responseType();
|
||||
}
|
||||
|
||||
const ok_nodata: u32 = @intFromEnum(vg.CmdType.resp_ok_nodata);
|
||||
|
||||
fn requestAt(comptime T: type) *T {
|
||||
return @ptrFromInt(command.virtual + request_offset);
|
||||
}
|
||||
|
||||
/// A deterministic, recognisable pixel so a read-back is a real check, not a tautology.
|
||||
fn testPixel(index: u32) u32 {
|
||||
return 0xFF00_0000 | (index *% 0x9E37_79B1);
|
||||
}
|
||||
|
||||
// --- bring-up --------------------------------------------------------------------------
|
||||
|
||||
fn initialise(endpoint: ipc.Handle) bool {
|
||||
_ = endpoint;
|
||||
if (!device.claim(device_id)) {
|
||||
log("virtio-gpu: unable to claim device {d}\n", .{device_id});
|
||||
return false;
|
||||
}
|
||||
|
||||
var descriptors: [64]device.DeviceDescriptor = undefined;
|
||||
const total = device.enumerate(&descriptors);
|
||||
const descriptor = for (descriptors[0..@min(total, descriptors.len)]) |*d| {
|
||||
if (d.id == device_id) break d;
|
||||
} else {
|
||||
log("virtio-gpu: device {d} not in the device tree\n", .{device_id});
|
||||
return false;
|
||||
};
|
||||
|
||||
// Config space is resource 0. Confirm it really is a virtio-gpu, then enable memory-space
|
||||
// decode + bus mastering (the device DMAs the ring and backing out of RAM); pci-bus only
|
||||
// preserves whatever the firmware left, and a secondary display is often left disabled.
|
||||
const config = device.mmioMap(device_id, 0) orelse {
|
||||
log("virtio-gpu: config-space map failed\n", .{});
|
||||
return false;
|
||||
};
|
||||
const vendor = mmio.read(u16, config + 0x00);
|
||||
const dev = mmio.read(u16, config + 0x02);
|
||||
if (vendor != virtio_vendor or dev != virtio_gpu_device) {
|
||||
log("virtio-gpu: not a virtio-gpu (vendor 0x{x} device 0x{x})\n", .{ vendor, dev });
|
||||
return false;
|
||||
}
|
||||
mmio.write(u16, config + 0x04, mmio.read(u16, config + 0x04) | 0x06); // MEM + bus master
|
||||
|
||||
if (!walkCapabilities(config, descriptor)) return false;
|
||||
|
||||
// Reset, then the modern feature handshake: acknowledge, take driver ownership, require
|
||||
// VERSION_1 and offer nothing else, and confirm the device accepts that.
|
||||
cfgWrite(u8, "device_status", 0);
|
||||
orStatus(vp.status_acknowledge);
|
||||
orStatus(vp.status_driver);
|
||||
|
||||
// Low feature word (device-specific): note whether the device offers EDID (bit 1).
|
||||
cfgWrite(u32, "device_feature_select", 0);
|
||||
edid_available = cfgRead(u32, "device_feature") & vg.feature_edid != 0;
|
||||
// High feature word: VERSION_1 (bit 32) is required for a modern device.
|
||||
cfgWrite(u32, "device_feature_select", vp.feature_version_1_word);
|
||||
if (cfgRead(u32, "device_feature") & vp.feature_version_1_bit == 0) {
|
||||
log("virtio-gpu: device does not offer VERSION_1 (not a modern device)\n", .{});
|
||||
return false;
|
||||
}
|
||||
// Accept exactly VERSION_1, plus EDID when the device offered it (never a feature it didn't).
|
||||
cfgWrite(u32, "driver_feature_select", 0);
|
||||
cfgWrite(u32, "driver_feature", if (edid_available) vg.feature_edid else 0);
|
||||
cfgWrite(u32, "driver_feature_select", vp.feature_version_1_word);
|
||||
cfgWrite(u32, "driver_feature", vp.feature_version_1_bit);
|
||||
orStatus(vp.status_features_ok);
|
||||
if (cfgRead(u8, "device_status") & vp.status_features_ok == 0) {
|
||||
log("virtio-gpu: device rejected the negotiated features\n", .{});
|
||||
return false;
|
||||
}
|
||||
|
||||
// Stand up the control virtqueue (queue 0) in coherent DMA memory.
|
||||
cfgWrite(u16, "queue_select", 0);
|
||||
const device_qsize = cfgRead(u16, "queue_size");
|
||||
if (device_qsize < queue_size) {
|
||||
log("virtio-gpu: control queue too small ({d})\n", .{device_qsize});
|
||||
return false;
|
||||
}
|
||||
ring = dma.alloc(4096, dma.coherent) orelse {
|
||||
log("virtio-gpu: virtqueue allocation failed\n", .{});
|
||||
return false;
|
||||
};
|
||||
command = dma.alloc(4096, dma.coherent) orelse {
|
||||
log("virtio-gpu: command-buffer allocation failed\n", .{});
|
||||
return false;
|
||||
};
|
||||
mmio.write(u16, ring.virtual + avail_offset, 1); // VIRTQ_AVAIL_F_NO_INTERRUPT: we poll
|
||||
cfgWrite(u16, "queue_size", queue_size);
|
||||
cfgWrite64("queue_desc", ring.physical + desc_offset);
|
||||
cfgWrite64("queue_driver", ring.physical + avail_offset);
|
||||
cfgWrite64("queue_device", ring.physical + used_offset);
|
||||
cfgWrite(u16, "queue_msix_vector", 0xFFFF); // VIRTIO_MSI_NO_VECTOR
|
||||
cfgWrite(u16, "queue_enable", 1);
|
||||
|
||||
cfgWrite(u16, "queue_select", 0);
|
||||
notify_addr = notify_base + @as(usize, cfgRead(u16, "queue_notify_off")) * notify_multiplier;
|
||||
|
||||
orStatus(vp.status_driver_ok);
|
||||
|
||||
// Drive the GPU: create a 2D resource at the *max* mode, back it with a shared surface, and
|
||||
// scan out the current-mode rectangle within it.
|
||||
{
|
||||
const request = requestAt(vg.ResourceCreate2d);
|
||||
request.* = .{
|
||||
.hdr = .{ .type = @intFromEnum(vg.CmdType.resource_create_2d) },
|
||||
.resource_id = resource_id,
|
||||
.format = vg.format_b8g8r8x8_unorm,
|
||||
.width = max_width,
|
||||
.height = max_height,
|
||||
};
|
||||
if (command_nodata(@sizeOf(vg.ResourceCreate2d)) != ok_nodata) {
|
||||
log("virtio-gpu: resource_create_2d failed\n", .{});
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
// Back the resource with a shared (shm) surface, so the compositor and the device work
|
||||
// the same physical pages. The device needs the guest-physical base for attach_backing.
|
||||
surface = shm.create(scanout_bytes) orelse {
|
||||
log("virtio-gpu: scanout surface allocation failed\n", .{});
|
||||
return false;
|
||||
};
|
||||
const surface_physical = shm.physical(surface.handle) orelse {
|
||||
log("virtio-gpu: could not resolve the scanout surface physical address\n", .{});
|
||||
return false;
|
||||
};
|
||||
{
|
||||
const request = requestAt(vg.ResourceAttachBacking);
|
||||
request.* = .{
|
||||
.hdr = .{ .type = @intFromEnum(vg.CmdType.resource_attach_backing) },
|
||||
.resource_id = resource_id,
|
||||
.nr_entries = 1,
|
||||
};
|
||||
const entry: *vg.MemEntry = @ptrFromInt(command.virtual + request_offset + @sizeOf(vg.ResourceAttachBacking));
|
||||
entry.* = .{ .addr = surface_physical, .length = @intCast(scanout_bytes) };
|
||||
if (command_nodata(@sizeOf(vg.ResourceAttachBacking) + @sizeOf(vg.MemEntry)) != ok_nodata) {
|
||||
log("virtio-gpu: resource_attach_backing failed\n", .{});
|
||||
return false;
|
||||
}
|
||||
}
|
||||
if (!setScanoutRect()) {
|
||||
log("virtio-gpu: set_scanout failed\n", .{});
|
||||
return false;
|
||||
}
|
||||
log("virtio-gpu: scanout {d}x{d} online\n", .{ current_width, current_height });
|
||||
|
||||
// Hello the device manager so it counts us as up (and does not stop us at the hello
|
||||
// deadline). A restarted instance re-hellos here and re-announces below — the compositor
|
||||
// re-attaches to the fresh scanout (V6).
|
||||
helloManager();
|
||||
|
||||
// Read the monitor's EDID (best-effort, when the device offers it) — the mode list a real
|
||||
// driver derives from it; we log the preferred mode and keep our fixed offered list.
|
||||
readEdid();
|
||||
|
||||
// Paint a known pattern, present it, and read it back — the V3 self-test that proves the
|
||||
// whole path (virtqueue, resource, shared backing, transfer, flush) before a client attaches.
|
||||
const pixels: [*]u32 = @ptrCast(@alignCast(surface.ptr));
|
||||
const pixel_count: usize = @as(usize, max_width) * max_height;
|
||||
for (0..pixel_count) |i| pixels[i] = testPixel(@intCast(i));
|
||||
|
||||
if (!presentFull()) {
|
||||
log("virtio-gpu: initial present failed\n", .{});
|
||||
return false;
|
||||
}
|
||||
// The scanout surface is CPU-visible RAM: read the pattern back to prove the mapping,
|
||||
// which together with the flush ack above is the automated stand-in for "it's on screen".
|
||||
mmio.rmb();
|
||||
if (pixels[0] != testPixel(0) or pixels[pixel_count / 2] != testPixel(@intCast(pixel_count / 2))) {
|
||||
log("virtio-gpu: pixel read-back mismatch\n", .{});
|
||||
return false;
|
||||
}
|
||||
log("virtio-gpu: flush acked, pixel check ok\n", .{});
|
||||
|
||||
// Offer the shared surface to the compositor so it upgrades off the GOP floor (V4).
|
||||
announce();
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Point scanout 0 at the current-mode rectangle of the resource. Reused by initial bring-up
|
||||
/// and by `set_mode`.
|
||||
fn setScanoutRect() bool {
|
||||
const request = requestAt(vg.SetScanout);
|
||||
request.* = .{
|
||||
.hdr = .{ .type = @intFromEnum(vg.CmdType.set_scanout) },
|
||||
.rect = .{ .x = 0, .y = 0, .width = current_width, .height = current_height },
|
||||
.scanout_id = 0,
|
||||
.resource_id = resource_id,
|
||||
};
|
||||
return command_nodata(@sizeOf(vg.SetScanout)) == ok_nodata;
|
||||
}
|
||||
|
||||
/// Read and log the monitor's preferred mode from its EDID (VIRTIO_GPU_F_EDID). Best-effort:
|
||||
/// a device that doesn't offer EDID, or a missing/short block, is logged and ignored.
|
||||
fn readEdid() void {
|
||||
if (!edid_available) {
|
||||
log("virtio-gpu: EDID not offered by device\n", .{});
|
||||
return;
|
||||
}
|
||||
const request = requestAt(vg.GetEdid);
|
||||
request.* = .{ .hdr = .{ .type = @intFromEnum(vg.CmdType.get_edid) }, .scanout = 0 };
|
||||
if (!submit(@sizeOf(vg.GetEdid), @sizeOf(vg.RespEdid))) {
|
||||
log("virtio-gpu: EDID request not acked\n", .{});
|
||||
return;
|
||||
}
|
||||
const response: *vg.RespEdid = @ptrFromInt(command.virtual + response_offset);
|
||||
if (response.hdr.type != @intFromEnum(vg.CmdType.resp_ok_edid) or response.size < 64) {
|
||||
log("virtio-gpu: EDID unavailable\n", .{});
|
||||
return;
|
||||
}
|
||||
// The first detailed timing descriptor (EDID base-block offset 54) is the preferred mode:
|
||||
// active pixels are 12-bit, low byte + high nibble (bytes 2/4 horizontal, 5/7 vertical).
|
||||
const e = &response.edid;
|
||||
const h_active = @as(u32, e[56]) | (@as(u32, e[58] & 0xF0) << 4);
|
||||
const v_active = @as(u32, e[59]) | (@as(u32, e[61] & 0xF0) << 4);
|
||||
log("virtio-gpu: EDID preferred mode {d}x{d}\n", .{ h_active, v_active });
|
||||
}
|
||||
|
||||
/// Present the whole surface: copy the guest backing into the host resource, then flush it to
|
||||
/// the panel. Reused by the V3 self-test and by every compositor present over `.scanout`. V4
|
||||
/// presents the full surface; the damage-rect fast path is a later refinement.
|
||||
fn presentFull() bool {
|
||||
mmio.wmb(); // the surface writes must be visible before the device transfers them
|
||||
{
|
||||
// Transfer the current-mode rectangle from the guest backing to the host resource. The
|
||||
// device uses the resource's (max) width as the row stride, so the top-left rect at
|
||||
// offset 0 is exactly the visible area — the compositor composes at that same stride.
|
||||
const request = requestAt(vg.TransferToHost2d);
|
||||
request.* = .{
|
||||
.hdr = .{ .type = @intFromEnum(vg.CmdType.transfer_to_host_2d) },
|
||||
.rect = .{ .x = 0, .y = 0, .width = current_width, .height = current_height },
|
||||
.offset = 0,
|
||||
.resource_id = resource_id,
|
||||
};
|
||||
if (command_nodata(@sizeOf(vg.TransferToHost2d)) != ok_nodata) return false;
|
||||
}
|
||||
{
|
||||
// A fenced flush (vsync): the device signals the fence when the frame is actually on
|
||||
// screen — which its used-ring ack, what our synchronous submit waits on, already gates.
|
||||
const request = requestAt(vg.ResourceFlush);
|
||||
request.* = .{
|
||||
.hdr = .{ .type = @intFromEnum(vg.CmdType.resource_flush), .flags = vg.flag_fence, .fence_id = fence_next },
|
||||
.rect = .{ .x = 0, .y = 0, .width = current_width, .height = current_height },
|
||||
.resource_id = resource_id,
|
||||
};
|
||||
fence_next += 1;
|
||||
if (command_nodata(@sizeOf(vg.ResourceFlush)) != ok_nodata) return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Hello the device manager (role: bus — we own a PCI function, though we report no children):
|
||||
/// the handshake that marks us up so the manager doesn't stop us at the hello deadline, and
|
||||
/// (as a supervised driver) restarts us if we die. Best-effort: without a manager we still run.
|
||||
fn helloManager() void {
|
||||
var tries: u32 = 0;
|
||||
const manager = while (tries < 100) : (tries += 1) {
|
||||
if (ipc.lookup(.device_manager)) |h| break h;
|
||||
system.sleep(20);
|
||||
} else {
|
||||
log("virtio-gpu: no device manager to hello\n", .{});
|
||||
return;
|
||||
};
|
||||
const hello = dm.Hello{ .role = @intFromEnum(dm.Role.bus), .device_id = device_id };
|
||||
var reply: [dm.reply_size]u8 = undefined;
|
||||
const n = ipc.call(manager, std.mem.asBytes(&hello), &reply) catch {
|
||||
log("virtio-gpu: hello call failed\n", .{});
|
||||
return;
|
||||
};
|
||||
if (n < dm.reply_size or std.mem.bytesToValue(dm.HelloReply, reply[0..dm.reply_size]).status != 0) {
|
||||
log("virtio-gpu: hello refused\n", .{});
|
||||
return;
|
||||
}
|
||||
log("virtio-gpu: hello acknowledged\n", .{});
|
||||
}
|
||||
|
||||
/// Announce the scanout to the display service so it upgrades off the GOP framebuffer: hand it
|
||||
/// the shared surface as a capability plus the geometry. Best-effort and non-fatal — without a
|
||||
/// display service (the standalone virtio-gpu bring-up test) the driver is still a valid
|
||||
/// scanout service; it just serves no one. The display replies immediately (it defers its
|
||||
/// first present to a timer), so this returns before we start serving `.scanout` — no deadlock.
|
||||
fn announce() void {
|
||||
var tries: u32 = 0;
|
||||
const display = while (tries < 50) : (tries += 1) {
|
||||
if (ipc.lookup(.display)) |h| break h;
|
||||
system.sleep(20);
|
||||
} else {
|
||||
log("virtio-gpu: no display service to announce to (scanout-only)\n", .{});
|
||||
return;
|
||||
};
|
||||
var request = dp.Request{
|
||||
.operation = @intFromEnum(dp.Operation.attach_scanout),
|
||||
.x = max_width, // the shared surface's row stride in pixels (it is sized to the max mode)
|
||||
.width = current_width,
|
||||
.height = current_height,
|
||||
.colour = display_format_bgrx,
|
||||
};
|
||||
var reply: [dp.reply_size]u8 = undefined;
|
||||
_ = ipc.callCap(display, std.mem.asBytes(&request), &reply, surface.handle) catch {
|
||||
log("virtio-gpu: announce to display failed\n", .{});
|
||||
return;
|
||||
};
|
||||
log("virtio-gpu: announced scanout to display\n", .{});
|
||||
}
|
||||
|
||||
/// A `sp.Reply{status}` written into `reply`.
|
||||
fn scanoutStatus(reply: []u8, ok: bool) usize {
|
||||
const response = sp.Reply{ .status = if (ok) 0 else -1 };
|
||||
@memcpy(reply[0..sp.reply_size], std.mem.asBytes(&response));
|
||||
return sp.reply_size;
|
||||
}
|
||||
|
||||
/// The `.scanout` service: the compositor drives present / mode queries here. The pixels are
|
||||
/// already in the shared surface, so a present is a transfer-to-host + fenced flush; a mode
|
||||
/// change just re-points the scanout rectangle (the surface is sized to the largest mode).
|
||||
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Handle) usize {
|
||||
_ = sender;
|
||||
_ = capability;
|
||||
if (message.len < sp.request_size) return 0;
|
||||
const request = std.mem.bytesToValue(sp.Request, message[0..sp.request_size]);
|
||||
switch (request.operation) {
|
||||
@intFromEnum(sp.Operation.present) => return scanoutStatus(reply, presentFull()),
|
||||
@intFromEnum(sp.Operation.get_modes) => {
|
||||
var response = sp.ModesReply{ .status = 0, .count = offered_modes.len, .modes = undefined };
|
||||
for (0..sp.max_modes) |i| {
|
||||
response.modes[i] = if (i < offered_modes.len)
|
||||
.{ .width = offered_modes[i].width, .height = offered_modes[i].height }
|
||||
else
|
||||
.{ .width = 0, .height = 0 };
|
||||
}
|
||||
@memcpy(reply[0..sp.modes_reply_size], std.mem.asBytes(&response));
|
||||
return sp.modes_reply_size;
|
||||
},
|
||||
@intFromEnum(sp.Operation.set_mode) => {
|
||||
const w = request.width;
|
||||
const h = request.height;
|
||||
if (w == 0 or h == 0 or w > max_width or h > max_height) return scanoutStatus(reply, false);
|
||||
current_width = w;
|
||||
current_height = h;
|
||||
return scanoutStatus(reply, setScanoutRect());
|
||||
},
|
||||
else => return 0,
|
||||
}
|
||||
}
|
||||
|
||||
pub fn main(init: runtime.process.Init) void {
|
||||
const argument = init.arguments.get(1) orelse {
|
||||
_ = system.write("virtio-gpu: missing device id (argv[1])\n");
|
||||
return;
|
||||
};
|
||||
device_id = std.fmt.parseInt(u64, argument, 10) catch {
|
||||
log("virtio-gpu: malformed device id '{s}'\n", .{argument});
|
||||
return;
|
||||
};
|
||||
runtime.service.run(256, .{
|
||||
.service = .scanout,
|
||||
.init = initialise,
|
||||
.on_message = onMessage,
|
||||
});
|
||||
}
|
||||
@@ -0,0 +1,105 @@
|
||||
//! virtio 1.0 PCI transport — the vendor capabilities in PCI config space that point at the
|
||||
//! device's structures (common config, notify, ISR) in a BAR, the common-config register
|
||||
//! block, and the split-virtqueue layout. `extern` structs matching the spec. Host-tested
|
||||
//! for size. See docs/display-v2.md.
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
/// PCI vendor-specific capability id (0x09) — virtio 1.0 structures are advertised as these.
|
||||
pub const pci_cap_vendor: u8 = 0x09;
|
||||
|
||||
/// virtio_pci_cap `cfg_type`: which structure a vendor capability points at.
|
||||
pub const cfg_common: u8 = 1;
|
||||
pub const cfg_notify: u8 = 2;
|
||||
pub const cfg_isr: u8 = 3;
|
||||
pub const cfg_device: u8 = 4;
|
||||
pub const cfg_pci: u8 = 5;
|
||||
|
||||
/// virtio_pci_cap — a vendor capability naming a structure at (bar, offset, length) within
|
||||
/// a PCI BAR. Read straight out of config space.
|
||||
pub const PciCap = extern struct {
|
||||
cap_vndr: u8, // 0x09
|
||||
cap_next: u8, // next capability's offset in config space (0 = end)
|
||||
cap_len: u8,
|
||||
cfg_type: u8, // cfg_common / cfg_notify / ...
|
||||
bar: u8, // which BAR the structure lives in
|
||||
padding: [3]u8,
|
||||
offset: u32, // offset within the BAR
|
||||
length: u32, // length of the structure
|
||||
};
|
||||
|
||||
/// virtio_pci_notify_cap: a notify capability carries a multiplier after the base cap; the
|
||||
/// per-queue notify address is `notify_base + queue_notify_off * notify_off_multiplier`.
|
||||
pub const NotifyCap = extern struct {
|
||||
cap: PciCap,
|
||||
notify_off_multiplier: u32,
|
||||
};
|
||||
|
||||
/// virtio_pci_common_cfg — the common configuration register block (little-endian MMIO).
|
||||
pub const CommonCfg = extern struct {
|
||||
device_feature_select: u32,
|
||||
device_feature: u32,
|
||||
driver_feature_select: u32,
|
||||
driver_feature: u32,
|
||||
msix_config: u16,
|
||||
num_queues: u16,
|
||||
device_status: u8,
|
||||
config_generation: u8,
|
||||
queue_select: u16,
|
||||
queue_size: u16,
|
||||
queue_msix_vector: u16,
|
||||
queue_enable: u16,
|
||||
queue_notify_off: u16,
|
||||
queue_desc: u64,
|
||||
queue_driver: u64,
|
||||
queue_device: u64,
|
||||
};
|
||||
|
||||
/// device_status bits (written to `CommonCfg.device_status` during bring-up).
|
||||
pub const status_acknowledge: u8 = 1;
|
||||
pub const status_driver: u8 = 2;
|
||||
pub const status_driver_ok: u8 = 4;
|
||||
pub const status_features_ok: u8 = 8;
|
||||
|
||||
/// VIRTIO_F_VERSION_1 — feature bit 32 (in the second 32-bit feature word). Required for a
|
||||
/// modern device; we negotiate exactly this bit and nothing else.
|
||||
pub const feature_version_1_word: u32 = 1; // device_feature_select value for bits 32..63
|
||||
pub const feature_version_1_bit: u32 = 1 << 0; // bit 32 within that word
|
||||
|
||||
// --- split virtqueue -------------------------------------------------------
|
||||
|
||||
pub const Desc = extern struct {
|
||||
addr: u64, // guest-physical
|
||||
len: u32,
|
||||
flags: u16,
|
||||
next: u16,
|
||||
};
|
||||
pub const desc_flag_next: u16 = 1; // buffer continues in `next`
|
||||
pub const desc_flag_write: u16 = 2; // device-writable (else driver-writable/device-readable)
|
||||
|
||||
/// The available ring's fixed header; a `[queue_size]u16` ring and a trailing `used_event`
|
||||
/// u16 follow it in memory (laid out by the driver).
|
||||
pub const AvailHdr = extern struct {
|
||||
flags: u16,
|
||||
idx: u16,
|
||||
};
|
||||
|
||||
/// One entry of the used ring.
|
||||
pub const UsedElem = extern struct {
|
||||
id: u32,
|
||||
len: u32,
|
||||
};
|
||||
|
||||
/// The used ring's fixed header; a `[queue_size]UsedElem` ring and a trailing `avail_event`
|
||||
/// u16 follow it.
|
||||
pub const UsedHdr = extern struct {
|
||||
flags: u16,
|
||||
idx: u16,
|
||||
};
|
||||
|
||||
test "virtio-pci struct sizes match the spec" {
|
||||
try std.testing.expectEqual(@as(usize, 16), @sizeOf(PciCap));
|
||||
try std.testing.expectEqual(@as(usize, 56), @sizeOf(CommonCfg));
|
||||
try std.testing.expectEqual(@as(usize, 16), @sizeOf(Desc));
|
||||
try std.testing.expectEqual(@as(usize, 8), @sizeOf(UsedElem));
|
||||
}
|
||||
@@ -188,6 +188,13 @@ pub fn mapUserDmaInto(root: u64, virtual: u64, physical: u64, len: u64) void {
|
||||
paging.mapUserDmaInto(root, virtual, physical, len);
|
||||
}
|
||||
|
||||
/// Map shared cacheable RAM into address space `root`: write-back cacheable, RW+NX, and
|
||||
/// marked so teardown won't free the frames (they're owned by a refcounted shm object,
|
||||
/// freed when its last capability drops). For shm_create/shm_map.
|
||||
pub fn mapUserSharedInto(root: u64, virtual: u64, physical: u64, len: u64) void {
|
||||
paging.mapUserSharedInto(root, virtual, physical, len);
|
||||
}
|
||||
|
||||
/// Map a page into the kernel address space (non-executable). For the heap, etc.
|
||||
pub fn mapPage(virtual: u64, physical: u64, writable: bool) void {
|
||||
paging.map(virtual, physical, writable);
|
||||
@@ -680,6 +687,15 @@ pub fn jumpToUser(entry: u64, stack_top: u64) noreturn {
|
||||
jump_to_user(entry, stack_top);
|
||||
}
|
||||
|
||||
/// As `jumpToUser`, but delivers `arg0` in the user's `rdi` — how a fresh thread
|
||||
/// receives its closure pointer (docs/threading.md). A normal process is dropped
|
||||
/// with `arg0 = 0`, which its `_start` ignores (it reads argv off the stack).
|
||||
extern fn jump_to_user_arg(rip: u64, rsp: u64, arg0: u64) callconv(.c) noreturn;
|
||||
|
||||
pub fn jumpToUserArg(entry: u64, stack_top: u64, arg0: u64) noreturn {
|
||||
jump_to_user_arg(entry, stack_top, arg0);
|
||||
}
|
||||
|
||||
/// Route CPU exceptions to `handler`, which receives the trap frame and does not
|
||||
/// return. Until set, faults just halt the core.
|
||||
pub fn setFaultHandler(handler: *const fn (*const CpuState) noreturn) void {
|
||||
|
||||
@@ -123,6 +123,22 @@ jump_to_user:
|
||||
swapgs # user GS base (isr_common/syscall swap back on entry)
|
||||
iretq
|
||||
|
||||
# jump_to_user_arg(rdi = user rip, rsi = user rsp, rdx = user rdi/arg0): as
|
||||
# jump_to_user, but delivers arg0 in the user's rdi — how a fresh **thread**
|
||||
# receives its closure pointer (docs/threading.md). rdi carries the rip only until
|
||||
# it is pushed into the iretq frame, after which we overwrite it with the arg.
|
||||
.global jump_to_user_arg
|
||||
jump_to_user_arg:
|
||||
cli
|
||||
push $0x1B # user SS (0x18 | RPL 3)
|
||||
push %rsi # user RSP
|
||||
push $0x202 # RFLAGS: IF | reserved-1
|
||||
push $0x23 # user CS (0x20 | RPL 3)
|
||||
push %rdi # user RIP (consumes rdi)
|
||||
mov %rdx, %rdi # user rdi = arg0 (the thread's closure pointer)
|
||||
swapgs # user GS base (isr_common/syscall swap back on entry)
|
||||
iretq
|
||||
|
||||
# --- ring 3 entry/exit ------------------------------------------------------
|
||||
|
||||
# enter_user(rdi = user rip, rsi = user rsp, rdx = &TSS.rsp0)
|
||||
@@ -205,7 +221,17 @@ syscall_entry:
|
||||
push %r14
|
||||
push %r15
|
||||
mov %rsp, %rdi # trap-frame pointer
|
||||
# Preserve the caller's SSE/x87 register file across the syscall — see the same
|
||||
# dance in isr_common. Without it a syscall (or a task the scheduler runs while
|
||||
# this one blocks) clobbers the caller's live XMM values, which the compiler is
|
||||
# free to hold across a syscall (its wrappers only clobber rcx/r11/memory).
|
||||
mov %rsp, %rbx
|
||||
and $-16, %rsp
|
||||
sub $512, %rsp
|
||||
fxsave (%rsp)
|
||||
call interruptDispatch
|
||||
fxrstor (%rsp)
|
||||
mov %rbx, %rsp # back to the trap frame (undo the fxsave scratch)
|
||||
pop %r15
|
||||
pop %r14
|
||||
pop %r13
|
||||
@@ -357,7 +383,21 @@ isr_common:
|
||||
push %r14
|
||||
push %r15
|
||||
mov %rsp, %rdi # first argument: pointer to the trap frame
|
||||
# Save the interrupted SSE/x87 register file before any kernel code runs, and
|
||||
# restore it on the way out — the kernel and user both keep live values in XMM
|
||||
# (a 16-byte struct copy is a movdqu), and the kernel never otherwise preserves
|
||||
# them, so an interrupt handler (and whatever the scheduler runs in its place)
|
||||
# would silently clobber the interrupted task's vector registers. rbx bridges the
|
||||
# exact rsp across the call: it is callee-saved (interruptDispatch and every
|
||||
# context switch preserve it), so it survives even a blocking dispatch, and the
|
||||
# `and`/`sub` gives fxsave its required 16-byte-aligned scratch on the kernel stack.
|
||||
mov %rsp, %rbx
|
||||
and $-16, %rsp
|
||||
sub $512, %rsp
|
||||
fxsave (%rsp)
|
||||
call interruptDispatch
|
||||
fxrstor (%rsp)
|
||||
mov %rbx, %rsp # back to the trap frame (undo the fxsave scratch)
|
||||
pop %r15
|
||||
pop %r14
|
||||
pop %r13
|
||||
|
||||
@@ -393,6 +393,30 @@ pub fn leafIsWriteCombining(pml4: u64, virtual: u64) ?bool {
|
||||
return (e & pte_pat != 0) and (e & pcd == 0) and (e & pwt == 0);
|
||||
}
|
||||
|
||||
/// Map `[physical, physical+len)` into the user half rooted at `pml4` as **shared cacheable
|
||||
/// RAM**: write-back cacheable (RW + NX) for CPU compositing, and carrying `device_grant`
|
||||
/// so teardown (`freeSubtree`) does **not** return the frames to the allocator. The frames
|
||||
/// are owned by a refcounted shared-memory object (system/kernel/ipc-synchronous.zig) and
|
||||
/// freed only when its last capability drops — not when one sharer's address space dies, or
|
||||
/// the other sharers would be left mapping freed RAM. The caller aligns `virtual`/`physical`.
|
||||
pub fn mapUserSharedInto(pml4: u64, virtual: u64, physical: u64, len: u64) void {
|
||||
const flags: u64 = present | user | writable | no_execute | device_grant; // WB cacheable
|
||||
const first = physical & ~@as(u64, page_size - 1);
|
||||
const last = (physical + (if (len == 0) 1 else len) - 1) & ~@as(u64, page_size - 1);
|
||||
var off: u64 = 0;
|
||||
while (first + off <= last) : (off += page_size) {
|
||||
const v = virtual + off;
|
||||
const pml4e = &tableAt(pml4)[(v >> 39) & 0x1FF];
|
||||
const pdpt = descendUser(pml4e);
|
||||
const pdpte = &tableAt(pdpt)[(v >> 30) & 0x1FF];
|
||||
const pd = descendUser(pdpte);
|
||||
const pde = &tableAt(pd)[(v >> 21) & 0x1FF];
|
||||
const pt = descendUser(pde);
|
||||
tableAt(pt)[(v >> 12) & 0x1FF] = ((first + off) & address_mask) | flags;
|
||||
invalidate(v);
|
||||
}
|
||||
}
|
||||
|
||||
/// Create a new address space: a fresh PML4 with an empty user half and the
|
||||
/// kernel's higher half shared in (copying PML4[256..512), whose entries point
|
||||
/// at the kernel's PDPTs — pre-created at init and never restaled, so growth in
|
||||
|
||||
@@ -2,12 +2,15 @@
|
||||
//! into the linear framebuffer the bootloader handed us. No firmware, no driver
|
||||
//! — just pixels.
|
||||
//!
|
||||
//! This is a **bootstrap** console — a stop-gap so early boot has something on
|
||||
//! screen. The framebuffer is a general graphics surface, *not* inherently a text
|
||||
//! terminal; once the driver machinery exists it becomes a proper graphics device
|
||||
//! driver and this text-grid crutch goes away. It is therefore kept **separate
|
||||
//! from the diagnostic [log](log.zig)** — the log fans out to serial/debugcon/file,
|
||||
//! while this only paints the handful of user-facing status lines and panics.
|
||||
//! This is a **bootstrap / fatal-fallback** console. The driver machinery now exists — the
|
||||
//! user-space **display service** ([../services/display](../services/display/display.zig),
|
||||
//! docs/display.md) owns the framebuffer in normal operation — so this no longer paints
|
||||
//! routine status. It exists for the two cases the display service can't cover: **early
|
||||
//! boot**, before the service has claimed the framebuffer, and **fatal errors** (a kernel
|
||||
//! panic or a kernel-mode fault), which force it back on (`setSuppressed`) so a dying
|
||||
//! machine's last words reach the screen even over a live display. It is kept **separate
|
||||
//! from the diagnostic [log](log.zig)** — the log fans out to serial/debugcon/file and
|
||||
//! carries all routine kernel output; this only paints those fatal cases.
|
||||
//!
|
||||
//! The module owns a single console and a `present` flag; `write` is a no-op when
|
||||
//! the firmware handed over no framebuffer (a headless machine), so the kernel
|
||||
|
||||
@@ -28,6 +28,7 @@ const architecture = @import("architecture");
|
||||
const scheduler = @import("scheduler.zig");
|
||||
const sync = @import("sync.zig");
|
||||
const heap = @import("heap.zig");
|
||||
const pmm = @import("pmm.zig");
|
||||
|
||||
const page_size = abi.page_size;
|
||||
const Task = scheduler.Task;
|
||||
@@ -89,6 +90,10 @@ const user_half_end: u64 = 0x0000_8000_0000_0000;
|
||||
/// (per process) and/or by a registry slot, counted by `refcount`.
|
||||
pub const Endpoint = struct {
|
||||
refcount: u32 = 1,
|
||||
// The task that created it. When that task dies, the endpoint is marked `dead` so a caller
|
||||
// gets -EPEER instead of blocking forever on a service that will never reply again (V6).
|
||||
owner: u32 = 0,
|
||||
dead: bool = false,
|
||||
// Callers blocked in `call`, awaiting receive, in FIFO order (threaded via
|
||||
// Task.next; each such task is .blocked and in no scheduler queue).
|
||||
sender_head: ?*Task = null,
|
||||
@@ -109,10 +114,30 @@ pub const Endpoint = struct {
|
||||
|
||||
pub fn createIpcEndpoint() ?*Endpoint {
|
||||
const endpoint = heap.allocator().create(Endpoint) catch return null;
|
||||
endpoint.* = .{};
|
||||
endpoint.* = .{ .owner = scheduler.currentId() };
|
||||
return endpoint;
|
||||
}
|
||||
|
||||
/// A task is dying: kill the endpoints it registered as services. Mark each `dead` (so a later
|
||||
/// `call` returns -EPEER rather than blocking on a reply that will never come), wake anyone
|
||||
/// already parked sending to it with that error, and vacate its registry slot. Only *registered*
|
||||
/// endpoints are reachable from here; unregistered ones drop with the task's handle table. The
|
||||
/// caller holds the big kernel lock (this runs on the death path). See docs/display-v2.md (V6).
|
||||
pub fn killOwnedEndpointsLocked(task_id: u32) void {
|
||||
for (®istry) |*slot| {
|
||||
const endpoint = slot.* orelse continue;
|
||||
if (endpoint.owner != task_id) continue;
|
||||
endpoint.dead = true;
|
||||
while (dequeueSender(endpoint)) |sender| {
|
||||
sender.ipc_status = -EPEER;
|
||||
sender.ipc_received_cap = abi.no_cap;
|
||||
scheduler.readyLocked(sender);
|
||||
}
|
||||
slot.* = null;
|
||||
dropRef(endpoint);
|
||||
}
|
||||
}
|
||||
|
||||
/// Drop a reference; free the endpoint when the last one goes. (Frames are leaked
|
||||
/// today like other kernel objects — but the refcount bookkeeping lands now.)
|
||||
pub fn dropRef(endpoint: *Endpoint) void {
|
||||
@@ -123,6 +148,43 @@ pub fn dropRef(endpoint: *Endpoint) void {
|
||||
}
|
||||
}
|
||||
|
||||
// --- capability objects: what a handle-table entry can name ------------------
|
||||
|
||||
/// The `kind` tag on a `scheduler.HandleObject` — which capability object a handle names.
|
||||
/// Defined here (not in scheduler) because the meaning is the IPC/capability layer's.
|
||||
pub const handle_kind_endpoint: u8 = 0;
|
||||
pub const handle_kind_shm: u8 = 1;
|
||||
|
||||
/// A page-aligned block of **shared cacheable RAM** (docs/display-v2.md), referenced by
|
||||
/// capability handles across processes and freed when the last one drops. `phys` is its
|
||||
/// contiguous physical base, `pages` its length. A sharer's address-space teardown never
|
||||
/// reclaims these frames (the mapping carries `device_grant`); this object owns them.
|
||||
pub const ShmObject = struct {
|
||||
refcount: u32 = 1,
|
||||
phys: u64,
|
||||
pages: usize,
|
||||
};
|
||||
|
||||
/// Wrap `pages` contiguous frames at `phys` (already allocated + zeroed by the caller) in a
|
||||
/// refcounted shm object, or null if the heap is out of room.
|
||||
pub fn createShm(phys: u64, pages: usize) ?*ShmObject {
|
||||
const shm = heap.allocator().create(ShmObject) catch return null;
|
||||
shm.* = .{ .phys = phys, .pages = pages };
|
||||
return shm;
|
||||
}
|
||||
|
||||
/// Drop a shared-memory reference; when the last one goes, return its frames to the
|
||||
/// allocator and free the object. (The mappings themselves are torn down with each
|
||||
/// sharer's address space; `device_grant` keeps that from freeing the frames early.)
|
||||
pub fn dropShmRef(shm: *ShmObject) void {
|
||||
if (shm.refcount > 1) {
|
||||
shm.refcount -= 1;
|
||||
} else {
|
||||
for (0..shm.pages) |i| pmm.free(shm.phys + i * page_size);
|
||||
heap.allocator().destroy(shm);
|
||||
}
|
||||
}
|
||||
|
||||
// --- sender FIFO (endpoint-local, via Task.next) ----------------------------
|
||||
|
||||
fn enqueueSender(endpoint: *Endpoint, t: *Task) void {
|
||||
@@ -222,11 +284,25 @@ pub fn copyFromUser(user_as: u64, user_va: u64, destination: []u8) bool {
|
||||
/// no live handle, or `-ENOSPC` if `to`'s table is full. Callers only invoke this when
|
||||
/// `cap != no_cap`. Used by both IPC directions to carry an endpoint with a message.
|
||||
fn shareCapability(from: *Task, to: *Task, cap: u64) i64 {
|
||||
const endpoint = resolveHandle(from, cap) orelse return -EBADF;
|
||||
endpoint.refcount += 1;
|
||||
const handle = installHandle(to, endpoint);
|
||||
if (cap >= from.handles.len) return -EBADF;
|
||||
const entry = from.handles[@intCast(cap)] orelse return -EBADF;
|
||||
// Bump the named object's refcount (a copy, not a move — the sender keeps its handle),
|
||||
// dispatching by kind so both endpoints and shared-memory regions can travel with a
|
||||
// message.
|
||||
switch (entry.kind) {
|
||||
handle_kind_endpoint => {
|
||||
const e: *Endpoint = @ptrCast(@alignCast(entry.ptr));
|
||||
e.refcount += 1;
|
||||
},
|
||||
handle_kind_shm => {
|
||||
const s: *ShmObject = @ptrCast(@alignCast(entry.ptr));
|
||||
s.refcount += 1;
|
||||
},
|
||||
else => return -EBADF,
|
||||
}
|
||||
const handle = installEntry(to, entry);
|
||||
if (handle < 0) {
|
||||
dropRef(endpoint); // undo the bump; the receiver had no room
|
||||
dropEntry(entry); // undo the bump; the receiver had no room
|
||||
return -ENOSPC;
|
||||
}
|
||||
return handle;
|
||||
@@ -241,6 +317,7 @@ pub fn call(endpoint: *Endpoint, message_ptr: u64, message_len: u64, reply_ptr:
|
||||
if (message_len > MESSAGE_MAXIMUM or reply_cap > MESSAGE_MAXIMUM) return -E2BIG;
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
if (endpoint.dead) return -EPEER; // the service that owned this endpoint is gone — don't block
|
||||
|
||||
const me = scheduler.current();
|
||||
me.ipc_send_ptr = message_ptr;
|
||||
@@ -416,36 +493,68 @@ pub fn notifyFromIsr(endpoint: *Endpoint, badge: u64) void {
|
||||
|
||||
// --- per-process handle table + name registry -------------------------------
|
||||
|
||||
/// Install `endpoint` in task `t`'s handle table; returns the small-int handle or
|
||||
/// -ENOSPC. The caller has already taken/holds the reference the slot represents.
|
||||
pub fn installHandle(t: *Task, endpoint: *Endpoint) i64 {
|
||||
/// Install a capability object (kind + pointer) in task `t`'s handle table; returns the
|
||||
/// small-int handle or -ENOSPC. The caller has already taken/holds the reference the slot
|
||||
/// represents.
|
||||
fn installEntry(t: *Task, entry: scheduler.HandleObject) i64 {
|
||||
for (&t.handles, 0..) |*slot, i| {
|
||||
if (slot.* == null) {
|
||||
slot.* = @ptrCast(endpoint);
|
||||
slot.* = entry;
|
||||
return @intCast(i);
|
||||
}
|
||||
}
|
||||
return -ENOSPC;
|
||||
}
|
||||
|
||||
/// Resolve a handle to its endpoint, or null if out of range / unused.
|
||||
pub fn resolveHandle(t: *Task, h: u64) ?*Endpoint {
|
||||
if (h >= t.handles.len) return null;
|
||||
const slot = t.handles[@intCast(h)] orelse return null;
|
||||
return @ptrCast(@alignCast(slot));
|
||||
/// Install an endpoint handle. The common case; keeps the endpoint callers' signature.
|
||||
pub fn installHandle(t: *Task, endpoint: *Endpoint) i64 {
|
||||
return installEntry(t, .{ .kind = handle_kind_endpoint, .ptr = @ptrCast(endpoint) });
|
||||
}
|
||||
|
||||
/// Drop every endpoint reference an exiting task holds. Called from the scheduler
|
||||
/// exit path so a dead server's endpoints don't linger referenced.
|
||||
/// Install a shared-memory handle.
|
||||
pub fn installShmHandle(t: *Task, shm: *ShmObject) i64 {
|
||||
return installEntry(t, .{ .kind = handle_kind_shm, .ptr = @ptrCast(shm) });
|
||||
}
|
||||
|
||||
/// Resolve a handle to its endpoint, or null if out of range, unused, or a different kind
|
||||
/// (e.g. an shm handle used where an endpoint is expected).
|
||||
pub fn resolveHandle(t: *Task, h: u64) ?*Endpoint {
|
||||
if (h >= t.handles.len) return null;
|
||||
const entry = t.handles[@intCast(h)] orelse return null;
|
||||
if (entry.kind != handle_kind_endpoint) return null;
|
||||
return @ptrCast(@alignCast(entry.ptr));
|
||||
}
|
||||
|
||||
/// Resolve a handle to its shared-memory object, or null if out of range, unused, or not
|
||||
/// an shm handle.
|
||||
pub fn resolveShm(t: *Task, h: u64) ?*ShmObject {
|
||||
if (h >= t.handles.len) return null;
|
||||
const entry = t.handles[@intCast(h)] orelse return null;
|
||||
if (entry.kind != handle_kind_shm) return null;
|
||||
return @ptrCast(@alignCast(entry.ptr));
|
||||
}
|
||||
|
||||
/// Drop every capability reference an exiting task holds, dispatching by kind so a dead
|
||||
/// task's endpoints *and* shared-memory regions are released correctly. Called from the
|
||||
/// scheduler exit path.
|
||||
pub fn closeHandles(t: *Task) void {
|
||||
for (&t.handles) |*slot| {
|
||||
if (slot.*) |p| {
|
||||
dropRef(@ptrCast(@alignCast(p)));
|
||||
if (slot.*) |entry| {
|
||||
dropEntry(entry);
|
||||
slot.* = null;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Drop the reference a handle-table entry represents, by kind.
|
||||
fn dropEntry(entry: scheduler.HandleObject) void {
|
||||
switch (entry.kind) {
|
||||
handle_kind_endpoint => dropRef(@ptrCast(@alignCast(entry.ptr))),
|
||||
handle_kind_shm => dropShmRef(@ptrCast(@alignCast(entry.ptr))),
|
||||
else => {},
|
||||
}
|
||||
}
|
||||
|
||||
var registry: [maximum_services]?*Endpoint = .{null} ** maximum_services;
|
||||
|
||||
/// Publish `endpoint` under well-known `id` (takes a reference). Returns 0 or -errno.
|
||||
|
||||
+46
-21
@@ -9,6 +9,7 @@ const wall_clock = @import("wall-clock.zig");
|
||||
const pmm = @import("pmm.zig");
|
||||
const heap = @import("heap.zig");
|
||||
const scheduler = @import("scheduler.zig");
|
||||
const sync = @import("sync.zig");
|
||||
const process = @import("process.zig");
|
||||
const devices_broker = @import("devices-broker.zig");
|
||||
const irq = @import("irq.zig");
|
||||
@@ -156,12 +157,13 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
||||
log.print(" page tables: root = 0x{x:0>16}\n", .{architecture.activePageTable()});
|
||||
log.print(" kernel segs: {d} (mapped with W^X permissions)\n", .{boot_information.kernel_segment_count});
|
||||
|
||||
// Now on our own tables, the framebuffer window is write-combining: bring up
|
||||
// the on-screen console and clear it (a fast burst here, not the loader's
|
||||
// uncached crawl). From here `status` reaches the screen as well as the log.
|
||||
// Now on our own tables, the framebuffer window is write-combining: bring up the
|
||||
// on-screen console and clear it to a blank canvas (a fast burst here, not the loader's
|
||||
// uncached crawl). Routine boot output goes only to the log; this console now exists for
|
||||
// early-boot and fatal (`fatal`/panic) output, until the display service takes over.
|
||||
console.init(fb);
|
||||
log.write(if (console.present())
|
||||
"/system/kernel: framebuffer console online (bootstrap; graphics driver later)\n"
|
||||
"/system/kernel: framebuffer ready (early-boot + fatal fallback; the display service drives it in normal operation)\n"
|
||||
else
|
||||
"/system/kernel: no framebuffer (headless) -> logging to serial/debugcon only\n");
|
||||
|
||||
@@ -414,11 +416,22 @@ fn bringUpSecondaries() void {
|
||||
log.print("/system/kernel: {d}/{d} cores online\n", .{ scheduler.onlineCount(), cores.len });
|
||||
}
|
||||
|
||||
/// A user-facing status line: to the diagnostic `log` *and* the on-screen console
|
||||
/// (if a framebuffer is present). The verbose log uses `log.*` directly and never
|
||||
/// touches the framebuffer.
|
||||
/// A user-facing status line. Now that the user-space **display service** owns the
|
||||
/// framebuffer in normal operation (docs/display.md), routine kernel output goes to the
|
||||
/// diagnostic `log` (serial/debugcon/RAM) *only* — never to the on-screen console, which
|
||||
/// the compositor is about to paint over. For a message that must reach the screen even so
|
||||
/// — a panic or a fatal fault, when the machine is going down — use `fatal`.
|
||||
fn status(message: []const u8) void {
|
||||
log.write(message);
|
||||
}
|
||||
|
||||
/// A fatal, user-facing message: to the diagnostic log *and* the on-screen console, forcing
|
||||
/// the console back on (`setSuppressed(false)`) first — a dying machine's last words outrank
|
||||
/// any display service holding the framebuffer. The console is otherwise silent in normal
|
||||
/// operation (see `status`); it exists now only for early-boot and fatal output.
|
||||
fn fatal(message: []const u8) void {
|
||||
log.write(message);
|
||||
console.setSuppressed(false);
|
||||
console.write(message);
|
||||
}
|
||||
|
||||
@@ -427,6 +440,11 @@ fn statusPrint(comptime fmt: []const u8, args: anytype) void {
|
||||
status(std.fmt.bufPrint(&buffer, fmt, args) catch return);
|
||||
}
|
||||
|
||||
fn fatalPrint(comptime fmt: []const u8, args: anytype) void {
|
||||
var buffer: [256]u8 = undefined;
|
||||
fatal(std.fmt.bufPrint(&buffer, fmt, args) catch return);
|
||||
}
|
||||
|
||||
/// Frames (4 KiB pages) to whole MiB.
|
||||
fn mib(pages: u64) u64 {
|
||||
return pages * abi.page_size / (1024 * 1024);
|
||||
@@ -486,20 +504,26 @@ fn onException(state: *const architecture.CpuState) noreturn {
|
||||
}
|
||||
|
||||
log.checkpoint(cp_exception);
|
||||
// The machine is going down: force the console back on even if a display service
|
||||
// was holding the framebuffer, so the exception actually reaches the screen.
|
||||
console.setSuppressed(false);
|
||||
const core = scheduler.currentCpuIndex();
|
||||
// A fault is user-facing enough to paint on screen too (via statusPrint), on
|
||||
// top of the diagnostic log.
|
||||
statusPrint("\nCPU EXCEPTION on core {d}: {s} (vector {d})\n", .{ core, architecture.exceptionName(state.vector), state.vector });
|
||||
statusPrint(" error code : 0x{x}\n", .{state.error_code});
|
||||
statusPrint(" IP : 0x{x:0>16}\n", .{architecture.instructionPointer(state)});
|
||||
statusPrint(" SP : 0x{x:0>16}\n", .{architecture.stackPointer(state)});
|
||||
if (architecture.faultAddress(state)) |address| statusPrint(" fault addr : 0x{x:0>16}\n", .{address});
|
||||
// The machine is going down: paint the exception on screen too — `fatalPrint` forces the
|
||||
// console back on even if a display service was holding the framebuffer — on top of the
|
||||
// diagnostic log.
|
||||
fatalPrint("\nCPU EXCEPTION on core {d}: {s} (vector {d})\n", .{ core, architecture.exceptionName(state.vector), state.vector });
|
||||
// Name the culprit: which task, and whether it faulted in ring 3 (a process the
|
||||
// kernel would normally kill — landing here means it had no address space) or ring 0
|
||||
// (the trusted base itself). Without this the fatal report is anonymous.
|
||||
fatalPrint(" task : {d} ({s}), {s}\n", .{ scheduler.currentIdSafe(), scheduler.currentNameSafe(), if (architecture.fromUser(state)) "ring 3 (user)" else "ring 0 (kernel)" });
|
||||
fatalPrint(" error code : 0x{x}\n", .{state.error_code});
|
||||
fatalPrint(" IP : 0x{x:0>16}\n", .{architecture.instructionPointer(state)});
|
||||
fatalPrint(" SP : 0x{x:0>16}\n", .{architecture.stackPointer(state)});
|
||||
if (architecture.faultAddress(state)) |address| fatalPrint(" fault addr : 0x{x:0>16}\n", .{address});
|
||||
|
||||
var buffer: [128]u8 = undefined;
|
||||
log.recordPanic(std.fmt.bufPrint(&buffer, "CPU exception {s} (vector {d}) on core {d} at IP 0x{x}", .{ architecture.exceptionName(state.vector), state.vector, core, architecture.instructionPointer(state) }) catch "cpu exception");
|
||||
// Free the BKL if this core held it (a kernel-mode fault, or a nested fault in the
|
||||
// recovery teardown), so halting this one core doesn't deadlock every other core on
|
||||
// the lock. Only that core stops; the rest — and the supervisor — keep running.
|
||||
sync.releaseIfHeldHere();
|
||||
architecture.halt();
|
||||
}
|
||||
|
||||
@@ -511,10 +535,11 @@ pub const panic = std.debug.FullPanic(struct {
|
||||
_ = first_trace_address;
|
||||
log.checkpoint(cp_panic);
|
||||
log.recordPanic(message);
|
||||
console.setSuppressed(false); // a panic outranks any display service holding the screen
|
||||
status("\nKERNEL PANIC: ");
|
||||
status(message);
|
||||
status("\n");
|
||||
fatal("\nKERNEL PANIC: "); // a panic outranks any display service holding the screen
|
||||
fatal(message);
|
||||
fatal("\n");
|
||||
fatalPrint(" task : {d} ({s})\n", .{ scheduler.currentIdSafe(), scheduler.currentNameSafe() });
|
||||
sync.releaseIfHeldHere(); // don't deadlock the other cores on the lock we may hold
|
||||
architecture.halt();
|
||||
}
|
||||
}.panic);
|
||||
|
||||
+197
-1
@@ -81,6 +81,18 @@ pub const device_arena_end: u64 = device_arena_base + (4 << 30);
|
||||
pub const dma_arena_base: u64 = 0x0000_7200_0000_0000;
|
||||
pub const dma_arena_end: u64 = dma_arena_base + (256 << 20); // 256 MiB per process
|
||||
|
||||
/// The shared-memory arena: where `shm_create`/`shm_map` place shared cacheable regions, in
|
||||
/// PML4[230] — a user-exclusive region distinct from the DMA arena. The frames are owned by
|
||||
/// a refcounted shm object and freed when its last capability drops, not on teardown, so the
|
||||
/// mapping carries `device_grant`. Per-process cursor in `Task.shm_map_next` (docs/display-v2.md).
|
||||
pub const shm_arena_base: u64 = 0x0000_7300_0000_0000;
|
||||
pub const shm_arena_end: u64 = shm_arena_base + (256 << 20); // 256 MiB per process
|
||||
|
||||
/// Largest single `shm_create`, in pages (32 MiB) — enough for a 4K framebuffer surface;
|
||||
/// also an overflow guard on the page count. shm frames are contiguous (like DMA), so this
|
||||
/// bounds the contiguous allocation asked of the frame allocator.
|
||||
const maximum_shm_pages = 8192;
|
||||
|
||||
/// Largest single `mmap` grant, in pages (32 MiB). Big enough for a display service's
|
||||
/// back buffer at up to 4K (3840x2160x4 ≈ 8100 pages); the user heap otherwise grows in
|
||||
/// small chunks. `systemMmap` maps page by page with rollback, so this is only a sanity
|
||||
@@ -212,6 +224,23 @@ fn system_call(state: *architecture.CpuState) void {
|
||||
.timer_bind => systemTimerBind(state),
|
||||
.klog_read => systemKlogRead(state),
|
||||
.wall_clock => systemWallClock(state),
|
||||
.shm_create => systemShmCreate(state),
|
||||
.shm_map => systemShmMap(state),
|
||||
.shm_physical => systemShmPhysical(state),
|
||||
.thread_spawn => systemThreadSpawn(state),
|
||||
.current_core => systemCurrentCore(state),
|
||||
.thread_self => systemThreadSelf(state),
|
||||
.futex_wait => systemFutexWait(state),
|
||||
.futex_wake => systemFutexWake(state),
|
||||
.thread_exit => {
|
||||
// A thread ends like a process exit(0), but only this task: its
|
||||
// resources are released and its address-space reference dropped (the
|
||||
// space survives while sibling threads hold it). docs/threading.md.
|
||||
if (scheduler.currentIsUserProcess()) {
|
||||
scheduler.current().exit_reason = .exited;
|
||||
terminateCurrent();
|
||||
} else architecture.userExit();
|
||||
},
|
||||
_ => fail(state),
|
||||
}
|
||||
}
|
||||
@@ -460,6 +489,81 @@ fn systemDmaFree(state: *architecture.CpuState) void {
|
||||
architecture.setSystemCallResult(state, 0);
|
||||
}
|
||||
|
||||
/// shm_create(len) -> vaddr (rax), handle (rdx): grant `len` bytes (rounded up to whole
|
||||
/// pages) of **shareable, zeroed, cacheable** RAM — contiguous frames mapped into the
|
||||
/// caller's shm arena — and hand back the virtual address plus a capability handle. Unlike
|
||||
/// `dma_alloc` the memory is write-back cacheable (for CPU compositing, not device DMA) and
|
||||
/// its frames are owned by a refcounted object: the handle is passed to another process as
|
||||
/// an `ipc_call` send_cap, that process `shm_map`s it, and the frames free only when the
|
||||
/// last capability drops (docs/display-v2.md — the compositor↔native-driver and
|
||||
/// app↔compositor surface path).
|
||||
fn systemShmCreate(state: *architecture.CpuState) void {
|
||||
const len = architecture.systemCallArg(state, 0);
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0 or len == 0) return fail(state);
|
||||
|
||||
const pages: usize = @intCast((len + page_size - 1) / page_size);
|
||||
if (pages == 0 or pages > maximum_shm_pages) return fail(state);
|
||||
|
||||
// Reserve arena virtual space up front, so a mapping failure needs no rollback.
|
||||
if (t.shm_map_next == 0) t.shm_map_next = shm_arena_base;
|
||||
const base_v = t.shm_map_next;
|
||||
if (base_v + pages * page_size > shm_arena_end) return fail(state); // arena exhausted
|
||||
|
||||
const phys = pmm.allocContiguous(pages, ~@as(u64, 0)) orelse return fail(state);
|
||||
// Zero through the physmap (the frames aren't mapped in the caller yet).
|
||||
const kernel_view: [*]u8 = @ptrFromInt(boot_handoff.physicalToVirtual(phys));
|
||||
@memset(kernel_view[0 .. pages * page_size], 0);
|
||||
|
||||
const shm = ipc.createShm(phys, pages) orelse {
|
||||
for (0..pages) |i| pmm.free(phys + i * page_size);
|
||||
return fail(state);
|
||||
};
|
||||
const handle = ipc.installShmHandle(t, shm);
|
||||
if (handle < 0) {
|
||||
ipc.dropShmRef(shm); // last ref: frees the object and its frames
|
||||
return fail(state);
|
||||
}
|
||||
|
||||
architecture.mapUserSharedInto(t.aspace, base_v, phys, pages * page_size);
|
||||
t.shm_map_next = base_v + pages * page_size;
|
||||
architecture.setSystemCallResult(state, base_v); // vaddr for the CPU
|
||||
architecture.setSystemCallResult2(state, @intCast(handle)); // capability handle to pass on
|
||||
}
|
||||
|
||||
/// shm_map(cap) -> vaddr: map the shared region named by a capability handle the caller
|
||||
/// received (via an `ipc_call` send_cap) into its shm arena — the same physical frames the
|
||||
/// creator sees — returning the virtual address. The handle already holds a reference (taken
|
||||
/// when the capability was shared), so this only adds a mapping; it never bumps the refcount.
|
||||
fn systemShmMap(state: *architecture.CpuState) void {
|
||||
const cap = architecture.systemCallArg(state, 0);
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state);
|
||||
|
||||
const shm = ipc.resolveShm(t, cap) orelse return fail(state); // not an shm handle we hold
|
||||
if (t.shm_map_next == 0) t.shm_map_next = shm_arena_base;
|
||||
const base_v = t.shm_map_next;
|
||||
const size = shm.pages * page_size;
|
||||
if (base_v + size > shm_arena_end) return fail(state);
|
||||
|
||||
architecture.mapUserSharedInto(t.aspace, base_v, shm.phys, size);
|
||||
t.shm_map_next = base_v + size;
|
||||
architecture.setSystemCallResult(state, base_v);
|
||||
}
|
||||
|
||||
/// shm_physical(cap) -> paddr: the guest-physical base of a shared region the caller holds a
|
||||
/// capability for. The frames are contiguous (allocated by `allocContiguous`), so a single
|
||||
/// physical base + length describes the whole region — which is exactly what a driver needs
|
||||
/// to hand a shm surface to a device (virtio-gpu `attach_backing`). Only a holder of the
|
||||
/// capability can ask; there is no ambient way to turn a virtual address into a physical one.
|
||||
fn systemShmPhysical(state: *architecture.CpuState) void {
|
||||
const cap = architecture.systemCallArg(state, 0);
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state);
|
||||
const shm = ipc.resolveShm(t, cap) orelse return fail(state); // not an shm handle we hold
|
||||
architecture.setSystemCallResult(state, shm.phys);
|
||||
}
|
||||
|
||||
/// device_register(parent_id, descriptor_ptr) -> id: publish a child device below a device
|
||||
/// this process has claimed. The bus-driver primitive: a process that owns a bus
|
||||
/// enumerates it and hands each device it finds to the table, where a class driver
|
||||
@@ -552,6 +656,97 @@ fn systemSpawn(state: *architecture.CpuState) void {
|
||||
fail(state); // no bundled binary by that name
|
||||
}
|
||||
|
||||
/// thread_spawn(entry, stack_top, arg) -> tid: start a task that shares the **caller's**
|
||||
/// address space (docs/threading.md). The runtime supplies `entry` (its thread
|
||||
/// trampoline), a stack it mmap'd, and the closure pointer, which the kernel delivers in
|
||||
/// the new thread's rdi. The entry and stack must lie in the user half; the new thread is
|
||||
/// supervised by the caller and inherits its priority. Only a user process may spawn.
|
||||
fn systemThreadSpawn(state: *architecture.CpuState) void {
|
||||
const entry = architecture.systemCallArg(state, 0);
|
||||
const stack_top = architecture.systemCallArg(state, 1);
|
||||
const arg = architecture.systemCallArg(state, 2);
|
||||
const exit_handle = architecture.systemCallArg(state, 3);
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state); // kernel tasks own no address space to share
|
||||
if (entry == 0 or entry >= user_half_end) return fail(state);
|
||||
if (stack_top == 0 or stack_top > user_half_end) return fail(state);
|
||||
// The endpoint the thread notifies on exit (how join waits), or none.
|
||||
const exit_endpoint: ?*ipc.Endpoint = if (exit_handle == abi.no_cap)
|
||||
null
|
||||
else
|
||||
ipc.resolveHandle(t, exit_handle) orelse return failErr(state, ipc.EBADF);
|
||||
const tid = spawnThreadSupervised(t.aspace, entry, stack_top, arg, t.priority, t.id, exit_endpoint) orelse return fail(state);
|
||||
architecture.setSystemCallResult(state, tid);
|
||||
}
|
||||
|
||||
/// Spawn a thread sharing `aspace`, taking the exit-endpoint reference under the **same**
|
||||
/// lock as the spawn (as `spawnProcessSupervised` does), so the thread cannot die before
|
||||
/// its reference exists. Returns the new thread id, or null on resource exhaustion.
|
||||
fn spawnThreadSupervised(aspace: u64, entry: u64, stack_top: u64, arg: u64, priority: scheduler.Priority, supervisor: u32, exit_endpoint: ?*ipc.Endpoint) ?u32 {
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
const tid = scheduler.spawnUserLocked(aspace, entry, stack_top, arg, priority, "thread", supervisor, if (exit_endpoint) |e| @ptrCast(e) else null) orelse return null;
|
||||
if (exit_endpoint) |endpoint| endpoint.refcount += 1; // the thread holds it birth-to-death
|
||||
return tid;
|
||||
}
|
||||
|
||||
/// current_core() -> index: the dense 0-based index of the core the caller runs on.
|
||||
fn systemCurrentCore(state: *architecture.CpuState) void {
|
||||
architecture.setSystemCallResult(state, scheduler.currentCpuIndex());
|
||||
}
|
||||
|
||||
/// thread_self() -> tid: the calling thread's kernel task id.
|
||||
fn systemThreadSelf(state: *architecture.CpuState) void {
|
||||
architecture.setSystemCallResult(state, scheduler.currentId());
|
||||
}
|
||||
|
||||
/// futex_wait(addr, expected, timeout_ns) -> status (docs/threading.md): if the 4-byte
|
||||
/// user word at `addr` still equals `expected`, block until a futex_wake on `addr` or
|
||||
/// (if timeout_ns > 0) the deadline. The compare and the block are one critical section,
|
||||
/// so a concurrent futex_wake cannot slip between them. Returns futex_woken / mismatch /
|
||||
/// timed_out.
|
||||
fn systemFutexWait(state: *architecture.CpuState) void {
|
||||
const addr = architecture.systemCallArg(state, 0);
|
||||
const expected: u32 = @truncate(architecture.systemCallArg(state, 1));
|
||||
const timeout_ns = architecture.systemCallArg(state, 2);
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state);
|
||||
if (addr == 0 or (addr & 3) != 0 or addr + 4 > user_half_end) return fail(state);
|
||||
|
||||
const flags = sync.enter();
|
||||
var word_bytes: [4]u8 = undefined;
|
||||
if (!ipc.copyFromUser(t.aspace, addr, &word_bytes)) {
|
||||
sync.leave(flags);
|
||||
return fail(state);
|
||||
}
|
||||
if (std.mem.readInt(u32, &word_bytes, .little) != expected) {
|
||||
sync.leave(flags);
|
||||
architecture.setSystemCallResult(state, abi.futex_mismatch);
|
||||
return;
|
||||
}
|
||||
const timeout_ms = if (timeout_ns == 0) 0 else (timeout_ns + 999_999) / 1_000_000;
|
||||
const result = scheduler.futexWaitLocked(addr, timeout_ms);
|
||||
sync.leave(flags);
|
||||
architecture.setSystemCallResult(state, switch (result) {
|
||||
.woken => abi.futex_woken,
|
||||
.timed_out => abi.futex_timed_out,
|
||||
});
|
||||
}
|
||||
|
||||
/// futex_wake(addr, count) -> woken: wake up to `count` tasks blocked in futex_wait on
|
||||
/// `addr` in the caller's address space.
|
||||
fn systemFutexWake(state: *architecture.CpuState) void {
|
||||
const addr = architecture.systemCallArg(state, 0);
|
||||
const count: u32 = @truncate(architecture.systemCallArg(state, 1));
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state);
|
||||
if (addr == 0 or (addr & 3) != 0 or addr + 4 > user_half_end) return fail(state);
|
||||
const flags = sync.enter();
|
||||
const woken = scheduler.futexWakeLocked(t.aspace, addr, count);
|
||||
sync.leave(flags);
|
||||
architecture.setSystemCallResult(state, woken);
|
||||
}
|
||||
|
||||
/// process_enumerate(buffer, maximum) -> total: snapshot the task table into the
|
||||
/// caller's buffer (up to `maximum` `abi.ProcessDescriptor` entries), returning
|
||||
/// the total live-task count — the exact shape of `device_enumerate`, so a `ps`
|
||||
@@ -647,6 +842,7 @@ fn releaseTaskResourcesLocked(t: *scheduler.Task) void {
|
||||
scheduler.readyLocked(client); // its blocked `call` now returns the error
|
||||
}
|
||||
ipc.abandonSenderLocked(t);
|
||||
ipc.killOwnedEndpointsLocked(t.id); // its registered services are gone: callers get -EPEER, not a hang
|
||||
scheduler.removeFromWaitQueueLocked(t);
|
||||
scheduler.forgetIpcClientLocked(t);
|
||||
ipc.closeHandles(t);
|
||||
@@ -1342,7 +1538,7 @@ pub fn spawnProcessSupervised(image: []const u8, priority: u3, argv: []const []c
|
||||
architecture.mapUserPageInto(aspace, page_virtual, stack_frame, true, false); // RW + NX
|
||||
}
|
||||
|
||||
const child = scheduler.spawnUserLocked(aspace, parsed.entry, user_sp, priority, argv[0], supervisor, if (exit_endpoint) |endpoint| @ptrCast(endpoint) else null) orelse
|
||||
const child = scheduler.spawnUserLocked(aspace, parsed.entry, user_sp, 0, priority, argv[0], supervisor, if (exit_endpoint) |endpoint| @ptrCast(endpoint) else null) orelse
|
||||
return error.OutOfMemory;
|
||||
// The child holds a reference to its exit endpoint from birth to death. Taken
|
||||
// only now, after nothing can fail; the lock is still held, so the child
|
||||
|
||||
+148
-7
@@ -78,6 +78,11 @@ pub const Task = struct {
|
||||
aspace: u64 = 0,
|
||||
user_ip: u64 = 0, // user-mode entry point (user task only)
|
||||
user_sp: u64 = 0, // user-mode stack pointer (user task only)
|
||||
user_arg: u64 = 0, // value delivered in the user's rdi at first entry: 0 for a
|
||||
// process (its _start ignores it), the closure pointer for a thread (docs/threading.md)
|
||||
// The user address this task is blocked on in futex_wait (0 = not futex-waiting).
|
||||
// Cleared to 0 by futexWakeLocked as the "woken, not timed out" signal (docs/threading.md).
|
||||
futex_addr: u64 = 0,
|
||||
// Next free virtual address in this task's mmap grant arena (0 = uninitialised;
|
||||
// process.zig lazily seeds it to the arena base on the first mmap). Bumped up
|
||||
// as the user heap grows; user task only.
|
||||
@@ -86,9 +91,11 @@ pub const Task = struct {
|
||||
// uninitialised, process.zig seeds it on the first mmio_map). User task only.
|
||||
device_map_next: u64 = 0,
|
||||
// --- synchronous IPC (ipc_sync.zig) ---
|
||||
// Per-process handle table: small-int handle -> *ipc_sync.Endpoint, kept
|
||||
// opaque here so the scheduler and IPC modules don't import each other.
|
||||
handles: [ipc_maximum_handles]?*anyopaque = .{null} ** ipc_maximum_handles,
|
||||
// Per-process handle table: a small-int handle names a kernel capability object.
|
||||
// Each entry tags its `kind` (an IPC endpoint or a shared-memory object) so the
|
||||
// close/exit and cap-passing paths reclaim the right type. Kept opaque here so the
|
||||
// scheduler and IPC modules don't import each other (ipc_sync.zig owns the kinds).
|
||||
handles: [ipc_maximum_handles]?HandleObject = .{null} ** ipc_maximum_handles,
|
||||
// A server holds the caller it currently owes a reply to (set by ReplyWait's
|
||||
// receive, cleared when it replies). A client, while blocked in Call, records
|
||||
// its message + reply buffers here and its result lands in `ipc_status`.
|
||||
@@ -99,6 +106,7 @@ pub const Task = struct {
|
||||
ipc_reply_cap: u64 = 0,
|
||||
ipc_status: i64 = 0, // client: reply length / -errno, written by the replier
|
||||
dma_map_next: u64 = 0, // bump pointer into this task's DMA arena (0 = unseeded)
|
||||
shm_map_next: u64 = 0, // bump pointer into this task's shared-memory arena (0 = unseeded)
|
||||
ipc_send_cap: u64 = ~@as(u64, 0), // handle to transfer with this message (abi.no_cap = none)
|
||||
ipc_received_cap: u64 = ~@as(u64, 0), // client: handle the reply's transferred cap landed at (abi.no_cap = none)
|
||||
next: ?*Task = null, // ready-queue link (also the endpoint sender-FIFO link)
|
||||
@@ -125,7 +133,75 @@ pub const maximum_task_name = abi.maximum_process_name;
|
||||
/// it dimensions a field of `Task`; ipc_sync.zig re-exports it.
|
||||
pub const ipc_maximum_handles = 16;
|
||||
|
||||
/// One handle-table entry: a capability object plus a `kind` tag saying what `ptr` points
|
||||
/// at (an ipc endpoint or a shared-memory object), so a task's exit path and the
|
||||
/// capability-passing path reclaim/share the right type. The `kind` values are defined by
|
||||
/// ipc_sync.zig (`handle_kind_*`); kept an opaque `u8` here so the scheduler doesn't import
|
||||
/// the IPC module.
|
||||
pub const HandleObject = struct { kind: u8, ptr: *anyopaque };
|
||||
|
||||
var tasks = [_]Task{.{}} ** maximum_tasks;
|
||||
|
||||
/// Address-space reference counts: one live entry per address space, counting the
|
||||
/// tasks that share it. An address space is 1:1 with a process today; threads
|
||||
/// (docs/threading.md) will push a count above 1, and `destroyAddressSpace` must run
|
||||
/// only when the **last** task on an address space exits. All access is under the big
|
||||
/// kernel lock. There can be no more live address spaces than tasks, so the table is
|
||||
/// sized to the task pool and never overflows in practice.
|
||||
const AspaceRef = struct { root: u64 = 0, count: u32 = 0 };
|
||||
var aspace_refs = [_]AspaceRef{.{}} ** maximum_tasks;
|
||||
var aspace_destroy_count: u64 = 0;
|
||||
|
||||
/// Take a reference to address space `root` (0 = a kernel task, which owns none).
|
||||
/// Returns false only if the ref table is full — bounded by `maximum_tasks`, so in
|
||||
/// practice it never is. Caller holds the kernel lock.
|
||||
fn retainAspace(root: u64) bool {
|
||||
if (root == 0) return true;
|
||||
var free: ?*AspaceRef = null;
|
||||
for (&aspace_refs) |*entry| {
|
||||
if (entry.count != 0 and entry.root == root) {
|
||||
entry.count += 1;
|
||||
return true;
|
||||
}
|
||||
if (entry.count == 0 and free == null) free = entry;
|
||||
}
|
||||
const slot = free orelse return false;
|
||||
slot.* = .{ .root = root, .count = 1 };
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Drop a reference to `root`; destroy the address space when the **last** one drops.
|
||||
/// A `root` with no entry — never retained, e.g. a hand-built test space — is
|
||||
/// destroyed directly, preserving the pre-refcount behaviour. Caller holds the lock.
|
||||
fn releaseAspace(root: u64) void {
|
||||
if (root == 0) return;
|
||||
for (&aspace_refs) |*entry| {
|
||||
if (entry.count == 0 or entry.root != root) continue;
|
||||
entry.count -= 1;
|
||||
if (entry.count == 0) {
|
||||
entry.root = 0;
|
||||
architecture.destroyAddressSpace(root);
|
||||
aspace_destroy_count += 1;
|
||||
}
|
||||
return;
|
||||
}
|
||||
architecture.destroyAddressSpace(root);
|
||||
aspace_destroy_count += 1;
|
||||
}
|
||||
|
||||
/// Test-observable: how many address spaces are live (entries with a nonzero count).
|
||||
pub fn liveAspaceCount() u32 {
|
||||
var live: u32 = 0;
|
||||
for (&aspace_refs) |*entry| {
|
||||
if (entry.count != 0) live += 1;
|
||||
}
|
||||
return live;
|
||||
}
|
||||
|
||||
/// Test-observable: total address-space destructions since boot.
|
||||
pub fn aspaceDestroyCount() u64 {
|
||||
return aspace_destroy_count;
|
||||
}
|
||||
var next_id: u32 = 1;
|
||||
|
||||
/// Per-CPU scheduler state: the task each core is running, its own idle task, and a
|
||||
@@ -317,9 +393,15 @@ pub fn spawnOn(entry: *const fn () void, priority: Priority, cpu: u32) bool {
|
||||
/// out of memory.
|
||||
/// **Caller must hold the kernel lock** (the loader that builds `aspace` holds it
|
||||
/// across the whole spawn, so the address space and the task appear atomically).
|
||||
pub fn spawnUserLocked(aspace: u64, entry: u64, user_sp: u64, priority: Priority, task_name: []const u8, supervisor: u32, exit_endpoint: ?*anyopaque) ?u32 {
|
||||
pub fn spawnUserLocked(aspace: u64, entry: u64, user_sp: u64, user_arg: u64, priority: Priority, task_name: []const u8, supervisor: u32, exit_endpoint: ?*anyopaque) ?u32 {
|
||||
const t = freeSlot() orelse return null;
|
||||
const stack = heap.allocator().alloc(u8, stack_size) catch return null;
|
||||
// Take this task's reference to the address space before we commit the slot, so a
|
||||
// failure here leaves nothing to unwind (the caller still owns the raw `aspace`).
|
||||
if (!retainAspace(aspace)) {
|
||||
heap.allocator().free(stack);
|
||||
return null;
|
||||
}
|
||||
t.* = .{
|
||||
.id = next_id,
|
||||
.state = .ready,
|
||||
@@ -328,6 +410,7 @@ pub fn spawnUserLocked(aspace: u64, entry: u64, user_sp: u64, priority: Priority
|
||||
.aspace = aspace,
|
||||
.user_ip = entry,
|
||||
.user_sp = user_sp,
|
||||
.user_arg = user_arg,
|
||||
.supervisor = supervisor,
|
||||
.exit_endpoint = exit_endpoint,
|
||||
};
|
||||
@@ -353,7 +436,7 @@ fn startUserTask() void {
|
||||
// No serial chatter here: this runs on every spawn, unserialized against
|
||||
// user-space writes, and its output used to shear concurrent log lines in
|
||||
// half — the largest source of corrupted markers in the QEMU scenarios.
|
||||
architecture.jumpToUser(t.user_ip, t.user_sp); // noreturn
|
||||
architecture.jumpToUserArg(t.user_ip, t.user_sp, t.user_arg); // noreturn (arg0 = 0 for a process)
|
||||
}
|
||||
|
||||
/// The unlocked task-creation primitive. Caller must hold the kernel lock (or be the
|
||||
@@ -436,6 +519,49 @@ pub fn sleep(ms: u64) void {
|
||||
sync.leave(flags);
|
||||
}
|
||||
|
||||
// --- futex: block/wake on a user address (docs/threading.md) ----------------
|
||||
//
|
||||
// A futex waiter is not linked into any queue — it is simply a `.blocked` task
|
||||
// tagged with the address it waits on (`futex_addr`). Waking scans the task table
|
||||
// (bounded) for matching waiters. A timed wait also sets `wake_at`, so the timer's
|
||||
// `wakeExpired` can wake it; `futex_addr` stays non-zero in that case, which is how
|
||||
// the waiter tells a timeout from a real wake.
|
||||
|
||||
pub const FutexResult = enum { woken, timed_out };
|
||||
|
||||
/// Block the current task on futex `addr` until woken, or (if `timeout_ms > 0`) the
|
||||
/// deadline. **Precondition:** the big kernel lock is held and the caller has already
|
||||
/// checked, under this same lock, that the futex word equals the expected value — so
|
||||
/// no wake can be missed. Returns with the lock still held.
|
||||
pub fn futexWaitLocked(addr: u64, timeout_ms: u64) FutexResult {
|
||||
const t = current();
|
||||
t.futex_addr = addr;
|
||||
t.wake_at = if (timeout_ms > 0) architecture.millis() + timeout_ms else 0;
|
||||
t.state = .blocked;
|
||||
schedule(); // woken by futexWakeLocked (clears futex_addr) or wakeExpired (timeout)
|
||||
const woken = t.futex_addr == 0;
|
||||
t.futex_addr = 0;
|
||||
t.wake_at = 0;
|
||||
return if (woken) .woken else .timed_out;
|
||||
}
|
||||
|
||||
/// Wake up to `count` tasks blocked in `futex_wait` on `addr` in address space
|
||||
/// `aspace`. Precondition: the big kernel lock is held. Returns how many woke.
|
||||
pub fn futexWakeLocked(aspace: u64, addr: u64, count: u32) u32 {
|
||||
var woken: u32 = 0;
|
||||
for (&tasks) |*t| {
|
||||
if (woken >= count) break;
|
||||
if (t.state == .blocked and t.aspace == aspace and t.futex_addr == addr) {
|
||||
t.futex_addr = 0; // the "woken, not timed out" signal to futexWaitLocked
|
||||
t.wake_at = 0;
|
||||
t.state = .ready;
|
||||
enqueue(t);
|
||||
woken += 1;
|
||||
}
|
||||
}
|
||||
return woken;
|
||||
}
|
||||
|
||||
// --- event-based blocking -------------------------------------------------
|
||||
//
|
||||
// A WaitQueue is a set of tasks blocked waiting for something (a resource, a
|
||||
@@ -710,7 +836,7 @@ pub fn exitUserLocked() noreturn {
|
||||
const kroot = architecture.kernelPageTable();
|
||||
architecture.loadPageTable(kroot); // off the process tables before freeing them
|
||||
pc.loaded_aspace = kroot;
|
||||
architecture.destroyAddressSpace(as);
|
||||
releaseAspace(as); // destroys only when this was the last task on the space
|
||||
}
|
||||
dying.state = .free;
|
||||
dying.aspace = 0;
|
||||
@@ -731,7 +857,7 @@ pub fn exitUserLocked() noreturn {
|
||||
/// task isn't running). The kernel stack is leaked, as in `exitUser` (no reaper
|
||||
/// yet). Precondition: the big kernel lock is held.
|
||||
pub fn destroyTaskLocked(t: *Task) void {
|
||||
if (t.aspace != 0) architecture.destroyAddressSpace(t.aspace);
|
||||
if (t.aspace != 0) releaseAspace(t.aspace); // destroys only on the last reference
|
||||
t.aspace = 0;
|
||||
t.kill_pending = false;
|
||||
t.in_system_call = false;
|
||||
@@ -791,6 +917,21 @@ pub fn currentCpuIndex() u32 {
|
||||
return thisCpu().index;
|
||||
}
|
||||
|
||||
/// The running task's id, or 0 if this core's scheduler isn't up yet (early boot, no GS
|
||||
/// base). Safe for a fault reporter to call unconditionally — like `currentCpuIndex`,
|
||||
/// it never dereferences an unpublished per-CPU pointer and so can't fault a second time.
|
||||
pub fn currentIdSafe() u32 {
|
||||
if (architecture.cpuLocal() == 0) return 0;
|
||||
return thisCpu().current.id;
|
||||
}
|
||||
|
||||
/// The running task's name (argv[0]), or "" if this core's scheduler isn't up yet.
|
||||
/// The companion to `currentIdSafe` for naming the culprit in a fatal fault report.
|
||||
pub fn currentNameSafe() []const u8 {
|
||||
if (architecture.cpuLocal() == 0) return "";
|
||||
return thisCpu().current.name();
|
||||
}
|
||||
|
||||
/// Change the running task's priority (takes effect next time it's enqueued).
|
||||
pub fn setPriority(p: Priority) void {
|
||||
current().priority = p;
|
||||
|
||||
@@ -33,6 +33,13 @@ const architecture = @import("architecture");
|
||||
/// 0 = free, 1 = held. A single global lock for the whole kernel.
|
||||
var held = std.atomic.Value(u32).init(0);
|
||||
|
||||
/// The per-CPU base pointer (`architecture.cpuLocal()`) of the core currently holding
|
||||
/// the lock, or 0 when free. Metadata only — `held` is what enforces exclusion — read
|
||||
/// solely by `releaseIfHeldHere` on the fatal-fault path. `cpuLocal()` is a unique,
|
||||
/// architecture-level token per core (0 before this core's GS base is published, which
|
||||
/// is fine: that window is single-core early boot, where no other core can deadlock).
|
||||
var owner = std.atomic.Value(usize).init(0);
|
||||
|
||||
/// Enter the kernel: disable interrupts on this core, then spin until we own the
|
||||
/// lock. Returns the caller's prior interrupt flags for `leave` to restore.
|
||||
/// Interrupts stay off for the whole critical section so this core's timer tick
|
||||
@@ -67,14 +74,29 @@ export fn releaseForFreshTask() callconv(.c) void {
|
||||
release();
|
||||
}
|
||||
|
||||
/// Release the big kernel lock **only if this core is the one holding it** — a no-op
|
||||
/// otherwise. For the fatal-fault path (a kernel-mode fault, or a nested fault inside the
|
||||
/// recovery teardown, both of which run under the lock): a core that dies holding the BKL
|
||||
/// must free it, or every other core spins forever in `acquire` and the whole machine
|
||||
/// deadlocks instead of just that core stopping. It must NOT free a lock another core
|
||||
/// owns, hence the owner check. Caveat: if we held it mid-mutation the shared state may be
|
||||
/// inconsistent — but letting the other cores (and the supervisor) run on possibly-degraded
|
||||
/// state is strictly more recoverable than a guaranteed total hang.
|
||||
pub fn releaseIfHeldHere() void {
|
||||
const me = architecture.cpuLocal();
|
||||
if (me != 0 and owner.load(.monotonic) == me) release();
|
||||
}
|
||||
|
||||
fn acquire() void {
|
||||
// Test-and-test-and-set: try once, then spin read-only until the lock looks
|
||||
// free before retrying the (bus-locked) swap — cheaper on the coherency fabric.
|
||||
while (held.swap(1, .acquire) != 0) {
|
||||
while (held.load(.monotonic) != 0) architecture.cpuRelax();
|
||||
}
|
||||
owner.store(architecture.cpuLocal(), .monotonic);
|
||||
}
|
||||
|
||||
fn release() void {
|
||||
owner.store(0, .monotonic);
|
||||
held.store(0, .release);
|
||||
}
|
||||
|
||||
+491
-68
@@ -101,6 +101,14 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void {
|
||||
displayServiceTest(boot_information);
|
||||
} else if (eql(case, "display-demo")) {
|
||||
displayDemoTest(boot_information);
|
||||
} else if (eql(case, "shm")) {
|
||||
shmTest(boot_information);
|
||||
} else if (eql(case, "virtio-gpu")) {
|
||||
virtioGpuTest(boot_information);
|
||||
} else if (eql(case, "display-native")) {
|
||||
displayNativeTest(boot_information);
|
||||
} else if (eql(case, "display-reattach")) {
|
||||
displayReattachTest(boot_information);
|
||||
} else if (eql(case, "clock")) {
|
||||
clockTest();
|
||||
} else if (eql(case, "smp")) {
|
||||
@@ -131,6 +139,18 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void {
|
||||
userPfTest();
|
||||
} else if (eql(case, "fault-recovery")) {
|
||||
faultRecoveryTest(boot_information);
|
||||
} else if (eql(case, "aspace-refcount")) {
|
||||
aspaceRefcountTest(boot_information);
|
||||
} else if (eql(case, "thread-spawn")) {
|
||||
threadSpawnTest(boot_information);
|
||||
} else if (eql(case, "thread-join")) {
|
||||
threadJoinTest(boot_information);
|
||||
} else if (eql(case, "thread-futex")) {
|
||||
threadFutexTest(boot_information);
|
||||
} else if (eql(case, "thread-mutex")) {
|
||||
threadMutexTest(boot_information);
|
||||
} else if (eql(case, "thread-id")) {
|
||||
threadIdTest(boot_information);
|
||||
} else if (eql(case, "args")) {
|
||||
argsTest(boot_information);
|
||||
} else if (eql(case, "init")) {
|
||||
@@ -230,6 +250,19 @@ fn bufferHas(needle: []const u8) bool {
|
||||
return std.mem.indexOf(u8, process.write_buffer[0..process.write_len], needle) != null;
|
||||
}
|
||||
|
||||
/// How many `pci_device` functions the devices broker currently holds. A durable
|
||||
/// snapshot, unlike a `bufferHas` poll of the single-latest write_buffer line, so a
|
||||
/// test can wait on it without racing transient log output. `scratch` is
|
||||
/// caller-owned to keep the (large) descriptor array off this helper's own frame.
|
||||
fn brokerPciCount(scratch: []device_abi.DeviceDescriptor) u32 {
|
||||
const k = @min(devices_broker.enumerate(scratch), scratch.len);
|
||||
var count: u32 = 0;
|
||||
for (scratch[0..k]) |d| {
|
||||
if (d.class == @intFromEnum(device_abi.DeviceClass.pci_device)) count += 1;
|
||||
}
|
||||
return count;
|
||||
}
|
||||
|
||||
/// Non-destructive checks of the memory map and frame allocator.
|
||||
fn smoke(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: smoke\n", .{});
|
||||
@@ -1353,7 +1386,7 @@ fn spawnFaultingProcess() ?u32 {
|
||||
architecture.mapUserPageInto(aspace, process.stack_base_virtual, stack_frame, true, false); // RW + NX
|
||||
|
||||
// Supervised by the calling test task, so exitReasonOf can read the verdict.
|
||||
const id = scheduler.spawnUserLocked(aspace, process.code_virtual, process.stack_base_virtual + abi.page_size, 4, "fault-probe", scheduler.currentId(), null) orelse {
|
||||
const id = scheduler.spawnUserLocked(aspace, process.code_virtual, process.stack_base_virtual + abi.page_size, 0, 4, "fault-probe", scheduler.currentId(), null) orelse {
|
||||
architecture.destroyAddressSpace(aspace);
|
||||
return null;
|
||||
};
|
||||
@@ -1408,6 +1441,254 @@ fn faultRecoveryTest(boot_information: *const BootInformation) void {
|
||||
result();
|
||||
}
|
||||
|
||||
/// Address-space refcount (docs/threading-plan.md M1): every process holds exactly one
|
||||
/// reference to its address space, released when it dies, so `destroyAddressSpace` runs
|
||||
/// exactly once per space — no leak, no double-free. Spawn and kill several ring-3
|
||||
/// processes (the faulting probe, reaped by the kernel) and confirm the count of live
|
||||
/// address spaces returns to baseline while destructions advance by exactly that many.
|
||||
/// This is the foundation threads (shared address spaces) build on: the refactor must be
|
||||
/// invisible while every space still has exactly one task.
|
||||
fn aspaceRefcountTest(boot_information: *const BootInformation) void {
|
||||
_ = boot_information;
|
||||
log("DANOS-TEST-BEGIN: aspace-refcount\n", .{});
|
||||
const base_live = scheduler.liveAspaceCount();
|
||||
const base_destroyed = scheduler.aspaceDestroyCount();
|
||||
const rounds: u32 = 5;
|
||||
var killed: u32 = 0;
|
||||
var round: u32 = 0;
|
||||
while (round < rounds) : (round += 1) {
|
||||
process.fault_kill_count = 0;
|
||||
const probe = spawnFaultingProcess() orelse break;
|
||||
_ = probe;
|
||||
// Let the probe fault on its first instruction and be reaped.
|
||||
scheduler.setPriority(1);
|
||||
const deadline = architecture.millis() + 5000;
|
||||
while (process.fault_kill_count < 1 and architecture.millis() < deadline) scheduler.yield();
|
||||
scheduler.setPriority(4);
|
||||
if (process.fault_kill_count >= 1) killed += 1;
|
||||
}
|
||||
check("all probes spawned and were killed", killed == rounds);
|
||||
check("live address-space count returned to baseline", scheduler.liveAspaceCount() == base_live);
|
||||
check("each address space destroyed exactly once", scheduler.aspaceDestroyCount() == base_destroyed + rounds);
|
||||
if (killed == rounds and scheduler.liveAspaceCount() == base_live and
|
||||
scheduler.aspaceDestroyCount() == base_destroyed + rounds)
|
||||
log("aspace-refcount: spaces released to baseline ok\n", .{});
|
||||
result();
|
||||
}
|
||||
|
||||
/// Thread spawn (docs/threading-plan.md M2): the `thread-test` service spawns a worker
|
||||
/// thread that writes a shared global; the main thread, polling that memory, observes the
|
||||
/// write — proving `runtime.Thread.spawn` started a task in the **same** address space
|
||||
/// (a separate process could not touch it). The service's own marker is the verdict.
|
||||
fn threadSpawnTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: thread-spawn\n", .{});
|
||||
if (boot_information.initial_ramdisk_len == 0) {
|
||||
check("bootloader handed over an initial_ramdisk", false);
|
||||
result();
|
||||
return;
|
||||
}
|
||||
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||
check("initial_ramdisk image is valid", false);
|
||||
result();
|
||||
return;
|
||||
};
|
||||
|
||||
check("thread-test spawned", spawnNamed(rd, "thread-test"));
|
||||
|
||||
// Wait for the service's verdict marker (it polls shared memory the worker wrote).
|
||||
const ok_marker = "thread-test: child ran in shared aspace ok";
|
||||
const fail_marker = "thread-test: FAIL";
|
||||
scheduler.setPriority(1);
|
||||
const deadline = architecture.millis() + 12000;
|
||||
while (architecture.millis() < deadline) {
|
||||
if (bufferHas(ok_marker) or bufferHas(fail_marker)) break;
|
||||
scheduler.yield();
|
||||
}
|
||||
scheduler.setPriority(4);
|
||||
|
||||
check("a worker thread ran in the shared address space (shared write observed)", bufferHas(ok_marker));
|
||||
check("the thread path reported no failure", !bufferHas(fail_marker));
|
||||
result();
|
||||
}
|
||||
|
||||
/// Thread join + parallelism (docs/threading-plan.md M3): `thread-test` in join mode
|
||||
/// spawns N workers that each do K atomic increments on a shared counter and stamp the
|
||||
/// core they ran on; it `join`s all N and asserts the total is exactly N*K (every worker
|
||||
/// ran, join waited for each) and that >1 core was used (genuine parallelism), then a
|
||||
/// detached worker proves `detach`. Its single verdict marker is the case result.
|
||||
fn threadJoinTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: thread-join\n", .{});
|
||||
if (boot_information.initial_ramdisk_len == 0) {
|
||||
check("bootloader handed over an initial_ramdisk", false);
|
||||
result();
|
||||
return;
|
||||
}
|
||||
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||
check("initial_ramdisk image is valid", false);
|
||||
result();
|
||||
return;
|
||||
};
|
||||
|
||||
// Spawn thread-test in join mode (argv selects the mode).
|
||||
var started = false;
|
||||
var i: u32 = 0;
|
||||
while (i < rd.count) : (i += 1) {
|
||||
const item = rd.entry(i) orelse continue;
|
||||
if (!eql(item.name, "thread-test")) continue;
|
||||
started = if (process.spawnProcess(item.blob, 4, &.{ "thread-test", "join" })) true else |_| false;
|
||||
break;
|
||||
}
|
||||
check("thread-test (join mode) spawned", started);
|
||||
|
||||
const ok_marker = "thread-test: join ok";
|
||||
const fail_marker = "thread-test: FAIL";
|
||||
scheduler.setPriority(1);
|
||||
const deadline = architecture.millis() + 15000;
|
||||
while (architecture.millis() < deadline) {
|
||||
if (bufferHas(ok_marker) or bufferHas(fail_marker)) break;
|
||||
scheduler.yield();
|
||||
}
|
||||
scheduler.setPriority(4);
|
||||
|
||||
check("N worker threads joined; counter exact (N*K) and >1 core used", bufferHas(ok_marker));
|
||||
check("no thread failure reported", !bufferHas(fail_marker));
|
||||
result();
|
||||
}
|
||||
|
||||
/// Futex (docs/threading-plan.md M4): `thread-test` in futex mode has a waiter thread
|
||||
/// block in `futex_wait` on a word; the main thread publishes the word and `futex_wake`s
|
||||
/// it. The serial order `waiting → waking → woke` shows the kernel handoff (the waiter
|
||||
/// parked and was woken, not spun), and a `timedWait` on an unwoken word reports a
|
||||
/// timeout. The verdict marker is emitted only after both hold.
|
||||
fn threadFutexTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: thread-futex\n", .{});
|
||||
if (boot_information.initial_ramdisk_len == 0) {
|
||||
check("bootloader handed over an initial_ramdisk", false);
|
||||
result();
|
||||
return;
|
||||
}
|
||||
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||
check("initial_ramdisk image is valid", false);
|
||||
result();
|
||||
return;
|
||||
};
|
||||
|
||||
var started = false;
|
||||
var i: u32 = 0;
|
||||
while (i < rd.count) : (i += 1) {
|
||||
const item = rd.entry(i) orelse continue;
|
||||
if (!eql(item.name, "thread-test")) continue;
|
||||
started = if (process.spawnProcess(item.blob, 4, &.{ "thread-test", "futex" })) true else |_| false;
|
||||
break;
|
||||
}
|
||||
check("thread-test (futex mode) spawned", started);
|
||||
|
||||
const ok_marker = "thread-futex: ok";
|
||||
const fail_marker = "thread-futex: FAIL";
|
||||
scheduler.setPriority(1);
|
||||
const deadline = architecture.millis() + 15000;
|
||||
while (architecture.millis() < deadline) {
|
||||
if (bufferHas(ok_marker) or bufferHas(fail_marker)) break;
|
||||
scheduler.yield();
|
||||
}
|
||||
scheduler.setPriority(4);
|
||||
|
||||
// Only the freshest marker is checked here — the kernel's log ring buffer may have
|
||||
// evicted the earlier ones by now. The waiting/waking/woke ordering (the handoff
|
||||
// proof) is asserted against the full serial stream by the qemu case's regex; the
|
||||
// verdict marker is emitted by thread-test only after the wake AND the timeout hold.
|
||||
check("futex handoff + timeout completed (verdict reached, no failure)", bufferHas(ok_marker) and !bufferHas(fail_marker));
|
||||
result();
|
||||
}
|
||||
|
||||
/// Mutex + Condition (docs/threading-plan.md M5): `thread-test` in mutex mode runs a
|
||||
/// bounded producer/consumer — P producers and C consumers over one `Mutex` and two
|
||||
/// `Condition`s move N unique items through a small ring. Every item is produced once;
|
||||
/// if the lock and condition variables are correct under real cross-core contention,
|
||||
/// the consumed checksum and tally match exactly (no lost or duplicated item, no
|
||||
/// overrun). The verdict marker is emitted only when both match.
|
||||
fn threadMutexTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: thread-mutex\n", .{});
|
||||
if (boot_information.initial_ramdisk_len == 0) {
|
||||
check("bootloader handed over an initial_ramdisk", false);
|
||||
result();
|
||||
return;
|
||||
}
|
||||
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||
check("initial_ramdisk image is valid", false);
|
||||
result();
|
||||
return;
|
||||
};
|
||||
|
||||
var started = false;
|
||||
var i: u32 = 0;
|
||||
while (i < rd.count) : (i += 1) {
|
||||
const item = rd.entry(i) orelse continue;
|
||||
if (!eql(item.name, "thread-test")) continue;
|
||||
started = if (process.spawnProcess(item.blob, 4, &.{ "thread-test", "mutex" })) true else |_| false;
|
||||
break;
|
||||
}
|
||||
check("thread-test (mutex mode) spawned", started);
|
||||
|
||||
const ok_marker = "thread-mutex: ok";
|
||||
const fail_marker = "thread-mutex: FAIL";
|
||||
scheduler.setPriority(1);
|
||||
const deadline = architecture.millis() + 20000;
|
||||
while (architecture.millis() < deadline) {
|
||||
if (bufferHas(ok_marker) or bufferHas(fail_marker)) break;
|
||||
scheduler.yield();
|
||||
}
|
||||
scheduler.setPriority(4);
|
||||
|
||||
check("producer/consumer over Mutex+Condition moved every item exactly once", bufferHas(ok_marker) and !bufferHas(fail_marker));
|
||||
result();
|
||||
}
|
||||
|
||||
/// Thread identity (docs/threading-plan.md M6): `thread-test` in id mode spawns two
|
||||
/// workers that each read `runtime.Thread.getCurrentId`; the main thread confirms all
|
||||
/// three ids are non-zero and distinct — proof each thread has its own kernel identity.
|
||||
fn threadIdTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: thread-id\n", .{});
|
||||
if (boot_information.initial_ramdisk_len == 0) {
|
||||
check("bootloader handed over an initial_ramdisk", false);
|
||||
result();
|
||||
return;
|
||||
}
|
||||
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||
check("initial_ramdisk image is valid", false);
|
||||
result();
|
||||
return;
|
||||
};
|
||||
|
||||
var started = false;
|
||||
var i: u32 = 0;
|
||||
while (i < rd.count) : (i += 1) {
|
||||
const item = rd.entry(i) orelse continue;
|
||||
if (!eql(item.name, "thread-test")) continue;
|
||||
started = if (process.spawnProcess(item.blob, 4, &.{ "thread-test", "id" })) true else |_| false;
|
||||
break;
|
||||
}
|
||||
check("thread-test (id mode) spawned", started);
|
||||
|
||||
const ok_marker = "thread-id: ok";
|
||||
const fail_marker = "thread-id: FAIL";
|
||||
scheduler.setPriority(1);
|
||||
const deadline = architecture.millis() + 12000;
|
||||
while (architecture.millis() < deadline) {
|
||||
if (bufferHas(ok_marker) or bufferHas(fail_marker)) break;
|
||||
scheduler.yield();
|
||||
}
|
||||
scheduler.setPriority(4);
|
||||
|
||||
check("each thread has a distinct, non-zero getCurrentId", bufferHas(ok_marker) and !bufferHas(fail_marker));
|
||||
result();
|
||||
}
|
||||
|
||||
/// The full PID-1 path: the bootloader read /system/services/init off the boot volume and
|
||||
/// handed it over; load it as a user ELF and spawn it as a real ring-3 process
|
||||
/// — the same call the normal boot path makes — then confirm it beats. init
|
||||
@@ -1895,18 +2176,14 @@ fn pciScanTest(boot_information: *const BootInformation) void {
|
||||
};
|
||||
|
||||
// Post-flip (M19.3) ground truth: the kernel no longer enumerates PCI
|
||||
// functions, so equivalence inverts — the broker's function count after
|
||||
// the scan must equal what the driver itself reported finding.
|
||||
// functions, so equivalence inverts — every PCI function the broker holds was
|
||||
// put there by the ring-3 driver, so before the driver runs the broker holds
|
||||
// none. One reusable descriptor buffer (each snapshot is ~20 KiB; three live
|
||||
// at once would overflow the 64 KiB bootstrap stack this test runs on).
|
||||
var buffer: [64]device_abi.DeviceDescriptor = undefined;
|
||||
const n = @min(devices_broker.enumerate(&buffer), buffer.len);
|
||||
var boot_pci: u32 = 0;
|
||||
for (buffer[0..n]) |d| {
|
||||
if (d.class == @intFromEnum(device_abi.DeviceClass.pci_device)) boot_pci += 1;
|
||||
}
|
||||
check("the kernel seeded no PCI functions (the walk retired)", boot_pci == 0);
|
||||
check("the kernel seeded no PCI functions (the walk retired)", brokerPciCount(&buffer) == 0);
|
||||
|
||||
process.setInitialRamdisk(image);
|
||||
process.write_count = 0;
|
||||
var manager: u32 = 0;
|
||||
var i: u32 = 0;
|
||||
while (i < rd.count) : (i += 1) {
|
||||
@@ -1917,68 +2194,49 @@ fn pciScanTest(boot_information: *const BootInformation) void {
|
||||
}
|
||||
check("device-manager spawned (test-pci-restart mode)", manager != 0);
|
||||
|
||||
// First scan: wait for the driver's count line and parse the number.
|
||||
const count_prefix = "pci-bus: ";
|
||||
const count_suffix = " functions found";
|
||||
var reported: u32 = 0;
|
||||
scheduler.setPriority(1);
|
||||
// The manager spawns pci-bus, which scans the ECAM window and registers every
|
||||
// function it finds, so the broker's PCI count climbs from zero and plateaus.
|
||||
// Wait for it to *settle*: latch N only once the count has held steady for a
|
||||
// stretch, so a mid-scan sample can't latch a low N that the rest of the same
|
||||
// scan then appears to exceed. The count is monotonic (registrations only add;
|
||||
// the table has no unregister) so the plateau is permanent — stability is
|
||||
// reached the moment the scan finishes and holds indefinitely. Sleep between
|
||||
// samples rather than busy-yield: the boot context outranks the drivers, and a
|
||||
// busy spin would starve the very processes it waits on; a sleeping task is
|
||||
// woken by the timer, so the drivers run in between.
|
||||
const poll_ms = 5;
|
||||
var registered: u32 = 0;
|
||||
var steady: u32 = 0;
|
||||
var deadline = architecture.millis() + 15000;
|
||||
while (architecture.millis() < deadline and reported == 0) {
|
||||
const line = process.write_buffer[0..process.write_len];
|
||||
if (std.mem.indexOf(u8, line, count_prefix)) |start| {
|
||||
if (std.mem.indexOf(u8, line, count_suffix)) |digits_end| {
|
||||
reported = std.fmt.parseInt(u32, line[start + count_prefix.len .. digits_end], 10) catch 0;
|
||||
}
|
||||
}
|
||||
scheduler.yield();
|
||||
while (architecture.millis() < deadline and steady < 60) { // 60 * 5ms = 300ms steady
|
||||
const now = brokerPciCount(&buffer);
|
||||
if (now != 0 and now == registered) steady += 1 else steady = 0;
|
||||
registered = now;
|
||||
scheduler.sleep(poll_ms);
|
||||
}
|
||||
scheduler.setPriority(4);
|
||||
check("the ring-3 scan reported a function count", reported >= 1);
|
||||
check("the ring-3 scan registered its PCI functions in the broker", registered >= 1);
|
||||
|
||||
// Every reported function was registered: the broker holds exactly them.
|
||||
var registered: [64]device_abi.DeviceDescriptor = undefined;
|
||||
const r = @min(devices_broker.enumerate(®istered), registered.len);
|
||||
var registered_pci: u32 = 0;
|
||||
for (registered[0..r]) |d| {
|
||||
if (d.class == @intFromEnum(device_abi.DeviceClass.pci_device)) registered_pci += 1;
|
||||
// The restart drill — the manager kills pci-bus ~1 s after its scan, prunes its
|
||||
// own child tree, and respawns it to re-claim, re-scan, and re-register — is
|
||||
// asserted by the harness's ordered regex over the whole serial log, the way
|
||||
// every restart drill is (see usbReportTest / driverRestartTest): the manager's
|
||||
// kill/prune/respawn lines are transient and would race a write_buffer poll, and
|
||||
// the *broker* count can't witness the restart at all — the table has no
|
||||
// unregister and register is idempotent (devices-broker.zig), so the kill leaves
|
||||
// the nodes in place and the respawn's re-registration dedupes against them.
|
||||
//
|
||||
// That idempotence is exactly this test's kernel-side claim: watch, across the
|
||||
// whole drill, that the count never grows past N. A broken dedup would append
|
||||
// the re-scanned functions as duplicates (N -> 2N), and with no unregister that
|
||||
// overshoot would persist — so a single late sample would catch it; the loop is
|
||||
// belt-and-braces over the ~2 s the kill + backoff + respawn takes.
|
||||
var duplicated = false;
|
||||
deadline = architecture.millis() + 5000;
|
||||
while (architecture.millis() < deadline and !duplicated) {
|
||||
if (brokerPciCount(&buffer) > registered) duplicated = true;
|
||||
scheduler.sleep(20);
|
||||
}
|
||||
check("the broker holds exactly the reported functions", registered_pci == reported);
|
||||
const kernel_count = reported; // the no-duplicate check below reuses it
|
||||
|
||||
// The restart drill: the manager kills pci-bus after its reports; the
|
||||
// respawn re-claims, re-scans, and re-registers.
|
||||
const restart_marker = "device-manager: restarting pci-bus";
|
||||
scheduler.setPriority(1);
|
||||
deadline = architecture.millis() + 15000;
|
||||
var restarted = false;
|
||||
while (architecture.millis() < deadline and !restarted) {
|
||||
if (bufferHas(restart_marker)) restarted = true;
|
||||
scheduler.yield();
|
||||
}
|
||||
scheduler.setPriority(4);
|
||||
check("the manager restarted pci-bus", restarted);
|
||||
|
||||
var marker_buffer: [48]u8 = undefined;
|
||||
const marker = std.fmt.bufPrint(&marker_buffer, "pci-bus: {d} functions found", .{reported}) catch "";
|
||||
scheduler.setPriority(1);
|
||||
deadline = architecture.millis() + 15000;
|
||||
var seen = false;
|
||||
while (architecture.millis() < deadline and !seen) {
|
||||
if (bufferHas(marker)) seen = true;
|
||||
scheduler.yield();
|
||||
}
|
||||
scheduler.setPriority(4);
|
||||
check("the respawned scan reported the same count", seen);
|
||||
|
||||
// No duplicates: the registrations deduped against the kernel's own nodes
|
||||
// on the first pass, and against themselves on the second.
|
||||
var after: [64]device_abi.DeviceDescriptor = undefined;
|
||||
const m = @min(devices_broker.enumerate(&after), after.len);
|
||||
var after_count: u32 = 0;
|
||||
for (after[0..m]) |d| {
|
||||
if (d.class == @intFromEnum(device_abi.DeviceClass.pci_device)) after_count += 1;
|
||||
}
|
||||
check("no duplicate PCI nodes after register + restart + re-register", after_count == kernel_count);
|
||||
check("no duplicate PCI nodes after the restart drill", !duplicated);
|
||||
result();
|
||||
}
|
||||
|
||||
@@ -2377,6 +2635,171 @@ fn displayDemoTest(boot_information: *const BootInformation) void {
|
||||
while (true) scheduler.yield();
|
||||
}
|
||||
|
||||
/// V2 — cross-process shared memory (docs/display-v2.md). Spawn shm-server and shm-client:
|
||||
/// the client shm_creates a region, writes a pattern, and passes the region's capability to
|
||||
/// the server as an ipc_call send_cap; the server shm_maps it and confirms the pattern is
|
||||
/// visible — proving the two processes share the same physical pages, and that the extended
|
||||
/// capability-passing (endpoints → memory objects) works. Its `shm: shared 4096 bytes ok`
|
||||
/// heartbeat is the marker.
|
||||
fn shmTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: shm\n", .{});
|
||||
if (boot_information.initial_ramdisk_len == 0) {
|
||||
check("bootloader handed over an initial_ramdisk", false);
|
||||
result();
|
||||
return;
|
||||
}
|
||||
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||
check("initial_ramdisk image is valid", false);
|
||||
result();
|
||||
return;
|
||||
};
|
||||
|
||||
if (!spawnNamed(rd, "shm-server")) {
|
||||
log("shm: could not spawn shm-server\n", .{});
|
||||
result();
|
||||
return;
|
||||
}
|
||||
_ = spawnNamed(rd, "shm-client");
|
||||
scheduler.setPriority(1); // below the two, so they run
|
||||
while (true) scheduler.yield();
|
||||
}
|
||||
|
||||
/// V3 — the virtio-gpu driver, end to end (docs/display-v2.md). Boot the device-manager
|
||||
/// stack (in its normal mode) so it discovers the virtio-gpu PCI function — present because
|
||||
/// the harness boots this case with QEMU's `-device virtio-gpu-pci` — and spawns the driver.
|
||||
/// The driver claims the device, brings up the control virtqueue, creates a 2D scanout,
|
||||
/// paints a known pattern, flushes it, and waits for the device's used-ring ack. Its serial
|
||||
/// heartbeats — `virtio-gpu: scanout WxH online` and `virtio-gpu: flush acked, pixel check
|
||||
/// ok` — are the harness's markers (it reads serial directly, like the display cases). The
|
||||
/// used-ring ack is the device confirming it consumed the frame; the pixel read-back proves
|
||||
/// the backing is CPU-visible RAM — together the automated stand-in for "it's on screen".
|
||||
fn virtioGpuTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: virtio-gpu\n", .{});
|
||||
if (boot_information.initial_ramdisk_len == 0) {
|
||||
check("bootloader handed over an initial_ramdisk", false);
|
||||
result();
|
||||
return;
|
||||
}
|
||||
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||
check("initial_ramdisk image is valid", false);
|
||||
result();
|
||||
return;
|
||||
};
|
||||
|
||||
// Spawn device-manager in its normal mode: its initialise discovers the PCI host bridge
|
||||
// from the kernel device tree, spawns pci-bus, and matches the virtio-gpu class triple to
|
||||
// spawn our driver with the function's device id as argv[1].
|
||||
process.setInitialRamdisk(image);
|
||||
var manager: u32 = 0;
|
||||
var i: u32 = 0;
|
||||
while (i < rd.count) : (i += 1) {
|
||||
const item = rd.entry(i) orelse continue;
|
||||
if (!eql(item.name, "device-manager")) continue;
|
||||
manager = process.spawnProcessSupervised(item.blob, 4, &.{"device-manager"}, scheduler.currentId(), null) catch 0;
|
||||
break;
|
||||
}
|
||||
if (manager == 0) {
|
||||
log("virtio-gpu: could not spawn device-manager\n", .{});
|
||||
result();
|
||||
return;
|
||||
}
|
||||
scheduler.setPriority(1); // below the manager and the driver it spawns, so they run
|
||||
while (true) scheduler.yield();
|
||||
}
|
||||
|
||||
/// V4 — the native backend + hot-attach (docs/display-v2.md). Boot the compositor and the
|
||||
/// hardware-free `display-demo` client (as displayDemoTest does), then the device-manager
|
||||
/// stack so it discovers the virtio-gpu function — present via QEMU's `-device
|
||||
/// virtio-gpu-pci` — and spawns the driver. The driver brings up its scanout, then announces
|
||||
/// the shared surface to the already-running compositor, which maps it, upgrades off the GOP
|
||||
/// floor, and presents the composited frame through the native backend. Its serial heartbeats
|
||||
/// — `display: scanout upgraded to virtio-gpu` and `display: native present verified` — plus
|
||||
/// the demo's own `display-demo: ok` are the harness's markers. Display is spawned first so
|
||||
/// it is registered on `.display` before the driver announces.
|
||||
fn displayNativeTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: display-native\n", .{});
|
||||
if (boot_information.initial_ramdisk_len == 0) {
|
||||
check("bootloader handed over an initial_ramdisk", false);
|
||||
result();
|
||||
return;
|
||||
}
|
||||
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||
check("initial_ramdisk image is valid", false);
|
||||
result();
|
||||
return;
|
||||
};
|
||||
|
||||
process.setInitialRamdisk(image);
|
||||
var manager: u32 = 0;
|
||||
var i: u32 = 0;
|
||||
while (i < rd.count) : (i += 1) {
|
||||
const item = rd.entry(i) orelse continue;
|
||||
if (!eql(item.name, "device-manager")) continue;
|
||||
manager = process.spawnProcessSupervised(item.blob, 4, &.{"device-manager"}, scheduler.currentId(), null) catch 0;
|
||||
break;
|
||||
}
|
||||
if (manager == 0) {
|
||||
log("display-native: could not spawn device-manager\n", .{});
|
||||
result();
|
||||
return;
|
||||
}
|
||||
if (!spawnNamed(rd, "display")) {
|
||||
log("display-native: could not spawn the display service\n", .{});
|
||||
result();
|
||||
return;
|
||||
}
|
||||
_ = spawnNamed(rd, "display-demo");
|
||||
scheduler.setPriority(1); // below the compositor, the demo, and the driver, so they run
|
||||
while (true) scheduler.yield();
|
||||
}
|
||||
|
||||
/// V6 — resilience: the compositor survives the virtio-gpu driver dying and re-attaches when
|
||||
/// device-manager restarts it (docs/display-v2.md). Same boot as display-native, but the
|
||||
/// manager runs in "test-scanout-restart" mode: a moment after the driver hellos, it kills it
|
||||
/// once; the normal restart policy respawns it, the restarted driver re-announces, and the
|
||||
/// compositor re-attaches to the fresh scanout — logging `display: scanout re-attached` after
|
||||
/// the initial `display: scanout upgraded to virtio-gpu`. The compositor must not crash.
|
||||
fn displayReattachTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: display-reattach\n", .{});
|
||||
if (boot_information.initial_ramdisk_len == 0) {
|
||||
check("bootloader handed over an initial_ramdisk", false);
|
||||
result();
|
||||
return;
|
||||
}
|
||||
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||
check("initial_ramdisk image is valid", false);
|
||||
result();
|
||||
return;
|
||||
};
|
||||
|
||||
process.setInitialRamdisk(image);
|
||||
var manager: u32 = 0;
|
||||
var i: u32 = 0;
|
||||
while (i < rd.count) : (i += 1) {
|
||||
const item = rd.entry(i) orelse continue;
|
||||
if (!eql(item.name, "device-manager")) continue;
|
||||
manager = process.spawnProcessSupervised(item.blob, 4, &.{ "device-manager", "test-scanout-restart" }, scheduler.currentId(), null) catch 0;
|
||||
break;
|
||||
}
|
||||
if (manager == 0) {
|
||||
log("display-reattach: could not spawn device-manager\n", .{});
|
||||
result();
|
||||
return;
|
||||
}
|
||||
if (!spawnNamed(rd, "display")) {
|
||||
log("display-reattach: could not spawn the display service\n", .{});
|
||||
result();
|
||||
return;
|
||||
}
|
||||
_ = spawnNamed(rd, "display-demo");
|
||||
scheduler.setPriority(1); // below the compositor, the demo, and the driver, so they run
|
||||
while (true) scheduler.yield();
|
||||
}
|
||||
|
||||
/// Process arguments, end to end: spawn args-echo bare (its argv[0] is the
|
||||
/// initial-ramdisk name). Instance 1 sees argc == 1 and respawns itself through
|
||||
/// `system_spawn` with the extra arguments "alpha beta-42" — the syscall argument
|
||||
|
||||
@@ -694,8 +694,3 @@ fn rd16(bytes: []const u8, off: usize) u64 {
|
||||
fn rd32(bytes: []const u8, off: usize) u64 {
|
||||
return rd16(bytes, off) | (rd16(bytes, off + 2) << 16);
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start; // pull the runtime entry shim into the image
|
||||
}
|
||||
|
||||
@@ -48,8 +48,3 @@ pub fn main(init: runtime.process.Init) void {
|
||||
len += 1;
|
||||
_ = runtime.system.write(buffer[0..len]);
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start; // pull the runtime entry shim into the image
|
||||
}
|
||||
|
||||
@@ -37,8 +37,3 @@ pub fn main(init: runtime.process.Init) void {
|
||||
const poison: *volatile u32 = @ptrFromInt(0xdead0000);
|
||||
poison.* = 1; // the restart machinery's fuel: a real segmentation fault
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start; // pull the runtime entry shim into the image
|
||||
}
|
||||
|
||||
@@ -81,8 +81,3 @@ pub fn main() void {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start; // pull the runtime entry shim into the image
|
||||
}
|
||||
|
||||
@@ -41,6 +41,15 @@ const xhci_pci_class: u64 = pci_class.ClassCode.pack(.{
|
||||
.prog_if = @intFromEnum(pci_class.serial_bus.usb.ProgIf.xhci),
|
||||
});
|
||||
|
||||
/// The PCI class triple of a virtio-gpu — Display Controller / Other (0x80) / 0. The class
|
||||
/// alone cannot tell it from any other display/other function, so the driver re-confirms
|
||||
/// vendor 0x1AF4 / device 0x1050 from config space once spawned; this only gets it spawned.
|
||||
const virtio_gpu_pci_class: u64 = pci_class.ClassCode.pack(.{
|
||||
.base = @intFromEnum(pci_class.BaseClass.display),
|
||||
.subclass = 0x80, // "Other" — no named SubClass member (PCI convention)
|
||||
.prog_if = 0,
|
||||
});
|
||||
|
||||
/// The driver that serves a *reported* PCI function (M19.3: matching moved
|
||||
/// from the boot snapshot to the bus reports), or null. A machine can carry
|
||||
/// several identical controllers — one driver instance per reported device,
|
||||
@@ -48,6 +57,7 @@ const xhci_pci_class: u64 = pci_class.ClassCode.pack(.{
|
||||
fn pciDriverForIdentity(identity: u64) ?[]const u8 {
|
||||
return switch (identity) {
|
||||
xhci_pci_class => "usb-xhci-bus",
|
||||
virtio_gpu_pci_class => "virtio-gpu",
|
||||
else => null,
|
||||
};
|
||||
}
|
||||
@@ -149,6 +159,8 @@ var test_restart_mode = false;
|
||||
var test_usb_restart_mode = false;
|
||||
var test_usb_killed = false;
|
||||
var test_pci_restart_mode = false;
|
||||
var test_scanout_restart_mode = false;
|
||||
var test_scanout_killed = false;
|
||||
var test_kill_pid: u32 = 0;
|
||||
var test_kill_due_ns: u64 = 0;
|
||||
|
||||
@@ -415,6 +427,14 @@ fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?runtime
|
||||
} else if (driverByProcess(sender)) |driver| {
|
||||
driver.state = .running;
|
||||
writeLine("/system/services/device-manager: hello from {s} (device {d})\n", .{ driver.name(), hello.device_id });
|
||||
// Resilience drill (V6): once, kill the virtio-gpu driver a moment after it hellos, so
|
||||
// the normal restart policy respawns it — the compositor must survive and re-attach.
|
||||
if (test_scanout_restart_mode and !test_scanout_killed and std.mem.eql(u8, driver.name(), "virtio-gpu")) {
|
||||
test_scanout_killed = true;
|
||||
test_kill_pid = sender;
|
||||
test_kill_due_ns = system.clock() + 1_500_000_000;
|
||||
_ = system.timerOnce(manager_endpoint, 1600);
|
||||
}
|
||||
} else {
|
||||
status = -1;
|
||||
writeLine("/system/services/device-manager: hello from unknown process {d}\n", .{sender});
|
||||
@@ -557,6 +577,7 @@ pub fn main(init: runtime.process.Init) void {
|
||||
test_restart_mode = std.mem.eql(u8, mode, "test-restart");
|
||||
test_usb_restart_mode = std.mem.eql(u8, mode, "test-usb-restart");
|
||||
test_pci_restart_mode = std.mem.eql(u8, mode, "test-pci-restart");
|
||||
test_scanout_restart_mode = std.mem.eql(u8, mode, "test-scanout-restart");
|
||||
}
|
||||
runtime.service.run(protocol.message_maximum, .{
|
||||
.service = .device_manager,
|
||||
@@ -565,8 +586,3 @@ pub fn main(init: runtime.process.Init) void {
|
||||
.on_notification = onNotification,
|
||||
});
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start; // pull the runtime entry shim into the image
|
||||
}
|
||||
|
||||
@@ -9,6 +9,7 @@ const runtime = @import("runtime");
|
||||
const display = runtime.display;
|
||||
const system = runtime.system;
|
||||
const time = runtime.time;
|
||||
const input = runtime.input;
|
||||
|
||||
pub fn main() void {
|
||||
const mode = display.info() orelse {
|
||||
@@ -27,8 +28,13 @@ pub fn main() void {
|
||||
const box = display.createLayer(0, box_y, box_w, box_h, 1) orelse return createFailed();
|
||||
_ = box.fill(0, 0, box_w, box_h, display.color(0xE0, 0x60, 0x40));
|
||||
|
||||
// A little cursor on top.
|
||||
const cursor = display.createLayer(40, 40, 12, 12, 2) orelse return createFailed();
|
||||
// A little cursor on top. Its position is signed (the layer API is i32) and clamped to
|
||||
// the screen; mouse motion arrives as relative deltas we accumulate below.
|
||||
var cursor_x: i32 = @intCast(mode.width / 2);
|
||||
var cursor_y: i32 = @intCast(mode.height / 2);
|
||||
const cursor_max_x: i32 = @as(i32, @intCast(mode.width)) - 12;
|
||||
const cursor_max_y: i32 = @as(i32, @intCast(mode.height)) - 12;
|
||||
const cursor = display.createLayer(cursor_x, cursor_y, 12, 12, 2) orelse return createFailed();
|
||||
_ = cursor.fill(0, 0, 12, 12, display.color(0xF0, 0xF0, 0xF0));
|
||||
|
||||
_ = display.present();
|
||||
@@ -38,7 +44,19 @@ pub fn main() void {
|
||||
var x: i32 = 0;
|
||||
var dx: i32 = 8;
|
||||
var frame: u32 = 0;
|
||||
|
||||
var mouse = input.subscribeMouse(); // type: ?input.MouseSubscriber
|
||||
if (mouse == null) _ = system.write("display-demo: no mouse; animating without it\n");
|
||||
|
||||
while (true) : (frame += 1) {
|
||||
if (mouse) |*ms| {
|
||||
if (ms.next()) |event| {
|
||||
cursor_x = clamp(cursor_x + event.dx, 0, cursor_max_x);
|
||||
cursor_y = clamp(cursor_y + event.dy, 0, cursor_max_y);
|
||||
_ = cursor.configure(cursor_x, cursor_y, 2, true);
|
||||
}
|
||||
}
|
||||
|
||||
x += dx;
|
||||
if (x <= 0) {
|
||||
x = 0;
|
||||
@@ -56,11 +74,13 @@ pub fn main() void {
|
||||
}
|
||||
}
|
||||
|
||||
/// Clamp `v` to the inclusive range [lo, hi].
|
||||
fn clamp(v: i32, lo: i32, hi: i32) i32 {
|
||||
if (v < lo) return lo;
|
||||
if (v > hi) return hi;
|
||||
return v;
|
||||
}
|
||||
|
||||
fn createFailed() void {
|
||||
_ = system.write("display-demo: create failed\n");
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start;
|
||||
}
|
||||
|
||||
@@ -0,0 +1,273 @@
|
||||
//! The compositor's **scanout backend** — how a finished frame reaches the panel
|
||||
//! (docs/display-v2.md). The compositor composes its layer stack into the backend's
|
||||
//! cacheable `surface()` and calls `present(damage)`; everything device-specific lives
|
||||
//! here. Today there is one backend, `Gop` — the firmware framebuffer: a cacheable back
|
||||
//! buffer streamed write-combining to the linear framebuffer. A native virtio-gpu backend
|
||||
//! slots in beside it later (V4); the compositor never learns which is active.
|
||||
|
||||
const std = @import("std");
|
||||
const runtime = @import("runtime");
|
||||
const compositor = @import("compositor.zig");
|
||||
|
||||
const system = runtime.system;
|
||||
const device = runtime.device;
|
||||
const ipc = runtime.ipc;
|
||||
const scanout_protocol = runtime.scanout_protocol;
|
||||
const Rect = compositor.Rect;
|
||||
const Surface = compositor.Surface;
|
||||
|
||||
/// The current display mode, as a backend reports it.
|
||||
pub const Info = struct { width: u32, height: u32, pitch: u32, format: u32 };
|
||||
|
||||
/// Enumeration scratch — a `DeviceDescriptor` is large, and only one scan is ever needed.
|
||||
var device_table: [64]device.DeviceDescriptor = undefined;
|
||||
|
||||
/// The GOP framebuffer backend: claims the kernel-seeded `display` device, maps the linear
|
||||
/// framebuffer write-combining as the front buffer, and keeps a cacheable back buffer of
|
||||
/// the same geometry as the compose target. `present` streams the damaged rectangle from
|
||||
/// the back buffer to the LFB (sequential WC writes; the LFB is never read). No mode-set,
|
||||
/// no vsync — the portable floor (docs/display-v2.md).
|
||||
pub const Gop = struct {
|
||||
device_id: u64,
|
||||
front: [*]volatile u8, // the LFB (write-combining)
|
||||
back: [*]u8, // cacheable compose target, same geometry
|
||||
width: u32,
|
||||
height: u32,
|
||||
pitch: u32,
|
||||
format: u32,
|
||||
|
||||
/// The framebuffer's id and geometry, captured together. `findDisplay` reads these out of
|
||||
/// the enumeration table and returns them by value, so the caller never re-reads the table
|
||||
/// across later syscalls (`device_enumerate` writes the whole table straight into this
|
||||
/// process's memory; reading a descriptor's tail again after other syscalls have run is a
|
||||
/// window we simply avoid by copying the few fields we need up front).
|
||||
const Found = struct { id: u64, width: u32, height: u32, pitch: u32, format: u32 };
|
||||
|
||||
/// The first `display`-class device with a *valid* (non-zero) geometry, or null. A zero
|
||||
/// geometry is treated as "not ready yet" so the caller retries — a real framebuffer always
|
||||
/// has a non-zero width, height, and pitch.
|
||||
fn findDisplay() ?Found {
|
||||
const total = device.enumerate(&device_table);
|
||||
const n = @min(total, device_table.len);
|
||||
for (device_table[0..n]) |*d| {
|
||||
if (d.class != @intFromEnum(device.DeviceClass.display)) continue;
|
||||
if (d.display.width == 0 or d.display.height == 0 or d.display.pitch == 0) continue;
|
||||
return .{ .id = d.id, .width = d.display.width, .height = d.display.height, .pitch = d.display.pitch, .format = d.display.format };
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Claim the framebuffer (retrying while discovery catches up), map the LFB, and
|
||||
/// allocate the back buffer. Null if there is no framebuffer or a mapping fails.
|
||||
pub fn init() ?Gop {
|
||||
var tries: u32 = 0;
|
||||
const found = while (tries < 100) : (tries += 1) {
|
||||
if (findDisplay()) |f| break f;
|
||||
system.sleep(50);
|
||||
} else {
|
||||
_ = system.write("display: no framebuffer device (headless?)\n");
|
||||
return null;
|
||||
};
|
||||
|
||||
if (!device.claim(found.id)) {
|
||||
_ = system.write("display: could not claim the framebuffer\n");
|
||||
return null;
|
||||
}
|
||||
// Resource 0 is the framebuffer memory window; the kernel maps it write-combining
|
||||
// because the resource carries that flag (docs/display-plan.md D1).
|
||||
const front_base = device.mmioMap(found.id, 0) orelse {
|
||||
_ = system.write("display: could not map the framebuffer\n");
|
||||
return null;
|
||||
};
|
||||
const size = @as(usize, found.height) * found.pitch;
|
||||
const back_base = system.mmap(size, system.PROT_READ | system.PROT_WRITE);
|
||||
if (system.mmapFailed(back_base)) {
|
||||
_ = system.write("display: could not allocate the back buffer\n");
|
||||
return null;
|
||||
}
|
||||
return .{
|
||||
.device_id = found.id,
|
||||
.front = @ptrFromInt(front_base),
|
||||
.back = @ptrFromInt(back_base),
|
||||
.width = found.width,
|
||||
.height = found.height,
|
||||
.pitch = found.pitch,
|
||||
.format = found.format,
|
||||
};
|
||||
}
|
||||
|
||||
pub fn info(self: *const Gop) Info {
|
||||
return .{ .width = self.width, .height = self.height, .pitch = self.pitch, .format = self.format };
|
||||
}
|
||||
|
||||
/// The cacheable compose target (the back buffer).
|
||||
pub fn surface(self: *const Gop) Surface {
|
||||
return .{
|
||||
.pixels = @ptrCast(@alignCast(self.back)),
|
||||
.stride = self.pitch / 4, // pitch is bytes; a 32-bpp row is pitch/4 pixels
|
||||
.width = self.width,
|
||||
.height = self.height,
|
||||
};
|
||||
}
|
||||
|
||||
/// Stream the damaged rectangle from the back buffer to the write-combining LFB, row by
|
||||
/// row (sequential writes — what WC memory wants; the LFB is never read).
|
||||
pub fn present(self: *const Gop, damage: Rect) void {
|
||||
const c = damage.intersect(.{ .x = 0, .y = 0, .w = @intCast(self.width), .h = @intCast(self.height) });
|
||||
if (c.isEmpty()) return;
|
||||
var y: i32 = c.y;
|
||||
while (y < c.bottom()) : (y += 1) {
|
||||
const off = @as(usize, @intCast(y)) * self.pitch;
|
||||
const src: [*]const u32 = @ptrCast(@alignCast(self.back + off));
|
||||
const dst: [*]volatile u32 = @ptrCast(@alignCast(self.front + off));
|
||||
var x: i32 = c.x;
|
||||
while (x < c.right()) : (x += 1) dst[@intCast(x)] = src[@intCast(x)];
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
/// A display mode the native backend can switch to.
|
||||
pub const Mode = scanout_protocol.Mode;
|
||||
|
||||
/// The native virtio-gpu backend: the compositor composes into a **shared** scanout surface
|
||||
/// (an `shm` region the driver created and handed over) and `present` asks the driver to put
|
||||
/// a frame on the panel over its `.scanout` endpoint. Unlike GOP there is no local copy — the
|
||||
/// surface *is* the device's resource backing, so compositing writes land straight where the
|
||||
/// driver transfers-and-flushes from (x86 DMA is cache-coherent, so the cacheable shared pages
|
||||
/// need no explicit flush). Built by the display service when a driver announces (V4). The
|
||||
/// surface is sized to the driver's largest mode, so `stride` (its row width) is fixed while
|
||||
/// `width`/`height` — the active mode — change under `setMode` (V5).
|
||||
pub const VirtioGpu = struct {
|
||||
pixels: [*]u32, // the shared scanout surface, mapped into the compositor
|
||||
stride: u32, // the surface's row stride in pixels (the driver's max mode width) — fixed
|
||||
width: u32, // the active mode
|
||||
height: u32,
|
||||
format: u32,
|
||||
scanout: ipc.Handle, // the driver's present + mode channel (looked up on `.scanout`)
|
||||
|
||||
pub fn info(self: *const VirtioGpu) Info {
|
||||
return .{ .width = self.width, .height = self.height, .pitch = self.stride * 4, .format = self.format };
|
||||
}
|
||||
pub fn surface(self: *const VirtioGpu) Surface {
|
||||
return .{ .pixels = self.pixels, .stride = self.stride, .width = self.width, .height = self.height };
|
||||
}
|
||||
/// Ask the driver to present. The composited pixels are already in the shared surface, so
|
||||
/// this is a single request over `.scanout`; the driver transfers + fenced-flushes.
|
||||
pub fn present(self: *const VirtioGpu, damage: Rect) void {
|
||||
_ = damage;
|
||||
var request = scanout_protocol.Request{
|
||||
.operation = @intFromEnum(scanout_protocol.Operation.present),
|
||||
.width = self.width,
|
||||
.height = self.height,
|
||||
};
|
||||
var reply: [scanout_protocol.reply_size]u8 = undefined;
|
||||
_ = ipc.call(self.scanout, std.mem.asBytes(&request), &reply) catch {};
|
||||
}
|
||||
/// Fill `out` with the driver's offered modes; returns how many were written.
|
||||
pub fn modes(self: *const VirtioGpu, out: []Mode) usize {
|
||||
var request = scanout_protocol.Request{ .operation = @intFromEnum(scanout_protocol.Operation.get_modes) };
|
||||
var reply: [scanout_protocol.modes_reply_size]u8 = undefined;
|
||||
const n = ipc.call(self.scanout, std.mem.asBytes(&request), &reply) catch return 0;
|
||||
if (n < scanout_protocol.modes_reply_size) return 0;
|
||||
const answer = std.mem.bytesToValue(scanout_protocol.ModesReply, reply[0..scanout_protocol.modes_reply_size]);
|
||||
if (answer.status != 0) return 0;
|
||||
const count = @min(@min(answer.count, scanout_protocol.max_modes), out.len);
|
||||
for (0..count) |i| out[i] = answer.modes[i];
|
||||
return count;
|
||||
}
|
||||
/// Change the scanout resolution. On success the active `width`/`height` update (the shared
|
||||
/// surface — sized to the max mode — is unchanged, so `stride` stays put).
|
||||
pub fn setMode(self: *VirtioGpu, w: u32, h: u32) bool {
|
||||
if (w == 0 or h == 0 or w > self.stride) return false;
|
||||
var request = scanout_protocol.Request{
|
||||
.operation = @intFromEnum(scanout_protocol.Operation.set_mode),
|
||||
.width = w,
|
||||
.height = h,
|
||||
};
|
||||
var reply: [scanout_protocol.reply_size]u8 = undefined;
|
||||
const n = ipc.call(self.scanout, std.mem.asBytes(&request), &reply) catch return false;
|
||||
if (n < scanout_protocol.reply_size) return false;
|
||||
if (std.mem.bytesToValue(scanout_protocol.Reply, reply[0..scanout_protocol.reply_size]).status != 0) return false;
|
||||
self.width = w;
|
||||
self.height = h;
|
||||
return true;
|
||||
}
|
||||
};
|
||||
|
||||
/// The pluggable scanout backend. A tagged union so the compositor holds one value and
|
||||
/// dispatches without caring which is active; the `virtio` native backend joins `gop` at V4.
|
||||
pub const Backend = union(enum) {
|
||||
gop: Gop,
|
||||
virtio: VirtioGpu,
|
||||
|
||||
pub fn info(self: *const Backend) Info {
|
||||
return switch (self.*) {
|
||||
inline else => |*b| b.info(),
|
||||
};
|
||||
}
|
||||
pub fn surface(self: *const Backend) Surface {
|
||||
return switch (self.*) {
|
||||
inline else => |*b| b.surface(),
|
||||
};
|
||||
}
|
||||
pub fn present(self: *const Backend, damage: Rect) void {
|
||||
switch (self.*) {
|
||||
inline else => |*b| b.present(damage),
|
||||
}
|
||||
}
|
||||
/// The modes this backend can switch to (none for GOP); returns how many were written.
|
||||
pub fn modes(self: *const Backend, out: []Mode) usize {
|
||||
return switch (self.*) {
|
||||
.virtio => |*v| v.modes(out),
|
||||
.gop => 0,
|
||||
};
|
||||
}
|
||||
/// Change the resolution; false if this backend can't mode-set or the mode was refused.
|
||||
pub fn setMode(self: *Backend, w: u32, h: u32) bool {
|
||||
return switch (self.*) {
|
||||
.virtio => |*v| v.setMode(w, h),
|
||||
.gop => false,
|
||||
};
|
||||
}
|
||||
/// Whether this backend supports runtime mode-setting (GOP: no; virtio-gpu: yes, V5).
|
||||
pub fn canModeSet(self: *const Backend) bool {
|
||||
return switch (self.*) {
|
||||
.gop => false,
|
||||
.virtio => true,
|
||||
};
|
||||
}
|
||||
/// Whether this backend has a vblank/fence for tear-free present (virtio-gpu: yes, V5 — every
|
||||
/// flush is fenced, so the device signals completion when the frame is actually on screen).
|
||||
pub fn hasVsync(self: *const Backend) bool {
|
||||
return switch (self.*) {
|
||||
.gop => false,
|
||||
.virtio => true,
|
||||
};
|
||||
}
|
||||
};
|
||||
|
||||
/// Which backend to use. The pure selection *decision* is `chooseKind`; `select` below
|
||||
/// binds it to the (syscall-bound) bring-up.
|
||||
pub const Kind = enum { gop, virtio };
|
||||
|
||||
/// The selection decision, factored out of bring-up so it stays pure and host-testable:
|
||||
/// prefer a native driver when one has announced itself (docs/display-v2.md V4), else the
|
||||
/// GOP floor. Trivial today; it grows real inputs when native detection lands.
|
||||
pub fn chooseKind(native_available: bool) Kind {
|
||||
return if (native_available) .virtio else .gop;
|
||||
}
|
||||
|
||||
/// Pick and bring up the best available backend. Today the GOP framebuffer is the only one
|
||||
/// (`chooseKind(false)` → `.gop`), so this is `Gop.init()`. V4 adds the native-if-present
|
||||
/// branch, with GOP as the floor.
|
||||
pub fn select() ?Backend {
|
||||
return switch (chooseKind(false)) {
|
||||
.gop => .{ .gop = Gop.init() orelse return null },
|
||||
.virtio => unreachable, // no native detection yet (V4)
|
||||
};
|
||||
}
|
||||
|
||||
test "selection prefers native when present, else the gop floor" {
|
||||
try std.testing.expectEqual(Kind.gop, chooseKind(false));
|
||||
try std.testing.expectEqual(Kind.virtio, chooseKind(true));
|
||||
}
|
||||
+199
-139
@@ -1,43 +1,48 @@
|
||||
//! /system/services/display — the display service (docs/display.md). A ring-3 process
|
||||
//! that claims the framebuffer the kernel seeded (docs/display-plan.md D1), owns it as a
|
||||
//! **write-combining front buffer**, composites an ordered stack of **layers** into a
|
||||
//! **cacheable back buffer**, and presents finished frames — the GUI track's compositor,
|
||||
//! the sibling of the input service. Reached by name over `ServiceId.display`.
|
||||
//! /system/services/display — the display service (docs/display.md, docs/display-v2.md).
|
||||
//! A ring-3 compositor: it composes an ordered stack of **layers** into a cacheable
|
||||
//! surface and presents finished frames. Scanout — how a frame reaches the panel — is a
|
||||
//! pluggable **backend** ([backend.zig](backend.zig)): the GOP framebuffer today, a native
|
||||
//! virtio-gpu driver later; this file never learns which is active. It owns the layer stack
|
||||
//! and damage tracking; the pixel math is the pure, host-tested
|
||||
//! [compositor.zig](compositor.zig).
|
||||
//!
|
||||
//! A layer is a server-owned surface (its own cacheable buffer) with a screen position,
|
||||
//! z-order, and visibility. Clients create layers and draw into them by command
|
||||
//! (`fill_rect`, `blit_tile`), mark `damage`, and ask for a `present`; the compositor
|
||||
//! repaints only the damaged region — clear it, paint the visible layers bottom-to-top,
|
||||
//! flush it to the screen. The pixel math lives in the pure, host-tested
|
||||
//! [compositor.zig](compositor.zig); this file wires real surfaces and the framebuffer to
|
||||
//! it. Shared-memory client surfaces are a later milestone (docs/display.md).
|
||||
//! z-order, and visibility. Clients create layers, draw into them by command (`fill_rect`,
|
||||
//! `blit_tile`), mark `damage`, and ask for a `present`; the compositor repaints only the
|
||||
//! damaged region — clear it, paint the visible layers bottom-to-top into the backend's
|
||||
//! surface, then `backend.present(damage)`. Shared-memory client surfaces are later
|
||||
//! (docs/display-v2.md).
|
||||
|
||||
const std = @import("std");
|
||||
const runtime = @import("runtime");
|
||||
const compositor = @import("compositor.zig");
|
||||
const backend_mod = @import("backend.zig");
|
||||
|
||||
const protocol = runtime.display_protocol;
|
||||
const ipc = runtime.ipc;
|
||||
const system = runtime.system;
|
||||
const device = runtime.device;
|
||||
const Rect = compositor.Rect;
|
||||
const Surface = compositor.Surface;
|
||||
|
||||
/// The claimed framebuffer and its off-screen twin. The front buffer is the LFB —
|
||||
/// write-combining, so it is **only ever written**, never read; all compositing happens
|
||||
/// in the cacheable back buffer, which is then streamed to the front (docs/display.md).
|
||||
const Display = struct {
|
||||
device_id: u64,
|
||||
front: [*]volatile u8, // the LFB (write-combining)
|
||||
back: [*]u8, // cacheable, same geometry
|
||||
width: u32,
|
||||
height: u32,
|
||||
pitch: u32, // bytes per row (shared by both buffers)
|
||||
format: u32, // a device-abi DisplayFormat value
|
||||
frames: u64 = 0,
|
||||
};
|
||||
/// The active scanout backend — the GOP framebuffer at boot, upgraded to a native driver
|
||||
/// (virtio-gpu) when one announces itself (V4).
|
||||
var backend: backend_mod.Backend = undefined;
|
||||
var frames: u64 = 0;
|
||||
|
||||
var display: Display = undefined;
|
||||
/// This service's endpoint, kept so `attach_scanout` can arm a one-shot timer: the very first
|
||||
/// native present must happen in a *later* loop iteration, after the reply to the driver's
|
||||
/// announce has unblocked it and it is serving its `.scanout` channel — presenting inline
|
||||
/// would deadlock (we'd call the driver while it waits on our reply).
|
||||
var service_endpoint: ipc.Handle = 0;
|
||||
|
||||
/// Set when the backend has just been upgraded to virtio-gpu: the next present repaints the
|
||||
/// whole screen into the shared surface and reads a pixel back to confirm the frame landed.
|
||||
var pending_native_verify: bool = false;
|
||||
|
||||
/// Set alongside it: after the native present is verified, run the mode-set self-check once
|
||||
/// (query the driver's modes, switch to a different one, confirm the geometry changed) — the
|
||||
/// serial proof the runtime-resolution-change + fenced-present paths work (V5).
|
||||
var pending_modeset_check: bool = false;
|
||||
|
||||
/// The wallpaper the compositor clears damaged regions to before painting layers.
|
||||
var background: u32 = 0;
|
||||
@@ -60,23 +65,11 @@ const Layer = struct {
|
||||
var layers: [maximum_layers]Layer = [_]Layer{.{}} ** maximum_layers;
|
||||
var damage: Rect = Rect.empty;
|
||||
|
||||
/// Enumeration buffer kept off the stack — a `DeviceDescriptor` is large, and this
|
||||
/// service only ever needs one scan.
|
||||
var device_table: [64]device.DeviceDescriptor = undefined;
|
||||
|
||||
// --- geometry helpers -------------------------------------------------------
|
||||
|
||||
fn screenRect() Rect {
|
||||
return .{ .x = 0, .y = 0, .w = @intCast(display.width), .h = @intCast(display.height) };
|
||||
}
|
||||
|
||||
fn backSurface() Surface {
|
||||
return .{
|
||||
.pixels = @ptrCast(@alignCast(display.back)),
|
||||
.stride = display.pitch / 4, // pitch is bytes; a 32-bpp row is pitch/4 pixels
|
||||
.width = display.width,
|
||||
.height = display.height,
|
||||
};
|
||||
const m = backend.info();
|
||||
return .{ .x = 0, .y = 0, .w = @intCast(m.width), .h = @intCast(m.height) };
|
||||
}
|
||||
|
||||
fn layerScreenRect(l: *const Layer) Rect {
|
||||
@@ -159,11 +152,11 @@ fn destroyLayer(id: u32) bool {
|
||||
|
||||
// --- compositing + present --------------------------------------------------
|
||||
|
||||
/// Repaint the damaged region `clip` of the back buffer: clear it to the background, then
|
||||
/// paint every visible layer that overlaps it, bottom to top (ascending z).
|
||||
/// Repaint the damaged region `clip` of the backend's compose surface: clear it to the
|
||||
/// background, then paint every visible layer that overlaps it, bottom to top (ascending z).
|
||||
fn compositeInto(clip: Rect) void {
|
||||
const back = backSurface();
|
||||
compositor.fillRect(back, clip, background);
|
||||
const target = backend.surface();
|
||||
compositor.fillRect(target, clip, background);
|
||||
|
||||
// z-order the used, visible layers (n ≤ 16; a plain insertion sort of indices).
|
||||
var order: [maximum_layers]u32 = undefined;
|
||||
@@ -184,56 +177,141 @@ fn compositeInto(clip: Rect) void {
|
||||
|
||||
for (order[0..n]) |i| {
|
||||
const l = layers[i];
|
||||
compositor.composite(back, l.x, l.y, l.surface, clip);
|
||||
compositor.composite(target, l.x, l.y, l.surface, clip);
|
||||
}
|
||||
}
|
||||
|
||||
/// Stream the damaged rectangle from the cacheable back buffer to the write-combining
|
||||
/// front buffer, row by row (sequential writes — what WC memory wants; we never read the
|
||||
/// front buffer). Only the visible width of each row is touched.
|
||||
fn flushRect(rect: Rect) void {
|
||||
const c = rect.intersect(screenRect());
|
||||
if (c.isEmpty()) return;
|
||||
var y: i32 = c.y;
|
||||
while (y < c.bottom()) : (y += 1) {
|
||||
const off = @as(usize, @intCast(y)) * display.pitch;
|
||||
const src: [*]const u32 = @ptrCast(@alignCast(display.back + off));
|
||||
const dst: [*]volatile u32 = @ptrCast(@alignCast(display.front + off));
|
||||
var x: i32 = c.x;
|
||||
while (x < c.right()) : (x += 1) dst[@intCast(x)] = src[@intCast(x)];
|
||||
}
|
||||
}
|
||||
|
||||
/// Composite and flush the accumulated damage, then clear it. A no-op when nothing is
|
||||
/// dirty. The frame counter advances regardless, so callers can name frames.
|
||||
/// Composite the accumulated damage into the backend's surface, hand it to the backend to
|
||||
/// put on screen, then clear the damage. A no-op when nothing is dirty. The frame counter
|
||||
/// advances regardless, so callers can name frames.
|
||||
fn present() void {
|
||||
const dirty = damage.intersect(screenRect());
|
||||
if (!dirty.isEmpty()) {
|
||||
compositeInto(dirty);
|
||||
flushRect(dirty);
|
||||
backend.present(dirty);
|
||||
}
|
||||
damage = Rect.empty;
|
||||
display.frames += 1;
|
||||
frames += 1;
|
||||
|
||||
// The first present after a native upgrade confirms the composited frame actually reached
|
||||
// the shared scanout surface (the automated stand-in for "it's on screen").
|
||||
if (pending_native_verify and !dirty.isEmpty()) {
|
||||
pending_native_verify = false;
|
||||
verifyNativePresent();
|
||||
}
|
||||
}
|
||||
|
||||
/// Read a pixel straight back from the shared scanout surface after a native present. The
|
||||
/// surface starts zeroed, so a non-zero centre pixel means the compositor wrote the frame into
|
||||
/// the pages the driver transfers-and-flushes from — that, plus the driver acking the present
|
||||
/// over `.scanout`, is the serial proof the native path works.
|
||||
fn verifyNativePresent() void {
|
||||
const s = backend.surface();
|
||||
const sample = s.pixels[@as(usize, s.height / 2) * s.stride + s.width / 2];
|
||||
if (sample != 0) {
|
||||
_ = system.write("display: native present verified\n");
|
||||
} else {
|
||||
_ = system.write("display: native present FAILED (blank surface)\n");
|
||||
}
|
||||
}
|
||||
|
||||
/// A native scanout driver announced itself: map the shared surface it handed over, find its
|
||||
/// present channel, switch the backend to virtio-gpu, and queue a full-screen repaint. The
|
||||
/// present is deferred to a timer (see `service_endpoint`) so it happens after this reply
|
||||
/// unblocks the driver and it starts serving `.scanout`.
|
||||
fn attachScanout(stride: u32, width: u32, height: u32, format: u32, capability: ?ipc.Handle, reply: []u8) usize {
|
||||
const cap = capability orelse return fail(reply);
|
||||
if (width == 0 or height == 0 or stride < width) return fail(reply);
|
||||
const mapped = runtime.shm.map(cap) orelse return fail(reply);
|
||||
const scanout = ipc.lookup(.scanout) orelse return fail(reply);
|
||||
// A second announce means the driver died and was restarted (V6): re-attach to its fresh
|
||||
// scanout. (The previous shared mapping leaks — there is no shm_unmap syscall yet — but the
|
||||
// frames are the dead driver's, reclaimed on its exit; a handful across a crash is benign.)
|
||||
const reattach = switch (backend) {
|
||||
.virtio => true,
|
||||
else => false,
|
||||
};
|
||||
|
||||
backend = .{ .virtio = .{
|
||||
.pixels = @ptrCast(@alignCast(mapped)),
|
||||
.stride = stride,
|
||||
.width = width,
|
||||
.height = height,
|
||||
.format = format,
|
||||
.scanout = scanout,
|
||||
} };
|
||||
background = protocol.pack(format, 0x20, 0x30, 0x48); // re-pack the wallpaper for the mode
|
||||
addDamage(screenRect()); // the whole new surface must be painted
|
||||
pending_native_verify = true;
|
||||
if (!reattach) pending_modeset_check = true; // the mode-set self-check runs once, on first upgrade
|
||||
_ = system.timerOnce(service_endpoint, 50); // present once the driver is serving .scanout
|
||||
_ = system.write(if (reattach)
|
||||
"display: scanout re-attached\n"
|
||||
else
|
||||
"display: scanout upgraded to virtio-gpu\n");
|
||||
return ok(reply);
|
||||
}
|
||||
|
||||
/// After the native upgrade is verified, prove the runtime-resolution-change and fenced-present
|
||||
/// paths: query the driver's modes, switch to one that differs from the current, re-composite
|
||||
/// the whole screen at the new size, and confirm the backend now reports that geometry. The
|
||||
/// present goes through the driver's fenced flush, so a clean present is a vsync present.
|
||||
fn modesetSelfCheck() void {
|
||||
if (!backend.canModeSet()) return;
|
||||
var mode_list: [4]backend_mod.Mode = undefined;
|
||||
const count = backend.modes(&mode_list);
|
||||
if (count == 0) {
|
||||
_ = system.write("display: mode-set self-check: no modes reported\n");
|
||||
return;
|
||||
}
|
||||
const current = backend.info();
|
||||
var target: ?backend_mod.Mode = null;
|
||||
for (mode_list[0..count]) |m| {
|
||||
if (m.width != current.width or m.height != current.height) {
|
||||
target = m;
|
||||
break;
|
||||
}
|
||||
}
|
||||
const wanted = target orelse {
|
||||
_ = system.write("display: mode-set self-check: no alternate mode offered\n");
|
||||
return;
|
||||
};
|
||||
if (!backend.setMode(wanted.width, wanted.height)) {
|
||||
_ = system.write("display: mode set FAILED\n");
|
||||
return;
|
||||
}
|
||||
addDamage(screenRect()); // repaint the whole screen at the new resolution, then present it
|
||||
present();
|
||||
|
||||
const now = backend.info();
|
||||
if (now.width == wanted.width and now.height == wanted.height) {
|
||||
var line: [80]u8 = undefined;
|
||||
_ = system.write(std.fmt.bufPrint(&line, "display: mode set to {d}x{d}, verified\n", .{ now.width, now.height }) catch "display: mode set, verified\n");
|
||||
if (backend.hasVsync()) _ = system.write("display: vsync present ok\n");
|
||||
} else {
|
||||
_ = system.write("display: mode set FAILED (geometry unchanged)\n");
|
||||
}
|
||||
}
|
||||
|
||||
// --- startup self-check -----------------------------------------------------
|
||||
|
||||
/// Prove the compositor wiring on the real framebuffer: two overlapping opaque layers,
|
||||
/// Prove the compositor wiring on the real backend: two overlapping opaque layers,
|
||||
/// composited, must show the top layer in the overlap and the bottom layer outside it.
|
||||
/// Exercises the whole path — mmap surfaces, the z-sort, damage, composite into the back
|
||||
/// buffer — and reads the composited result back. Cleans up after itself.
|
||||
/// Exercises the whole path — mmap surfaces, the z-sort, damage, composite into the
|
||||
/// backend surface — and reads the composited result back. Cleans up after itself.
|
||||
fn selfCheck() void {
|
||||
const red = protocol.pack(display.format, 0xC0, 0x20, 0x20);
|
||||
const green = protocol.pack(display.format, 0x20, 0xC0, 0x20);
|
||||
const format = backend.info().format;
|
||||
const red = protocol.pack(format, 0xC0, 0x20, 0x20);
|
||||
const green = protocol.pack(format, 0x20, 0xC0, 0x20);
|
||||
const bottom = createLayer(100, 100, 80, 80, 0, true) orelse return fail_check("create");
|
||||
const top = createLayer(140, 140, 80, 80, 1, true) orelse return fail_check("create");
|
||||
_ = fillLayer(bottom, Rect.init(0, 0, 80, 80), red);
|
||||
_ = fillLayer(top, Rect.init(0, 0, 80, 80), green);
|
||||
present();
|
||||
|
||||
const back = backSurface();
|
||||
const overlap = back.pixels[@as(usize, 150) * back.stride + 150]; // in both layers → top
|
||||
const bottom_only = back.pixels[@as(usize, 110) * back.stride + 110]; // bottom only
|
||||
const surface = backend.surface();
|
||||
const overlap = surface.pixels[@as(usize, 150) * surface.stride + 150]; // in both → top
|
||||
const bottom_only = surface.pixels[@as(usize, 110) * surface.stride + 110]; // bottom only
|
||||
|
||||
_ = destroyLayer(top);
|
||||
_ = destroyLayer(bottom);
|
||||
@@ -252,67 +330,22 @@ fn fail_check(_: []const u8) void {
|
||||
|
||||
// --- service ----------------------------------------------------------------
|
||||
|
||||
/// The framebuffer node the kernel seeded (`DeviceClass.display`), or null if none.
|
||||
fn findDisplay() ?device.DeviceDescriptor {
|
||||
const total = device.enumerate(&device_table);
|
||||
const n = @min(total, device_table.len);
|
||||
for (device_table[0..n]) |d| {
|
||||
if (d.class == @intFromEnum(device.DeviceClass.display)) return d;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
fn initialise(endpoint: ipc.Handle) bool {
|
||||
_ = endpoint;
|
||||
service_endpoint = endpoint;
|
||||
|
||||
// Find the framebuffer, retrying while device discovery catches up with our spawn.
|
||||
var tries: u32 = 0;
|
||||
const found = while (tries < 100) : (tries += 1) {
|
||||
if (findDisplay()) |d| break d;
|
||||
system.sleep(50);
|
||||
} else {
|
||||
_ = system.write("display: no framebuffer device (headless?)\n");
|
||||
return false; // clean exit: nothing to drive
|
||||
};
|
||||
// Pick the scanout backend (GOP today). It logs the reason on failure.
|
||||
backend = backend_mod.select() orelse return false;
|
||||
const mode = backend.info();
|
||||
background = protocol.pack(mode.format, 0x20, 0x30, 0x48); // a dark slate wallpaper
|
||||
|
||||
if (!device.claim(found.id)) {
|
||||
_ = system.write("display: could not claim the framebuffer\n");
|
||||
return false;
|
||||
}
|
||||
// Resource 0 is the framebuffer memory window; the kernel maps it write-combining
|
||||
// because the resource carries that flag (docs/display-plan.md D1).
|
||||
const front_base = device.mmioMap(found.id, 0) orelse {
|
||||
_ = system.write("display: could not map the framebuffer\n");
|
||||
return false;
|
||||
};
|
||||
|
||||
const geometry = found.display;
|
||||
const size = @as(usize, geometry.height) * geometry.pitch;
|
||||
const back_base = system.mmap(size, system.PROT_READ | system.PROT_WRITE);
|
||||
if (system.mmapFailed(back_base)) {
|
||||
_ = system.write("display: could not allocate the back buffer\n");
|
||||
return false;
|
||||
}
|
||||
|
||||
display = .{
|
||||
.device_id = found.id,
|
||||
.front = @ptrFromInt(front_base),
|
||||
.back = @ptrFromInt(back_base),
|
||||
.width = geometry.width,
|
||||
.height = geometry.height,
|
||||
.pitch = geometry.pitch,
|
||||
.format = geometry.format,
|
||||
};
|
||||
background = protocol.pack(display.format, 0x20, 0x30, 0x48); // a dark slate wallpaper
|
||||
|
||||
// Clear the whole screen through the back buffer → present path (double buffering:
|
||||
// no direct-to-LFB drawing).
|
||||
// Clear the whole screen through the compose surface → present path (double buffering:
|
||||
// no direct-to-scanout drawing).
|
||||
addDamage(screenRect());
|
||||
present();
|
||||
|
||||
var line: [96]u8 = undefined;
|
||||
_ = system.write(std.fmt.bufPrint(&line, "display: online {d}x{d} pitch {d} format {d}\n", .{
|
||||
display.width, display.height, display.pitch, display.format,
|
||||
mode.width, mode.height, mode.pitch, mode.format,
|
||||
}) catch "display: online\n");
|
||||
_ = system.write("display: presented frame 0\n");
|
||||
|
||||
@@ -336,20 +369,16 @@ fn fail(reply: []u8) usize {
|
||||
|
||||
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Handle) usize {
|
||||
_ = sender;
|
||||
_ = capability;
|
||||
if (message.len < protocol.request_size) return fail(reply);
|
||||
const request = std.mem.bytesToValue(protocol.Request, message[0..protocol.request_size]);
|
||||
const payload = message[protocol.request_size..];
|
||||
// Switch on the raw operation value — an out-of-range one must fail cleanly, not
|
||||
// panic an `@enumFromInt`.
|
||||
switch (request.operation) {
|
||||
@intFromEnum(protocol.Operation.info) => return writeReply(reply, .{
|
||||
.status = 0,
|
||||
.width = display.width,
|
||||
.height = display.height,
|
||||
.pitch = display.pitch,
|
||||
.format = display.format,
|
||||
}),
|
||||
@intFromEnum(protocol.Operation.info) => {
|
||||
const m = backend.info();
|
||||
return writeReply(reply, .{ .status = 0, .width = m.width, .height = m.height, .pitch = m.pitch, .format = m.format });
|
||||
},
|
||||
@intFromEnum(protocol.Operation.create_layer) => {
|
||||
// x/y are signed coordinates carried in the u32 wire fields — reinterpret the
|
||||
// bits (@bitCast), don't range-check (@intCast) which a negative would fail.
|
||||
@@ -379,19 +408,50 @@ fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Han
|
||||
present();
|
||||
return ok(reply);
|
||||
},
|
||||
@intFromEnum(protocol.Operation.attach_scanout) => {
|
||||
return attachScanout(request.x, request.width, request.height, request.colour, capability, reply);
|
||||
},
|
||||
@intFromEnum(protocol.Operation.set_mode) => {
|
||||
if (!backend.setMode(request.width, request.height)) return fail(reply);
|
||||
addDamage(screenRect()); // repaint the whole screen at the new resolution
|
||||
present();
|
||||
return ok(reply);
|
||||
},
|
||||
@intFromEnum(protocol.Operation.get_modes) => {
|
||||
var list: [4]backend_mod.Mode = undefined;
|
||||
const count = backend.modes(&list);
|
||||
var response = protocol.ModesReply{ .status = 0, .count = @intCast(count), .modes = undefined };
|
||||
for (0..protocol.max_modes) |i| {
|
||||
response.modes[i] = if (i < count)
|
||||
.{ .width = list[i].width, .height = list[i].height }
|
||||
else
|
||||
.{ .width = 0, .height = 0 };
|
||||
}
|
||||
const bytes = std.mem.asBytes(&response);
|
||||
@memcpy(reply[0..bytes.len], bytes);
|
||||
return bytes.len;
|
||||
},
|
||||
else => return fail(reply),
|
||||
}
|
||||
}
|
||||
|
||||
/// The only notification the compositor arms is the post-attach present timer: repaint the
|
||||
/// screen into the freshly attached native surface, verify the frame landed, then run the
|
||||
/// one-shot mode-set self-check (V5).
|
||||
fn onNotification(badge: u64) void {
|
||||
_ = badge;
|
||||
present(); // native present + verify (first timer fire after the upgrade)
|
||||
if (pending_modeset_check) {
|
||||
pending_modeset_check = false;
|
||||
modesetSelfCheck();
|
||||
}
|
||||
}
|
||||
|
||||
pub fn main() void {
|
||||
runtime.service.run(protocol.message_maximum, .{
|
||||
.service = .display,
|
||||
.init = initialise,
|
||||
.on_message = onMessage,
|
||||
.on_notification = onNotification,
|
||||
});
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start;
|
||||
}
|
||||
|
||||
@@ -24,6 +24,17 @@ pub const Operation = enum(u32) {
|
||||
damage = 6,
|
||||
/// present(): composite the dirty layers and flush to the screen.
|
||||
present = 7,
|
||||
/// attach_scanout(x=stride, width, height, colour=format) + <surface capability>: a native
|
||||
/// scanout driver announces itself, handing over the shared scanout surface as an `ipc_call`
|
||||
/// send_cap. The compositor maps it, looks up the driver's `.scanout` present channel, and
|
||||
/// upgrades off the GOP floor (docs/display-v2.md V4). `x` is the surface's row stride in
|
||||
/// pixels, `colour` the DisplayFormat.
|
||||
attach_scanout = 8,
|
||||
/// set_mode(width, height): change the display resolution — only a native backend that
|
||||
/// reports `canModeSet` honours it; on the GOP floor it fails (docs/display-v2.md V5).
|
||||
set_mode = 9,
|
||||
/// get_modes() -> ModesReply: the resolutions the display can switch to (empty on GOP).
|
||||
get_modes = 10,
|
||||
};
|
||||
|
||||
/// The fixed request header. A `blit_tile`'s pixel payload (width*height 32-bit pixels)
|
||||
@@ -54,6 +65,18 @@ pub const Reply = extern struct {
|
||||
reserved2: u32 = 0,
|
||||
};
|
||||
|
||||
/// One selectable display mode.
|
||||
pub const Mode = extern struct { width: u32, height: u32 };
|
||||
pub const max_modes = 4;
|
||||
|
||||
/// The reply to `get_modes`: a small fixed list of resolutions the display can switch to.
|
||||
pub const ModesReply = extern struct {
|
||||
status: i32,
|
||||
count: u32,
|
||||
modes: [max_modes]Mode,
|
||||
};
|
||||
pub const modes_reply_size: usize = @sizeOf(ModesReply);
|
||||
|
||||
/// The IPC message size — the kernel caps every message at `MESSAGE_MAXIMUM` (256 bytes,
|
||||
/// system/kernel/ipc-synchronous.zig), so this matches it (a larger receive/reply buffer
|
||||
/// is rejected with -E2BIG). A `blit_tile` therefore carries only a *small* tile inline —
|
||||
|
||||
@@ -0,0 +1,49 @@
|
||||
//! The scanout wire protocol — what the compositor says to a native scanout driver (e.g.
|
||||
//! virtio-gpu) over its well-known `.scanout` endpoint to put a composited frame on screen.
|
||||
//! The driver owns the panel and the shared scanout surface it handed the compositor (via the
|
||||
//! display service's `attach_scanout`); the compositor composites into that surface, then asks
|
||||
//! the driver to present a damaged rectangle. Tiny by design — one present request. Separate
|
||||
//! from the display protocol because the directions differ: clients call the compositor over
|
||||
//! `.display`; the compositor calls the driver over `.scanout`. See docs/display-v2.md.
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
pub const Operation = enum(u32) {
|
||||
/// present(x, y, width, height): put the given rectangle of the shared scanout surface on
|
||||
/// the panel (on virtio-gpu: transfer-to-host of the region, then a fenced resource flush).
|
||||
present = 0,
|
||||
/// get_modes() -> ModesReply: the display modes this scanout can switch to (V5).
|
||||
get_modes = 1,
|
||||
/// set_mode(width, height): change the scanout resolution — the shared surface is sized to
|
||||
/// the largest mode, so this just re-points the scanout rectangle; the surface is unchanged.
|
||||
set_mode = 2,
|
||||
};
|
||||
|
||||
pub const Request = extern struct {
|
||||
operation: u32,
|
||||
x: u32 = 0,
|
||||
y: u32 = 0,
|
||||
width: u32 = 0,
|
||||
height: u32 = 0,
|
||||
};
|
||||
|
||||
pub const Reply = extern struct {
|
||||
status: i32, // 0 on success, negative on failure
|
||||
reserved: u32 = 0,
|
||||
};
|
||||
|
||||
/// One offered display mode.
|
||||
pub const Mode = extern struct { width: u32, height: u32 };
|
||||
pub const max_modes = 4;
|
||||
|
||||
/// The reply to `get_modes`: a small fixed list of modes.
|
||||
pub const ModesReply = extern struct {
|
||||
status: i32,
|
||||
count: u32,
|
||||
modes: [max_modes]Mode,
|
||||
};
|
||||
|
||||
pub const message_maximum: usize = 64;
|
||||
pub const request_size: usize = @sizeOf(Request);
|
||||
pub const reply_size: usize = @sizeOf(Reply);
|
||||
pub const modes_reply_size: usize = @sizeOf(ModesReply);
|
||||
@@ -101,8 +101,3 @@ pub fn main(init: runtime.process.Init) void {
|
||||
}
|
||||
_ = runtime.system.write("fat-test: root listing was empty\n");
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start;
|
||||
}
|
||||
|
||||
@@ -242,8 +242,3 @@ pub fn main() void {
|
||||
.on_message = onMessage,
|
||||
});
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start;
|
||||
}
|
||||
|
||||
@@ -28,8 +28,3 @@ pub fn main(init: runtime.process.Init) void {
|
||||
// supervisor reads a clean exit as "meant to stop" — correct for a
|
||||
// placeholder.
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start; // pull the runtime entry shim into the image
|
||||
}
|
||||
|
||||
@@ -20,6 +20,7 @@
|
||||
const std = @import("std");
|
||||
const runtime = @import("runtime");
|
||||
const power = runtime.power_protocol;
|
||||
const build_options = @import("build_options");
|
||||
|
||||
/// Where the kernel boot log is persisted on the USB FAT volume — an 8.3 name at
|
||||
/// the mount root (see system/services/log-flush). init writes it at shutdown;
|
||||
@@ -30,12 +31,22 @@ const log_path = "/mnt/usb/DANOS.LOG";
|
||||
/// microkernel keeps such choices in user space, not the kernel. Drivers are absent
|
||||
/// on purpose: the device manager owns those. (A future init reads this from a
|
||||
/// manifest under /system/services instead of a hardcoded list.)
|
||||
const boot_services = [_][]const u8{ "vfs", "input", "device-manager", "fat", "display" };
|
||||
const boot_services = [_][]const u8{ "vfs", "input", "device-manager", "fat", "display", "display-demo" };
|
||||
|
||||
var children: [boot_services.len]u32 = .{0} ** boot_services.len;
|
||||
var child_count: usize = 0;
|
||||
/// The live process id of each boot service (0 = not running), indexed by its position
|
||||
/// in `boot_services`, plus how many times init has restarted it. init supervises these:
|
||||
/// it spawns them against `supervision_endpoint` and, on a child's death, restarts it (up
|
||||
/// to `maximum_restarts`) — the reincarnation half of resilience (docs/resilience.md), the
|
||||
/// service-level counterpart to the device manager's driver restarts.
|
||||
var child_ids: [boot_services.len]u32 = .{0} ** boot_services.len;
|
||||
var restart_counts: [boot_services.len]u32 = .{0} ** boot_services.len;
|
||||
var shutting_down = false;
|
||||
var supervision_endpoint: runtime.ipc.Handle = 0;
|
||||
|
||||
/// Give up restarting a service after this many crashes — a crash-loop cap, so a service
|
||||
/// that faults immediately on every spawn doesn't respawn forever.
|
||||
const maximum_restarts = 3;
|
||||
|
||||
pub fn main() void {
|
||||
// Prove the heap end to end: allocate through the runtime allocator (which
|
||||
// mmaps pages from the kernel and carves them with the free list), write into
|
||||
@@ -63,11 +74,8 @@ pub fn main() void {
|
||||
// Bring up the boot services, supervised so init can stop them cleanly.
|
||||
// Best-effort and silent: each service announces its own readiness, and in
|
||||
// an isolation test with no initial-ramdisk the spawns simply no-op.
|
||||
for (boot_services) |service| {
|
||||
if (runtime.system.spawnSupervised(service, &.{}, supervision_endpoint)) |id| {
|
||||
children[child_count] = id;
|
||||
child_count += 1;
|
||||
}
|
||||
for (boot_services, 0..) |service, i| {
|
||||
if (runtime.system.spawnSupervised(service, &.{}, supervision_endpoint)) |id| child_ids[i] = id;
|
||||
}
|
||||
|
||||
// Once the storage stack is up, a one-shot copies the boot log to the USB
|
||||
@@ -83,10 +91,13 @@ pub fn main() void {
|
||||
// triggers the same shutdown path.
|
||||
subscribePower();
|
||||
|
||||
// A re-arming timer drives the liveness heartbeat: proof PID 1 is alive
|
||||
// (the init test's marker) while the loop stays free to receive signals,
|
||||
// power events, and children's exit notifications.
|
||||
_ = runtime.system.timerOnce(supervision_endpoint, 1000);
|
||||
// A re-arming timer drives the liveness heartbeat — proof PID 1 is alive (the
|
||||
// init test's marker) and a -Dserial diagnostic. It is a serial/test-build-only
|
||||
// concern: a flashable (serial-off) image runs a purely event-driven PID 1 that
|
||||
// wakes only for real work (signals, power events, children's exits), never for a
|
||||
// periodic beat. `build_options.serial` is comptime, so the heartbeat — its timer
|
||||
// and the handler below — folds away entirely when serial is off.
|
||||
if (build_options.serial) _ = runtime.system.timerOnce(supervision_endpoint, 1000);
|
||||
|
||||
var receive: [power.message_maximum]u8 = undefined;
|
||||
while (true) {
|
||||
@@ -95,7 +106,7 @@ pub fn main() void {
|
||||
if (signals.has(.terminate)) shutDown();
|
||||
continue;
|
||||
}
|
||||
if (got.isTimer()) {
|
||||
if (build_options.serial and got.isTimer()) {
|
||||
_ = runtime.system.write("/system/services/init: heartbeat\n");
|
||||
_ = runtime.system.timerOnce(supervision_endpoint, 1000);
|
||||
continue;
|
||||
@@ -105,11 +116,47 @@ pub fn main() void {
|
||||
if (receive[1] == @intFromEnum(power.Event.power_button)) shutDown();
|
||||
continue;
|
||||
}
|
||||
// Child-exit notifications and anything else: keep waiting.
|
||||
if (got.isChildExit()) {
|
||||
restartChild(got.childProcessId());
|
||||
continue;
|
||||
}
|
||||
// Anything else: keep waiting.
|
||||
if (got.isNotification()) continue;
|
||||
}
|
||||
}
|
||||
|
||||
/// A supervised boot service died. Find which one and restart it — unless it exited
|
||||
/// cleanly (it chose to stop, e.g. a driver with no hardware) or has hit the crash-loop
|
||||
/// cap. Reclaiming the dead process is already the kernel's job (docs/process-lifecycle.md
|
||||
/// iron rule 1); init only decides whether to bring it back.
|
||||
fn restartChild(id: u32) void {
|
||||
if (shutting_down) return; // deaths during the stop sequence are expected, not crashes
|
||||
for (boot_services, 0..) |service, i| {
|
||||
if (child_ids[i] != id) continue;
|
||||
child_ids[i] = 0;
|
||||
// An unknown reason (the record aged out) is treated as a crash worth restarting.
|
||||
const reason = runtime.process.exitReason(id) orelse .fault;
|
||||
if (reason == .exited) {
|
||||
logLine("/system/services/init: {s} exited cleanly; not restarting\n", .{service});
|
||||
return;
|
||||
}
|
||||
restart_counts[i] += 1;
|
||||
if (restart_counts[i] > maximum_restarts) {
|
||||
logLine("/system/services/init: {s} keeps crashing; giving up after {d} restarts\n", .{ service, maximum_restarts });
|
||||
return;
|
||||
}
|
||||
logLine("/system/services/init: {s} died ({s}); restarting ({d}/{d})\n", .{ service, @tagName(reason), restart_counts[i], maximum_restarts });
|
||||
if (runtime.system.spawnSupervised(service, &.{}, supervision_endpoint)) |new_id| child_ids[i] = new_id;
|
||||
return;
|
||||
}
|
||||
// An untracked child (e.g. the log-flush one-shot): nothing to restart.
|
||||
}
|
||||
|
||||
fn logLine(comptime fmt: []const u8, args: anytype) void {
|
||||
var line: [128]u8 = undefined;
|
||||
_ = runtime.system.write(std.fmt.bufPrint(&line, fmt, args) catch return);
|
||||
}
|
||||
|
||||
/// Look up the power service and subscribe our endpoint (handed over as the
|
||||
/// call's capability) so events arrive as buffered messages here.
|
||||
fn subscribePower() void {
|
||||
@@ -153,15 +200,16 @@ fn flushKernelLog() void {
|
||||
/// it), waiting up to a deadline for each to exit before killing it, then ask the
|
||||
/// power service to enter S5.
|
||||
fn shutDown() void {
|
||||
shutting_down = true; // the stop loop below kills children — those deaths aren't crashes
|
||||
_ = runtime.system.write("/system/services/init: shutting down\n");
|
||||
// Persist the fullest log to the USB volume BEFORE tearing anything down: the
|
||||
// reverse-order stop loop below kills the fat server (children[3]) first, so
|
||||
// /mnt/usb must be written while it is still mounted.
|
||||
// reverse-order stop loop below kills the fat server first, so /mnt/usb must be
|
||||
// written while it is still mounted.
|
||||
flushKernelLog();
|
||||
var i = child_count;
|
||||
var i = boot_services.len;
|
||||
while (i > 0) {
|
||||
i -= 1;
|
||||
if (children[i] != 0) runtime.process.stop(children[i], 2000, supervision_endpoint);
|
||||
if (child_ids[i] != 0) runtime.process.stop(child_ids[i], 2000, supervision_endpoint);
|
||||
}
|
||||
if (runtime.ipc.lookup(.power)) |h| {
|
||||
const request = power.Shutdown{};
|
||||
@@ -171,8 +219,3 @@ fn shutDown() void {
|
||||
// If S5 did not take, init has nothing left to do but idle.
|
||||
while (true) runtime.system.sleep(1000);
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start; // pull the runtime entry shim into the image
|
||||
}
|
||||
|
||||
@@ -33,8 +33,3 @@ pub fn main() void {
|
||||
system.sleep(200);
|
||||
}
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start;
|
||||
}
|
||||
|
||||
@@ -49,8 +49,3 @@ pub fn main() void {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start;
|
||||
}
|
||||
|
||||
@@ -138,8 +138,3 @@ pub fn main() void {
|
||||
reply_len = handle(receive[0..got.len], got, &reply_buffer);
|
||||
}
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start;
|
||||
}
|
||||
|
||||
@@ -60,8 +60,3 @@ pub fn main() void {
|
||||
var line: [96]u8 = undefined;
|
||||
_ = runtime.system.write(std.fmt.bufPrint(&line, "log-flush: wrote {d} bytes to {s}\n", .{ written, log_path }) catch return);
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start; // pull the runtime entry shim into the image
|
||||
}
|
||||
|
||||
@@ -178,8 +178,3 @@ pub fn main(init: runtime.process.Init) void {
|
||||
|
||||
_ = runtime.system.write("process-test: ok\n");
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start; // pull the runtime entry shim into the image
|
||||
}
|
||||
|
||||
@@ -0,0 +1,46 @@
|
||||
//! system/services/shm-client — the creating half of the shm test (docs/display-v2.md V2).
|
||||
//! It `shm_create`s a shared region, writes a known pattern into it, and hands the region's
|
||||
//! capability to `shm-server` as an `ipc_call` send_cap. The server maps that capability and
|
||||
//! confirms the pattern is visible — proving cross-process shared memory over the extended
|
||||
//! capability-passing path.
|
||||
|
||||
const runtime = @import("runtime");
|
||||
const system = runtime.system;
|
||||
const shm = runtime.shm;
|
||||
const ipc = runtime.ipc;
|
||||
|
||||
const pattern_len = 4096;
|
||||
|
||||
/// The pattern the server checks — must match shm-server.zig.
|
||||
fn expected(i: usize) u8 {
|
||||
return @truncate(i *% 7 +% 3);
|
||||
}
|
||||
|
||||
fn lookupServer() ?ipc.Handle {
|
||||
var attempts: usize = 0;
|
||||
while (attempts < 100) : (attempts += 1) {
|
||||
if (ipc.lookup(.shm_test)) |h| return h;
|
||||
system.sleep(50);
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
pub fn main() void {
|
||||
const region = shm.create(pattern_len) orelse {
|
||||
_ = system.write("shm: create failed\n");
|
||||
return;
|
||||
};
|
||||
var i: usize = 0;
|
||||
while (i < pattern_len) : (i += 1) region.ptr[i] = expected(i);
|
||||
|
||||
const server = lookupServer() orelse {
|
||||
_ = system.write("shm: no server\n");
|
||||
return;
|
||||
};
|
||||
// A non-empty message (so it reaches on_message, not the ping path), carrying the shm
|
||||
// region's capability. The reply is empty; we just need the round trip.
|
||||
var reply: [64]u8 = undefined;
|
||||
_ = ipc.callCap(server, "shm", &reply, region.handle) catch {
|
||||
_ = system.write("shm: call failed\n");
|
||||
};
|
||||
}
|
||||
@@ -0,0 +1,44 @@
|
||||
//! system/services/shm-server — the receiving half of the shm test (docs/display-v2.md V2).
|
||||
//! It registers under `ServiceId.shm_test`; when `shm-client` calls it carrying a
|
||||
//! shared-memory capability, it `shm_map`s that capability and checks the client's pattern
|
||||
//! is visible through the mapping — proving the two processes share the same physical pages
|
||||
//! (not a copy). On success it prints `shm: shared 4096 bytes ok`, the test's marker.
|
||||
|
||||
const runtime = @import("runtime");
|
||||
const system = runtime.system;
|
||||
const shm = runtime.shm;
|
||||
const ipc = runtime.ipc;
|
||||
|
||||
const pattern_len = 4096;
|
||||
|
||||
/// The pattern the client writes — must match shm-client.zig.
|
||||
fn expected(i: usize) u8 {
|
||||
return @truncate(i *% 7 +% 3);
|
||||
}
|
||||
|
||||
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Handle) usize {
|
||||
_ = message;
|
||||
_ = reply;
|
||||
_ = sender;
|
||||
const cap = capability orelse {
|
||||
_ = system.write("shm: shared FAILED (no capability)\n");
|
||||
return 0;
|
||||
};
|
||||
const ptr = shm.map(cap) orelse {
|
||||
_ = system.write("shm: shared FAILED (map)\n");
|
||||
return 0;
|
||||
};
|
||||
var i: usize = 0;
|
||||
while (i < pattern_len) : (i += 1) {
|
||||
if (ptr[i] != expected(i)) {
|
||||
_ = system.write("shm: shared FAILED (mismatch)\n");
|
||||
return 0;
|
||||
}
|
||||
}
|
||||
_ = system.write("shm: shared 4096 bytes ok\n");
|
||||
return 0; // empty reply — the client only needs the round trip to unblock
|
||||
}
|
||||
|
||||
pub fn main() void {
|
||||
runtime.service.run(64, .{ .service = .shm_test, .on_message = onMessage });
|
||||
}
|
||||
@@ -0,0 +1,310 @@
|
||||
//! thread-test — danos's multi-threaded exerciser (docs/threading-plan.md M2, M3).
|
||||
//!
|
||||
//! Two modes, chosen by argv[1] (default "spawn"):
|
||||
//! spawn — M2: one worker writes a shared global; the main thread observes it, proving
|
||||
//! `runtime.Thread.spawn` started a task in the **same** address space.
|
||||
//! join — M3: N workers each do K atomic increments on a shared counter and stamp the
|
||||
//! core they ran on; the main thread `join`s all N and checks the total is
|
||||
//! exactly N*K (every worker ran, join waited) and that >1 core was used
|
||||
//! (genuine parallelism). Then a detached worker proves `detach` runs and
|
||||
//! needs no join.
|
||||
//!
|
||||
//! Built multi-threaded (`addThreadedUserBinary`) so atomics/shared reads are real.
|
||||
|
||||
const std = @import("std");
|
||||
const runtime = @import("runtime");
|
||||
|
||||
fn write(comptime s: []const u8) void {
|
||||
_ = runtime.system.write(s);
|
||||
}
|
||||
|
||||
// --- M2: spawn mode ---------------------------------------------------------
|
||||
|
||||
var shared_value: u32 = 0;
|
||||
var spawn_done = std.atomic.Value(u32).init(0);
|
||||
const sentinel: u32 = 0xA5A5;
|
||||
|
||||
fn spawnWorker() void {
|
||||
shared_value = sentinel;
|
||||
spawn_done.store(1, .release);
|
||||
}
|
||||
|
||||
fn runSpawnMode() void {
|
||||
write("thread-test: starting\n");
|
||||
_ = runtime.Thread.spawn(.{}, spawnWorker, .{}) catch {
|
||||
write("thread-test: FAIL spawn refused\n");
|
||||
return;
|
||||
};
|
||||
var spins: usize = 0;
|
||||
while (spawn_done.load(.acquire) == 0 and spins < 50_000_000) : (spins += 1) {
|
||||
runtime.system.yield();
|
||||
}
|
||||
if (spawn_done.load(.acquire) == 1 and shared_value == sentinel) {
|
||||
write("thread-test: child ran in shared aspace ok\n");
|
||||
} else {
|
||||
write("thread-test: FAIL worker did not update shared memory\n");
|
||||
}
|
||||
}
|
||||
|
||||
// --- M3: join mode ----------------------------------------------------------
|
||||
|
||||
const worker_count: u32 = 4;
|
||||
const iterations: u64 = 100_000;
|
||||
|
||||
var counter = std.atomic.Value(u64).init(0);
|
||||
var cores_seen = std.atomic.Value(u32).init(0);
|
||||
|
||||
fn joinWorker() void {
|
||||
var i: u64 = 0;
|
||||
while (i < iterations) : (i += 1) {
|
||||
_ = counter.fetchAdd(1, .monotonic);
|
||||
if (i % 1000 == 0) stampCore(); // periodic: catches cross-core migration too
|
||||
}
|
||||
stampCore();
|
||||
}
|
||||
|
||||
fn stampCore() void {
|
||||
const core = runtime.Thread.currentCore();
|
||||
if (core < 32) _ = cores_seen.fetchOr(@as(u32, 1) << @intCast(core), .monotonic);
|
||||
}
|
||||
|
||||
var detach_done = std.atomic.Value(u32).init(0);
|
||||
|
||||
fn detachWorker() void {
|
||||
detach_done.store(1, .release);
|
||||
}
|
||||
|
||||
fn runJoinMode() void {
|
||||
write("thread-test: join mode starting\n");
|
||||
|
||||
var threads: [worker_count]runtime.Thread = undefined;
|
||||
var spawned: u32 = 0;
|
||||
while (spawned < worker_count) : (spawned += 1) {
|
||||
threads[spawned] = runtime.Thread.spawn(.{}, joinWorker, .{}) catch break;
|
||||
}
|
||||
if (spawned != worker_count) {
|
||||
write("thread-test: FAIL could not spawn all workers\n");
|
||||
return;
|
||||
}
|
||||
for (threads[0..spawned]) |t| t.join();
|
||||
|
||||
const total = counter.load(.acquire);
|
||||
const cores = @popCount(cores_seen.load(.acquire));
|
||||
if (total != worker_count * iterations) {
|
||||
write("thread-test: FAIL counter mismatch (a worker was lost or join did not wait)\n");
|
||||
return;
|
||||
}
|
||||
if (cores <= 1) {
|
||||
write("thread-test: FAIL workers never ran on more than one core\n");
|
||||
return;
|
||||
}
|
||||
|
||||
// detach: the worker runs and we never join it.
|
||||
const dt = runtime.Thread.spawn(.{}, detachWorker, .{}) catch {
|
||||
write("thread-test: FAIL detach spawn refused\n");
|
||||
return;
|
||||
};
|
||||
dt.detach();
|
||||
var spins: usize = 0;
|
||||
while (detach_done.load(.acquire) == 0 and spins < 50_000_000) : (spins += 1) {
|
||||
runtime.system.yield();
|
||||
}
|
||||
if (detach_done.load(.acquire) != 1) {
|
||||
write("thread-test: FAIL detached worker did not run\n");
|
||||
return;
|
||||
}
|
||||
|
||||
write("thread-test: join ok\n"); // the M3 verdict marker
|
||||
}
|
||||
|
||||
// --- M4: futex mode ---------------------------------------------------------
|
||||
|
||||
const Futex = runtime.Thread.Futex;
|
||||
|
||||
var futex_word = std.atomic.Value(u32).init(0);
|
||||
var waiter_parked = std.atomic.Value(u32).init(0);
|
||||
|
||||
fn futexWaiter() void {
|
||||
write("thread-futex: waiting\n");
|
||||
waiter_parked.store(1, .release);
|
||||
// Block while the word is still 0; the waker sets it to 1 and wakes us.
|
||||
while (futex_word.load(.acquire) == 0) {
|
||||
Futex.wait(&futex_word, 0);
|
||||
}
|
||||
write("thread-futex: woke\n");
|
||||
}
|
||||
|
||||
fn runFutexMode() void {
|
||||
write("thread-futex: starting\n");
|
||||
|
||||
const waiter = runtime.Thread.spawn(.{}, futexWaiter, .{}) catch {
|
||||
write("thread-futex: FAIL spawn refused\n");
|
||||
return;
|
||||
};
|
||||
// Let the waiter reach its wait, then give it a beat to actually park in-kernel.
|
||||
var spins: usize = 0;
|
||||
while (waiter_parked.load(.acquire) == 0 and spins < 50_000_000) : (spins += 1) {
|
||||
runtime.system.yield();
|
||||
}
|
||||
runtime.system.sleep(50);
|
||||
|
||||
// The handshake: publish the value, then wake the parked waiter.
|
||||
futex_word.store(1, .release);
|
||||
write("thread-futex: waking\n");
|
||||
Futex.wake(&futex_word, 1);
|
||||
|
||||
waiter.join(); // returns once the waiter woke and printed "woke"
|
||||
|
||||
// Timeout: nobody ever wakes this word, so timedWait must report a timeout.
|
||||
var lonely = std.atomic.Value(u32).init(0);
|
||||
if (Futex.timedWait(&lonely, 0, 100_000_000)) |_| {
|
||||
write("thread-futex: FAIL timedWait did not time out\n");
|
||||
return;
|
||||
} else |_| {}
|
||||
write("thread-futex: timeout ok\n");
|
||||
|
||||
write("thread-futex: ok\n"); // the M4 verdict marker
|
||||
}
|
||||
|
||||
// --- M5: mutex mode (bounded producer/consumer over Mutex + Condition) ------
|
||||
|
||||
const Mutex = runtime.Thread.Mutex;
|
||||
const Condition = runtime.Thread.Condition;
|
||||
|
||||
const producers: u32 = 2;
|
||||
const consumers: u32 = 2;
|
||||
const per_producer: u32 = 1000;
|
||||
const per_consumer: u32 = 1000; // producers*per_producer == consumers*per_consumer (balanced)
|
||||
const total_items: u32 = producers * per_producer;
|
||||
const ring_cap: usize = 8; // small, so producers block on full and consumers on empty
|
||||
|
||||
var ring: [ring_cap]u32 = undefined;
|
||||
var ring_count: usize = 0;
|
||||
var ring_head: usize = 0;
|
||||
var ring_tail: usize = 0;
|
||||
|
||||
var pc_mutex = Mutex{};
|
||||
var not_full = Condition{};
|
||||
var not_empty = Condition{};
|
||||
|
||||
// Verified outside the lock: the checksum and tally of everything consumed.
|
||||
var consumed_sum = std.atomic.Value(u64).init(0);
|
||||
var consumed_count = std.atomic.Value(u32).init(0);
|
||||
|
||||
fn producer(base: u32) void {
|
||||
var i: u32 = 0;
|
||||
while (i < per_producer) : (i += 1) {
|
||||
const item = base + i;
|
||||
pc_mutex.lock();
|
||||
while (ring_count == ring_cap) not_full.wait(&pc_mutex);
|
||||
ring[ring_tail] = item;
|
||||
ring_tail = (ring_tail + 1) % ring_cap;
|
||||
ring_count += 1;
|
||||
pc_mutex.unlock();
|
||||
not_empty.signal();
|
||||
}
|
||||
}
|
||||
|
||||
fn consumer() void {
|
||||
var i: u32 = 0;
|
||||
while (i < per_consumer) : (i += 1) {
|
||||
pc_mutex.lock();
|
||||
while (ring_count == 0) not_empty.wait(&pc_mutex);
|
||||
const item = ring[ring_head];
|
||||
ring_head = (ring_head + 1) % ring_cap;
|
||||
ring_count -= 1;
|
||||
pc_mutex.unlock();
|
||||
not_full.signal();
|
||||
_ = consumed_sum.fetchAdd(item, .monotonic);
|
||||
_ = consumed_count.fetchAdd(1, .monotonic);
|
||||
}
|
||||
}
|
||||
|
||||
fn runMutexMode() void {
|
||||
write("thread-mutex: starting\n");
|
||||
|
||||
var threads: [producers + consumers]runtime.Thread = undefined;
|
||||
var n: usize = 0;
|
||||
var p: u32 = 0;
|
||||
while (p < producers) : (p += 1) {
|
||||
threads[n] = runtime.Thread.spawn(.{}, producer, .{p * per_producer}) catch {
|
||||
write("thread-mutex: FAIL producer spawn\n");
|
||||
return;
|
||||
};
|
||||
n += 1;
|
||||
}
|
||||
var c: u32 = 0;
|
||||
while (c < consumers) : (c += 1) {
|
||||
threads[n] = runtime.Thread.spawn(.{}, consumer, .{}) catch {
|
||||
write("thread-mutex: FAIL consumer spawn\n");
|
||||
return;
|
||||
};
|
||||
n += 1;
|
||||
}
|
||||
for (threads[0..n]) |t| t.join();
|
||||
|
||||
// Every item 0..total_items-1 was produced exactly once; if the mutex/condition are
|
||||
// correct, each was consumed exactly once, so the checksum matches.
|
||||
const expected_sum: u64 = @as(u64, total_items) * (total_items - 1) / 2;
|
||||
if (consumed_count.load(.acquire) != total_items) {
|
||||
write("thread-mutex: FAIL wrong number of items consumed\n");
|
||||
return;
|
||||
}
|
||||
if (consumed_sum.load(.acquire) != expected_sum) {
|
||||
write("thread-mutex: FAIL checksum mismatch (item lost or duplicated)\n");
|
||||
return;
|
||||
}
|
||||
write("thread-mutex: ok\n"); // the M5 verdict marker
|
||||
}
|
||||
|
||||
// --- M6: id mode (getCurrentId identity) ------------------------------------
|
||||
|
||||
var worker_ids: [2]std.atomic.Value(u32) = .{ std.atomic.Value(u32).init(0), std.atomic.Value(u32).init(0) };
|
||||
|
||||
fn idWorker(slot: usize) void {
|
||||
worker_ids[slot].store(runtime.Thread.getCurrentId(), .release);
|
||||
}
|
||||
|
||||
fn runIdMode() void {
|
||||
write("thread-id: starting\n");
|
||||
const main_id = runtime.Thread.getCurrentId();
|
||||
|
||||
const t0 = runtime.Thread.spawn(.{}, idWorker, .{@as(usize, 0)}) catch {
|
||||
write("thread-id: FAIL spawn\n");
|
||||
return;
|
||||
};
|
||||
const t1 = runtime.Thread.spawn(.{}, idWorker, .{@as(usize, 1)}) catch {
|
||||
write("thread-id: FAIL spawn\n");
|
||||
return;
|
||||
};
|
||||
t0.join();
|
||||
t1.join();
|
||||
|
||||
const id0 = worker_ids[0].load(.acquire);
|
||||
const id1 = worker_ids[1].load(.acquire);
|
||||
// Each thread has its own kernel task id: all three distinct and non-zero.
|
||||
if (main_id == 0 or id0 == 0 or id1 == 0) {
|
||||
write("thread-id: FAIL a thread reported id 0\n");
|
||||
return;
|
||||
}
|
||||
if (id0 == id1 or id0 == main_id or id1 == main_id) {
|
||||
write("thread-id: FAIL thread ids collided\n");
|
||||
return;
|
||||
}
|
||||
write("thread-id: ok\n"); // the M6 verdict marker
|
||||
}
|
||||
|
||||
pub fn main(init: runtime.process.Init) void {
|
||||
const mode = init.arguments.get(1) orelse "spawn";
|
||||
if (std.mem.eql(u8, mode, "join")) {
|
||||
runJoinMode();
|
||||
} else if (std.mem.eql(u8, mode, "futex")) {
|
||||
runFutexMode();
|
||||
} else if (std.mem.eql(u8, mode, "mutex")) {
|
||||
runMutexMode();
|
||||
} else if (std.mem.eql(u8, mode, "id")) {
|
||||
runIdMode();
|
||||
} else {
|
||||
runSpawnMode();
|
||||
}
|
||||
}
|
||||
@@ -60,8 +60,3 @@ pub fn main(init: runtime.process.Init) void {
|
||||
}
|
||||
_ = runtime.system.write("vfstest: mismatch\n");
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start;
|
||||
}
|
||||
|
||||
@@ -394,8 +394,3 @@ pub fn main() void {
|
||||
.on_notification = onNotification,
|
||||
});
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start;
|
||||
}
|
||||
|
||||
+107
-3
@@ -185,6 +185,51 @@ CASES = [
|
||||
{"name": "display-demo",
|
||||
"expect": r"display-demo: scene up[\s\S]*display-demo: ok",
|
||||
"fail": r"display-demo: (no display|create failed)|display: could not|CPU EXCEPTION|KERNEL PANIC"},
|
||||
# Shared memory (v2 V2): shm-client creates a region, writes a pattern, and passes its
|
||||
# capability to shm-server, which maps it and confirms the same bytes — proving
|
||||
# cross-process shared pages over the extended capability passing.
|
||||
{"name": "shm",
|
||||
"expect": r"shm: shared 4096 bytes ok",
|
||||
"fail": r"shm: (shared FAILED|create failed|no server|call failed|map)|CPU EXCEPTION|KERNEL PANIC"},
|
||||
# virtio-gpu driver (v2 V3): boot with an emulated virtio-gpu. The device-manager stack
|
||||
# discovers the PCI function and spawns the driver, which brings up the control virtqueue,
|
||||
# creates a 2D scanout resource backed by DMA memory, set_scanouts it, paints a test
|
||||
# pattern, transfers + flushes it, and waits for the device's used-ring ack, then reads
|
||||
# the backing back. `scanout WxH online` + `flush acked, pixel check ok` are the markers.
|
||||
{"name": "virtio-gpu",
|
||||
"qemu_extra": ["-device", "virtio-gpu-pci"],
|
||||
"expect": r"virtio-gpu: scanout \d+x\d+ online[\s\S]*virtio-gpu: flush acked, pixel check ok",
|
||||
"fail": r"virtio-gpu:.*(failed|not acked|mismatch|unable to claim|not a virtio-gpu|too small|no PCI capability|does not offer|rejected|missing common-config|not a mapped resource|could not spawn)|CPU EXCEPTION|KERNEL PANIC"},
|
||||
# Native backend + hot-attach (v2 V4): boot the compositor + display-demo with an emulated
|
||||
# virtio-gpu. The driver announces its shared scanout surface to the compositor, which maps
|
||||
# it, upgrades off the GOP floor, and drives frames through the native backend — reading a
|
||||
# pixel back to confirm the composited frame reached the shared surface, while the demo runs.
|
||||
{"name": "display-native",
|
||||
"qemu_extra": ["-device", "virtio-gpu-pci"],
|
||||
"mem": "512M", # boots the compositor + demo + the whole device-manager driver stack at once
|
||||
# Order-independent: the demo's `ok` may print before or after the driver announces, so
|
||||
# require all three markers to appear somewhere rather than in a fixed order.
|
||||
"expect": r"(?s)(?=.*display: scanout upgraded to virtio-gpu)(?=.*display: native present verified)(?=.*display-demo: ok)",
|
||||
"fail": r"display: native present FAILED|display: could not|display-demo: (no display|create failed)|CPU EXCEPTION|KERNEL PANIC"},
|
||||
# Mode-set + EDID + vsync (v2 V5): same boot as display-native. After upgrading, the
|
||||
# compositor queries the driver's modes, switches to a different resolution, and confirms the
|
||||
# backend now reports it; the fenced present path makes it a vsync present. (The driver also
|
||||
# logs the EDID preferred mode during bring-up.) Reuses the display-native kernel scenario.
|
||||
{"name": "display-modeset",
|
||||
"build_case": "display-native",
|
||||
"qemu_extra": ["-device", "virtio-gpu-pci"],
|
||||
"mem": "512M",
|
||||
"expect": r"(?s)(?=.*display: mode set to \d+x\d+, verified)(?=.*display: vsync present ok)",
|
||||
"fail": r"display: mode set FAILED|display: mode-set self-check: |display: native present FAILED|CPU EXCEPTION|KERNEL PANIC"},
|
||||
# Resilience: driver restart + re-attach (v2 V6). device-manager (in test-scanout-restart
|
||||
# mode) kills the virtio-gpu driver once after it hellos; the restart policy respawns it, it
|
||||
# re-announces, and the compositor re-attaches — surviving the loss. Expect the initial
|
||||
# upgrade AND the re-attach; any CPU exception / panic (the compositor crashing) is a fail.
|
||||
{"name": "display-reattach",
|
||||
"qemu_extra": ["-device", "virtio-gpu-pci"],
|
||||
"mem": "512M",
|
||||
"expect": r"(?s)(?=.*display: scanout upgraded to virtio-gpu)(?=.*display: scanout re-attached)",
|
||||
"fail": r"CPU EXCEPTION|KERNEL PANIC|display: could not"},
|
||||
# Monotonic clock (clock() syscall source): calibrated, advancing, never backwards.
|
||||
{"name": "clock",
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
@@ -248,6 +293,52 @@ CASES = [
|
||||
"timeout": 60,
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
|
||||
# docs/threading-plan.md M1: address-space refcount — spaces destroyed exactly
|
||||
# once per process, no leak/double-free (the foundation shared-aspace threads need).
|
||||
{"name": "aspace-refcount",
|
||||
"timeout": 60,
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
|
||||
# docs/threading-plan.md M2: runtime.Thread.spawn — a worker thread runs in the
|
||||
# caller's address space (a shared-memory write, observed by the main thread).
|
||||
{"name": "thread-spawn",
|
||||
"timeout": 60,
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
|
||||
# docs/threading-plan.md M3: join + parallelism — N workers each do K atomic
|
||||
# increments (total exactly N*K after join) and run on >1 core; plus detach.
|
||||
{"name": "thread-join",
|
||||
"smp": 4,
|
||||
"timeout": 60,
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
|
||||
# docs/threading-plan.md M4: futex — a thread parks in futex_wait and is woken by
|
||||
# futex_wake (serial order waiting/waking/woke), and timedWait reports a timeout.
|
||||
{"name": "thread-futex",
|
||||
"smp": 4,
|
||||
"timeout": 60,
|
||||
"expect": r"thread-futex: waiting[\s\S]*thread-futex: waking[\s\S]*thread-futex: woke[\s\S]*DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
|
||||
# docs/threading-plan.md M5: Mutex + Condition — a bounded producer/consumer moves
|
||||
# N unique items across cores; the consumed checksum matches exactly (no loss).
|
||||
{"name": "thread-mutex",
|
||||
"smp": 4,
|
||||
"timeout": 60,
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
|
||||
# docs/threading-plan.md M6: thread identity — getCurrentId is distinct and non-zero
|
||||
# for the main thread and two workers.
|
||||
{"name": "thread-id",
|
||||
"smp": 4,
|
||||
"timeout": 60,
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# Process arguments: argv arrives on the SysV entry stack (argv[0] = the spawned
|
||||
# name, argv[1..] = the system_spawn argument blob) and echoes back intact.
|
||||
{"name": "args",
|
||||
@@ -441,12 +532,23 @@ CASES = [
|
||||
"expect": r"acpi: reported PNP0303 \(device \d+, 3 resources\)[\s\S]*"
|
||||
r"acpi: reported PNP0F13 \(device \d+, 1 resources\)",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# M19.1: the ring-3 PCI scan (pci-bus walks the ECAM through its mmio_map
|
||||
# grant) finds exactly the functions the kernel's own walk recorded.
|
||||
# M19.1/M19.3: the ring-3 PCI scan. pci-bus walks the ECAM through its mmio_map
|
||||
# grant and registers every function it finds; the kernel's own walk retired, so
|
||||
# the broker starts empty and the driver populates it. The manager then runs the
|
||||
# restart drill: ~1 s after the scan it kills pci-bus, prunes its child tree, and
|
||||
# respawns it to re-claim, re-scan, and re-register the same functions. The kernel
|
||||
# test asserts the broker equivalence (empty before, populated after, no
|
||||
# duplicates); this ordered regex asserts the drill itself over the whole serial
|
||||
# log — the backreference requires the respawn to re-scan the same count, and the
|
||||
# full-capture match is immune to the transient-line races an in-kernel poll hits.
|
||||
{"name": "pci-scan",
|
||||
"smp": 4,
|
||||
"timeout": 60,
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"expect": r"pci-bus: (\d+) functions found[\s\S]*"
|
||||
r"device-manager: test mode: killing the reporter[\s\S]*"
|
||||
r"device-manager: restarting pci-bus[\s\S]*"
|
||||
r"pci-bus: \1 functions found[\s\S]*"
|
||||
r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# M18.3: the application surface — device-list enumerates the tree over IPC,
|
||||
# subscribes (endpoint as capability), and observes the removed/added events
|
||||
@@ -594,6 +696,8 @@ def run_case(arch, case):
|
||||
cmd = [arch["qemu"]] + arch["qemu_args"](arch, boot_volume, vars_fd, serial)
|
||||
if case.get("smp"): # some cases need more than one core (e.g. parallelism)
|
||||
cmd += ["-smp", str(case["smp"])]
|
||||
if case.get("mem"): # a case that boots the whole system at once needs more than the 128M floor
|
||||
cmd[cmd.index("-m") + 1] = case["mem"]
|
||||
if case.get("qemu_extra"): # extra qemu args, e.g. -device intel-iommu for the IOMMU case
|
||||
cmd += case["qemu_extra"]
|
||||
# A QMP control socket, always present (additive): how a case's `qmp_after`
|
||||
|
||||
@@ -0,0 +1,261 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Wrap the FAT32 boot volume in a hybrid ISO — the flashable danos release image.
|
||||
|
||||
Mirrors tools/make-fat-image.py in spirit: pure Python 3 standard library, no
|
||||
external tools (no xorriso / mkisofs / isohybrid). The output is one file that
|
||||
boots both ways release media is consumed:
|
||||
|
||||
* Flashed raw to a USB stick (Etcher, dd): the ISO's system area carries an
|
||||
MBR whose single partition (type 0xEF, "EFI System") points at the FAT32
|
||||
image embedded in the ISO, so UEFI firmware finds the ESP and runs
|
||||
\\EFI\\BOOT\\BOOTX64.efi off it.
|
||||
* Burned to optical media: an El Torito boot catalog with an EFI platform
|
||||
entry points at the same embedded FAT image.
|
||||
|
||||
The ISO9660 filesystem itself is minimal but valid — a primary volume
|
||||
descriptor, the El Torito boot record, path tables, and a root directory that
|
||||
lists the boot catalog and the FAT image — so inspection tools can open it.
|
||||
|
||||
make-iso-image.py <out.iso> <esp.img>
|
||||
make-iso-image.py --verify <out.iso>
|
||||
|
||||
All timestamp fields are zero ("not specified") so the build is reproducible.
|
||||
"""
|
||||
|
||||
import struct
|
||||
import sys
|
||||
|
||||
ISO_SECTOR = 2048
|
||||
|
||||
# Fixed layout, in ISO sectors (LBA). Sectors 0-15 are the system area (the
|
||||
# hybrid MBR lives in its first 512 bytes); volume descriptors start at 16.
|
||||
PVD_LBA = 16 # primary volume descriptor
|
||||
BOOT_RECORD_LBA = 17 # El Torito boot record volume descriptor
|
||||
TERMINATOR_LBA = 18 # volume descriptor set terminator
|
||||
PATH_TABLE_L_LBA = 19
|
||||
PATH_TABLE_M_LBA = 20
|
||||
ROOT_DIR_LBA = 21 # root directory (one sector holds our four records)
|
||||
CATALOG_LBA = 22 # El Torito boot catalog
|
||||
ESP_LBA = 23 # the embedded FAT32 image starts here
|
||||
|
||||
MBR_PARTITION_TYPE_ESP = 0xEF
|
||||
|
||||
|
||||
def both16(value):
|
||||
"""ISO9660 both-byte-order encoding: little-endian then big-endian."""
|
||||
return struct.pack("<H", value) + struct.pack(">H", value)
|
||||
|
||||
|
||||
def both32(value):
|
||||
return struct.pack("<I", value) + struct.pack(">I", value)
|
||||
|
||||
|
||||
def directory_record(identifier, lba, size, flags):
|
||||
length = 33 + len(identifier)
|
||||
if length % 2:
|
||||
length += 1 # records are padded to even length
|
||||
record = bytearray(length)
|
||||
record[0] = length
|
||||
record[2:10] = both32(lba)
|
||||
record[10:18] = both32(size)
|
||||
# record[18:25] is the recording date; zero = unspecified (reproducible).
|
||||
record[25] = flags # 0x02 = directory
|
||||
record[28:32] = both16(1) # volume sequence number
|
||||
record[32] = len(identifier)
|
||||
record[33:33 + len(identifier)] = identifier
|
||||
return bytes(record)
|
||||
|
||||
|
||||
def primary_volume_descriptor(total_sectors, path_table_size):
|
||||
sector = bytearray(ISO_SECTOR)
|
||||
sector[0] = 1 # type: primary
|
||||
sector[1:6] = b"CD001"
|
||||
sector[6] = 1 # version
|
||||
sector[8:40] = b"DANOS".ljust(32) # system identifier
|
||||
sector[40:72] = b"DANOS".ljust(32) # volume identifier
|
||||
sector[80:88] = both32(total_sectors)
|
||||
sector[120:124] = both16(1) # volume set size
|
||||
sector[124:128] = both16(1) # volume sequence number
|
||||
sector[128:132] = both16(ISO_SECTOR) # logical block size
|
||||
sector[132:140] = both32(path_table_size)
|
||||
sector[140:144] = struct.pack("<I", PATH_TABLE_L_LBA)
|
||||
sector[148:152] = struct.pack(">I", PATH_TABLE_M_LBA)
|
||||
sector[156:190] = directory_record(b"\x00", ROOT_DIR_LBA, ISO_SECTOR, 0x02)
|
||||
sector[190:318] = b" " * 128 # volume set identifier
|
||||
sector[318:446] = b" " * 128 # publisher
|
||||
sector[446:574] = b" " * 128 # data preparer
|
||||
sector[574:702] = b"DANOS MAKE-ISO-IMAGE".ljust(128) # application
|
||||
sector[702:739] = b" " * 37 # copyright file
|
||||
sector[739:776] = b" " * 37 # abstract file
|
||||
sector[776:813] = b" " * 37 # bibliographic file
|
||||
unspecified_date = b"0" * 16 + b"\x00"
|
||||
for offset in (813, 830, 847, 864): # creation/modification/expiry/effective
|
||||
sector[offset:offset + 17] = unspecified_date
|
||||
sector[881] = 1 # file structure version
|
||||
return bytes(sector)
|
||||
|
||||
|
||||
def boot_record_descriptor():
|
||||
sector = bytearray(ISO_SECTOR)
|
||||
sector[0] = 0 # type: boot record
|
||||
sector[1:6] = b"CD001"
|
||||
sector[6] = 1
|
||||
sector[7:39] = b"EL TORITO SPECIFICATION".ljust(32, b"\x00")
|
||||
sector[71:75] = struct.pack("<I", CATALOG_LBA)
|
||||
return bytes(sector)
|
||||
|
||||
|
||||
def terminator_descriptor():
|
||||
sector = bytearray(ISO_SECTOR)
|
||||
sector[0] = 255
|
||||
sector[1:6] = b"CD001"
|
||||
sector[6] = 1
|
||||
return bytes(sector)
|
||||
|
||||
|
||||
def path_table(byte_order):
|
||||
# A single entry: the root directory.
|
||||
return (struct.pack("BB", 1, 0)
|
||||
+ struct.pack(byte_order + "I", ROOT_DIR_LBA)
|
||||
+ struct.pack(byte_order + "H", 1)
|
||||
+ b"\x00\x00") # identifier 0x00 + pad to even
|
||||
|
||||
|
||||
def root_directory(esp_size):
|
||||
# Records must be sorted by identifier; BOOT.CAT < EFI.IMG holds.
|
||||
entries = (directory_record(b"\x00", ROOT_DIR_LBA, ISO_SECTOR, 0x02)
|
||||
+ directory_record(b"\x01", ROOT_DIR_LBA, ISO_SECTOR, 0x02)
|
||||
+ directory_record(b"BOOT.CAT;1", CATALOG_LBA, ISO_SECTOR, 0)
|
||||
+ directory_record(b"EFI.IMG;1", ESP_LBA, esp_size, 0))
|
||||
return entries + b"\x00" * (ISO_SECTOR - len(entries))
|
||||
|
||||
|
||||
def boot_catalog(esp_size):
|
||||
# Validation entry: EFI platform (0xEF), checksummed so its 16-bit words sum
|
||||
# to zero, closed by the 0x55AA key bytes.
|
||||
validation = bytearray(32)
|
||||
validation[0] = 0x01
|
||||
validation[1] = 0xEF
|
||||
validation[4:28] = b"danos".ljust(24, b"\x00")
|
||||
validation[30] = 0x55
|
||||
validation[31] = 0xAA
|
||||
checksum = (-sum(struct.unpack("<16H", validation))) & 0xFFFF
|
||||
validation[28:30] = struct.pack("<H", checksum)
|
||||
# Initial/default entry: bootable, no emulation, image at ESP_LBA. The
|
||||
# sector-count field is 16-bit (units of 512 bytes) so it can't span a large
|
||||
# ESP; UEFI firmware sizes the FAT filesystem from its own BPB, and the
|
||||
# image's boot files sit well inside the capped span regardless.
|
||||
default = bytearray(32)
|
||||
default[0] = 0x88 # bootable
|
||||
default[1] = 0x00 # no emulation
|
||||
sector_count = min(0xFFFF, esp_size // 512)
|
||||
default[6:8] = struct.pack("<H", sector_count)
|
||||
default[8:12] = struct.pack("<I", ESP_LBA)
|
||||
catalog = bytes(validation) + bytes(default)
|
||||
return catalog + b"\x00" * (ISO_SECTOR - len(catalog))
|
||||
|
||||
|
||||
def hybrid_mbr(esp_size):
|
||||
"""The system-area MBR that makes the ISO flashable: one ESP partition."""
|
||||
mbr = bytearray(512)
|
||||
mbr[440:444] = b"dano" # disk signature (fixed: reproducible builds)
|
||||
start_lba = ESP_LBA * (ISO_SECTOR // 512)
|
||||
partition = struct.pack(
|
||||
"<B3sB3sII",
|
||||
0x80, # status: active (harmless; helps picky firmware)
|
||||
b"\xFE\xFF\xFF", # CHS start: maxed out, LBA is authoritative
|
||||
MBR_PARTITION_TYPE_ESP, # type: EFI System
|
||||
b"\xFE\xFF\xFF", # CHS end
|
||||
start_lba,
|
||||
esp_size // 512,
|
||||
)
|
||||
mbr[446:462] = partition
|
||||
mbr[510] = 0x55
|
||||
mbr[511] = 0xAA
|
||||
return bytes(mbr)
|
||||
|
||||
|
||||
def build(out_path, esp_path):
|
||||
with open(esp_path, "rb") as handle:
|
||||
esp = handle.read()
|
||||
if len(esp) % ISO_SECTOR:
|
||||
esp += b"\x00" * (ISO_SECTOR - len(esp) % ISO_SECTOR)
|
||||
esp_sectors = len(esp) // ISO_SECTOR
|
||||
total_sectors = ESP_LBA + esp_sectors
|
||||
|
||||
table_l = path_table("<")
|
||||
image = bytearray(total_sectors * ISO_SECTOR)
|
||||
image[0:512] = hybrid_mbr(len(esp))
|
||||
image[PVD_LBA * ISO_SECTOR:(PVD_LBA + 1) * ISO_SECTOR] = \
|
||||
primary_volume_descriptor(total_sectors, len(table_l))
|
||||
image[BOOT_RECORD_LBA * ISO_SECTOR:(BOOT_RECORD_LBA + 1) * ISO_SECTOR] = \
|
||||
boot_record_descriptor()
|
||||
image[TERMINATOR_LBA * ISO_SECTOR:(TERMINATOR_LBA + 1) * ISO_SECTOR] = \
|
||||
terminator_descriptor()
|
||||
image[PATH_TABLE_L_LBA * ISO_SECTOR:PATH_TABLE_L_LBA * ISO_SECTOR + len(table_l)] = table_l
|
||||
table_m = path_table(">")
|
||||
image[PATH_TABLE_M_LBA * ISO_SECTOR:PATH_TABLE_M_LBA * ISO_SECTOR + len(table_m)] = table_m
|
||||
image[ROOT_DIR_LBA * ISO_SECTOR:(ROOT_DIR_LBA + 1) * ISO_SECTOR] = root_directory(len(esp))
|
||||
image[CATALOG_LBA * ISO_SECTOR:(CATALOG_LBA + 1) * ISO_SECTOR] = boot_catalog(len(esp))
|
||||
image[ESP_LBA * ISO_SECTOR:] = esp
|
||||
|
||||
with open(out_path, "wb") as handle:
|
||||
handle.write(image)
|
||||
print(f"make-iso-image: wrote {out_path} "
|
||||
f"({total_sectors * ISO_SECTOR // (1024 * 1024)} MiB hybrid ISO, "
|
||||
f"ESP at LBA {ESP_LBA}, {esp_sectors} sectors)")
|
||||
|
||||
|
||||
def verify(path):
|
||||
with open(path, "rb") as handle:
|
||||
data = handle.read()
|
||||
# The hybrid MBR (the Etcher/dd boot path).
|
||||
if data[510] != 0x55 or data[511] != 0xAA:
|
||||
sys.exit("verify: missing MBR 0x55AA signature")
|
||||
status, _, part_type, _, part_start, part_sectors = \
|
||||
struct.unpack_from("<B3sB3sII", data, 446)
|
||||
if part_type != MBR_PARTITION_TYPE_ESP:
|
||||
sys.exit(f"verify: MBR partition type 0x{part_type:02X}, expected 0xEF (ESP)")
|
||||
# The ISO9660 descriptors (the optical boot path).
|
||||
if data[PVD_LBA * ISO_SECTOR + 1:PVD_LBA * ISO_SECTOR + 6] != b"CD001":
|
||||
sys.exit("verify: no primary volume descriptor")
|
||||
boot_record = data[BOOT_RECORD_LBA * ISO_SECTOR:(BOOT_RECORD_LBA + 1) * ISO_SECTOR]
|
||||
if not boot_record.startswith(b"\x00CD001") or \
|
||||
not boot_record[7:30].startswith(b"EL TORITO SPECIFICATION"):
|
||||
sys.exit("verify: no El Torito boot record")
|
||||
catalog_lba = struct.unpack_from("<I", boot_record, 71)[0]
|
||||
catalog = data[catalog_lba * ISO_SECTOR:(catalog_lba + 1) * ISO_SECTOR]
|
||||
if catalog[0] != 0x01 or catalog[1] != 0xEF or catalog[30:32] != b"\x55\xAA":
|
||||
sys.exit("verify: boot catalog validation entry is not an EFI entry")
|
||||
if sum(struct.unpack("<16H", catalog[0:32])) & 0xFFFF != 0:
|
||||
sys.exit("verify: boot catalog validation checksum is wrong")
|
||||
if catalog[32] != 0x88:
|
||||
sys.exit("verify: default catalog entry is not bootable")
|
||||
boot_lba = struct.unpack_from("<I", catalog, 40)[0]
|
||||
# Both paths must agree on where the ESP lives, and it must be a FAT32 image.
|
||||
if boot_lba * (ISO_SECTOR // 512) != part_start:
|
||||
sys.exit(f"verify: catalog boot image (LBA {boot_lba}) and MBR partition "
|
||||
f"(sector {part_start}) disagree")
|
||||
esp = data[boot_lba * ISO_SECTOR:]
|
||||
if len(esp) < part_sectors * 512:
|
||||
sys.exit("verify: MBR partition extends past the end of the file")
|
||||
if esp[510] != 0x55 or esp[511] != 0xAA or esp[82:90] != b"FAT32 ":
|
||||
sys.exit("verify: embedded image is not a FAT32 boot volume")
|
||||
print(f"verify: {path} is a hybrid ISO — MBR ESP partition (sector {part_start}, "
|
||||
f"{part_sectors} sectors, active={status == 0x80}) and El Torito EFI entry "
|
||||
f"both point at the embedded FAT32 image")
|
||||
|
||||
|
||||
def main(argv):
|
||||
if len(argv) == 3 and argv[1] == "--verify":
|
||||
verify(argv[2])
|
||||
return 0
|
||||
if len(argv) != 3:
|
||||
sys.exit("usage: make-iso-image.py <out.iso> <esp.img>\n"
|
||||
" make-iso-image.py --verify <out.iso>")
|
||||
build(argv[1], argv[2])
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main(sys.argv))
|
||||
Binary file not shown.
@@ -0,0 +1,93 @@
|
||||
Copyright 2018 The Lexend Project Authors (https://github.com/googlefonts/lexend), with Reserved Font Name “RevReading Lexend”.
|
||||
|
||||
This Font Software is licensed under the SIL Open Font License, Version 1.1.
|
||||
This license is copied below, and is also available with a FAQ at:
|
||||
https://openfontlicense.org
|
||||
|
||||
|
||||
-----------------------------------------------------------
|
||||
SIL OPEN FONT LICENSE Version 1.1 - 26 February 2007
|
||||
-----------------------------------------------------------
|
||||
|
||||
PREAMBLE
|
||||
The goals of the Open Font License (OFL) are to stimulate worldwide
|
||||
development of collaborative font projects, to support the font creation
|
||||
efforts of academic and linguistic communities, and to provide a free and
|
||||
open framework in which fonts may be shared and improved in partnership
|
||||
with others.
|
||||
|
||||
The OFL allows the licensed fonts to be used, studied, modified and
|
||||
redistributed freely as long as they are not sold by themselves. The
|
||||
fonts, including any derivative works, can be bundled, embedded,
|
||||
redistributed and/or sold with any software provided that any reserved
|
||||
names are not used by derivative works. The fonts and derivatives,
|
||||
however, cannot be released under any other type of license. The
|
||||
requirement for fonts to remain under this license does not apply
|
||||
to any document created using the fonts or their derivatives.
|
||||
|
||||
DEFINITIONS
|
||||
"Font Software" refers to the set of files released by the Copyright
|
||||
Holder(s) under this license and clearly marked as such. This may
|
||||
include source files, build scripts and documentation.
|
||||
|
||||
"Reserved Font Name" refers to any names specified as such after the
|
||||
copyright statement(s).
|
||||
|
||||
"Original Version" refers to the collection of Font Software components as
|
||||
distributed by the Copyright Holder(s).
|
||||
|
||||
"Modified Version" refers to any derivative made by adding to, deleting,
|
||||
or substituting -- in part or in whole -- any of the components of the
|
||||
Original Version, by changing formats or by porting the Font Software to a
|
||||
new environment.
|
||||
|
||||
"Author" refers to any designer, engineer, programmer, technical
|
||||
writer or other person who contributed to the Font Software.
|
||||
|
||||
PERMISSION & CONDITIONS
|
||||
Permission is hereby granted, free of charge, to any person obtaining
|
||||
a copy of the Font Software, to use, study, copy, merge, embed, modify,
|
||||
redistribute, and sell modified and unmodified copies of the Font
|
||||
Software, subject to the following conditions:
|
||||
|
||||
1) Neither the Font Software nor any of its individual components,
|
||||
in Original or Modified Versions, may be sold by itself.
|
||||
|
||||
2) Original or Modified Versions of the Font Software may be bundled,
|
||||
redistributed and/or sold with any software, provided that each copy
|
||||
contains the above copyright notice and this license. These can be
|
||||
included either as stand-alone text files, human-readable headers or
|
||||
in the appropriate machine-readable metadata fields within text or
|
||||
binary files as long as those fields can be easily viewed by the user.
|
||||
|
||||
3) No Modified Version of the Font Software may use the Reserved Font
|
||||
Name(s) unless explicit written permission is granted by the corresponding
|
||||
Copyright Holder. This restriction only applies to the primary font name as
|
||||
presented to the users.
|
||||
|
||||
4) The name(s) of the Copyright Holder(s) or the Author(s) of the Font
|
||||
Software shall not be used to promote, endorse or advertise any
|
||||
Modified Version, except to acknowledge the contribution(s) of the
|
||||
Copyright Holder(s) and the Author(s) or with their explicit written
|
||||
permission.
|
||||
|
||||
5) The Font Software, modified or unmodified, in part or in whole,
|
||||
must be distributed entirely under this license, and must not be
|
||||
distributed under any other license. The requirement for fonts to
|
||||
remain under this license does not apply to any document created
|
||||
using the Font Software.
|
||||
|
||||
TERMINATION
|
||||
This license becomes null and void if any of the above conditions are
|
||||
not met.
|
||||
|
||||
DISCLAIMER
|
||||
THE FONT SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
||||
EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO ANY WARRANTIES OF
|
||||
MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT
|
||||
OF COPYRIGHT, PATENT, TRADEMARK, OR OTHER RIGHT. IN NO EVENT SHALL THE
|
||||
COPYRIGHT HOLDER BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY,
|
||||
INCLUDING ANY GENERAL, SPECIAL, INDIRECT, INCIDENTAL, OR CONSEQUENTIAL
|
||||
DAMAGES, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING
|
||||
FROM, OUT OF THE USE OR INABILITY TO USE THE FONT SOFTWARE OR FROM
|
||||
OTHER DEALINGS IN THE FONT SOFTWARE.
|
||||
@@ -0,0 +1,71 @@
|
||||
Lexend Variable Font
|
||||
====================
|
||||
|
||||
This download contains Lexend as both a variable font and static fonts.
|
||||
|
||||
Lexend is a variable font with this axis:
|
||||
wght
|
||||
|
||||
This means all the styles are contained in a single file:
|
||||
Lexend-VariableFont_wght.ttf
|
||||
|
||||
If your app fully supports variable fonts, you can now pick intermediate styles
|
||||
that aren’t available as static fonts. Not all apps support variable fonts, and
|
||||
in those cases you can use the static font files for Lexend:
|
||||
static/Lexend-Thin.ttf
|
||||
static/Lexend-ExtraLight.ttf
|
||||
static/Lexend-Light.ttf
|
||||
static/Lexend-Regular.ttf
|
||||
static/Lexend-Medium.ttf
|
||||
static/Lexend-SemiBold.ttf
|
||||
static/Lexend-Bold.ttf
|
||||
static/Lexend-ExtraBold.ttf
|
||||
static/Lexend-Black.ttf
|
||||
|
||||
Get started
|
||||
-----------
|
||||
|
||||
1. Install the font files you want to use
|
||||
|
||||
2. Use your app's font picker to view the font family and all the
|
||||
available styles
|
||||
|
||||
Learn more about variable fonts
|
||||
-------------------------------
|
||||
|
||||
https://developers.google.com/web/fundamentals/design-and-ux/typography/variable-fonts
|
||||
https://variablefonts.typenetwork.com
|
||||
https://medium.com/variable-fonts
|
||||
|
||||
In desktop apps
|
||||
|
||||
https://theblog.adobe.com/can-variable-fonts-illustrator-cc
|
||||
https://helpx.adobe.com/nz/photoshop/using/fonts.html#variable_fonts
|
||||
|
||||
Online
|
||||
|
||||
https://developers.google.com/fonts/docs/getting_started
|
||||
https://developer.mozilla.org/en-US/docs/Web/CSS/CSS_Fonts/Variable_Fonts_Guide
|
||||
https://developer.microsoft.com/en-us/microsoft-edge/testdrive/demos/variable-fonts
|
||||
|
||||
Installing fonts
|
||||
|
||||
MacOS: https://support.apple.com/en-us/HT201749
|
||||
Linux: https://www.google.com/search?q=how+to+install+a+font+on+gnu%2Blinux
|
||||
Windows: https://support.microsoft.com/en-us/help/314960/how-to-install-or-remove-a-font-in-windows
|
||||
|
||||
Android Apps
|
||||
|
||||
https://developers.google.com/fonts/docs/android
|
||||
https://developer.android.com/guide/topics/ui/look-and-feel/downloadable-fonts
|
||||
|
||||
License
|
||||
-------
|
||||
Please read the full license text (OFL.txt) to understand the permissions,
|
||||
restrictions and requirements for usage, redistribution, and modification.
|
||||
|
||||
You can use them in your products & projects – print or digital,
|
||||
commercial or otherwise.
|
||||
|
||||
This isn't legal advice, please consider consulting a lawyer and see the full
|
||||
license for all details.
|
||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
@@ -0,0 +1,43 @@
|
||||
ISC License
|
||||
|
||||
Copyright (c) 2026 Lucide Icons and Contributors
|
||||
|
||||
Permission to use, copy, modify, and/or distribute this software for any
|
||||
purpose with or without fee is hereby granted, provided that the above
|
||||
copyright notice and this permission notice appear in all copies.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS" AND THE AUTHOR DISCLAIMS ALL WARRANTIES
|
||||
WITH REGARD TO THIS SOFTWARE INCLUDING ALL IMPLIED WARRANTIES OF
|
||||
MERCHANTABILITY AND FITNESS. IN NO EVENT SHALL THE AUTHOR BE LIABLE FOR
|
||||
ANY SPECIAL, DIRECT, INDIRECT, OR CONSEQUENTIAL DAMAGES OR ANY DAMAGES
|
||||
WHATSOEVER RESULTING FROM LOSS OF USE, DATA OR PROFITS, WHETHER IN AN
|
||||
ACTION OF CONTRACT, NEGLIGENCE OR OTHER TORTIOUS ACTION, ARISING OUT OF
|
||||
OR IN CONNECTION WITH THE USE OR PERFORMANCE OF THIS SOFTWARE.
|
||||
|
||||
---
|
||||
|
||||
The following Lucide icons are derived from the Feather project:
|
||||
|
||||
airplay, alert-circle, alert-octagon, alert-triangle, aperture, arrow-down-circle, arrow-down-left, arrow-down-right, arrow-down, arrow-left-circle, arrow-left, arrow-right-circle, arrow-right, arrow-up-circle, arrow-up-left, arrow-up-right, arrow-up, at-sign, calendar, cast, check, chevron-down, chevron-left, chevron-right, chevron-up, chevrons-down, chevrons-left, chevrons-right, chevrons-up, circle, clipboard, clock, code, columns, command, compass, corner-down-left, corner-down-right, corner-left-down, corner-left-up, corner-right-down, corner-right-up, corner-up-left, corner-up-right, crosshair, database, divide-circle, divide-square, dollar-sign, download, external-link, feather, frown, hash, headphones, help-circle, info, italic, key, layout, life-buoy, link-2, link, loader, lock, log-in, log-out, maximize, meh, minimize, minimize-2, minus-circle, minus-square, minus, monitor, moon, more-horizontal, more-vertical, move, music, navigation-2, navigation, octagon, pause-circle, percent, plus-circle, plus-square, plus, power, radio, rss, search, server, share, shopping-bag, sidebar, smartphone, smile, square, table-2, tablet, target, terminal, trash-2, trash, triangle, tv, type, upload, x-circle, x-octagon, x-square, x, zoom-in, zoom-out
|
||||
|
||||
The MIT License (MIT) (for the icons listed above)
|
||||
|
||||
Copyright (c) 2013-present Cole Bemis
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in all
|
||||
copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||
SOFTWARE.
|
||||
Binary file not shown.
@@ -0,0 +1,7 @@
|
||||
# Lucide Font
|
||||
|
||||
Lucide is an open-source icon library that provides 1600+ vector (svg) files for displaying icons and symbols in digital and non-digital projects.
|
||||
|
||||
## Source
|
||||
|
||||
This font was taken from the lucid-static package.
|
||||
File diff suppressed because it is too large
Load Diff
Binary file not shown.
Binary file not shown.
@@ -0,0 +1,93 @@
|
||||
Copyright 2022 The Noto Project Authors (https://github.com/notofonts/latin-greek-cyrillic)
|
||||
|
||||
This Font Software is licensed under the SIL Open Font License, Version 1.1.
|
||||
This license is copied below, and is also available with a FAQ at:
|
||||
https://openfontlicense.org
|
||||
|
||||
|
||||
-----------------------------------------------------------
|
||||
SIL OPEN FONT LICENSE Version 1.1 - 26 February 2007
|
||||
-----------------------------------------------------------
|
||||
|
||||
PREAMBLE
|
||||
The goals of the Open Font License (OFL) are to stimulate worldwide
|
||||
development of collaborative font projects, to support the font creation
|
||||
efforts of academic and linguistic communities, and to provide a free and
|
||||
open framework in which fonts may be shared and improved in partnership
|
||||
with others.
|
||||
|
||||
The OFL allows the licensed fonts to be used, studied, modified and
|
||||
redistributed freely as long as they are not sold by themselves. The
|
||||
fonts, including any derivative works, can be bundled, embedded,
|
||||
redistributed and/or sold with any software provided that any reserved
|
||||
names are not used by derivative works. The fonts and derivatives,
|
||||
however, cannot be released under any other type of license. The
|
||||
requirement for fonts to remain under this license does not apply
|
||||
to any document created using the fonts or their derivatives.
|
||||
|
||||
DEFINITIONS
|
||||
"Font Software" refers to the set of files released by the Copyright
|
||||
Holder(s) under this license and clearly marked as such. This may
|
||||
include source files, build scripts and documentation.
|
||||
|
||||
"Reserved Font Name" refers to any names specified as such after the
|
||||
copyright statement(s).
|
||||
|
||||
"Original Version" refers to the collection of Font Software components as
|
||||
distributed by the Copyright Holder(s).
|
||||
|
||||
"Modified Version" refers to any derivative made by adding to, deleting,
|
||||
or substituting -- in part or in whole -- any of the components of the
|
||||
Original Version, by changing formats or by porting the Font Software to a
|
||||
new environment.
|
||||
|
||||
"Author" refers to any designer, engineer, programmer, technical
|
||||
writer or other person who contributed to the Font Software.
|
||||
|
||||
PERMISSION & CONDITIONS
|
||||
Permission is hereby granted, free of charge, to any person obtaining
|
||||
a copy of the Font Software, to use, study, copy, merge, embed, modify,
|
||||
redistribute, and sell modified and unmodified copies of the Font
|
||||
Software, subject to the following conditions:
|
||||
|
||||
1) Neither the Font Software nor any of its individual components,
|
||||
in Original or Modified Versions, may be sold by itself.
|
||||
|
||||
2) Original or Modified Versions of the Font Software may be bundled,
|
||||
redistributed and/or sold with any software, provided that each copy
|
||||
contains the above copyright notice and this license. These can be
|
||||
included either as stand-alone text files, human-readable headers or
|
||||
in the appropriate machine-readable metadata fields within text or
|
||||
binary files as long as those fields can be easily viewed by the user.
|
||||
|
||||
3) No Modified Version of the Font Software may use the Reserved Font
|
||||
Name(s) unless explicit written permission is granted by the corresponding
|
||||
Copyright Holder. This restriction only applies to the primary font name as
|
||||
presented to the users.
|
||||
|
||||
4) The name(s) of the Copyright Holder(s) or the Author(s) of the Font
|
||||
Software shall not be used to promote, endorse or advertise any
|
||||
Modified Version, except to acknowledge the contribution(s) of the
|
||||
Copyright Holder(s) and the Author(s) or with their explicit written
|
||||
permission.
|
||||
|
||||
5) The Font Software, modified or unmodified, in part or in whole,
|
||||
must be distributed entirely under this license, and must not be
|
||||
distributed under any other license. The requirement for fonts to
|
||||
remain under this license does not apply to any document created
|
||||
using the Font Software.
|
||||
|
||||
TERMINATION
|
||||
This license becomes null and void if any of the above conditions are
|
||||
not met.
|
||||
|
||||
DISCLAIMER
|
||||
THE FONT SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
||||
EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO ANY WARRANTIES OF
|
||||
MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT
|
||||
OF COPYRIGHT, PATENT, TRADEMARK, OR OTHER RIGHT. IN NO EVENT SHALL THE
|
||||
COPYRIGHT HOLDER BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY,
|
||||
INCLUDING ANY GENERAL, SPECIAL, INDIRECT, INCIDENTAL, OR CONSEQUENTIAL
|
||||
DAMAGES, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING
|
||||
FROM, OUT OF THE USE OR INABILITY TO USE THE FONT SOFTWARE OR FROM
|
||||
OTHER DEALINGS IN THE FONT SOFTWARE.
|
||||
@@ -0,0 +1,136 @@
|
||||
Noto Serif Variable Font
|
||||
========================
|
||||
|
||||
This download contains Noto Serif as both variable fonts and static fonts.
|
||||
|
||||
Noto Serif is a variable font with these axes:
|
||||
wdth
|
||||
wght
|
||||
|
||||
This means all the styles are contained in these files:
|
||||
NotoSerif-VariableFont_wdth,wght.ttf
|
||||
NotoSerif-Italic-VariableFont_wdth,wght.ttf
|
||||
|
||||
If your app fully supports variable fonts, you can now pick intermediate styles
|
||||
that aren’t available as static fonts. Not all apps support variable fonts, and
|
||||
in those cases you can use the static font files for Noto Serif:
|
||||
static/NotoSerif_ExtraCondensed-Thin.ttf
|
||||
static/NotoSerif_ExtraCondensed-ExtraLight.ttf
|
||||
static/NotoSerif_ExtraCondensed-Light.ttf
|
||||
static/NotoSerif_ExtraCondensed-Regular.ttf
|
||||
static/NotoSerif_ExtraCondensed-Medium.ttf
|
||||
static/NotoSerif_ExtraCondensed-SemiBold.ttf
|
||||
static/NotoSerif_ExtraCondensed-Bold.ttf
|
||||
static/NotoSerif_ExtraCondensed-ExtraBold.ttf
|
||||
static/NotoSerif_ExtraCondensed-Black.ttf
|
||||
static/NotoSerif_Condensed-Thin.ttf
|
||||
static/NotoSerif_Condensed-ExtraLight.ttf
|
||||
static/NotoSerif_Condensed-Light.ttf
|
||||
static/NotoSerif_Condensed-Regular.ttf
|
||||
static/NotoSerif_Condensed-Medium.ttf
|
||||
static/NotoSerif_Condensed-SemiBold.ttf
|
||||
static/NotoSerif_Condensed-Bold.ttf
|
||||
static/NotoSerif_Condensed-ExtraBold.ttf
|
||||
static/NotoSerif_Condensed-Black.ttf
|
||||
static/NotoSerif_SemiCondensed-Thin.ttf
|
||||
static/NotoSerif_SemiCondensed-ExtraLight.ttf
|
||||
static/NotoSerif_SemiCondensed-Light.ttf
|
||||
static/NotoSerif_SemiCondensed-Regular.ttf
|
||||
static/NotoSerif_SemiCondensed-Medium.ttf
|
||||
static/NotoSerif_SemiCondensed-SemiBold.ttf
|
||||
static/NotoSerif_SemiCondensed-Bold.ttf
|
||||
static/NotoSerif_SemiCondensed-ExtraBold.ttf
|
||||
static/NotoSerif_SemiCondensed-Black.ttf
|
||||
static/NotoSerif-Thin.ttf
|
||||
static/NotoSerif-ExtraLight.ttf
|
||||
static/NotoSerif-Light.ttf
|
||||
static/NotoSerif-Regular.ttf
|
||||
static/NotoSerif-Medium.ttf
|
||||
static/NotoSerif-SemiBold.ttf
|
||||
static/NotoSerif-Bold.ttf
|
||||
static/NotoSerif-ExtraBold.ttf
|
||||
static/NotoSerif-Black.ttf
|
||||
static/NotoSerif_ExtraCondensed-ThinItalic.ttf
|
||||
static/NotoSerif_ExtraCondensed-ExtraLightItalic.ttf
|
||||
static/NotoSerif_ExtraCondensed-LightItalic.ttf
|
||||
static/NotoSerif_ExtraCondensed-Italic.ttf
|
||||
static/NotoSerif_ExtraCondensed-MediumItalic.ttf
|
||||
static/NotoSerif_ExtraCondensed-SemiBoldItalic.ttf
|
||||
static/NotoSerif_ExtraCondensed-BoldItalic.ttf
|
||||
static/NotoSerif_ExtraCondensed-ExtraBoldItalic.ttf
|
||||
static/NotoSerif_ExtraCondensed-BlackItalic.ttf
|
||||
static/NotoSerif_Condensed-ThinItalic.ttf
|
||||
static/NotoSerif_Condensed-ExtraLightItalic.ttf
|
||||
static/NotoSerif_Condensed-LightItalic.ttf
|
||||
static/NotoSerif_Condensed-Italic.ttf
|
||||
static/NotoSerif_Condensed-MediumItalic.ttf
|
||||
static/NotoSerif_Condensed-SemiBoldItalic.ttf
|
||||
static/NotoSerif_Condensed-BoldItalic.ttf
|
||||
static/NotoSerif_Condensed-ExtraBoldItalic.ttf
|
||||
static/NotoSerif_Condensed-BlackItalic.ttf
|
||||
static/NotoSerif_SemiCondensed-ThinItalic.ttf
|
||||
static/NotoSerif_SemiCondensed-ExtraLightItalic.ttf
|
||||
static/NotoSerif_SemiCondensed-LightItalic.ttf
|
||||
static/NotoSerif_SemiCondensed-Italic.ttf
|
||||
static/NotoSerif_SemiCondensed-MediumItalic.ttf
|
||||
static/NotoSerif_SemiCondensed-SemiBoldItalic.ttf
|
||||
static/NotoSerif_SemiCondensed-BoldItalic.ttf
|
||||
static/NotoSerif_SemiCondensed-ExtraBoldItalic.ttf
|
||||
static/NotoSerif_SemiCondensed-BlackItalic.ttf
|
||||
static/NotoSerif-ThinItalic.ttf
|
||||
static/NotoSerif-ExtraLightItalic.ttf
|
||||
static/NotoSerif-LightItalic.ttf
|
||||
static/NotoSerif-Italic.ttf
|
||||
static/NotoSerif-MediumItalic.ttf
|
||||
static/NotoSerif-SemiBoldItalic.ttf
|
||||
static/NotoSerif-BoldItalic.ttf
|
||||
static/NotoSerif-ExtraBoldItalic.ttf
|
||||
static/NotoSerif-BlackItalic.ttf
|
||||
|
||||
Get started
|
||||
-----------
|
||||
|
||||
1. Install the font files you want to use
|
||||
|
||||
2. Use your app's font picker to view the font family and all the
|
||||
available styles
|
||||
|
||||
Learn more about variable fonts
|
||||
-------------------------------
|
||||
|
||||
https://developers.google.com/web/fundamentals/design-and-ux/typography/variable-fonts
|
||||
https://variablefonts.typenetwork.com
|
||||
https://medium.com/variable-fonts
|
||||
|
||||
In desktop apps
|
||||
|
||||
https://theblog.adobe.com/can-variable-fonts-illustrator-cc
|
||||
https://helpx.adobe.com/nz/photoshop/using/fonts.html#variable_fonts
|
||||
|
||||
Online
|
||||
|
||||
https://developers.google.com/fonts/docs/getting_started
|
||||
https://developer.mozilla.org/en-US/docs/Web/CSS/CSS_Fonts/Variable_Fonts_Guide
|
||||
https://developer.microsoft.com/en-us/microsoft-edge/testdrive/demos/variable-fonts
|
||||
|
||||
Installing fonts
|
||||
|
||||
MacOS: https://support.apple.com/en-us/HT201749
|
||||
Linux: https://www.google.com/search?q=how+to+install+a+font+on+gnu%2Blinux
|
||||
Windows: https://support.microsoft.com/en-us/help/314960/how-to-install-or-remove-a-font-in-windows
|
||||
|
||||
Android Apps
|
||||
|
||||
https://developers.google.com/fonts/docs/android
|
||||
https://developer.android.com/guide/topics/ui/look-and-feel/downloadable-fonts
|
||||
|
||||
License
|
||||
-------
|
||||
Please read the full license text (OFL.txt) to understand the permissions,
|
||||
restrictions and requirements for usage, redistribution, and modification.
|
||||
|
||||
You can use them in your products & projects – print or digital,
|
||||
commercial or otherwise.
|
||||
|
||||
This isn't legal advice, please consider consulting a lawyer and see the full
|
||||
license for all details.
|
||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user