2 Commits
7 changed files with 769 additions and 10 deletions
+11
View File
@@ -54,6 +54,17 @@ the UEFI bootloader at `zig-out/EFI/BOOT/BOOTX64.efi`, the kernel at
`zig-out/system/kernel`, init at `zig-out/system/services/init`, drivers under
`zig-out/system/drivers/`, and the initial-ramdisk at `zig-out/boot/`.
## Release media
```sh
zig build release-x86-64
```
Produces `zig-out/danos-x86-64.iso`, a hybrid ISO that boots flashed raw to a
USB stick (balenaEtcher, dd) or burned to optical media — see
[docs/release-iso.md](docs/release-iso.md). `zig build check-iso-image`
validates it without booting.
## Run
Boot it in QEMU with OVMF (opens a display window):
+25
View File
@@ -684,6 +684,31 @@ pub fn build(b: *std.Build) void {
const check_fat_step = b.step("check-fat-image", "Verify the FAT32 USB image is valid and bootable");
check_fat_step.dependOn(&check_fat.step);
// --- release-x86-64: danos-x86-64.iso, the flashable release image ---
// Wrap the FAT32 boot volume in a hybrid ISO (the in-repo Python builder
// again, no xorriso/isohybrid): an ISO9660 whose El Torito EFI boot entry
// and MBR ESP partition entry both point at the embedded FAT image. One
// file then boots every way release media is consumed — flashed raw to a
// USB stick with Etcher or dd, or burned to optical media — while
// danos-usb.img stays the raw superfloppy QEMU and the test harness boot.
const mk_iso = b.addSystemCommand(&.{"python3"});
mk_iso.addFileArg(b.path("tools/make-iso-image.py"));
const iso_image = mk_iso.addOutputFileArg("danos-x86-64.iso");
mk_iso.addFileArg(fat_image);
const iso_install = b.addInstallFile(iso_image, "danos-x86-64.iso");
const release_step = b.step("release-x86-64", "Build the flashable x86-64 release ISO (zig-out/danos-x86-64.iso; flash with Etcher or dd)");
release_step.dependOn(&iso_install.step);
// `zig build check-iso-image` — the ISO builder's own --verify (mirroring
// check-fat-image): the MBR partition, the El Torito catalog, and the
// embedded FAT32 image must all agree.
const check_iso = b.addSystemCommand(&.{"python3"});
check_iso.addFileArg(b.path("tools/make-iso-image.py"));
check_iso.addArg("--verify");
check_iso.addFileArg(iso_image);
const check_iso_step = b.step("check-iso-image", "Verify the release ISO is a valid hybrid (MBR ESP partition + El Torito EFI entry)");
check_iso_step.dependOn(&check_iso.step);
// --- run-x86-64: boot the x86-64 kernel in QEMU via UEFI/OVMF ---
// Firmware lives in different places per OS/distro, so probe the known
// layouts (Architecture, Debian/Ubuntu, Fedora, macOS Homebrew) and use the first
+26 -10
View File
@@ -39,37 +39,42 @@ rather than restate it. Roughly in the order things happen at runtime:
endpoints — the backbone the microkernel's isolated servers talk over.
12. **[syscall.md](syscall.md) — system calls.** How ring 3 asks the kernel for
something: the `syscall`/`sysret` fast path, the trap frame, and why the table is
deliberately tiny.
13. **[drivers.md](drivers.md) — writing a driver.** The payoff: a driver is an
deliberately tiny. The numbers are a **private** ABI — [vdso.md](vdso.md) designs
the public boundary that will hide them.
13. **[vfs-protocol.md](vfs-protocol.md) — the VFS wire protocol.** The language-neutral
byte-level spec of the file protocol spoken over IPC: request/reply headers,
the operation table, mount routing, and the append-only evolution rules — the
first IPC protocol documented as public ABI.
14. **[drivers.md](drivers.md) — writing a driver.** The payoff: a driver is an
ordinary ring-3 process that claims a device, maps its registers, and **sleeps
until its hardware interrupts it**. The claim is the capability; `irq_ack` is the
unmask.
14. **[driver-model.md](driver-model.md) — buses, classes and host controllers.** How
15. **[driver-model.md](driver-model.md) — buses, classes and host controllers.** How
real driver stacks factor into three shapes and how families share code. The
three primitives it proposed are long since built (M13 capability passing,
M14 DMA + barriers, M15 MSI), and the driver *contract* on top of them —
hello, supervision, restart — is built too (device-manager.md, M18).
15. **[process-management.md](process-management.md) — process management.** The
16. **[process-management.md](process-management.md) — process management.** The
microkernel's `ps`/`kill`/SIGCHLD: enumerate as a table snapshot, the
supervision link as the kill authority, and child-exit notifications over the
same endpoints IRQs arrive on.
16. **[process-lifecycle.md](process-lifecycle.md) — the process lifecycle.** Built
17. **[process-lifecycle.md](process-lifecycle.md) — the process lifecycle.** Built
(M17): signals over IPC as the one lifecycle vocabulary every process speaks — the
POSIX.1-1990 words with message delivery instead of stack hijack, the stable
`runtime.process` interface, exit reasons, published exit events any stateful
service can subscribe to (the VFS releasing dead clients' handles), and the two
iron rules (cleanup is the kernel's job; kill is not a signal).
17. **[device-manager.md](device-manager.md) — the device manager.** Built (M18,
18. **[device-manager.md](device-manager.md) — the device manager.** Built (M18,
through the app surface): the
tree, the matcher, and the supervisor. Tree structure lives in the manager,
authority stays in the kernel; bus drivers report what they see; drivers are
restarted through the lifecycle vocabulary — the plan that turns
[resilience.md](resilience.md)'s restart goal into increments.
18. **[input.md](input.md) — the input module.** Broadcasting input events (keyboard,
19. **[input.md](input.md) — the input module.** Broadcasting input events (keyboard,
mouse, joystick): why a synchronous rendezvous can't fan out to many listeners, the
asynchronous `ipc_send` primitive built to fix it, and the per-device subscribe/publish
service layered on top.
19. **[display.md](display.md) — the display service.** The display half of the GUI
20. **[display.md](display.md) — the display service.** The display half of the GUI
track: a user-space compositor that owns the framebuffer, composes a layer stack into
a double buffer, and presents it. Why GOP and the PCI display device are two views of
one controller, the device-node + write-combining handoff, and what flicker-free buys
@@ -80,7 +85,7 @@ rather than restate it. Roughly in the order things happen at runtime:
further out, two research snapshots survey what a *native* driver for real GPU silicon
would take as another `.scanout` backend: [nvidia-gpus.md](nvidia-gpus.md) (RTX 3060 /
Ampere) and [intel-igpu.md](intel-igpu.md) (Intel iGPU).
20. **[halting.md](halting.md) — halting.** Why a kernel can't just "exit", and
21. **[halting.md](halting.md) — halting.** Why a kernel can't just "exit", and
how `while (true) hlt` parks the CPU safely once there's nothing left to do.
Start with the north star:
@@ -99,6 +104,12 @@ Start with the north star:
port to **one seam** (`std.os.danos`), so we build `runtime.os` (→ that seam) plus a
thin `runtime.fs`, retire the `posix` shim, and follow a phased path to
`zig build-exe hello.zig` running on danos — **not** Linux-ABI emulation.
- **[vdso.md](vdso.md) — the vDSO, the public system-call boundary.** A design note
(not built yet) on keeping `abi.zig` genuinely private: a kernel-supplied, C-ABI
entry blob mapped into every process as the *only* way into the kernel — so the
syscall numbers can be renumbered or randomised at will, and Rust/C binaries get a
stable boundary without danos growing a dynamic linker. danos's public ABI = the
vDSO + the documented IPC wire protocols ([vfs-protocol.md](vfs-protocol.md) first).
Cutting across all of these:
@@ -106,6 +117,11 @@ Cutting across all of these:
hardware needed to run danos: minimum specs (UEFI x86-64, ACPI, PCIe ECAM,
xHCI, ~128 MiB RAM) grounded in what the boot path actually assumes, plus a
plain-language guide matching Intel/AMD CPU generations by name.
- **[release-iso.md](release-iso.md) — the release ISO.** The flashable boot
media: `zig build release-x86-64` wraps the FAT32 boot volume in a hybrid ISO
(MBR ESP partition + El Torito EFI entry, one embedded image) that Etcher/dd
flash to USB or a burner writes to disc — built by an in-repo pure-Python
tool, like the FAT image itself.
- **[arch.md](arch.md) — the architecture split.** How CPU-specific code is kept
behind a build-time `arch` module so the generic kernel never names x86_64,
leaving room for other systems (e.g. an AArch64 Raspberry Pi) later.
@@ -250,5 +266,5 @@ exception in [coding-standards.md](coding-standards.md) applies to that seam.
| danos-native runtime (`runtime`): syscall wrappers, heap, IPC, device access, the file API (`fs`) — the stable application ABI | `library/runtime/` |
| System services (init, the VFS server + `protocol`, the device-manager) | `system/services/` |
| Device drivers, one sub-project each (`pci-bus`, `ps2-bus`, `usb-xhci-bus` bus drivers) | `system/drivers/` |
| Build + `run-x86-64` (QEMU/OVMF) | `build.zig` |
| Build + `run-x86-64` (QEMU/OVMF) + `release-x86-64` (the flashable ISO) | `build.zig` |
| QEMU integration test harness | `test/qemu_test.py` |
+70
View File
@@ -0,0 +1,70 @@
# The release ISO — flashable boot media
`zig build release-x86-64` produces **`zig-out/danos-x86-64.iso`**, the file you
hand to someone who wants to try danos on a real machine: point
[balenaEtcher](https://etcher.balena.io) (or Raspberry Pi Imager, or plain `dd`)
at it, flash a USB stick, and boot the stick. The same file also burns to
optical media. `zig build check-iso-image` validates it without booting.
```
zig build release-x86-64
# Etcher: select danos-x86-64.iso → select the stick → Flash
# or: sudo dd if=zig-out/danos-x86-64.iso of=/dev/rdiskN bs=4m (macOS; triple-check N)
```
## Why an ISO when danos-usb.img already boots
`danos-usb.img` is a raw FAT32 **superfloppy** — a filesystem starting at
sector 0, no partition table. UEFI firmware accepts that from a USB stick (it
probes whole-disk FAT before giving up), which is why `dd`-ing the .img works
and why QEMU and the test harness boot it directly. But it is a
developer-shaped artifact: flashing apps expect an ISO, and a superfloppy
can't be burned to a CD/DVD or carry a partition table for pickier firmware.
The ISO wraps that same FAT image — bit-identical, built by the same
`tools/make-fat-image.py` — in a container that boots everywhere release media
gets consumed. One payload, two images: the .img stays the raw volume the QEMU
harness mounts and boots, the .iso is what leaves the building.
## How a hybrid ISO boots twice
The trick (the same one Linux distribution ISOs use, usually via `xorriso
-isohybrid…`) is that ISO9660 reserves its first 32 KiB as a **system area** it
never touches — exactly where an MBR lives on a disk. So one file can carry two
tables of contents, both pointing at the same embedded FAT image:
* **Flashed to USB (Etcher, dd):** firmware sees a disk whose sector 0 is an
MBR with one partition of type `0xEF` (EFI System Partition) covering the
embedded FAT image. It mounts that ESP and runs `\EFI\BOOT\BOOTX64.efi` —
the standard removable-media path ([efi.md](efi.md)).
* **Burned to optical media:** firmware reads the ISO9660 volume descriptors
at sector 16 and finds an **El Torito** boot record. Its catalog has one
entry, platform ID `0xEF` (EFI), whose start LBA is — again — the embedded
FAT image. The firmware exposes that image as a virtual disk and runs the
same `BOOTX64.efi` off it.
Neither path involves the legacy BIOS boot-sector machinery: danos is
UEFI-only ([system-requirements.md](system-requirements.md)), so the MBR holds
no boot code, just the partition entry, and the El Torito entry is EFI-class,
not floppy emulation.
One El Torito wrinkle: the catalog's sector-count field is 16-bit (units of
512 bytes), so it can name at most 32 MiB — less than the 64 MiB FAT image.
That is fine in practice: firmware sizes the FAT filesystem from its own BPB,
and the boot files sit in the first few MiB of the image (clusters are
allocated from the front) either way. The USB path has no such cap.
## The builder
`tools/make-iso-image.py` follows the house rule of
[make-fat-image.py](../tools/make-fat-image.py): pure Python 3 standard
library, no external tools (no xorriso, mkisofs, or isohybrid), with a
`--verify` mode the `check-iso-image` step runs — it checks that the MBR
partition and the El Torito catalog agree on where the FAT image lives and
that a FAT32 boot sector is actually there. Every timestamp field in the ISO
is zeroed, so the build is reproducible byte-for-byte.
The ISO9660 filesystem around the boot machinery is minimal but real: a root
directory listing `BOOT.CAT` (the catalog) and `EFI.IMG` (the FAT image), so
`file`, mount tools, and archive browsers can open the ISO and see what's in
it.
+206
View File
@@ -0,0 +1,206 @@
# The vDSO — the public system-call boundary
> **Status:** design note, not built. The runtime today issues raw `syscall`
> instructions from `library/runtime/system-call.zig` using the numbers in
> `system/abi.zig`. This note designs the layer that replaces that arrangement:
> a **kernel-supplied, C-ABI entry library** mapped into every process — the
> only supported way into the kernel — so the raw numbers can stay private,
> be renumbered at will, and eventually be randomised per boot.
## Why: the ABI danos promises, and the one it doesn't
`system/abi.zig` is the **private** kernel ↔ runtime contract. Its header says
so: the numbers are an implementation detail the runtime hides and may
renumber, the same split as libSystem over the XNU syscalls on macOS or win32
over the NT syscalls on Windows. Linux — with its world-visible, frozen
syscall table — is the outlier, not the norm.
That stance has consequences the moment binaries exist that we don't rebuild
ourselves:
1. **Third-party binaries** (docs/zig-self-hosting.md) must keep working across
kernel updates. If they contain raw `syscall` instructions with today's
numbers baked in, every renumbering breaks the world — the ABI would be
*de facto* public no matter what the header says. Go on macOS made exactly
this mistake: it issued XNU syscalls directly instead of going through
libSystem, and macOS updates repeatedly broke every Go binary until Go
switched to the library like everyone else.
2. **Not everything is Zig.** A Rust or C program can't import the `runtime`
module. The public boundary has to be expressible in the one calling
convention every language speaks: the C ABI.
3. **Randomised syscall numbers** — a hardening option we want open — only
work if no user binary anywhere knows a number at build time. The binding
must happen at *load time*, from something the kernel controls.
All three point at the same well-known shape: a **vDSO** (virtual dynamic
shared object). The kernel carries a small blob of user-mode code, maps it
into every process at spawn, and that blob — not the application — contains
the `syscall` instructions. Fuchsia works exactly this way: its vDSO is the
*only* kernel entry, version-matched by construction because the kernel itself
injects it. Because the kernel and the blob ship as one artifact, there is
**no version skew, no loader, no search path, and no shared file on disk** —
which is what makes this the resilient way to have a private ABI
(docs/resilience.md), where a conventional `ld.so` + `/lib/libdanos.so`
arrangement would add a loader to every spawn and a single shared point of
failure.
The public danos ABI then has exactly two layers, neither of which is
`abi.zig`:
| Layer | Contract | Spoken by |
|-------|----------|-----------|
| **vDSO** | C-ABI functions, this note | every language's thin shim (`runtime.system` for Zig, a `-sys` crate for Rust, a header for C) |
| **IPC wire protocols** | byte layouts over `ipc_call` ([vfs-protocol.md](vfs-protocol.md) is the first one documented) | any client that can lay out bytes |
Everything above those — the heap, `runtime.fs`, the service harness — is
per-language convenience, compiled into each binary from source, exactly as
today. Nothing about the Zig runtime's shape changes; it just stops being the
*only* door.
## The blob
A single copy of the vDSO code lives in the kernel image (built by
`build.zig` as a tiny freestanding object, embedded like the AP trampoline).
At boot the kernel finalises it once — this is where randomised numbers would
be patched in — and thereafter maps the **same physical pages** read-execute
into every process's address space. The blob is:
- **Position-independent.** It is mapped at a per-process randomised base, so
it must be PIC (rip-relative addressing only — no relocations to process).
- **Stateless and re-entrant.** No writable data. Anything stateful belongs to
the process, not the vDSO.
- **Architecture-specific.** The x86-64 blob wraps `syscall`; an aarch64 blob
wraps `svc #0`. It lives beside the other per-architecture kernel sources
(`system/kernel/architecture/<arch>/`), selected the same way the
`architecture` module is (docs/arch.md).
### Shape: a function table, not an ELF
A real `.so` with a dynamic symbol table is the conventional vDSO shape, but
linking against one at load time needs a dynamic linker in every binary —
machinery danos deliberately doesn't have. Instead the v1 shape is the
simplest thing that is still a stable contract — a **function-pointer table**
at the vDSO base:
```
offset 0 u64 magic 'danosVDS' — a mapped-the-wrong-thing guard
offset 8 u64 api_level incremented when the table grows
offset 16 u64 count number of table entries that follow
offset 24 u64 table[count] function pointers into the vDSO's own code
```
Table *indices* are the public constants (published in a C header,
`danos.h`), assigned once and append-only — the same discipline the IPC
protocols use for operation values. The pointers point at stubs inside the
blob; what those stubs put in `rax` is nobody's business but the kernel's.
A language shim binds in one step: read the base from the init block, check
the magic, keep the table pointer. Feature detection for a binary built
against older headers is `count`/`api_level` — a kernel never removes or
reorders entries.
(If danos ever grows a real dynamic linker, the same blob can additionally
present an ELF `dynsym` without breaking the table — Fuchsia's vDSO is
likewise both a mappable blob and a linkable `.so`. That is a later
convenience, not a requirement.)
### Delivery: the auxiliary vector
The kernel already builds a System V entry block — argc, argv, envp
terminator, **auxiliary vector** — on every new process's stack
(`buildEntryStack`, read by `runtime.start`). The vDSO base rides in a new
auxv entry, exactly Linux's `AT_SYSINFO_EHDR` move. No new syscall, no magic
address, and a language shim finds it the same portable way on every
architecture.
## The function surface
One table entry per kernel call, C ABI (System V AMD64), names prefixed
`danos_`. The current `SystemCall` set maps directly; integer arguments and
returns are `u64`, errors return as negative values exactly as today.
The calls that return two values in `rax:rdx` today — `dma_alloc`
(vaddr + paddr), `msi_bind` (address + data), `shm_create` (vaddr + handle) —
become functions returning a two-`u64` struct. The System V ABI returns a
16-byte struct in `rax:rdx`, so the stub is a plain `syscall; ret` — the
C-ABI spelling of the existing convention, at zero cost.
Grouped as `abi.zig` groups them:
| Group | Functions |
|-------|-----------|
| process | `danos_exit`, `danos_yield`, `danos_sleep`, `danos_spawn`, `danos_process_enumerate`, `danos_process_kill`, `danos_process_exit_reason`, `danos_process_subscribe`, `danos_process_signal`, `danos_signal_bind` |
| memory | `danos_mmap`, `danos_munmap`, `danos_dma_alloc`, `danos_dma_free`, `danos_shm_create`, `danos_shm_map`, `danos_shm_physical` |
| ipc | `danos_endpoint_create`, `danos_ipc_register`, `danos_ipc_lookup`, `danos_ipc_call`, `danos_ipc_reply_wait`, `danos_ipc_send` |
| devices | `danos_device_enumerate`, `danos_device_claim`, `danos_device_register`, `danos_mmio_map`, `danos_irq_bind`, `danos_irq_ack`, `danos_msi_bind`, `danos_io_read`, `danos_io_write` |
| time | `danos_clock`, `danos_wall_clock`, `danos_timer_bind` |
| diagnostics | `danos_debug_write`, `danos_klog_read` |
The constants that ride alongside the calls — mmap protection bits, DMA
flags, notification badge bits, `ExitReason`, `Signal`, well-known service
ids, `page_size`, the IPC message maximum — move to the public header too:
they are wire values a Rust program needs verbatim. What stays private in
`abi.zig` is exactly the thing the vDSO exists to hide: the `SystemCall`
numbers and the trap convention.
## Enforcement, and an honest threat model
Renumbering only has teeth if the kernel **refuses syscalls that don't come
from the vDSO**. The check is cheap: on kernel entry, the saved user `rip`
must lie inside the calling process's vDSO mapping; otherwise the process is
killed with a fault-class exit reason (its supervisor restarts or gives up,
docs/process-lifecycle.md — a foreign-syscall attempt is a bug or an attack,
never something to limp past). Fuchsia enforces exactly this.
What this buys, precisely:
- **ABI freedom** — the real prize. The numbers can change per release or per
boot and nothing outside the kernel image cares. The private ABI stays
actually private, permanently.
- **A single audited chokepoint** for kernel entry, per process, at a
randomised address.
- **Raised bar for exploits**: shellcode can't issue a hard-coded `syscall`;
it must first discover the per-process vDSO base (ASLR) and call through
it.
What it does *not* buy: an attacker with arbitrary code execution in a
process can still *call* the vDSO functions — they are mapped executable in
that process, and return-oriented chains reach them. Syscall randomisation is
hardening, not a security boundary; the security boundary remains the
capability model (what the process's endpoints and device claims let it do).
It is worth building anyway — for the ABI freedom first and the hardening
second — but the design should never be sold as more than that.
## Migration
Phased so every step ships alone (the M-milestone discipline):
1. **The blob + the table.** Build the vDSO, map it at spawn, deliver the
base via auxv. `runtime.system-call.zig` binds through the table when the
auxv entry is present, falls back to raw `syscall` when absent — the whole
tree keeps booting during the transition.
2. **Cut the runtime over.** Delete the raw stubs; `runtime` no longer
imports the `SystemCall` numbers at all (`abi.zig`'s enum becomes
kernel-internal). The QEMU suite passing proves the table carries the
whole system.
3. **Enforce + randomise.** Add the `rip`-range check, then per-boot number
randomisation patched into the blob at kernel init. A test boots with
randomisation on and runs the full suite.
4. **The other languages.** Publish `danos.h`; a Rust `danos-sys` crate wraps
the table. This is also the seam `std.os.danos` calls through when the Zig
self-hosting fork lands (docs/zig-self-hosting.md) — the vDSO is what
makes that seam stable across kernel versions.
## What deliberately stays out
- **No dynamic linker, no `/lib/*.so`.** The vDSO is kernel-injected precisely
so danos binaries can stay fully static above it. Sharing *library code*
across processes stays what it is today: a service behind IPC, or source
compiled into each binary.
- **No file/device I/O in the vDSO.** The microkernel line doesn't move: the
vDSO wraps the same deliberately tiny table (docs/syscall.md); files are
still the VFS server's business over IPC.
- **No fast-path user-mode implementations yet.** Linux's vDSO exists mostly
to answer `gettimeofday` without a kernel entry. `danos_clock` could one
day read the calibrated TSC in user mode the same way — the blob is where
such an optimisation would live — but that is an optimisation, not part of
this design's contract.
+170
View File
@@ -0,0 +1,170 @@
# The VFS wire protocol
> **Status:** built and spoken today between `runtime.fs` (the client) and the
> VFS server (`system/services/vfs`), with mounted backends (the FAT server)
> speaking the same protocol behind the router. The Zig source of truth is
> `system/services/vfs/protocol.zig` (the `vfs-protocol` module), whose unit
> tests pin the sizes and values below. This page is the **language-neutral
> wire specification** of that contract — what a Rust or C client implements
> ([vdso.md](vdso.md) explains why the IPC protocols, not the syscall
> numbers, are danos's public ABI).
## Transport
A VFS exchange is one synchronous IPC rendezvous (`ipc_call`,
docs/ipc.md): the client sends one message and blocks; the server replies
with one message. The endpoint is found by well-known service id
(`ipc_lookup`, service id **1** = vfs).
- A message is at most **256 bytes** (`message_maximum`).
- A request is a fixed 32-byte **Request** header followed by an inline
payload of at most **224 bytes** (`maximum_payload`) — a path, or write
bytes. There is no multi-message request: paths and single reads/writes
must fit, and larger transfers loop (see *read* / *write*).
- A reply is a fixed 24-byte **Reply** header followed by an inline payload —
read bytes, a `FileStatus`, or a `DirectoryEntry`.
- All integers are **little-endian**; layouts are C layout for x86-64
(`extern struct`), offsets given below so nothing need be inferred.
The kernel never parses any of this — it only moves the bytes
(docs/syscall.md); files are entirely a user-space affair.
## Request header — 32 bytes
| offset | size | field | meaning |
|-------:|-----:|-------|---------|
| 0 | 4 | `operation` | an **Operation** value (below) |
| 4 | 4 | — | padding |
| 8 | 8 | `node` | the server-side open-node id from a prior `open`; 0 for path-based operations |
| 16 | 8 | `offset` | byte position for read/write; entry index (cursor) for readdir; else 0 |
| 24 | 4 | `len` | payload length for path/write operations; requested byte count for read |
| 28 | 4 | `flags` | open flags (below); else 0 |
## Reply header — 24 bytes
| offset | size | field | meaning |
|-------:|-----:|-------|---------|
| 0 | 4 | `status` | **0 = success**, negative = failure (signed) |
| 4 | 4 | — | padding |
| 8 | 8 | `node` | the new open-node id (for `open`); else 0 |
| 16 | 4 | `len` | reply payload length in bytes |
| 20 | 4 | — | padding |
On failure the router replies `status = -1`; a mounted backend's negative
status is forwarded to the client verbatim. A richer errno vocabulary is
future work — clients must treat *any* negative status as failure, not match
on -1.
## Operations
Values are append-only and never renumbered (the same evolution rule every
danos protocol follows); an unrecognised operation gets a `status = -1`
reply.
| value | operation | request payload | reply |
|------:|-----------|-----------------|-------|
| 0 | `open` | the path (`len` = its length), `flags` as below | `node` = open-node id |
| 1 | `close` | — (`node` set) | status only |
| 2 | `read` | — (`node`, `offset`, `len` = wanted count) | `len` bytes read, payload = the bytes; `len` 0 at end of file |
| 3 | `write` | the bytes (`node`, `offset`, `len` = count) | `len` = bytes accepted (may be short — loop) |
| 4 | `status` | — (`node` set) | payload = **FileStatus** (24 bytes) |
| 5 | `readdir` | — (`node` = a directory, `offset` = cursor) | payload = one **DirectoryEntry** + name; `len` 0 at end |
| 6 | `mount` | the mount-point path; the backend endpoint rides as the call's **capability** | status only |
| 7 | `unmount` | the mount-point path | status only |
| 8 | `mkdir` | the path | status only |
| 9 | `unlink` | the path | status only |
| 10 | `rename` | old path, one `0x00`, new path (`len` = total) | status only |
Notes per operation:
- **open** — paths are absolute (`/mnt/usb/notes.txt`) or bare names
(`greeting`); bare names resolve in the VFS's flat ramfs, absolute paths
route through the mount table (below). The returned `node` is an id in the
*router's* open table; clients never see a backend's own ids.
- **read / write** — a single exchange moves at most 224 bytes
(`maximum_payload`); the client loops, advancing `offset` by the returned
`len`, until done (read) or the slice is written (write). A `write` reply
shorter than requested is progress, not an error; a `len` of 0 means no
forward progress — stop rather than spin.
- **readdir** — `offset` is a **cursor: the entry index**, not a byte
position. Each call returns exactly one entry; the client increments the
cursor by 1. A reply with `len` 0 is end-of-directory. The directory must
have been opened with the `directory` flag.
- **mount** — the one operation that passes a **capability**: the caller
(a filesystem server, e.g. FAT) sends its own request endpoint as the
`ipc_call` capability argument, and the router forwards everything under
the mount point to it — speaking this same protocol, with paths rewritten
relative to the mount. Prefixes match at path boundaries only
(`/mnt/usb` never captures `/mnt/usbextra`); the longest matching prefix
wins.
- **rename** — same-directory rename only (the router requires old and new to
resolve under one mount).
## Open flags
Bitwise OR in `Request.flags`, meaningful for `open` only:
| bit | name | meaning |
|----:|------|---------|
| 1 | `create` | create the file if it does not exist |
| 2 | `directory` | open a directory node for `readdir` rather than a file |
| 4 | `truncate` | truncate an existing file to zero length on open (replace, don't overwrite in place) |
## FileStatus — 24 bytes (the `status` reply payload)
| offset | size | field | meaning |
|-------:|-----:|-------|---------|
| 0 | 8 | `size` | file size in bytes |
| 8 | 4 | `kind` | a **NodeKind** value |
| 12 | 4 | — | padding |
| 16 | 8 | `mtime` | modification time, Unix epoch seconds UTC; 0 if the backend keeps none |
## DirectoryEntry — 16 bytes + name (the `readdir` reply payload)
| offset | size | field | meaning |
|-------:|-----:|-------|---------|
| 0 | 4 | `kind` | a **NodeKind** value |
| 4 | 4 | `name_len` | length of the name that follows |
| 8 | 8 | `size` | the entry's size in bytes |
| 16 | `name_len` | name | the entry's name, not NUL-terminated |
## NodeKind
Aligned to the FSH file-type table
(docs/danos-file-system-hierarchy-FSH.md):
| value | kind |
|------:|------|
| 0 | regular file |
| 1 | directory |
| 2 | character device |
| 3 | block device |
| 4 | symbolic link |
| 5 | fifo |
| 6 | socket |
Clients should map unknown values to *regular* rather than reject — the
table can grow.
## Lifetimes and trust
Open-node ids live in the server. A client that dies without closing leaks
nothing permanently: the VFS subscribes to the kernel's published process-exit
events (docs/process-lifecycle.md) and releases a dead client's handles,
closing forwarded backend nodes best-effort. Ids are plain integers, not
capabilities — the VFS trusts its callers with each other's ids today, which
is acceptable while every client is part of the system image and worth
revisiting (per-client id namespaces) before third-party binaries arrive.
## Evolution rules
What a non-Zig implementation may rely on, and what it must not:
- Operation values, flag bits, `NodeKind` values, and struct layouts are
**append-only and frozen once shipped** — the unit tests in `protocol.zig`
pin them exactly so a refactor can't silently move them.
- The 256-byte message ceiling is a property of the current IPC transport,
not a promise; clients should read `maximum_payload`-shaped limits from the
reply lengths they actually get (loop-until-done), not hard-code 224.
- Negative statuses beyond -1 will appear (an errno vocabulary); success is
exactly 0.
+261
View File
@@ -0,0 +1,261 @@
#!/usr/bin/env python3
"""Wrap the FAT32 boot volume in a hybrid ISO — the flashable danos release image.
Mirrors tools/make-fat-image.py in spirit: pure Python 3 standard library, no
external tools (no xorriso / mkisofs / isohybrid). The output is one file that
boots both ways release media is consumed:
* Flashed raw to a USB stick (Etcher, dd): the ISO's system area carries an
MBR whose single partition (type 0xEF, "EFI System") points at the FAT32
image embedded in the ISO, so UEFI firmware finds the ESP and runs
\\EFI\\BOOT\\BOOTX64.efi off it.
* Burned to optical media: an El Torito boot catalog with an EFI platform
entry points at the same embedded FAT image.
The ISO9660 filesystem itself is minimal but valid — a primary volume
descriptor, the El Torito boot record, path tables, and a root directory that
lists the boot catalog and the FAT image — so inspection tools can open it.
make-iso-image.py <out.iso> <esp.img>
make-iso-image.py --verify <out.iso>
All timestamp fields are zero ("not specified") so the build is reproducible.
"""
import struct
import sys
ISO_SECTOR = 2048
# Fixed layout, in ISO sectors (LBA). Sectors 0-15 are the system area (the
# hybrid MBR lives in its first 512 bytes); volume descriptors start at 16.
PVD_LBA = 16 # primary volume descriptor
BOOT_RECORD_LBA = 17 # El Torito boot record volume descriptor
TERMINATOR_LBA = 18 # volume descriptor set terminator
PATH_TABLE_L_LBA = 19
PATH_TABLE_M_LBA = 20
ROOT_DIR_LBA = 21 # root directory (one sector holds our four records)
CATALOG_LBA = 22 # El Torito boot catalog
ESP_LBA = 23 # the embedded FAT32 image starts here
MBR_PARTITION_TYPE_ESP = 0xEF
def both16(value):
"""ISO9660 both-byte-order encoding: little-endian then big-endian."""
return struct.pack("<H", value) + struct.pack(">H", value)
def both32(value):
return struct.pack("<I", value) + struct.pack(">I", value)
def directory_record(identifier, lba, size, flags):
length = 33 + len(identifier)
if length % 2:
length += 1 # records are padded to even length
record = bytearray(length)
record[0] = length
record[2:10] = both32(lba)
record[10:18] = both32(size)
# record[18:25] is the recording date; zero = unspecified (reproducible).
record[25] = flags # 0x02 = directory
record[28:32] = both16(1) # volume sequence number
record[32] = len(identifier)
record[33:33 + len(identifier)] = identifier
return bytes(record)
def primary_volume_descriptor(total_sectors, path_table_size):
sector = bytearray(ISO_SECTOR)
sector[0] = 1 # type: primary
sector[1:6] = b"CD001"
sector[6] = 1 # version
sector[8:40] = b"DANOS".ljust(32) # system identifier
sector[40:72] = b"DANOS".ljust(32) # volume identifier
sector[80:88] = both32(total_sectors)
sector[120:124] = both16(1) # volume set size
sector[124:128] = both16(1) # volume sequence number
sector[128:132] = both16(ISO_SECTOR) # logical block size
sector[132:140] = both32(path_table_size)
sector[140:144] = struct.pack("<I", PATH_TABLE_L_LBA)
sector[148:152] = struct.pack(">I", PATH_TABLE_M_LBA)
sector[156:190] = directory_record(b"\x00", ROOT_DIR_LBA, ISO_SECTOR, 0x02)
sector[190:318] = b" " * 128 # volume set identifier
sector[318:446] = b" " * 128 # publisher
sector[446:574] = b" " * 128 # data preparer
sector[574:702] = b"DANOS MAKE-ISO-IMAGE".ljust(128) # application
sector[702:739] = b" " * 37 # copyright file
sector[739:776] = b" " * 37 # abstract file
sector[776:813] = b" " * 37 # bibliographic file
unspecified_date = b"0" * 16 + b"\x00"
for offset in (813, 830, 847, 864): # creation/modification/expiry/effective
sector[offset:offset + 17] = unspecified_date
sector[881] = 1 # file structure version
return bytes(sector)
def boot_record_descriptor():
sector = bytearray(ISO_SECTOR)
sector[0] = 0 # type: boot record
sector[1:6] = b"CD001"
sector[6] = 1
sector[7:39] = b"EL TORITO SPECIFICATION".ljust(32, b"\x00")
sector[71:75] = struct.pack("<I", CATALOG_LBA)
return bytes(sector)
def terminator_descriptor():
sector = bytearray(ISO_SECTOR)
sector[0] = 255
sector[1:6] = b"CD001"
sector[6] = 1
return bytes(sector)
def path_table(byte_order):
# A single entry: the root directory.
return (struct.pack("BB", 1, 0)
+ struct.pack(byte_order + "I", ROOT_DIR_LBA)
+ struct.pack(byte_order + "H", 1)
+ b"\x00\x00") # identifier 0x00 + pad to even
def root_directory(esp_size):
# Records must be sorted by identifier; BOOT.CAT < EFI.IMG holds.
entries = (directory_record(b"\x00", ROOT_DIR_LBA, ISO_SECTOR, 0x02)
+ directory_record(b"\x01", ROOT_DIR_LBA, ISO_SECTOR, 0x02)
+ directory_record(b"BOOT.CAT;1", CATALOG_LBA, ISO_SECTOR, 0)
+ directory_record(b"EFI.IMG;1", ESP_LBA, esp_size, 0))
return entries + b"\x00" * (ISO_SECTOR - len(entries))
def boot_catalog(esp_size):
# Validation entry: EFI platform (0xEF), checksummed so its 16-bit words sum
# to zero, closed by the 0x55AA key bytes.
validation = bytearray(32)
validation[0] = 0x01
validation[1] = 0xEF
validation[4:28] = b"danos".ljust(24, b"\x00")
validation[30] = 0x55
validation[31] = 0xAA
checksum = (-sum(struct.unpack("<16H", validation))) & 0xFFFF
validation[28:30] = struct.pack("<H", checksum)
# Initial/default entry: bootable, no emulation, image at ESP_LBA. The
# sector-count field is 16-bit (units of 512 bytes) so it can't span a large
# ESP; UEFI firmware sizes the FAT filesystem from its own BPB, and the
# image's boot files sit well inside the capped span regardless.
default = bytearray(32)
default[0] = 0x88 # bootable
default[1] = 0x00 # no emulation
sector_count = min(0xFFFF, esp_size // 512)
default[6:8] = struct.pack("<H", sector_count)
default[8:12] = struct.pack("<I", ESP_LBA)
catalog = bytes(validation) + bytes(default)
return catalog + b"\x00" * (ISO_SECTOR - len(catalog))
def hybrid_mbr(esp_size):
"""The system-area MBR that makes the ISO flashable: one ESP partition."""
mbr = bytearray(512)
mbr[440:444] = b"dano" # disk signature (fixed: reproducible builds)
start_lba = ESP_LBA * (ISO_SECTOR // 512)
partition = struct.pack(
"<B3sB3sII",
0x80, # status: active (harmless; helps picky firmware)
b"\xFE\xFF\xFF", # CHS start: maxed out, LBA is authoritative
MBR_PARTITION_TYPE_ESP, # type: EFI System
b"\xFE\xFF\xFF", # CHS end
start_lba,
esp_size // 512,
)
mbr[446:462] = partition
mbr[510] = 0x55
mbr[511] = 0xAA
return bytes(mbr)
def build(out_path, esp_path):
with open(esp_path, "rb") as handle:
esp = handle.read()
if len(esp) % ISO_SECTOR:
esp += b"\x00" * (ISO_SECTOR - len(esp) % ISO_SECTOR)
esp_sectors = len(esp) // ISO_SECTOR
total_sectors = ESP_LBA + esp_sectors
table_l = path_table("<")
image = bytearray(total_sectors * ISO_SECTOR)
image[0:512] = hybrid_mbr(len(esp))
image[PVD_LBA * ISO_SECTOR:(PVD_LBA + 1) * ISO_SECTOR] = \
primary_volume_descriptor(total_sectors, len(table_l))
image[BOOT_RECORD_LBA * ISO_SECTOR:(BOOT_RECORD_LBA + 1) * ISO_SECTOR] = \
boot_record_descriptor()
image[TERMINATOR_LBA * ISO_SECTOR:(TERMINATOR_LBA + 1) * ISO_SECTOR] = \
terminator_descriptor()
image[PATH_TABLE_L_LBA * ISO_SECTOR:PATH_TABLE_L_LBA * ISO_SECTOR + len(table_l)] = table_l
table_m = path_table(">")
image[PATH_TABLE_M_LBA * ISO_SECTOR:PATH_TABLE_M_LBA * ISO_SECTOR + len(table_m)] = table_m
image[ROOT_DIR_LBA * ISO_SECTOR:(ROOT_DIR_LBA + 1) * ISO_SECTOR] = root_directory(len(esp))
image[CATALOG_LBA * ISO_SECTOR:(CATALOG_LBA + 1) * ISO_SECTOR] = boot_catalog(len(esp))
image[ESP_LBA * ISO_SECTOR:] = esp
with open(out_path, "wb") as handle:
handle.write(image)
print(f"make-iso-image: wrote {out_path} "
f"({total_sectors * ISO_SECTOR // (1024 * 1024)} MiB hybrid ISO, "
f"ESP at LBA {ESP_LBA}, {esp_sectors} sectors)")
def verify(path):
with open(path, "rb") as handle:
data = handle.read()
# The hybrid MBR (the Etcher/dd boot path).
if data[510] != 0x55 or data[511] != 0xAA:
sys.exit("verify: missing MBR 0x55AA signature")
status, _, part_type, _, part_start, part_sectors = \
struct.unpack_from("<B3sB3sII", data, 446)
if part_type != MBR_PARTITION_TYPE_ESP:
sys.exit(f"verify: MBR partition type 0x{part_type:02X}, expected 0xEF (ESP)")
# The ISO9660 descriptors (the optical boot path).
if data[PVD_LBA * ISO_SECTOR + 1:PVD_LBA * ISO_SECTOR + 6] != b"CD001":
sys.exit("verify: no primary volume descriptor")
boot_record = data[BOOT_RECORD_LBA * ISO_SECTOR:(BOOT_RECORD_LBA + 1) * ISO_SECTOR]
if not boot_record.startswith(b"\x00CD001") or \
not boot_record[7:30].startswith(b"EL TORITO SPECIFICATION"):
sys.exit("verify: no El Torito boot record")
catalog_lba = struct.unpack_from("<I", boot_record, 71)[0]
catalog = data[catalog_lba * ISO_SECTOR:(catalog_lba + 1) * ISO_SECTOR]
if catalog[0] != 0x01 or catalog[1] != 0xEF or catalog[30:32] != b"\x55\xAA":
sys.exit("verify: boot catalog validation entry is not an EFI entry")
if sum(struct.unpack("<16H", catalog[0:32])) & 0xFFFF != 0:
sys.exit("verify: boot catalog validation checksum is wrong")
if catalog[32] != 0x88:
sys.exit("verify: default catalog entry is not bootable")
boot_lba = struct.unpack_from("<I", catalog, 40)[0]
# Both paths must agree on where the ESP lives, and it must be a FAT32 image.
if boot_lba * (ISO_SECTOR // 512) != part_start:
sys.exit(f"verify: catalog boot image (LBA {boot_lba}) and MBR partition "
f"(sector {part_start}) disagree")
esp = data[boot_lba * ISO_SECTOR:]
if len(esp) < part_sectors * 512:
sys.exit("verify: MBR partition extends past the end of the file")
if esp[510] != 0x55 or esp[511] != 0xAA or esp[82:90] != b"FAT32 ":
sys.exit("verify: embedded image is not a FAT32 boot volume")
print(f"verify: {path} is a hybrid ISO — MBR ESP partition (sector {part_start}, "
f"{part_sectors} sectors, active={status == 0x80}) and El Torito EFI entry "
f"both point at the embedded FAT32 image")
def main(argv):
if len(argv) == 3 and argv[1] == "--verify":
verify(argv[2])
return 0
if len(argv) != 3:
sys.exit("usage: make-iso-image.py <out.iso> <esp.img>\n"
" make-iso-image.py --verify <out.iso>")
build(argv[1], argv[2])
return 0
if __name__ == "__main__":
sys.exit(main(sys.argv))