Compare commits
2
Commits
f309ce04f4
...
914af52b94
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
914af52b94 | ||
|
|
bdd8a48476 |
@@ -54,6 +54,17 @@ the UEFI bootloader at `zig-out/EFI/BOOT/BOOTX64.efi`, the kernel at
|
|||||||
`zig-out/system/kernel`, init at `zig-out/system/services/init`, drivers under
|
`zig-out/system/kernel`, init at `zig-out/system/services/init`, drivers under
|
||||||
`zig-out/system/drivers/`, and the initial-ramdisk at `zig-out/boot/`.
|
`zig-out/system/drivers/`, and the initial-ramdisk at `zig-out/boot/`.
|
||||||
|
|
||||||
|
## Release media
|
||||||
|
|
||||||
|
```sh
|
||||||
|
zig build release-x86-64
|
||||||
|
```
|
||||||
|
|
||||||
|
Produces `zig-out/danos-x86-64.iso`, a hybrid ISO that boots flashed raw to a
|
||||||
|
USB stick (balenaEtcher, dd) or burned to optical media — see
|
||||||
|
[docs/release-iso.md](docs/release-iso.md). `zig build check-iso-image`
|
||||||
|
validates it without booting.
|
||||||
|
|
||||||
## Run
|
## Run
|
||||||
|
|
||||||
Boot it in QEMU with OVMF (opens a display window):
|
Boot it in QEMU with OVMF (opens a display window):
|
||||||
|
|||||||
@@ -684,6 +684,31 @@ pub fn build(b: *std.Build) void {
|
|||||||
const check_fat_step = b.step("check-fat-image", "Verify the FAT32 USB image is valid and bootable");
|
const check_fat_step = b.step("check-fat-image", "Verify the FAT32 USB image is valid and bootable");
|
||||||
check_fat_step.dependOn(&check_fat.step);
|
check_fat_step.dependOn(&check_fat.step);
|
||||||
|
|
||||||
|
// --- release-x86-64: danos-x86-64.iso, the flashable release image ---
|
||||||
|
// Wrap the FAT32 boot volume in a hybrid ISO (the in-repo Python builder
|
||||||
|
// again, no xorriso/isohybrid): an ISO9660 whose El Torito EFI boot entry
|
||||||
|
// and MBR ESP partition entry both point at the embedded FAT image. One
|
||||||
|
// file then boots every way release media is consumed — flashed raw to a
|
||||||
|
// USB stick with Etcher or dd, or burned to optical media — while
|
||||||
|
// danos-usb.img stays the raw superfloppy QEMU and the test harness boot.
|
||||||
|
const mk_iso = b.addSystemCommand(&.{"python3"});
|
||||||
|
mk_iso.addFileArg(b.path("tools/make-iso-image.py"));
|
||||||
|
const iso_image = mk_iso.addOutputFileArg("danos-x86-64.iso");
|
||||||
|
mk_iso.addFileArg(fat_image);
|
||||||
|
const iso_install = b.addInstallFile(iso_image, "danos-x86-64.iso");
|
||||||
|
const release_step = b.step("release-x86-64", "Build the flashable x86-64 release ISO (zig-out/danos-x86-64.iso; flash with Etcher or dd)");
|
||||||
|
release_step.dependOn(&iso_install.step);
|
||||||
|
|
||||||
|
// `zig build check-iso-image` — the ISO builder's own --verify (mirroring
|
||||||
|
// check-fat-image): the MBR partition, the El Torito catalog, and the
|
||||||
|
// embedded FAT32 image must all agree.
|
||||||
|
const check_iso = b.addSystemCommand(&.{"python3"});
|
||||||
|
check_iso.addFileArg(b.path("tools/make-iso-image.py"));
|
||||||
|
check_iso.addArg("--verify");
|
||||||
|
check_iso.addFileArg(iso_image);
|
||||||
|
const check_iso_step = b.step("check-iso-image", "Verify the release ISO is a valid hybrid (MBR ESP partition + El Torito EFI entry)");
|
||||||
|
check_iso_step.dependOn(&check_iso.step);
|
||||||
|
|
||||||
// --- run-x86-64: boot the x86-64 kernel in QEMU via UEFI/OVMF ---
|
// --- run-x86-64: boot the x86-64 kernel in QEMU via UEFI/OVMF ---
|
||||||
// Firmware lives in different places per OS/distro, so probe the known
|
// Firmware lives in different places per OS/distro, so probe the known
|
||||||
// layouts (Architecture, Debian/Ubuntu, Fedora, macOS Homebrew) and use the first
|
// layouts (Architecture, Debian/Ubuntu, Fedora, macOS Homebrew) and use the first
|
||||||
|
|||||||
+26
-10
@@ -39,37 +39,42 @@ rather than restate it. Roughly in the order things happen at runtime:
|
|||||||
endpoints — the backbone the microkernel's isolated servers talk over.
|
endpoints — the backbone the microkernel's isolated servers talk over.
|
||||||
12. **[syscall.md](syscall.md) — system calls.** How ring 3 asks the kernel for
|
12. **[syscall.md](syscall.md) — system calls.** How ring 3 asks the kernel for
|
||||||
something: the `syscall`/`sysret` fast path, the trap frame, and why the table is
|
something: the `syscall`/`sysret` fast path, the trap frame, and why the table is
|
||||||
deliberately tiny.
|
deliberately tiny. The numbers are a **private** ABI — [vdso.md](vdso.md) designs
|
||||||
13. **[drivers.md](drivers.md) — writing a driver.** The payoff: a driver is an
|
the public boundary that will hide them.
|
||||||
|
13. **[vfs-protocol.md](vfs-protocol.md) — the VFS wire protocol.** The language-neutral
|
||||||
|
byte-level spec of the file protocol spoken over IPC: request/reply headers,
|
||||||
|
the operation table, mount routing, and the append-only evolution rules — the
|
||||||
|
first IPC protocol documented as public ABI.
|
||||||
|
14. **[drivers.md](drivers.md) — writing a driver.** The payoff: a driver is an
|
||||||
ordinary ring-3 process that claims a device, maps its registers, and **sleeps
|
ordinary ring-3 process that claims a device, maps its registers, and **sleeps
|
||||||
until its hardware interrupts it**. The claim is the capability; `irq_ack` is the
|
until its hardware interrupts it**. The claim is the capability; `irq_ack` is the
|
||||||
unmask.
|
unmask.
|
||||||
14. **[driver-model.md](driver-model.md) — buses, classes and host controllers.** How
|
15. **[driver-model.md](driver-model.md) — buses, classes and host controllers.** How
|
||||||
real driver stacks factor into three shapes and how families share code. The
|
real driver stacks factor into three shapes and how families share code. The
|
||||||
three primitives it proposed are long since built (M13 capability passing,
|
three primitives it proposed are long since built (M13 capability passing,
|
||||||
M14 DMA + barriers, M15 MSI), and the driver *contract* on top of them —
|
M14 DMA + barriers, M15 MSI), and the driver *contract* on top of them —
|
||||||
hello, supervision, restart — is built too (device-manager.md, M18).
|
hello, supervision, restart — is built too (device-manager.md, M18).
|
||||||
15. **[process-management.md](process-management.md) — process management.** The
|
16. **[process-management.md](process-management.md) — process management.** The
|
||||||
microkernel's `ps`/`kill`/SIGCHLD: enumerate as a table snapshot, the
|
microkernel's `ps`/`kill`/SIGCHLD: enumerate as a table snapshot, the
|
||||||
supervision link as the kill authority, and child-exit notifications over the
|
supervision link as the kill authority, and child-exit notifications over the
|
||||||
same endpoints IRQs arrive on.
|
same endpoints IRQs arrive on.
|
||||||
16. **[process-lifecycle.md](process-lifecycle.md) — the process lifecycle.** Built
|
17. **[process-lifecycle.md](process-lifecycle.md) — the process lifecycle.** Built
|
||||||
(M17): signals over IPC as the one lifecycle vocabulary every process speaks — the
|
(M17): signals over IPC as the one lifecycle vocabulary every process speaks — the
|
||||||
POSIX.1-1990 words with message delivery instead of stack hijack, the stable
|
POSIX.1-1990 words with message delivery instead of stack hijack, the stable
|
||||||
`runtime.process` interface, exit reasons, published exit events any stateful
|
`runtime.process` interface, exit reasons, published exit events any stateful
|
||||||
service can subscribe to (the VFS releasing dead clients' handles), and the two
|
service can subscribe to (the VFS releasing dead clients' handles), and the two
|
||||||
iron rules (cleanup is the kernel's job; kill is not a signal).
|
iron rules (cleanup is the kernel's job; kill is not a signal).
|
||||||
17. **[device-manager.md](device-manager.md) — the device manager.** Built (M18,
|
18. **[device-manager.md](device-manager.md) — the device manager.** Built (M18,
|
||||||
through the app surface): the
|
through the app surface): the
|
||||||
tree, the matcher, and the supervisor. Tree structure lives in the manager,
|
tree, the matcher, and the supervisor. Tree structure lives in the manager,
|
||||||
authority stays in the kernel; bus drivers report what they see; drivers are
|
authority stays in the kernel; bus drivers report what they see; drivers are
|
||||||
restarted through the lifecycle vocabulary — the plan that turns
|
restarted through the lifecycle vocabulary — the plan that turns
|
||||||
[resilience.md](resilience.md)'s restart goal into increments.
|
[resilience.md](resilience.md)'s restart goal into increments.
|
||||||
18. **[input.md](input.md) — the input module.** Broadcasting input events (keyboard,
|
19. **[input.md](input.md) — the input module.** Broadcasting input events (keyboard,
|
||||||
mouse, joystick): why a synchronous rendezvous can't fan out to many listeners, the
|
mouse, joystick): why a synchronous rendezvous can't fan out to many listeners, the
|
||||||
asynchronous `ipc_send` primitive built to fix it, and the per-device subscribe/publish
|
asynchronous `ipc_send` primitive built to fix it, and the per-device subscribe/publish
|
||||||
service layered on top.
|
service layered on top.
|
||||||
19. **[display.md](display.md) — the display service.** The display half of the GUI
|
20. **[display.md](display.md) — the display service.** The display half of the GUI
|
||||||
track: a user-space compositor that owns the framebuffer, composes a layer stack into
|
track: a user-space compositor that owns the framebuffer, composes a layer stack into
|
||||||
a double buffer, and presents it. Why GOP and the PCI display device are two views of
|
a double buffer, and presents it. Why GOP and the PCI display device are two views of
|
||||||
one controller, the device-node + write-combining handoff, and what flicker-free buys
|
one controller, the device-node + write-combining handoff, and what flicker-free buys
|
||||||
@@ -80,7 +85,7 @@ rather than restate it. Roughly in the order things happen at runtime:
|
|||||||
further out, two research snapshots survey what a *native* driver for real GPU silicon
|
further out, two research snapshots survey what a *native* driver for real GPU silicon
|
||||||
would take as another `.scanout` backend: [nvidia-gpus.md](nvidia-gpus.md) (RTX 3060 /
|
would take as another `.scanout` backend: [nvidia-gpus.md](nvidia-gpus.md) (RTX 3060 /
|
||||||
Ampere) and [intel-igpu.md](intel-igpu.md) (Intel iGPU).
|
Ampere) and [intel-igpu.md](intel-igpu.md) (Intel iGPU).
|
||||||
20. **[halting.md](halting.md) — halting.** Why a kernel can't just "exit", and
|
21. **[halting.md](halting.md) — halting.** Why a kernel can't just "exit", and
|
||||||
how `while (true) hlt` parks the CPU safely once there's nothing left to do.
|
how `while (true) hlt` parks the CPU safely once there's nothing left to do.
|
||||||
|
|
||||||
Start with the north star:
|
Start with the north star:
|
||||||
@@ -99,6 +104,12 @@ Start with the north star:
|
|||||||
port to **one seam** (`std.os.danos`), so we build `runtime.os` (→ that seam) plus a
|
port to **one seam** (`std.os.danos`), so we build `runtime.os` (→ that seam) plus a
|
||||||
thin `runtime.fs`, retire the `posix` shim, and follow a phased path to
|
thin `runtime.fs`, retire the `posix` shim, and follow a phased path to
|
||||||
`zig build-exe hello.zig` running on danos — **not** Linux-ABI emulation.
|
`zig build-exe hello.zig` running on danos — **not** Linux-ABI emulation.
|
||||||
|
- **[vdso.md](vdso.md) — the vDSO, the public system-call boundary.** A design note
|
||||||
|
(not built yet) on keeping `abi.zig` genuinely private: a kernel-supplied, C-ABI
|
||||||
|
entry blob mapped into every process as the *only* way into the kernel — so the
|
||||||
|
syscall numbers can be renumbered or randomised at will, and Rust/C binaries get a
|
||||||
|
stable boundary without danos growing a dynamic linker. danos's public ABI = the
|
||||||
|
vDSO + the documented IPC wire protocols ([vfs-protocol.md](vfs-protocol.md) first).
|
||||||
|
|
||||||
Cutting across all of these:
|
Cutting across all of these:
|
||||||
|
|
||||||
@@ -106,6 +117,11 @@ Cutting across all of these:
|
|||||||
hardware needed to run danos: minimum specs (UEFI x86-64, ACPI, PCIe ECAM,
|
hardware needed to run danos: minimum specs (UEFI x86-64, ACPI, PCIe ECAM,
|
||||||
xHCI, ~128 MiB RAM) grounded in what the boot path actually assumes, plus a
|
xHCI, ~128 MiB RAM) grounded in what the boot path actually assumes, plus a
|
||||||
plain-language guide matching Intel/AMD CPU generations by name.
|
plain-language guide matching Intel/AMD CPU generations by name.
|
||||||
|
- **[release-iso.md](release-iso.md) — the release ISO.** The flashable boot
|
||||||
|
media: `zig build release-x86-64` wraps the FAT32 boot volume in a hybrid ISO
|
||||||
|
(MBR ESP partition + El Torito EFI entry, one embedded image) that Etcher/dd
|
||||||
|
flash to USB or a burner writes to disc — built by an in-repo pure-Python
|
||||||
|
tool, like the FAT image itself.
|
||||||
- **[arch.md](arch.md) — the architecture split.** How CPU-specific code is kept
|
- **[arch.md](arch.md) — the architecture split.** How CPU-specific code is kept
|
||||||
behind a build-time `arch` module so the generic kernel never names x86_64,
|
behind a build-time `arch` module so the generic kernel never names x86_64,
|
||||||
leaving room for other systems (e.g. an AArch64 Raspberry Pi) later.
|
leaving room for other systems (e.g. an AArch64 Raspberry Pi) later.
|
||||||
@@ -250,5 +266,5 @@ exception in [coding-standards.md](coding-standards.md) applies to that seam.
|
|||||||
| danos-native runtime (`runtime`): syscall wrappers, heap, IPC, device access, the file API (`fs`) — the stable application ABI | `library/runtime/` |
|
| danos-native runtime (`runtime`): syscall wrappers, heap, IPC, device access, the file API (`fs`) — the stable application ABI | `library/runtime/` |
|
||||||
| System services (init, the VFS server + `protocol`, the device-manager) | `system/services/` |
|
| System services (init, the VFS server + `protocol`, the device-manager) | `system/services/` |
|
||||||
| Device drivers, one sub-project each (`pci-bus`, `ps2-bus`, `usb-xhci-bus` bus drivers) | `system/drivers/` |
|
| Device drivers, one sub-project each (`pci-bus`, `ps2-bus`, `usb-xhci-bus` bus drivers) | `system/drivers/` |
|
||||||
| Build + `run-x86-64` (QEMU/OVMF) | `build.zig` |
|
| Build + `run-x86-64` (QEMU/OVMF) + `release-x86-64` (the flashable ISO) | `build.zig` |
|
||||||
| QEMU integration test harness | `test/qemu_test.py` |
|
| QEMU integration test harness | `test/qemu_test.py` |
|
||||||
|
|||||||
@@ -0,0 +1,70 @@
|
|||||||
|
# The release ISO — flashable boot media
|
||||||
|
|
||||||
|
`zig build release-x86-64` produces **`zig-out/danos-x86-64.iso`**, the file you
|
||||||
|
hand to someone who wants to try danos on a real machine: point
|
||||||
|
[balenaEtcher](https://etcher.balena.io) (or Raspberry Pi Imager, or plain `dd`)
|
||||||
|
at it, flash a USB stick, and boot the stick. The same file also burns to
|
||||||
|
optical media. `zig build check-iso-image` validates it without booting.
|
||||||
|
|
||||||
|
```
|
||||||
|
zig build release-x86-64
|
||||||
|
# Etcher: select danos-x86-64.iso → select the stick → Flash
|
||||||
|
# or: sudo dd if=zig-out/danos-x86-64.iso of=/dev/rdiskN bs=4m (macOS; triple-check N)
|
||||||
|
```
|
||||||
|
|
||||||
|
## Why an ISO when danos-usb.img already boots
|
||||||
|
|
||||||
|
`danos-usb.img` is a raw FAT32 **superfloppy** — a filesystem starting at
|
||||||
|
sector 0, no partition table. UEFI firmware accepts that from a USB stick (it
|
||||||
|
probes whole-disk FAT before giving up), which is why `dd`-ing the .img works
|
||||||
|
and why QEMU and the test harness boot it directly. But it is a
|
||||||
|
developer-shaped artifact: flashing apps expect an ISO, and a superfloppy
|
||||||
|
can't be burned to a CD/DVD or carry a partition table for pickier firmware.
|
||||||
|
|
||||||
|
The ISO wraps that same FAT image — bit-identical, built by the same
|
||||||
|
`tools/make-fat-image.py` — in a container that boots everywhere release media
|
||||||
|
gets consumed. One payload, two images: the .img stays the raw volume the QEMU
|
||||||
|
harness mounts and boots, the .iso is what leaves the building.
|
||||||
|
|
||||||
|
## How a hybrid ISO boots twice
|
||||||
|
|
||||||
|
The trick (the same one Linux distribution ISOs use, usually via `xorriso
|
||||||
|
-isohybrid…`) is that ISO9660 reserves its first 32 KiB as a **system area** it
|
||||||
|
never touches — exactly where an MBR lives on a disk. So one file can carry two
|
||||||
|
tables of contents, both pointing at the same embedded FAT image:
|
||||||
|
|
||||||
|
* **Flashed to USB (Etcher, dd):** firmware sees a disk whose sector 0 is an
|
||||||
|
MBR with one partition of type `0xEF` (EFI System Partition) covering the
|
||||||
|
embedded FAT image. It mounts that ESP and runs `\EFI\BOOT\BOOTX64.efi` —
|
||||||
|
the standard removable-media path ([efi.md](efi.md)).
|
||||||
|
* **Burned to optical media:** firmware reads the ISO9660 volume descriptors
|
||||||
|
at sector 16 and finds an **El Torito** boot record. Its catalog has one
|
||||||
|
entry, platform ID `0xEF` (EFI), whose start LBA is — again — the embedded
|
||||||
|
FAT image. The firmware exposes that image as a virtual disk and runs the
|
||||||
|
same `BOOTX64.efi` off it.
|
||||||
|
|
||||||
|
Neither path involves the legacy BIOS boot-sector machinery: danos is
|
||||||
|
UEFI-only ([system-requirements.md](system-requirements.md)), so the MBR holds
|
||||||
|
no boot code, just the partition entry, and the El Torito entry is EFI-class,
|
||||||
|
not floppy emulation.
|
||||||
|
|
||||||
|
One El Torito wrinkle: the catalog's sector-count field is 16-bit (units of
|
||||||
|
512 bytes), so it can name at most 32 MiB — less than the 64 MiB FAT image.
|
||||||
|
That is fine in practice: firmware sizes the FAT filesystem from its own BPB,
|
||||||
|
and the boot files sit in the first few MiB of the image (clusters are
|
||||||
|
allocated from the front) either way. The USB path has no such cap.
|
||||||
|
|
||||||
|
## The builder
|
||||||
|
|
||||||
|
`tools/make-iso-image.py` follows the house rule of
|
||||||
|
[make-fat-image.py](../tools/make-fat-image.py): pure Python 3 standard
|
||||||
|
library, no external tools (no xorriso, mkisofs, or isohybrid), with a
|
||||||
|
`--verify` mode the `check-iso-image` step runs — it checks that the MBR
|
||||||
|
partition and the El Torito catalog agree on where the FAT image lives and
|
||||||
|
that a FAT32 boot sector is actually there. Every timestamp field in the ISO
|
||||||
|
is zeroed, so the build is reproducible byte-for-byte.
|
||||||
|
|
||||||
|
The ISO9660 filesystem around the boot machinery is minimal but real: a root
|
||||||
|
directory listing `BOOT.CAT` (the catalog) and `EFI.IMG` (the FAT image), so
|
||||||
|
`file`, mount tools, and archive browsers can open the ISO and see what's in
|
||||||
|
it.
|
||||||
+206
@@ -0,0 +1,206 @@
|
|||||||
|
# The vDSO — the public system-call boundary
|
||||||
|
|
||||||
|
> **Status:** design note, not built. The runtime today issues raw `syscall`
|
||||||
|
> instructions from `library/runtime/system-call.zig` using the numbers in
|
||||||
|
> `system/abi.zig`. This note designs the layer that replaces that arrangement:
|
||||||
|
> a **kernel-supplied, C-ABI entry library** mapped into every process — the
|
||||||
|
> only supported way into the kernel — so the raw numbers can stay private,
|
||||||
|
> be renumbered at will, and eventually be randomised per boot.
|
||||||
|
|
||||||
|
## Why: the ABI danos promises, and the one it doesn't
|
||||||
|
|
||||||
|
`system/abi.zig` is the **private** kernel ↔ runtime contract. Its header says
|
||||||
|
so: the numbers are an implementation detail the runtime hides and may
|
||||||
|
renumber, the same split as libSystem over the XNU syscalls on macOS or win32
|
||||||
|
over the NT syscalls on Windows. Linux — with its world-visible, frozen
|
||||||
|
syscall table — is the outlier, not the norm.
|
||||||
|
|
||||||
|
That stance has consequences the moment binaries exist that we don't rebuild
|
||||||
|
ourselves:
|
||||||
|
|
||||||
|
1. **Third-party binaries** (docs/zig-self-hosting.md) must keep working across
|
||||||
|
kernel updates. If they contain raw `syscall` instructions with today's
|
||||||
|
numbers baked in, every renumbering breaks the world — the ABI would be
|
||||||
|
*de facto* public no matter what the header says. Go on macOS made exactly
|
||||||
|
this mistake: it issued XNU syscalls directly instead of going through
|
||||||
|
libSystem, and macOS updates repeatedly broke every Go binary until Go
|
||||||
|
switched to the library like everyone else.
|
||||||
|
2. **Not everything is Zig.** A Rust or C program can't import the `runtime`
|
||||||
|
module. The public boundary has to be expressible in the one calling
|
||||||
|
convention every language speaks: the C ABI.
|
||||||
|
3. **Randomised syscall numbers** — a hardening option we want open — only
|
||||||
|
work if no user binary anywhere knows a number at build time. The binding
|
||||||
|
must happen at *load time*, from something the kernel controls.
|
||||||
|
|
||||||
|
All three point at the same well-known shape: a **vDSO** (virtual dynamic
|
||||||
|
shared object). The kernel carries a small blob of user-mode code, maps it
|
||||||
|
into every process at spawn, and that blob — not the application — contains
|
||||||
|
the `syscall` instructions. Fuchsia works exactly this way: its vDSO is the
|
||||||
|
*only* kernel entry, version-matched by construction because the kernel itself
|
||||||
|
injects it. Because the kernel and the blob ship as one artifact, there is
|
||||||
|
**no version skew, no loader, no search path, and no shared file on disk** —
|
||||||
|
which is what makes this the resilient way to have a private ABI
|
||||||
|
(docs/resilience.md), where a conventional `ld.so` + `/lib/libdanos.so`
|
||||||
|
arrangement would add a loader to every spawn and a single shared point of
|
||||||
|
failure.
|
||||||
|
|
||||||
|
The public danos ABI then has exactly two layers, neither of which is
|
||||||
|
`abi.zig`:
|
||||||
|
|
||||||
|
| Layer | Contract | Spoken by |
|
||||||
|
|-------|----------|-----------|
|
||||||
|
| **vDSO** | C-ABI functions, this note | every language's thin shim (`runtime.system` for Zig, a `-sys` crate for Rust, a header for C) |
|
||||||
|
| **IPC wire protocols** | byte layouts over `ipc_call` ([vfs-protocol.md](vfs-protocol.md) is the first one documented) | any client that can lay out bytes |
|
||||||
|
|
||||||
|
Everything above those — the heap, `runtime.fs`, the service harness — is
|
||||||
|
per-language convenience, compiled into each binary from source, exactly as
|
||||||
|
today. Nothing about the Zig runtime's shape changes; it just stops being the
|
||||||
|
*only* door.
|
||||||
|
|
||||||
|
## The blob
|
||||||
|
|
||||||
|
A single copy of the vDSO code lives in the kernel image (built by
|
||||||
|
`build.zig` as a tiny freestanding object, embedded like the AP trampoline).
|
||||||
|
At boot the kernel finalises it once — this is where randomised numbers would
|
||||||
|
be patched in — and thereafter maps the **same physical pages** read-execute
|
||||||
|
into every process's address space. The blob is:
|
||||||
|
|
||||||
|
- **Position-independent.** It is mapped at a per-process randomised base, so
|
||||||
|
it must be PIC (rip-relative addressing only — no relocations to process).
|
||||||
|
- **Stateless and re-entrant.** No writable data. Anything stateful belongs to
|
||||||
|
the process, not the vDSO.
|
||||||
|
- **Architecture-specific.** The x86-64 blob wraps `syscall`; an aarch64 blob
|
||||||
|
wraps `svc #0`. It lives beside the other per-architecture kernel sources
|
||||||
|
(`system/kernel/architecture/<arch>/`), selected the same way the
|
||||||
|
`architecture` module is (docs/arch.md).
|
||||||
|
|
||||||
|
### Shape: a function table, not an ELF
|
||||||
|
|
||||||
|
A real `.so` with a dynamic symbol table is the conventional vDSO shape, but
|
||||||
|
linking against one at load time needs a dynamic linker in every binary —
|
||||||
|
machinery danos deliberately doesn't have. Instead the v1 shape is the
|
||||||
|
simplest thing that is still a stable contract — a **function-pointer table**
|
||||||
|
at the vDSO base:
|
||||||
|
|
||||||
|
```
|
||||||
|
offset 0 u64 magic 'danosVDS' — a mapped-the-wrong-thing guard
|
||||||
|
offset 8 u64 api_level incremented when the table grows
|
||||||
|
offset 16 u64 count number of table entries that follow
|
||||||
|
offset 24 u64 table[count] function pointers into the vDSO's own code
|
||||||
|
```
|
||||||
|
|
||||||
|
Table *indices* are the public constants (published in a C header,
|
||||||
|
`danos.h`), assigned once and append-only — the same discipline the IPC
|
||||||
|
protocols use for operation values. The pointers point at stubs inside the
|
||||||
|
blob; what those stubs put in `rax` is nobody's business but the kernel's.
|
||||||
|
A language shim binds in one step: read the base from the init block, check
|
||||||
|
the magic, keep the table pointer. Feature detection for a binary built
|
||||||
|
against older headers is `count`/`api_level` — a kernel never removes or
|
||||||
|
reorders entries.
|
||||||
|
|
||||||
|
(If danos ever grows a real dynamic linker, the same blob can additionally
|
||||||
|
present an ELF `dynsym` without breaking the table — Fuchsia's vDSO is
|
||||||
|
likewise both a mappable blob and a linkable `.so`. That is a later
|
||||||
|
convenience, not a requirement.)
|
||||||
|
|
||||||
|
### Delivery: the auxiliary vector
|
||||||
|
|
||||||
|
The kernel already builds a System V entry block — argc, argv, envp
|
||||||
|
terminator, **auxiliary vector** — on every new process's stack
|
||||||
|
(`buildEntryStack`, read by `runtime.start`). The vDSO base rides in a new
|
||||||
|
auxv entry, exactly Linux's `AT_SYSINFO_EHDR` move. No new syscall, no magic
|
||||||
|
address, and a language shim finds it the same portable way on every
|
||||||
|
architecture.
|
||||||
|
|
||||||
|
## The function surface
|
||||||
|
|
||||||
|
One table entry per kernel call, C ABI (System V AMD64), names prefixed
|
||||||
|
`danos_`. The current `SystemCall` set maps directly; integer arguments and
|
||||||
|
returns are `u64`, errors return as negative values exactly as today.
|
||||||
|
|
||||||
|
The calls that return two values in `rax:rdx` today — `dma_alloc`
|
||||||
|
(vaddr + paddr), `msi_bind` (address + data), `shm_create` (vaddr + handle) —
|
||||||
|
become functions returning a two-`u64` struct. The System V ABI returns a
|
||||||
|
16-byte struct in `rax:rdx`, so the stub is a plain `syscall; ret` — the
|
||||||
|
C-ABI spelling of the existing convention, at zero cost.
|
||||||
|
|
||||||
|
Grouped as `abi.zig` groups them:
|
||||||
|
|
||||||
|
| Group | Functions |
|
||||||
|
|-------|-----------|
|
||||||
|
| process | `danos_exit`, `danos_yield`, `danos_sleep`, `danos_spawn`, `danos_process_enumerate`, `danos_process_kill`, `danos_process_exit_reason`, `danos_process_subscribe`, `danos_process_signal`, `danos_signal_bind` |
|
||||||
|
| memory | `danos_mmap`, `danos_munmap`, `danos_dma_alloc`, `danos_dma_free`, `danos_shm_create`, `danos_shm_map`, `danos_shm_physical` |
|
||||||
|
| ipc | `danos_endpoint_create`, `danos_ipc_register`, `danos_ipc_lookup`, `danos_ipc_call`, `danos_ipc_reply_wait`, `danos_ipc_send` |
|
||||||
|
| devices | `danos_device_enumerate`, `danos_device_claim`, `danos_device_register`, `danos_mmio_map`, `danos_irq_bind`, `danos_irq_ack`, `danos_msi_bind`, `danos_io_read`, `danos_io_write` |
|
||||||
|
| time | `danos_clock`, `danos_wall_clock`, `danos_timer_bind` |
|
||||||
|
| diagnostics | `danos_debug_write`, `danos_klog_read` |
|
||||||
|
|
||||||
|
The constants that ride alongside the calls — mmap protection bits, DMA
|
||||||
|
flags, notification badge bits, `ExitReason`, `Signal`, well-known service
|
||||||
|
ids, `page_size`, the IPC message maximum — move to the public header too:
|
||||||
|
they are wire values a Rust program needs verbatim. What stays private in
|
||||||
|
`abi.zig` is exactly the thing the vDSO exists to hide: the `SystemCall`
|
||||||
|
numbers and the trap convention.
|
||||||
|
|
||||||
|
## Enforcement, and an honest threat model
|
||||||
|
|
||||||
|
Renumbering only has teeth if the kernel **refuses syscalls that don't come
|
||||||
|
from the vDSO**. The check is cheap: on kernel entry, the saved user `rip`
|
||||||
|
must lie inside the calling process's vDSO mapping; otherwise the process is
|
||||||
|
killed with a fault-class exit reason (its supervisor restarts or gives up,
|
||||||
|
docs/process-lifecycle.md — a foreign-syscall attempt is a bug or an attack,
|
||||||
|
never something to limp past). Fuchsia enforces exactly this.
|
||||||
|
|
||||||
|
What this buys, precisely:
|
||||||
|
|
||||||
|
- **ABI freedom** — the real prize. The numbers can change per release or per
|
||||||
|
boot and nothing outside the kernel image cares. The private ABI stays
|
||||||
|
actually private, permanently.
|
||||||
|
- **A single audited chokepoint** for kernel entry, per process, at a
|
||||||
|
randomised address.
|
||||||
|
- **Raised bar for exploits**: shellcode can't issue a hard-coded `syscall`;
|
||||||
|
it must first discover the per-process vDSO base (ASLR) and call through
|
||||||
|
it.
|
||||||
|
|
||||||
|
What it does *not* buy: an attacker with arbitrary code execution in a
|
||||||
|
process can still *call* the vDSO functions — they are mapped executable in
|
||||||
|
that process, and return-oriented chains reach them. Syscall randomisation is
|
||||||
|
hardening, not a security boundary; the security boundary remains the
|
||||||
|
capability model (what the process's endpoints and device claims let it do).
|
||||||
|
It is worth building anyway — for the ABI freedom first and the hardening
|
||||||
|
second — but the design should never be sold as more than that.
|
||||||
|
|
||||||
|
## Migration
|
||||||
|
|
||||||
|
Phased so every step ships alone (the M-milestone discipline):
|
||||||
|
|
||||||
|
1. **The blob + the table.** Build the vDSO, map it at spawn, deliver the
|
||||||
|
base via auxv. `runtime.system-call.zig` binds through the table when the
|
||||||
|
auxv entry is present, falls back to raw `syscall` when absent — the whole
|
||||||
|
tree keeps booting during the transition.
|
||||||
|
2. **Cut the runtime over.** Delete the raw stubs; `runtime` no longer
|
||||||
|
imports the `SystemCall` numbers at all (`abi.zig`'s enum becomes
|
||||||
|
kernel-internal). The QEMU suite passing proves the table carries the
|
||||||
|
whole system.
|
||||||
|
3. **Enforce + randomise.** Add the `rip`-range check, then per-boot number
|
||||||
|
randomisation patched into the blob at kernel init. A test boots with
|
||||||
|
randomisation on and runs the full suite.
|
||||||
|
4. **The other languages.** Publish `danos.h`; a Rust `danos-sys` crate wraps
|
||||||
|
the table. This is also the seam `std.os.danos` calls through when the Zig
|
||||||
|
self-hosting fork lands (docs/zig-self-hosting.md) — the vDSO is what
|
||||||
|
makes that seam stable across kernel versions.
|
||||||
|
|
||||||
|
## What deliberately stays out
|
||||||
|
|
||||||
|
- **No dynamic linker, no `/lib/*.so`.** The vDSO is kernel-injected precisely
|
||||||
|
so danos binaries can stay fully static above it. Sharing *library code*
|
||||||
|
across processes stays what it is today: a service behind IPC, or source
|
||||||
|
compiled into each binary.
|
||||||
|
- **No file/device I/O in the vDSO.** The microkernel line doesn't move: the
|
||||||
|
vDSO wraps the same deliberately tiny table (docs/syscall.md); files are
|
||||||
|
still the VFS server's business over IPC.
|
||||||
|
- **No fast-path user-mode implementations yet.** Linux's vDSO exists mostly
|
||||||
|
to answer `gettimeofday` without a kernel entry. `danos_clock` could one
|
||||||
|
day read the calibrated TSC in user mode the same way — the blob is where
|
||||||
|
such an optimisation would live — but that is an optimisation, not part of
|
||||||
|
this design's contract.
|
||||||
@@ -0,0 +1,170 @@
|
|||||||
|
# The VFS wire protocol
|
||||||
|
|
||||||
|
> **Status:** built and spoken today between `runtime.fs` (the client) and the
|
||||||
|
> VFS server (`system/services/vfs`), with mounted backends (the FAT server)
|
||||||
|
> speaking the same protocol behind the router. The Zig source of truth is
|
||||||
|
> `system/services/vfs/protocol.zig` (the `vfs-protocol` module), whose unit
|
||||||
|
> tests pin the sizes and values below. This page is the **language-neutral
|
||||||
|
> wire specification** of that contract — what a Rust or C client implements
|
||||||
|
> ([vdso.md](vdso.md) explains why the IPC protocols, not the syscall
|
||||||
|
> numbers, are danos's public ABI).
|
||||||
|
|
||||||
|
## Transport
|
||||||
|
|
||||||
|
A VFS exchange is one synchronous IPC rendezvous (`ipc_call`,
|
||||||
|
docs/ipc.md): the client sends one message and blocks; the server replies
|
||||||
|
with one message. The endpoint is found by well-known service id
|
||||||
|
(`ipc_lookup`, service id **1** = vfs).
|
||||||
|
|
||||||
|
- A message is at most **256 bytes** (`message_maximum`).
|
||||||
|
- A request is a fixed 32-byte **Request** header followed by an inline
|
||||||
|
payload of at most **224 bytes** (`maximum_payload`) — a path, or write
|
||||||
|
bytes. There is no multi-message request: paths and single reads/writes
|
||||||
|
must fit, and larger transfers loop (see *read* / *write*).
|
||||||
|
- A reply is a fixed 24-byte **Reply** header followed by an inline payload —
|
||||||
|
read bytes, a `FileStatus`, or a `DirectoryEntry`.
|
||||||
|
- All integers are **little-endian**; layouts are C layout for x86-64
|
||||||
|
(`extern struct`), offsets given below so nothing need be inferred.
|
||||||
|
|
||||||
|
The kernel never parses any of this — it only moves the bytes
|
||||||
|
(docs/syscall.md); files are entirely a user-space affair.
|
||||||
|
|
||||||
|
## Request header — 32 bytes
|
||||||
|
|
||||||
|
| offset | size | field | meaning |
|
||||||
|
|-------:|-----:|-------|---------|
|
||||||
|
| 0 | 4 | `operation` | an **Operation** value (below) |
|
||||||
|
| 4 | 4 | — | padding |
|
||||||
|
| 8 | 8 | `node` | the server-side open-node id from a prior `open`; 0 for path-based operations |
|
||||||
|
| 16 | 8 | `offset` | byte position for read/write; entry index (cursor) for readdir; else 0 |
|
||||||
|
| 24 | 4 | `len` | payload length for path/write operations; requested byte count for read |
|
||||||
|
| 28 | 4 | `flags` | open flags (below); else 0 |
|
||||||
|
|
||||||
|
## Reply header — 24 bytes
|
||||||
|
|
||||||
|
| offset | size | field | meaning |
|
||||||
|
|-------:|-----:|-------|---------|
|
||||||
|
| 0 | 4 | `status` | **0 = success**, negative = failure (signed) |
|
||||||
|
| 4 | 4 | — | padding |
|
||||||
|
| 8 | 8 | `node` | the new open-node id (for `open`); else 0 |
|
||||||
|
| 16 | 4 | `len` | reply payload length in bytes |
|
||||||
|
| 20 | 4 | — | padding |
|
||||||
|
|
||||||
|
On failure the router replies `status = -1`; a mounted backend's negative
|
||||||
|
status is forwarded to the client verbatim. A richer errno vocabulary is
|
||||||
|
future work — clients must treat *any* negative status as failure, not match
|
||||||
|
on -1.
|
||||||
|
|
||||||
|
## Operations
|
||||||
|
|
||||||
|
Values are append-only and never renumbered (the same evolution rule every
|
||||||
|
danos protocol follows); an unrecognised operation gets a `status = -1`
|
||||||
|
reply.
|
||||||
|
|
||||||
|
| value | operation | request payload | reply |
|
||||||
|
|------:|-----------|-----------------|-------|
|
||||||
|
| 0 | `open` | the path (`len` = its length), `flags` as below | `node` = open-node id |
|
||||||
|
| 1 | `close` | — (`node` set) | status only |
|
||||||
|
| 2 | `read` | — (`node`, `offset`, `len` = wanted count) | `len` bytes read, payload = the bytes; `len` 0 at end of file |
|
||||||
|
| 3 | `write` | the bytes (`node`, `offset`, `len` = count) | `len` = bytes accepted (may be short — loop) |
|
||||||
|
| 4 | `status` | — (`node` set) | payload = **FileStatus** (24 bytes) |
|
||||||
|
| 5 | `readdir` | — (`node` = a directory, `offset` = cursor) | payload = one **DirectoryEntry** + name; `len` 0 at end |
|
||||||
|
| 6 | `mount` | the mount-point path; the backend endpoint rides as the call's **capability** | status only |
|
||||||
|
| 7 | `unmount` | the mount-point path | status only |
|
||||||
|
| 8 | `mkdir` | the path | status only |
|
||||||
|
| 9 | `unlink` | the path | status only |
|
||||||
|
| 10 | `rename` | old path, one `0x00`, new path (`len` = total) | status only |
|
||||||
|
|
||||||
|
Notes per operation:
|
||||||
|
|
||||||
|
- **open** — paths are absolute (`/mnt/usb/notes.txt`) or bare names
|
||||||
|
(`greeting`); bare names resolve in the VFS's flat ramfs, absolute paths
|
||||||
|
route through the mount table (below). The returned `node` is an id in the
|
||||||
|
*router's* open table; clients never see a backend's own ids.
|
||||||
|
- **read / write** — a single exchange moves at most 224 bytes
|
||||||
|
(`maximum_payload`); the client loops, advancing `offset` by the returned
|
||||||
|
`len`, until done (read) or the slice is written (write). A `write` reply
|
||||||
|
shorter than requested is progress, not an error; a `len` of 0 means no
|
||||||
|
forward progress — stop rather than spin.
|
||||||
|
- **readdir** — `offset` is a **cursor: the entry index**, not a byte
|
||||||
|
position. Each call returns exactly one entry; the client increments the
|
||||||
|
cursor by 1. A reply with `len` 0 is end-of-directory. The directory must
|
||||||
|
have been opened with the `directory` flag.
|
||||||
|
- **mount** — the one operation that passes a **capability**: the caller
|
||||||
|
(a filesystem server, e.g. FAT) sends its own request endpoint as the
|
||||||
|
`ipc_call` capability argument, and the router forwards everything under
|
||||||
|
the mount point to it — speaking this same protocol, with paths rewritten
|
||||||
|
relative to the mount. Prefixes match at path boundaries only
|
||||||
|
(`/mnt/usb` never captures `/mnt/usbextra`); the longest matching prefix
|
||||||
|
wins.
|
||||||
|
- **rename** — same-directory rename only (the router requires old and new to
|
||||||
|
resolve under one mount).
|
||||||
|
|
||||||
|
## Open flags
|
||||||
|
|
||||||
|
Bitwise OR in `Request.flags`, meaningful for `open` only:
|
||||||
|
|
||||||
|
| bit | name | meaning |
|
||||||
|
|----:|------|---------|
|
||||||
|
| 1 | `create` | create the file if it does not exist |
|
||||||
|
| 2 | `directory` | open a directory node for `readdir` rather than a file |
|
||||||
|
| 4 | `truncate` | truncate an existing file to zero length on open (replace, don't overwrite in place) |
|
||||||
|
|
||||||
|
## FileStatus — 24 bytes (the `status` reply payload)
|
||||||
|
|
||||||
|
| offset | size | field | meaning |
|
||||||
|
|-------:|-----:|-------|---------|
|
||||||
|
| 0 | 8 | `size` | file size in bytes |
|
||||||
|
| 8 | 4 | `kind` | a **NodeKind** value |
|
||||||
|
| 12 | 4 | — | padding |
|
||||||
|
| 16 | 8 | `mtime` | modification time, Unix epoch seconds UTC; 0 if the backend keeps none |
|
||||||
|
|
||||||
|
## DirectoryEntry — 16 bytes + name (the `readdir` reply payload)
|
||||||
|
|
||||||
|
| offset | size | field | meaning |
|
||||||
|
|-------:|-----:|-------|---------|
|
||||||
|
| 0 | 4 | `kind` | a **NodeKind** value |
|
||||||
|
| 4 | 4 | `name_len` | length of the name that follows |
|
||||||
|
| 8 | 8 | `size` | the entry's size in bytes |
|
||||||
|
| 16 | `name_len` | name | the entry's name, not NUL-terminated |
|
||||||
|
|
||||||
|
## NodeKind
|
||||||
|
|
||||||
|
Aligned to the FSH file-type table
|
||||||
|
(docs/danos-file-system-hierarchy-FSH.md):
|
||||||
|
|
||||||
|
| value | kind |
|
||||||
|
|------:|------|
|
||||||
|
| 0 | regular file |
|
||||||
|
| 1 | directory |
|
||||||
|
| 2 | character device |
|
||||||
|
| 3 | block device |
|
||||||
|
| 4 | symbolic link |
|
||||||
|
| 5 | fifo |
|
||||||
|
| 6 | socket |
|
||||||
|
|
||||||
|
Clients should map unknown values to *regular* rather than reject — the
|
||||||
|
table can grow.
|
||||||
|
|
||||||
|
## Lifetimes and trust
|
||||||
|
|
||||||
|
Open-node ids live in the server. A client that dies without closing leaks
|
||||||
|
nothing permanently: the VFS subscribes to the kernel's published process-exit
|
||||||
|
events (docs/process-lifecycle.md) and releases a dead client's handles,
|
||||||
|
closing forwarded backend nodes best-effort. Ids are plain integers, not
|
||||||
|
capabilities — the VFS trusts its callers with each other's ids today, which
|
||||||
|
is acceptable while every client is part of the system image and worth
|
||||||
|
revisiting (per-client id namespaces) before third-party binaries arrive.
|
||||||
|
|
||||||
|
## Evolution rules
|
||||||
|
|
||||||
|
What a non-Zig implementation may rely on, and what it must not:
|
||||||
|
|
||||||
|
- Operation values, flag bits, `NodeKind` values, and struct layouts are
|
||||||
|
**append-only and frozen once shipped** — the unit tests in `protocol.zig`
|
||||||
|
pin them exactly so a refactor can't silently move them.
|
||||||
|
- The 256-byte message ceiling is a property of the current IPC transport,
|
||||||
|
not a promise; clients should read `maximum_payload`-shaped limits from the
|
||||||
|
reply lengths they actually get (loop-until-done), not hard-code 224.
|
||||||
|
- Negative statuses beyond -1 will appear (an errno vocabulary); success is
|
||||||
|
exactly 0.
|
||||||
@@ -0,0 +1,261 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""Wrap the FAT32 boot volume in a hybrid ISO — the flashable danos release image.
|
||||||
|
|
||||||
|
Mirrors tools/make-fat-image.py in spirit: pure Python 3 standard library, no
|
||||||
|
external tools (no xorriso / mkisofs / isohybrid). The output is one file that
|
||||||
|
boots both ways release media is consumed:
|
||||||
|
|
||||||
|
* Flashed raw to a USB stick (Etcher, dd): the ISO's system area carries an
|
||||||
|
MBR whose single partition (type 0xEF, "EFI System") points at the FAT32
|
||||||
|
image embedded in the ISO, so UEFI firmware finds the ESP and runs
|
||||||
|
\\EFI\\BOOT\\BOOTX64.efi off it.
|
||||||
|
* Burned to optical media: an El Torito boot catalog with an EFI platform
|
||||||
|
entry points at the same embedded FAT image.
|
||||||
|
|
||||||
|
The ISO9660 filesystem itself is minimal but valid — a primary volume
|
||||||
|
descriptor, the El Torito boot record, path tables, and a root directory that
|
||||||
|
lists the boot catalog and the FAT image — so inspection tools can open it.
|
||||||
|
|
||||||
|
make-iso-image.py <out.iso> <esp.img>
|
||||||
|
make-iso-image.py --verify <out.iso>
|
||||||
|
|
||||||
|
All timestamp fields are zero ("not specified") so the build is reproducible.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import struct
|
||||||
|
import sys
|
||||||
|
|
||||||
|
ISO_SECTOR = 2048
|
||||||
|
|
||||||
|
# Fixed layout, in ISO sectors (LBA). Sectors 0-15 are the system area (the
|
||||||
|
# hybrid MBR lives in its first 512 bytes); volume descriptors start at 16.
|
||||||
|
PVD_LBA = 16 # primary volume descriptor
|
||||||
|
BOOT_RECORD_LBA = 17 # El Torito boot record volume descriptor
|
||||||
|
TERMINATOR_LBA = 18 # volume descriptor set terminator
|
||||||
|
PATH_TABLE_L_LBA = 19
|
||||||
|
PATH_TABLE_M_LBA = 20
|
||||||
|
ROOT_DIR_LBA = 21 # root directory (one sector holds our four records)
|
||||||
|
CATALOG_LBA = 22 # El Torito boot catalog
|
||||||
|
ESP_LBA = 23 # the embedded FAT32 image starts here
|
||||||
|
|
||||||
|
MBR_PARTITION_TYPE_ESP = 0xEF
|
||||||
|
|
||||||
|
|
||||||
|
def both16(value):
|
||||||
|
"""ISO9660 both-byte-order encoding: little-endian then big-endian."""
|
||||||
|
return struct.pack("<H", value) + struct.pack(">H", value)
|
||||||
|
|
||||||
|
|
||||||
|
def both32(value):
|
||||||
|
return struct.pack("<I", value) + struct.pack(">I", value)
|
||||||
|
|
||||||
|
|
||||||
|
def directory_record(identifier, lba, size, flags):
|
||||||
|
length = 33 + len(identifier)
|
||||||
|
if length % 2:
|
||||||
|
length += 1 # records are padded to even length
|
||||||
|
record = bytearray(length)
|
||||||
|
record[0] = length
|
||||||
|
record[2:10] = both32(lba)
|
||||||
|
record[10:18] = both32(size)
|
||||||
|
# record[18:25] is the recording date; zero = unspecified (reproducible).
|
||||||
|
record[25] = flags # 0x02 = directory
|
||||||
|
record[28:32] = both16(1) # volume sequence number
|
||||||
|
record[32] = len(identifier)
|
||||||
|
record[33:33 + len(identifier)] = identifier
|
||||||
|
return bytes(record)
|
||||||
|
|
||||||
|
|
||||||
|
def primary_volume_descriptor(total_sectors, path_table_size):
|
||||||
|
sector = bytearray(ISO_SECTOR)
|
||||||
|
sector[0] = 1 # type: primary
|
||||||
|
sector[1:6] = b"CD001"
|
||||||
|
sector[6] = 1 # version
|
||||||
|
sector[8:40] = b"DANOS".ljust(32) # system identifier
|
||||||
|
sector[40:72] = b"DANOS".ljust(32) # volume identifier
|
||||||
|
sector[80:88] = both32(total_sectors)
|
||||||
|
sector[120:124] = both16(1) # volume set size
|
||||||
|
sector[124:128] = both16(1) # volume sequence number
|
||||||
|
sector[128:132] = both16(ISO_SECTOR) # logical block size
|
||||||
|
sector[132:140] = both32(path_table_size)
|
||||||
|
sector[140:144] = struct.pack("<I", PATH_TABLE_L_LBA)
|
||||||
|
sector[148:152] = struct.pack(">I", PATH_TABLE_M_LBA)
|
||||||
|
sector[156:190] = directory_record(b"\x00", ROOT_DIR_LBA, ISO_SECTOR, 0x02)
|
||||||
|
sector[190:318] = b" " * 128 # volume set identifier
|
||||||
|
sector[318:446] = b" " * 128 # publisher
|
||||||
|
sector[446:574] = b" " * 128 # data preparer
|
||||||
|
sector[574:702] = b"DANOS MAKE-ISO-IMAGE".ljust(128) # application
|
||||||
|
sector[702:739] = b" " * 37 # copyright file
|
||||||
|
sector[739:776] = b" " * 37 # abstract file
|
||||||
|
sector[776:813] = b" " * 37 # bibliographic file
|
||||||
|
unspecified_date = b"0" * 16 + b"\x00"
|
||||||
|
for offset in (813, 830, 847, 864): # creation/modification/expiry/effective
|
||||||
|
sector[offset:offset + 17] = unspecified_date
|
||||||
|
sector[881] = 1 # file structure version
|
||||||
|
return bytes(sector)
|
||||||
|
|
||||||
|
|
||||||
|
def boot_record_descriptor():
|
||||||
|
sector = bytearray(ISO_SECTOR)
|
||||||
|
sector[0] = 0 # type: boot record
|
||||||
|
sector[1:6] = b"CD001"
|
||||||
|
sector[6] = 1
|
||||||
|
sector[7:39] = b"EL TORITO SPECIFICATION".ljust(32, b"\x00")
|
||||||
|
sector[71:75] = struct.pack("<I", CATALOG_LBA)
|
||||||
|
return bytes(sector)
|
||||||
|
|
||||||
|
|
||||||
|
def terminator_descriptor():
|
||||||
|
sector = bytearray(ISO_SECTOR)
|
||||||
|
sector[0] = 255
|
||||||
|
sector[1:6] = b"CD001"
|
||||||
|
sector[6] = 1
|
||||||
|
return bytes(sector)
|
||||||
|
|
||||||
|
|
||||||
|
def path_table(byte_order):
|
||||||
|
# A single entry: the root directory.
|
||||||
|
return (struct.pack("BB", 1, 0)
|
||||||
|
+ struct.pack(byte_order + "I", ROOT_DIR_LBA)
|
||||||
|
+ struct.pack(byte_order + "H", 1)
|
||||||
|
+ b"\x00\x00") # identifier 0x00 + pad to even
|
||||||
|
|
||||||
|
|
||||||
|
def root_directory(esp_size):
|
||||||
|
# Records must be sorted by identifier; BOOT.CAT < EFI.IMG holds.
|
||||||
|
entries = (directory_record(b"\x00", ROOT_DIR_LBA, ISO_SECTOR, 0x02)
|
||||||
|
+ directory_record(b"\x01", ROOT_DIR_LBA, ISO_SECTOR, 0x02)
|
||||||
|
+ directory_record(b"BOOT.CAT;1", CATALOG_LBA, ISO_SECTOR, 0)
|
||||||
|
+ directory_record(b"EFI.IMG;1", ESP_LBA, esp_size, 0))
|
||||||
|
return entries + b"\x00" * (ISO_SECTOR - len(entries))
|
||||||
|
|
||||||
|
|
||||||
|
def boot_catalog(esp_size):
|
||||||
|
# Validation entry: EFI platform (0xEF), checksummed so its 16-bit words sum
|
||||||
|
# to zero, closed by the 0x55AA key bytes.
|
||||||
|
validation = bytearray(32)
|
||||||
|
validation[0] = 0x01
|
||||||
|
validation[1] = 0xEF
|
||||||
|
validation[4:28] = b"danos".ljust(24, b"\x00")
|
||||||
|
validation[30] = 0x55
|
||||||
|
validation[31] = 0xAA
|
||||||
|
checksum = (-sum(struct.unpack("<16H", validation))) & 0xFFFF
|
||||||
|
validation[28:30] = struct.pack("<H", checksum)
|
||||||
|
# Initial/default entry: bootable, no emulation, image at ESP_LBA. The
|
||||||
|
# sector-count field is 16-bit (units of 512 bytes) so it can't span a large
|
||||||
|
# ESP; UEFI firmware sizes the FAT filesystem from its own BPB, and the
|
||||||
|
# image's boot files sit well inside the capped span regardless.
|
||||||
|
default = bytearray(32)
|
||||||
|
default[0] = 0x88 # bootable
|
||||||
|
default[1] = 0x00 # no emulation
|
||||||
|
sector_count = min(0xFFFF, esp_size // 512)
|
||||||
|
default[6:8] = struct.pack("<H", sector_count)
|
||||||
|
default[8:12] = struct.pack("<I", ESP_LBA)
|
||||||
|
catalog = bytes(validation) + bytes(default)
|
||||||
|
return catalog + b"\x00" * (ISO_SECTOR - len(catalog))
|
||||||
|
|
||||||
|
|
||||||
|
def hybrid_mbr(esp_size):
|
||||||
|
"""The system-area MBR that makes the ISO flashable: one ESP partition."""
|
||||||
|
mbr = bytearray(512)
|
||||||
|
mbr[440:444] = b"dano" # disk signature (fixed: reproducible builds)
|
||||||
|
start_lba = ESP_LBA * (ISO_SECTOR // 512)
|
||||||
|
partition = struct.pack(
|
||||||
|
"<B3sB3sII",
|
||||||
|
0x80, # status: active (harmless; helps picky firmware)
|
||||||
|
b"\xFE\xFF\xFF", # CHS start: maxed out, LBA is authoritative
|
||||||
|
MBR_PARTITION_TYPE_ESP, # type: EFI System
|
||||||
|
b"\xFE\xFF\xFF", # CHS end
|
||||||
|
start_lba,
|
||||||
|
esp_size // 512,
|
||||||
|
)
|
||||||
|
mbr[446:462] = partition
|
||||||
|
mbr[510] = 0x55
|
||||||
|
mbr[511] = 0xAA
|
||||||
|
return bytes(mbr)
|
||||||
|
|
||||||
|
|
||||||
|
def build(out_path, esp_path):
|
||||||
|
with open(esp_path, "rb") as handle:
|
||||||
|
esp = handle.read()
|
||||||
|
if len(esp) % ISO_SECTOR:
|
||||||
|
esp += b"\x00" * (ISO_SECTOR - len(esp) % ISO_SECTOR)
|
||||||
|
esp_sectors = len(esp) // ISO_SECTOR
|
||||||
|
total_sectors = ESP_LBA + esp_sectors
|
||||||
|
|
||||||
|
table_l = path_table("<")
|
||||||
|
image = bytearray(total_sectors * ISO_SECTOR)
|
||||||
|
image[0:512] = hybrid_mbr(len(esp))
|
||||||
|
image[PVD_LBA * ISO_SECTOR:(PVD_LBA + 1) * ISO_SECTOR] = \
|
||||||
|
primary_volume_descriptor(total_sectors, len(table_l))
|
||||||
|
image[BOOT_RECORD_LBA * ISO_SECTOR:(BOOT_RECORD_LBA + 1) * ISO_SECTOR] = \
|
||||||
|
boot_record_descriptor()
|
||||||
|
image[TERMINATOR_LBA * ISO_SECTOR:(TERMINATOR_LBA + 1) * ISO_SECTOR] = \
|
||||||
|
terminator_descriptor()
|
||||||
|
image[PATH_TABLE_L_LBA * ISO_SECTOR:PATH_TABLE_L_LBA * ISO_SECTOR + len(table_l)] = table_l
|
||||||
|
table_m = path_table(">")
|
||||||
|
image[PATH_TABLE_M_LBA * ISO_SECTOR:PATH_TABLE_M_LBA * ISO_SECTOR + len(table_m)] = table_m
|
||||||
|
image[ROOT_DIR_LBA * ISO_SECTOR:(ROOT_DIR_LBA + 1) * ISO_SECTOR] = root_directory(len(esp))
|
||||||
|
image[CATALOG_LBA * ISO_SECTOR:(CATALOG_LBA + 1) * ISO_SECTOR] = boot_catalog(len(esp))
|
||||||
|
image[ESP_LBA * ISO_SECTOR:] = esp
|
||||||
|
|
||||||
|
with open(out_path, "wb") as handle:
|
||||||
|
handle.write(image)
|
||||||
|
print(f"make-iso-image: wrote {out_path} "
|
||||||
|
f"({total_sectors * ISO_SECTOR // (1024 * 1024)} MiB hybrid ISO, "
|
||||||
|
f"ESP at LBA {ESP_LBA}, {esp_sectors} sectors)")
|
||||||
|
|
||||||
|
|
||||||
|
def verify(path):
|
||||||
|
with open(path, "rb") as handle:
|
||||||
|
data = handle.read()
|
||||||
|
# The hybrid MBR (the Etcher/dd boot path).
|
||||||
|
if data[510] != 0x55 or data[511] != 0xAA:
|
||||||
|
sys.exit("verify: missing MBR 0x55AA signature")
|
||||||
|
status, _, part_type, _, part_start, part_sectors = \
|
||||||
|
struct.unpack_from("<B3sB3sII", data, 446)
|
||||||
|
if part_type != MBR_PARTITION_TYPE_ESP:
|
||||||
|
sys.exit(f"verify: MBR partition type 0x{part_type:02X}, expected 0xEF (ESP)")
|
||||||
|
# The ISO9660 descriptors (the optical boot path).
|
||||||
|
if data[PVD_LBA * ISO_SECTOR + 1:PVD_LBA * ISO_SECTOR + 6] != b"CD001":
|
||||||
|
sys.exit("verify: no primary volume descriptor")
|
||||||
|
boot_record = data[BOOT_RECORD_LBA * ISO_SECTOR:(BOOT_RECORD_LBA + 1) * ISO_SECTOR]
|
||||||
|
if not boot_record.startswith(b"\x00CD001") or \
|
||||||
|
not boot_record[7:30].startswith(b"EL TORITO SPECIFICATION"):
|
||||||
|
sys.exit("verify: no El Torito boot record")
|
||||||
|
catalog_lba = struct.unpack_from("<I", boot_record, 71)[0]
|
||||||
|
catalog = data[catalog_lba * ISO_SECTOR:(catalog_lba + 1) * ISO_SECTOR]
|
||||||
|
if catalog[0] != 0x01 or catalog[1] != 0xEF or catalog[30:32] != b"\x55\xAA":
|
||||||
|
sys.exit("verify: boot catalog validation entry is not an EFI entry")
|
||||||
|
if sum(struct.unpack("<16H", catalog[0:32])) & 0xFFFF != 0:
|
||||||
|
sys.exit("verify: boot catalog validation checksum is wrong")
|
||||||
|
if catalog[32] != 0x88:
|
||||||
|
sys.exit("verify: default catalog entry is not bootable")
|
||||||
|
boot_lba = struct.unpack_from("<I", catalog, 40)[0]
|
||||||
|
# Both paths must agree on where the ESP lives, and it must be a FAT32 image.
|
||||||
|
if boot_lba * (ISO_SECTOR // 512) != part_start:
|
||||||
|
sys.exit(f"verify: catalog boot image (LBA {boot_lba}) and MBR partition "
|
||||||
|
f"(sector {part_start}) disagree")
|
||||||
|
esp = data[boot_lba * ISO_SECTOR:]
|
||||||
|
if len(esp) < part_sectors * 512:
|
||||||
|
sys.exit("verify: MBR partition extends past the end of the file")
|
||||||
|
if esp[510] != 0x55 or esp[511] != 0xAA or esp[82:90] != b"FAT32 ":
|
||||||
|
sys.exit("verify: embedded image is not a FAT32 boot volume")
|
||||||
|
print(f"verify: {path} is a hybrid ISO — MBR ESP partition (sector {part_start}, "
|
||||||
|
f"{part_sectors} sectors, active={status == 0x80}) and El Torito EFI entry "
|
||||||
|
f"both point at the embedded FAT32 image")
|
||||||
|
|
||||||
|
|
||||||
|
def main(argv):
|
||||||
|
if len(argv) == 3 and argv[1] == "--verify":
|
||||||
|
verify(argv[2])
|
||||||
|
return 0
|
||||||
|
if len(argv) != 3:
|
||||||
|
sys.exit("usage: make-iso-image.py <out.iso> <esp.img>\n"
|
||||||
|
" make-iso-image.py --verify <out.iso>")
|
||||||
|
build(argv[1], argv[2])
|
||||||
|
return 0
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
sys.exit(main(sys.argv))
|
||||||
Reference in New Issue
Block a user