Compare commits
42
Commits
68e65803eb
...
main
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
4037e746aa | ||
|
|
4dfb5012c0 | ||
|
|
061eb7c004 | ||
|
|
5dc966838a | ||
|
|
700452dc4e | ||
|
|
0faa0fd21b | ||
|
|
c47215821c | ||
|
|
6ddb08091d | ||
|
|
77b64229c2 | ||
|
|
90906bcefe | ||
|
|
2c2745e9e5 | ||
|
|
2a6d604577 | ||
|
|
e81a4e6f1d | ||
|
|
e240341bfb | ||
|
|
62eb2a748a | ||
|
|
56bd2e7678 | ||
|
|
0b25cd2c94 | ||
|
|
d59279422e | ||
|
|
bf9f8560c6 | ||
|
|
da7dcce64e | ||
|
|
9750db14da | ||
|
|
b2a5a0a3c6 | ||
|
|
7efe7b72d8 | ||
|
|
d4b544d66b | ||
|
|
6d4992ae02 | ||
|
|
df61693065 | ||
|
|
f1e79d0eeb | ||
|
|
167e9c7a9e | ||
|
|
5bfdb75e12 | ||
|
|
3fb8a9b96f | ||
|
|
ea8ccf65d0 | ||
|
|
aca3d5855a | ||
|
|
5637d0e5fc | ||
|
|
48ab12e262 | ||
|
|
020e31bc8f | ||
|
|
c81120ef0f | ||
|
|
4f9196c03e | ||
|
|
eff95416d0 | ||
|
|
addd264880 | ||
|
|
fca41b351e | ||
|
|
bf0595763e | ||
|
|
8216be991d |
@@ -122,6 +122,7 @@ fn driverArtifact(comptime package: []const u8, comptime artifact: []const u8) S
|
||||
/// (docs/build-packages-plan.md).
|
||||
const production_ship = [_]ShipRow{
|
||||
service("fat"),
|
||||
service("exfat"),
|
||||
service("display"),
|
||||
service("display-demo"),
|
||||
service("device-manager"),
|
||||
@@ -317,6 +318,11 @@ pub fn build(b: *std.Build) void {
|
||||
// out of the same read-only initrd, before it spawns anything — the registrar
|
||||
// has to know its policy before the first provider asks.
|
||||
bundled_list.append(b.allocator, .{ .path = "system/configuration/protocol.csv", .binary = b.path("system/configuration/protocol.csv") }) catch @panic("OOM");
|
||||
// The storage mount map (docs/file-system-development/storage-architecture.md):
|
||||
// filesystems.csv (content signature -> service binary) and volumes.csv (the
|
||||
// optional id -> mount-prefix override), both read by the volume manager.
|
||||
bundled_list.append(b.allocator, .{ .path = "system/configuration/filesystems.csv", .binary = b.path("system/configuration/filesystems.csv") }) catch @panic("OOM");
|
||||
bundled_list.append(b.allocator, .{ .path = "system/configuration/volumes.csv", .binary = b.path("system/configuration/volumes.csv") }) catch @panic("OOM");
|
||||
// A no-option build assumes neither -Dtest-case nor -Ddiagnose: it ships the
|
||||
// production set only. The userspace test fixtures under /test join in only
|
||||
// for a test build — which the QEMU harness signals by passing
|
||||
@@ -327,6 +333,7 @@ pub fn build(b: *std.Build) void {
|
||||
if (test_case != null) for ([_][]const u8{
|
||||
"vfs-test", // the user-space VFS round-trip client
|
||||
"fat-test",
|
||||
"exfat-test", // the exFAT mount round-trip client (S4)
|
||||
"badge-scope-test", // the guessable-id probe: a second process names the first's node and layer
|
||||
"shared-memory-server",
|
||||
"shared-memory-client",
|
||||
@@ -442,6 +449,8 @@ pub fn build(b: *std.Build) void {
|
||||
csv_library,
|
||||
xkeyboard_config_library,
|
||||
b.dependency("fat", .{}),
|
||||
b.dependency("exfat", .{}),
|
||||
b.dependency("volume-manager", .{}),
|
||||
b.dependency("display", .{}),
|
||||
b.dependency("ps2-bus", .{}),
|
||||
b.dependency("usb-hid", .{}),
|
||||
|
||||
@@ -46,6 +46,7 @@
|
||||
.@"pci-bus" = .{ .path = "system/drivers/pci-bus" },
|
||||
.init = .{ .path = "system/services/init" },
|
||||
.fat = .{ .path = "system/services/fat" },
|
||||
.exfat = .{ .path = "system/services/exfat" },
|
||||
.display = .{ .path = "system/services/display" },
|
||||
.@"display-demo" = .{ .path = "system/services/display-demo" },
|
||||
.@"device-manager" = .{ .path = "system/services/device-manager" },
|
||||
@@ -64,6 +65,7 @@
|
||||
.@"virtio-gpu" = .{ .path = "system/drivers/virtio-gpu" },
|
||||
.@"vfs-test" = .{ .path = "test/system/services/vfs-test", .lazy = true },
|
||||
.@"fat-test" = .{ .path = "test/system/services/fat-test", .lazy = true },
|
||||
.@"exfat-test" = .{ .path = "test/system/services/exfat-test", .lazy = true },
|
||||
.@"badge-scope-test" = .{ .path = "test/system/services/badge-scope-test", .lazy = true },
|
||||
.@"shared-memory-server" = .{ .path = "test/system/services/shared-memory-server", .lazy = true },
|
||||
.@"shared-memory-client" = .{ .path = "test/system/services/shared-memory-client", .lazy = true },
|
||||
|
||||
@@ -136,7 +136,7 @@ xHCI match already uses); USB children carry the (class, subclass, protocol) tri
|
||||
from usb-ids.zig — each bus's native language, decoded by the shared ids modules.
|
||||
(Since the registry landed, `child_added` also carries a `bus` discriminator and
|
||||
the numeric `vendor`/`device`/`subsystem` ids the finer match levels need —
|
||||
see [/etc/devices.csv](devices-csv.md).)
|
||||
see [/system/configuration/devices.csv](devices-csv.md).)
|
||||
|
||||
## Supervision and restart
|
||||
|
||||
@@ -227,7 +227,7 @@ published exit events, signals + `process`). On top of those:
|
||||
class triple alone, so a virtio-gpu could only be matched as a generic display
|
||||
function and the driver had to re-confirm its `1AF4:1050` identity from config
|
||||
space after being spawned. The manifest the earlier note anticipated landed as a
|
||||
human-readable registry: **[/etc/devices.csv](devices-csv.md)**, parsed by the
|
||||
human-readable registry: **[/system/configuration/devices.csv](devices-csv.md)**, parsed by the
|
||||
pure `device-registry` module and read by the manager at boot. A row binds a
|
||||
driver to a device by any of base / subclass / prog-IF / vendor / device /
|
||||
subsystem / `_HID`, most-specific match winning; it is authoritative (no
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
# /etc/devices.csv — the device registry
|
||||
# /system/configuration/devices.csv — the device registry
|
||||
|
||||
**Status: built (2026-07-26).** The device manager reads `/etc/devices.csv` at
|
||||
**Status: built (2026-07-26).** The device manager reads `/system/configuration/devices.csv` at
|
||||
boot and binds every device a bus driver reports to the driver the registry
|
||||
names. It replaces the three hand-written `switch` tables that used to live in
|
||||
the manager (`pciDriverForIdentity`, `hidDriverFor`, `usbDriverForIdentity`) —
|
||||
@@ -73,15 +73,15 @@ the first, so the shadowed rule is visible rather than silently dropped.
|
||||
|
||||
There is no compiled-in default table behind the registry. A device that no row
|
||||
matches goes **unbound** and is logged; the manager never guesses. A missing or
|
||||
empty `/etc/devices.csv` therefore means nothing matches — which is loud at boot,
|
||||
empty `/system/configuration/devices.csv` therefore means nothing matches — which is loud at boot,
|
||||
not a silent half-working system.
|
||||
|
||||
## How the manager reads it
|
||||
|
||||
`/etc/devices.csv` is bundled into the initial ramdisk (`build.zig`'s `bundled`
|
||||
list). The kernel serves the initrd's `/etc` tree directly — the `fat` service is
|
||||
spawned *after* the device manager and is irrelevant to `/etc` — so the manager
|
||||
reads the file with a plain `fs.open("/etc/devices.csv")` + `read`, with no
|
||||
`/system/configuration/devices.csv` is bundled into the initial ramdisk (`build.zig`'s `bundled`
|
||||
list). The kernel serves the initrd's `/system/configuration` tree directly — the `fat` service is
|
||||
spawned *after* the device manager and is irrelevant to `/system/configuration` — so the manager
|
||||
reads the file with a plain `fs.open("/system/configuration/devices.csv")` + `read`, with no
|
||||
filesystem service running and no boot-ordering dependency. It parses the bytes
|
||||
once in `initialise`, before any bus driver can report a device to match.
|
||||
|
||||
@@ -105,6 +105,6 @@ in different namespaces — against the right `bus` column.
|
||||
`build_support.programModule`). Then bundle it at `/system/drivers/<name>`:
|
||||
one dependency + one bundled entry in the root `build.zig`, one line in the
|
||||
root `build.zig.zon`.
|
||||
2. Add a row to `etc/devices.csv` naming the identity it binds and its full path.
|
||||
2. Add a row to `system/configuration/devices.csv` naming the identity it binds and its full path.
|
||||
|
||||
No device-manager change is required — the registry is the seam.
|
||||
|
||||
@@ -193,12 +193,14 @@ class driver, the device manager, or the kernel may share them freely.
|
||||
a PS/2 or 16550 driver possible; the low-rate legacy hardware that needs it is fine with
|
||||
a syscall per access. `io_port` resources were recorded by discovery and ignored — now
|
||||
they're used.
|
||||
- **M16 (detection)** — the IOMMU is now *found*: discovery parses the ACPI DMAR table,
|
||||
maps the first VT-d unit, and reads its version + capabilities (`iommu_present` in the
|
||||
platform info). This is detection only — **no translation domains are programmed, so
|
||||
DMA is still unprotected** (the caveat below). Enforcement lands with the first DMA
|
||||
driver, which is what there is to protect and test against. Proven in the `iommu` test,
|
||||
booted with an emulated `intel-iommu`.
|
||||
- **M16 (enforcement)** — the IOMMU is now *found and used*: discovery parses the ACPI
|
||||
DMAR table, maps the first VT-d unit, and reads its version + capabilities
|
||||
(`iommu_present` in the platform info), and **`device_claim` programs a private
|
||||
per-device translation domain** for the claimed function — rolling the claim back with
|
||||
`-ECONFINE` if it cannot confine it — so `dma_alloc` buffers are bound into that domain
|
||||
and torn down at process death. DMA is protected on any IOMMU-equipped machine; the
|
||||
system fails open only when there is no IOMMU at all. Proven in the `iommu` test, booted
|
||||
with an emulated `intel-iommu`.
|
||||
- **`system_spawn`** — a user-space supervisor starts a driver:
|
||||
`system_spawn(name, arguments)` loads a binary bundled in the initial-ramdisk as a
|
||||
fresh ring-3 process; `name` becomes the child's argv[0] and the optional
|
||||
@@ -369,25 +371,29 @@ capability walk (MSI, MSI-X, PCIe extended caps) without any new syscall.
|
||||
Note QEMU's HPET reports `Tn_FSB_INT_DEL_CAP = 0` — no MSI — so an HPET timer could never
|
||||
exercise this path. The first MSI driver will be the first PCI driver.
|
||||
|
||||
## M16 — the IOMMU, and the honest caveat ◑ detection done, enforcement pending
|
||||
## M16 — the IOMMU, and the honest caveat ✅ done
|
||||
|
||||
*The IOMMU is now detected (DMAR parsed, VT-d unit mapped and read — see the `iommu`
|
||||
test), but **enforcement is not built**: no translation domains are programmed, so the
|
||||
caveat below still holds in full. Detection can't be taken further usefully until there
|
||||
is a DMA driver to protect and QEMU's `intel-iommu` to test the protection against —
|
||||
building the per-device domains alongside that first driver is both the natural order
|
||||
and the only way to verify them. The rest of this section is the original caveat.*
|
||||
test) **and enforced**: `device_claim` programs a private per-device VT-d/AMD-Vi
|
||||
translation domain for the claimed function and rolls the claim back with `-ECONFINE` if
|
||||
it cannot confine it, `dma_alloc` buffers are bound into that domain and torn down at
|
||||
process death, and the machine fails open only when it has no IOMMU at all. So the caveat
|
||||
below no longer holds except on IOMMU-less hardware. The rest of this section is the
|
||||
original caveat, kept for the reasoning.*
|
||||
|
||||
Everything above is capability-gated at the *CPU*. None of it is gated at the *device*.
|
||||
A driver that can program a bus-mastering engine can make that device write to any
|
||||
physical address, because page tables sit between the CPU and RAM, not between a device
|
||||
and RAM. Until VT-d/DMAR (or SMMU on ARM) is programmed from the DMAR table, **`device_claim`
|
||||
on any DMA-capable device is equivalent to granting ring 0.**
|
||||
Everything above is capability-gated at the *CPU*. CPU page tables alone do not gate the
|
||||
*device*: a driver that can program a bus-mastering engine could make that device write to
|
||||
any physical address, because those page tables sit between the CPU and RAM, not between a
|
||||
device and RAM. That is exactly what the IOMMU closes. Now that VT-d/DMAR (and AMD-Vi; SMMU
|
||||
on ARM) is programmed, **`device_claim` confines the function into a private translation
|
||||
domain** and rolls the claim back with `-ECONFINE` if it cannot — so a claimed DMA-capable
|
||||
device is no longer equivalent to granting ring 0.
|
||||
|
||||
This does not make the model useless — it's the same position Linux is in with the
|
||||
IOMMU off, and every other guarantee (crash isolation, restart, no shared address
|
||||
space) still holds. But "user-space drivers are memory-safe" is not true yet, and the
|
||||
gap should be named rather than implied.
|
||||
This puts the model ahead of Linux-with-the-IOMMU-off: with an IOMMU present,
|
||||
"user-space drivers are memory-safe" now holds, alongside every other guarantee (crash
|
||||
isolation, restart, no shared address space). The one remaining gap — a machine with no
|
||||
IOMMU at all, where the system deliberately fails open — should be named rather than
|
||||
implied.
|
||||
|
||||
## Ordering
|
||||
|
||||
|
||||
@@ -32,7 +32,7 @@ kernel ──spawns──► init (PID 1) ──spawns──► device-manag
|
||||
spawns only init, the service supervisor: the driver supervisor: enumerates
|
||||
publishes the starts the system /system/devices, matches each device
|
||||
initial-ramdisk services (device-manager, to a driver, and system_spawn's it
|
||||
so user space can fat, logger, ...). Its
|
||||
so user space can volume-manager, logger, ...). Its
|
||||
system_spawn from it list is init policy.
|
||||
```
|
||||
|
||||
@@ -43,7 +43,7 @@ the optional arguments its argv[1..], on a SysV entry stack, see sysv.md). Every
|
||||
|
||||
- **init** ([system/services/init](system/services/init/init.zig)) is the **service
|
||||
supervisor**. It spawns the system services danos brings up at boot — today `input`,
|
||||
the `device-manager`, `fat`, `display`, `display-demo`, and the `logger` — from a
|
||||
the `device-manager`, `volume-manager`, `display`, `display-demo`, and the `logger` — from a
|
||||
small list. Drivers are deliberately *not* its job. (An earlier draft listed a `vfs`
|
||||
service here; that service is retired — the router moved into the kernel as
|
||||
`fs_resolve`.)
|
||||
@@ -66,9 +66,10 @@ the optional arguments its argv[1..], on a SysV entry stack, see sysv.md). Every
|
||||
|
||||
So "how is a driver discovered and configured" has two halves: **discovery** is the
|
||||
kernel's device table, read by anyone; **configuration** is two user-space policies —
|
||||
init's service list and the device-manager's match table. Both are hardcoded in their
|
||||
respective programs today; the natural next step is to move them into `/etc` (see the
|
||||
milestone notes in [driver-model.md](driver-model.md)). `system_spawn` is currently
|
||||
init's service list and the device-manager's match table. Both are now data, not code —
|
||||
init reads `/system/configuration/init.csv` and the device manager reads
|
||||
`/system/configuration/devices.csv`, each parsed at startup (the compiled-in switch
|
||||
tables are gone). `system_spawn` is currently
|
||||
ungated — any process may spawn any bundled binary — because there is no spawn
|
||||
capability yet.
|
||||
|
||||
@@ -301,12 +302,13 @@ process releases its claims and IRQ/MSI bindings — `releaseAllOwnedBy`,
|
||||
- **Page granularity.** `mmio_map` rounds to 4 KiB. Two devices sharing a page means
|
||||
granting one grants the other. A `device_register`ed child's *resource* can be narrower
|
||||
than a page, but its *mapping* can't.
|
||||
- **DMA is not contained.** A driver that can program a bus-mastering device can make
|
||||
that device write to *any* physical address — page tables don't sit between a device
|
||||
and RAM; an IOMMU does. The IOMMU is now *detected* (M16), but no translation domains
|
||||
are programmed, so `device_claim` on a DMA-capable device is still effectively
|
||||
equivalent to granting ring 0. This is the largest gap between the design's promise and
|
||||
what it delivers; enforcement lands with the first DMA driver.
|
||||
- **DMA is contained — except on a machine with no IOMMU.** A driver that can program a
|
||||
bus-mastering device could make that device write to *any* physical address — page
|
||||
tables don't sit between a device and RAM; an IOMMU does. `device_claim` now confines
|
||||
each claimed PCI function into its own VT-d/AMD-Vi translation domain and rolls the
|
||||
claim back with `-ECONFINE` if it can't (`system/kernel/process.zig`); DMA buffers are
|
||||
bound into that domain and torn down at process death. The residual gap is fail-open:
|
||||
where the machine exposes **no IOMMU at all**, a DMA-capable claim still reaches RAM.
|
||||
- **No voluntary `dev_release`.** A *live* driver can't drop a claim — only exit
|
||||
releases it (any path out of a process runs `releaseAllOwnedBy`) — so handing a
|
||||
device between running drivers still means exiting.
|
||||
@@ -365,10 +367,9 @@ $ python3 test/qemu_test.py device-manager acpi-ps2 pci-scan containment irqfree
|
||||
## What's next (not done here)
|
||||
|
||||
The big driver-model pieces — capability passing (class drivers), DMA + barriers, MSI,
|
||||
and IOMMU detection — are **now done** ([driver-model.md](driver-model.md), M13–M16), as
|
||||
is **port I/O** (`io_read`/`io_write`, the claim-gated syscalls that make a PS/2 or 16550
|
||||
driver possible). What's left is IOMMU *enforcement* (per-device domains — it waits on
|
||||
the first DMA driver to protect and test against) and these smaller items:
|
||||
and per-device IOMMU confinement — are **now done** ([driver-model.md](driver-model.md),
|
||||
M13–M16), as is **port I/O** (`io_read`/`io_write`, the claim-gated syscalls that make a
|
||||
PS/2 or 16550 driver possible). What's left is a handful of smaller items:
|
||||
|
||||
- **Releasing a claim** — half done. The kernel now drops *all* of a dead driver's
|
||||
claims on every path out of a process (`releaseAllOwnedBy`, called from process
|
||||
|
||||
@@ -162,7 +162,7 @@ Without the boot-tree row the binary never reaches the image and the
|
||||
device-manager has nothing to spawn. (The package also builds standalone:
|
||||
`cd system/drivers/intel-uhd-graphics-750 && zig build`.)
|
||||
|
||||
## 3. Add the match rule to `etc/devices.csv`
|
||||
## 3. Add the match rule to `system/configuration/devices.csv`
|
||||
|
||||
One row: bus, class triplet, vendor/device, driver path. **Copy the class
|
||||
triplet from the pci-bus boot log line, not from another row** — for the iGPU
|
||||
@@ -195,9 +195,9 @@ mapping — with zero risk to the hardware.
|
||||
## 5. Verify the plumbing
|
||||
|
||||
- `zig build test` still passes.
|
||||
- On the image: `/var/log/<boot-stamp>/system/services/device-manager.log`
|
||||
- On the image: `/system/logs/<boot-stamp>/system/services/device-manager.log`
|
||||
shows `spawned <name> for device <N>`, and
|
||||
`/var/log/<boot-stamp>/system/drivers/<name>.log` holds the resource list and
|
||||
`/system/logs/<boot-stamp>/system/drivers/<name>.log` holds the resource list and
|
||||
your first read.
|
||||
- If the driver did not spawn, diagnose in this order: binary on the image
|
||||
(step 2) → CSV row matches the log line exactly (step 3) → path identical in
|
||||
|
||||
@@ -32,8 +32,9 @@ Read-only and writable halves of `/system`: the program subtrees (`kernel`,
|
||||
initrd-backed or synthetic — while `configuration` and `logs` are mutable
|
||||
machine state served by the boot-volume FAT backend. The kernel's
|
||||
reserved-prefix rule (no mount may shadow `/system`, `/test`, or `/protocol`)
|
||||
needs a carve-out for exactly these two writable subtrees; that lands with the
|
||||
path migration below.
|
||||
carves out exactly these two writable subtrees (`initrd_carve_outs` in
|
||||
`system/kernel/vfs.zig`), so the boot volume mounts them while every other
|
||||
`/system` path stays initrd-served.
|
||||
|
||||
Deliberately not defined yet: a temporary-files location and per-application
|
||||
mutable storage. Both belong to the `/applications` design and will be
|
||||
@@ -74,10 +75,10 @@ expect, mapped onto the real tree; the tree itself stays danos-native.
|
||||
|
||||
## Migration
|
||||
|
||||
The tree above is the specification; some code still writes the unix paths it
|
||||
replaced. The flag-day converting them:
|
||||
The tree above is the specification, and the code writes it. A flag-day already
|
||||
converted the unix paths it replaced:
|
||||
|
||||
| Today (in code) | Becomes | Where |
|
||||
| Was | Now | Where |
|
||||
|------------------------------------------|-------------------------------------------|-----------------------------------------------------------------|
|
||||
| `/etc/init.csv` | `/system/configuration/init.csv` | `system/services/init/init.zig` |
|
||||
| `/etc/devices.csv` | `/system/configuration/devices.csv` | `system/services/device-manager/device-manager.zig` |
|
||||
@@ -85,6 +86,6 @@ replaced. The flag-day converting them:
|
||||
| `/mnt/usb` | `/volumes/usb` | `system/services/fat/fat.zig`, the fat/vfs tests |
|
||||
| `ServiceId` lookup | resolve + open under `/protocol` | every service and client; [protocol-namespace.md](../os-development/protocol-namespace.md) |
|
||||
|
||||
The boot-image builder and the on-volume directory layout move in the same
|
||||
The boot-image builder and the on-volume directory layout moved in the same
|
||||
change, so a freshly written image and the paths the services expect never
|
||||
disagree.
|
||||
|
||||
@@ -3,18 +3,37 @@
|
||||
> **Status:** the layered model below is the settled design
|
||||
> ([storage-design-rationale.md](storage-design-rationale.md) records how it was
|
||||
> reached, and [volume-manager-plan.md](../volume-manager-plan.md) how it was
|
||||
> built). **Built** (the volume-manager track, V0–V4): the data path, the driver
|
||||
> built). **Built** (V0–V4 + the storage-stack S1/S2): the data path, the driver
|
||||
> range confinement (per-sender clamp + the confinement gate), the `medium_changed`
|
||||
> presence event, the volume manager itself — it probes the partition table,
|
||||
> confines each filesystem to its partition, spawns one filesystem per volume, and
|
||||
> supervises it — and the removal half of the lifecycle (a pulled stick unmounts).
|
||||
> **Still pending**: the fuller identity ladder and the `volumes.csv` mount map,
|
||||
> multi-volume (one FAT volume today), the volume manager *consuming*
|
||||
> `medium_changed` (removal is detected by device-presence polling; the event is
|
||||
> published but only a card-reader medium change needs the subscription), and the
|
||||
> remount-on-replug end-to-end (the logic is in place; QEMU can't re-present the
|
||||
> boot-controller device, so it is bench-verified). A few markers below are left
|
||||
> where a duty is still pending.
|
||||
> supervises it — the removal half of the lifecycle (a pulled stick unmounts), the
|
||||
> identity ladder (GPT GUID + name, FAT serial + label, MBR), and the mount map:
|
||||
> `filesystems.csv` (signature → binary) + `volumes.csv` (identity → optional
|
||||
> override), a volume's mount path IS its content id (`/volumes/<id>`), with the
|
||||
> label as display metadata a `volumes` query returns. Multi-volume is **built**:
|
||||
> the manager adopts every storage device, probes each device's whole partition
|
||||
> table, and spawns one range-confined FAT per volume — several volumes across
|
||||
> several devices, or several partitions sharing one device's channel — each at
|
||||
> its own `/volumes/<id>` path with its own supervision. The boot volume is
|
||||
> identified by **content** (a volume backs `/system/configuration` + `/system/logs`
|
||||
> only when it resolves `/system/configuration` on its own media), so it works as
|
||||
> any partition of any device. exFAT is **built** as a second engine
|
||||
> (`system/services/exfat`): full read + write, directories, rename, and on-disk
|
||||
> up-case folding, reusing `library/kernel/file-system-harness` wholesale — the
|
||||
> reuse claim, proven — and a volume routes to fat or exfat by its VBR, at an
|
||||
> `exfat-<serial>` id-path. Removal is robust to all three triggers now: a
|
||||
> pulled device (presence polling), a medium that leaves while its device stays
|
||||
> (the volume manager CONSUMES `medium_changed`), and a storage driver that
|
||||
> crashes while its device stays present (a channel-liveness `geometry()` probe
|
||||
> reaps the volume and rebuilds it on the restarted driver's fresh channel). The
|
||||
> re-adopt-and-remount path is QEMU-proven by the driver-crash rebuild; a physical
|
||||
> unplug/replug exercises the same path but is bench-pending (QEMU cannot
|
||||
> re-present a usb-storage `device_add`). **Still pending**: the `filesystem UUID`
|
||||
> rung (ext-family superblocks, which need such an engine); and arbitration when
|
||||
> two volumes both resolve the boot markers (S3 mounts both and logs each claim;
|
||||
> picking one is deferred). A few
|
||||
> markers below are left where a duty is still pending.
|
||||
|
||||
## The model
|
||||
|
||||
@@ -65,12 +84,13 @@ own children's class drivers, routed by lineage.
|
||||
|
||||
**Storage driver** (usb-storage, per device; later nvme, ahci): hardware →
|
||||
blocks. Speaks its transport upward from its device; serves the block
|
||||
protocol (geometry, read, write, flush, attach, detach — and, *planned*, the
|
||||
pushed `medium_changed` presence event, translated from the transport's
|
||||
native signal). **Content-blind, permanently**:
|
||||
protocol (geometry, read, write, flush, attach, detach — and the pushed
|
||||
`medium_changed` presence event, published today from a slow TEST UNIT READY
|
||||
poll; *planned*: translating it from the transport's native signal instead of
|
||||
polling). **Content-blind, permanently**:
|
||||
it reads block 0 only as a bring-up self-check and parses nothing — no MBR, no
|
||||
GPT, no filesystem magic, ever. *(Planned)* it gains one content-blind
|
||||
mechanism: **named sub-ranges** — "serve blocks [a, b) as a channel of the
|
||||
GPT, no filesystem magic, ever. It has one content-blind mechanism *(built)*:
|
||||
**named sub-ranges** — "serve blocks [a, b) as a channel of the
|
||||
same block contract". It clamps and offsets; it never knows the numbers came
|
||||
from a partition table. The clamp lives here and nowhere else because a
|
||||
channel must carry exactly the authority it grants: handing a filesystem the
|
||||
@@ -84,16 +104,17 @@ manager's tree for a storage provider; when one appears it consumer-hellos for
|
||||
the block channel, reads the partition table and the first blocks itself
|
||||
(**it** is the prober), defines the volume's sub-range on the driver, spawns the
|
||||
matching filesystem service confined to that range, and supervises it (backoff,
|
||||
crash-loop cap). *(Pending)*: it decides mount placement from `volumes.csv` and
|
||||
picks the filesystem binary from `filesystems.csv` — today it hands every
|
||||
FAT-shaped volume to the FAT service and the FAT service carries hardcoded mount
|
||||
prefixes. Those tables are CSV configuration, read by it (the policy), enforced
|
||||
by nobody else:
|
||||
crash-loop cap). *(Built)*: it picks the filesystem binary from
|
||||
`filesystems.csv` by the volume's content signature, and mounts the volume at its
|
||||
content id (`/volumes/<id>`) — or a `volumes.csv` override. The label is display
|
||||
metadata the `volumes` query returns, never the path. Those tables are CSV
|
||||
configuration, read by it (the policy), enforced by nobody else:
|
||||
|
||||
- `filesystems.csv` *(pending)* — content signature → filesystem binary. Adding
|
||||
- `filesystems.csv` *(built)* — content signature → filesystem binary. Adding
|
||||
a filesystem adds a row.
|
||||
- `volumes.csv` *(pending)* — the mount map, danos's fstab: **volume identity → mount
|
||||
prefix**, keyed on content identity and never on port, path, or arrival
|
||||
- `volumes.csv` *(built)* — the mount map, danos's fstab: an OPTIONAL **volume
|
||||
identity → mount prefix** override (a volume with no row mounts at its default
|
||||
`/volumes/<id>`), keyed on content identity and never on port, path, or arrival
|
||||
order (the lesson of Linux's `/dev/sda1`-era fstab, which broke on every
|
||||
port move until `UUID=` replaced it). Identity is read off the medium by
|
||||
the prober, strongest first: GPT partition GUID → filesystem UUID → FAT
|
||||
@@ -104,8 +125,8 @@ by nobody else:
|
||||
identity (cloned sticks, together) is policy: first keeps the name, the
|
||||
second mounts suffixed and is logged loudly. The boot volume is the
|
||||
recorded identity of the volume carrying `/system/configuration` and
|
||||
`/system/logs`, findable on any port. Unknown volumes mount under
|
||||
`/volumes/<derived name>`.
|
||||
`/system/logs`, findable on any port. Every volume's default mount is
|
||||
`/volumes/<id>` — its rendered content identity.
|
||||
|
||||
**Filesystem service** (the FAT service today; one process per volume): the
|
||||
proven unit — block-client + engine + file-protocol provider in one binary. It
|
||||
@@ -113,16 +134,25 @@ receives its block channel at spawn; it never discovers devices. It registers
|
||||
its own mounts with the kernel; its write cache lives inside the process, so a
|
||||
write error is observed by the code that owns the volume and surfaces on the
|
||||
owning channel (the anti-fsyncgate rule — never a system-wide dirty pool).
|
||||
*(Today, interim:)* fat also acquires its own volume (first mass-storage child
|
||||
by enumeration order) and hardcodes its mount prefixes; both migrate to the
|
||||
volume manager.
|
||||
*(Built:)* fat receives its mount path as `argv[2]` from the volume manager (the
|
||||
volume's id-path, e.g. `/volumes/fat-12345678`) and mounts its root there. It
|
||||
installs the two `/system` hierarchy rewrites (`/system/configuration`,
|
||||
`/system/logs`) only when it is the boot volume — decided by **content**: it
|
||||
resolves `/system/configuration` on its own media at mount, so a data volume
|
||||
mounts at its id-path alone and never shadows the running system. It no longer
|
||||
self-acquires a volume — the V3b flip made it receive its volume id and block
|
||||
channel from the volume manager, consistent with "it never discovers devices"
|
||||
above. Because several volumes now serve at once, no filesystem binds a shared
|
||||
service name; clients reach each through the kernel mount table (`fs_resolve`
|
||||
routes by prefix to the backing endpoint).
|
||||
|
||||
**Kernel** (mechanism only): the mount table routes paths to backend
|
||||
endpoints — resolve and redirect, never data. Remount-replace is the restart
|
||||
story; a dead backend's slot is swept lazily on the next resolution.
|
||||
*(Planned fix:)* `fs_unmount` gains ownership — only the mounting endpoint's
|
||||
holder may unmount; today it is ungated, which is safe with one mount owner
|
||||
and wrong with several.
|
||||
`fs_unmount` is ownership-gated *(built, V0)* — only the mounting task may
|
||||
unmount its own prefix; anyone else is refused (`EPERM`). No dead-owner
|
||||
exception: a dead owner's slot is swept lazily by resolution, and restart goes
|
||||
through remount-replace, never through a stranger's unmount.
|
||||
|
||||
## Adding a filesystem
|
||||
|
||||
@@ -135,8 +165,9 @@ and wrong with several.
|
||||
2. **Reuse the shell**: the filesystem harness — establishment, the
|
||||
badge-scoped open-node table, the nine vfs-protocol handlers, mount
|
||||
registration, the removal path — is shared code, not per-filesystem code.
|
||||
*(Planned: extracted from fat's 434-line shell into a library before the
|
||||
second engine is written.)* An engine plus a `main` wiring it into the
|
||||
*(Done: extracted from fat's original 434-line shell into
|
||||
`library/kernel/file-system-harness.zig`; fat imports it and instantiates
|
||||
`harness.Server(engine.FileSystem)`.)* An engine plus a `main` wiring it into the
|
||||
harness is a complete filesystem service.
|
||||
3. **Add the configuration row**: one line in `filesystems.csv` mapping the
|
||||
on-disk signature to the binary. No other component changes: the volume
|
||||
@@ -159,24 +190,27 @@ surprise-removal path — kill the filesystem process, retire its mounts,
|
||||
respawn on return. No half-alive states, no `remount-ro`, no mounts that
|
||||
error forever (Plan 9's dead-server wart).
|
||||
|
||||
The path has **two triggers, one lifecycle**: the *device* leaving (the
|
||||
storage driver dies — channel death, the table below), and the *medium*
|
||||
leaving while the device stays (an SD card pulled from its reader, an ATAPI
|
||||
tray opened — including USB card readers today). The second trigger is a
|
||||
pushed `medium_changed` event on the block protocol *(planned)*: the storage
|
||||
driver translates its transport's native signal (SCSI UNIT ATTENTION, AHCI
|
||||
PxSSTS, NVMe namespace-change AER) into presence-changed — never content —
|
||||
and the volume manager runs the same kill-retire path, then re-probes on
|
||||
medium return exactly as on device return. Without it, a swapped card would
|
||||
be served with the previous card's filesystem state.
|
||||
The path folds **three triggers into one lifecycle**: the *device* leaving (a
|
||||
pulled stick — presence polling); the *medium* leaving while the device stays
|
||||
(an SD card pulled from its reader, an ATAPI tray opened, a USB card reader);
|
||||
and a storage *driver crashing* while its device stays in the tree. The second
|
||||
trigger is the pushed `medium_changed` event on the block protocol, published
|
||||
from a TEST UNIT READY poll — the volume manager now **consumes** it (subscribed
|
||||
per device), running the same kill-retire path and re-probing on medium return,
|
||||
so a swapped card is never served with the previous card's filesystem state. The
|
||||
third is caught by a channel-liveness `geometry()` probe: presence polling alone
|
||||
sees the device still present, but the channel is dead, so the manager reaps the
|
||||
volume and rebuilds it on the restarted driver's fresh channel. Still *planned*
|
||||
is translating the transport's native signal (SCSI UNIT ATTENTION, AHCI PxSSTS,
|
||||
NVMe namespace-change AER) in place of the presence poll.
|
||||
|
||||
| Layer | Observes | Must do | Guarantees |
|
||||
|---|---|---|---|
|
||||
| Bus driver | port/hub status change | tear down the device's slots (children first, recursively — built, hot-plug matrix), report `child_removed` per interface | the device tree is honest within one reconcile tick |
|
||||
| Device manager | `child_removed` / reporter death | prune the child; **reap the bound driver** (built) — the storage driver for that stick dies now, not never | no zombie storage processes; re-report rebinds |
|
||||
| Storage driver | its own death (it IS the removed device's driver) | nothing — dying is its removal handling; DMA/IOMMU/claims release mechanically at death | in-flight transfers fail visibly to callers, never hang |
|
||||
| Volume manager *(removal built; remount bench-pending)* | the storage device leaving the device-manager tree (poll) | kill the filesystem service of that device's volume; its kernel mounts retire | one removal path; mounts never dangle; log persistence stops *cleanly* |
|
||||
| Filesystem service | its block channel dies (`EPEER`) mid-operation, or it is killed by the volume manager | if it observes the death first: flush nothing (the medium is gone), answer in-flight requests with errors, exit; dirty write-back data is **lost and said to be lost** | the FAT dirty flag on disk marks the unclean removal; the process never serves from behind a dead channel |
|
||||
| Volume manager *(built)* | a device leaving the tree (poll), a `medium_changed` event, or a dead channel under a still-present device (a crashed driver — `geometry()` liveness probe) | kill that volume's filesystem service (its mounts retire), then re-adopt + remount on return or on the restarted driver's fresh channel | one removal path for all three triggers; mounts never dangle; the manager never serves from behind a dead channel |
|
||||
| Filesystem service | its block channel dies (`EPEER`) mid-operation, or it is killed by the volume manager | if it observes the death first: flush nothing (the medium is gone), answer in-flight requests with errors, exit; dirty write-back data is **lost and said to be lost** | the unflushed write-back window is dropped on a surprise yank — danos writes no on-disk dirty/clean-shutdown marker today; the process never serves from behind a dead channel |
|
||||
| Kernel | backend endpoint death | lazy mount-slot sweep on next resolve (built); ownership-gated `fs_unmount` (built, V0) | resolution under a dead mount is `not_found`, not a hang |
|
||||
| Application | `not_found` / error on paths under the vanished mount | its own error handling — the contract is honest absence, identical to the path never existing | no operation blocks forever on removed media |
|
||||
|
||||
@@ -222,7 +256,8 @@ never abandoned after the first `not_found`.
|
||||
|
||||
What is lost on a surprise yank is exactly the write-back window of the
|
||||
filesystem service, no more: the engine owns its cache, so the blast radius of
|
||||
a yank is one volume's unflushed writes, marked by the on-disk dirty flag. A
|
||||
a yank is one volume's unflushed writes, which are simply lost — danos records
|
||||
no on-disk dirty/clean-shutdown bit yet. A
|
||||
filesystem format with better crash honesty (journaling, copy-on-write — the
|
||||
lesson of QNX's Power-Safe) narrows that window further and slots in as
|
||||
implementation N+1 through the table above, changing nothing else.
|
||||
@@ -249,8 +284,9 @@ the device lifecycle already uses, mapped one-to-one:
|
||||
obsolescence (the reap argument, proven on the device tree).
|
||||
3. **The shared harness is how every filesystem inherits the lifecycle by
|
||||
construction.** The harness — not the engine — owns the state machine:
|
||||
establishment at spawn, mount registration, the dirty-flag set/clear
|
||||
bracket, error-out-and-exit on channel death. The engine sits behind the
|
||||
establishment at spawn, mount registration, the flush-on-close hook (an
|
||||
in-memory device-dirty check that commits the device write cache on close),
|
||||
error-out-and-exit on channel death. The engine sits behind the
|
||||
four-function vtable and never sees a channel; it cannot opt out of the
|
||||
lifecycle for the same reason it cannot find a device. This is why the
|
||||
harness is extracted BEFORE the second engine is written.
|
||||
|
||||
@@ -104,11 +104,19 @@ and /system/logs), closing the two-sticks question honestly.
|
||||
**Filesystems (per volume, one process).** The proven unit everywhere from
|
||||
Plan 9's `dossrv` to Minix to Fuchsia: block-client + engine + file-protocol
|
||||
provider in one binary, one process per volume (9front practice; per-volume
|
||||
fault isolation is what our supervision makes cheap). fat's shell becomes a
|
||||
shared *filesystem harness* library before a second engine is written; the MBR
|
||||
walk moves out of the engine into the volume manager; write caching stays
|
||||
inside the process (the anti-fsyncgate rule). Each mounts its prefixes into the
|
||||
kernel mount table itself, exactly as today.
|
||||
fault isolation is what our supervision makes cheap). fat's shell became a
|
||||
shared *filesystem harness* library (`library/kernel/file-system-harness`), and
|
||||
the second engine — **exFAT**, `system/services/exfat` — now reuses it wholesale:
|
||||
the reuse this design promised, proven. exfat is nothing but the exFAT engine +
|
||||
a near-clone of fat's thin service, full read + write + directories + rename +
|
||||
on-disk up-case folding, differing only in the format it wraps. A partition walk
|
||||
lives in the volume manager (`partition.zig`), which recognizes fat vs exFAT by
|
||||
VBR and routes each to its engine; write caching stays inside the process (the
|
||||
anti-fsyncgate rule). Each mounts its prefixes into the kernel mount table
|
||||
itself. Two surface limits are shared across both engines and are the vfs
|
||||
layer's, not an exFAT shortcut: file offsets are u32 (a 4 GiB addressable cap),
|
||||
and file names are ASCII bytes (a non-ASCII unit becomes `?`) — teaching the vfs
|
||||
name layer UTF-8 is a separate cross-cutting change.
|
||||
|
||||
**Kernel: two small changes only.** `fs_unmount` gains ownership (only the
|
||||
mounting endpoint's holder may unmount — possession-is-capability, consistent
|
||||
@@ -132,8 +140,10 @@ what the architecture promises: one driver binary plus one `devices.csv` row.
|
||||
**NVMe** shortens the stack — no bus/class split; the driver IS the
|
||||
controller driver, one hop fewer than USB. Its structural novelty,
|
||||
**namespaces** (hardware-native multiple volumes behind one controller), is
|
||||
exactly the case the block protocol reserved on day one and decision 4 below
|
||||
settles as endpoint-per-volume. **AHCI** is a shape choice, not a problem: an
|
||||
exactly the case the block protocol reserved on day one — still reserved, not
|
||||
implemented (pressure point 1 below): decision 4 settles per-volume addressing
|
||||
as per-sender confinement on a single endpoint, and names endpoint-per-volume
|
||||
only as an unbuilt future refactor. **AHCI** is a shape choice, not a problem: an
|
||||
HBA fronts up to 32 ports plus port multipliers — structurally a bus — so
|
||||
either mirror USB (ahci-bus + a per-port disk driver: maximum restart
|
||||
granularity, the proven shape) or mirror NVMe (one driver per controller, one
|
||||
@@ -142,9 +152,13 @@ matrix-proven shape; genuinely open.
|
||||
|
||||
**The pressure points, honestly:**
|
||||
|
||||
1. **Multi-volume providers are reserved, not implemented.** The volume
|
||||
manager flow assumes one provider, one volume; NVMe namespaces make
|
||||
endpoint-per-volume real work with hardware demanding it.
|
||||
1. **Multi-volume is built; multi-namespace-per-provider is untried.** The
|
||||
volume manager adopts every device and spawns one range-confined FAT per
|
||||
partition — several volumes across several devices, or several partitions
|
||||
sharing one device's channel, both proven on USB. What is untried is a single
|
||||
provider exposing several volumes as *namespaces* (NVMe): the endpoint and
|
||||
per-badge range machinery generalizes, but no such driver exists yet to
|
||||
exercise it.
|
||||
2. **The current transport will bottleneck NVMe.** Synchronous call/reply,
|
||||
one operation in flight, one bounce buffer — fine for a USB2 stick,
|
||||
forfeits an NVMe drive's queue depth and per-queue MSI-X. Correctness
|
||||
@@ -180,7 +194,10 @@ matrix-proven shape; genuinely open.
|
||||
same pattern applied to blocks. The volume manager sets each filesystem
|
||||
process's range on the driver; the driver clamps AND translates every
|
||||
transfer by the sender's range, so filesystems address volume-relative
|
||||
LBAs from 0 and the FAT engine's `base_lba` is deleted rather than moved.
|
||||
LBAs from 0. On the confined path the FAT engine's `base_lba` therefore
|
||||
resolves to 0 on every access; the field and the engine's own MBR walk
|
||||
remain in `engine.zig` as now-inert legacy code (the authoritative partition
|
||||
walk lives in the volume manager's `partition.zig`), not yet deleted.
|
||||
The enforcement point (the clamp at the provider, never in the consumer)
|
||||
is what carries the security property; endpoint-per-volume would deliver
|
||||
the same property only by inventing a multi-endpoint harness the pattern
|
||||
@@ -192,41 +209,56 @@ matrix-proven shape; genuinely open.
|
||||
names it as the 256-byte ceiling's unlock — Fuchsia's FIFO+VMO is the
|
||||
precedent); format-level crash honesty (a Power-Safe-style journaling or COW
|
||||
filesystem) once danos outgrows FAT; per-process namespaces.
|
||||
7. **The media-presence event** (settled in principle; lands with the volume
|
||||
manager): the block protocol gains a pushed event — `medium_changed`, with
|
||||
present/absent and a change counter — produced by the storage driver from
|
||||
its transport's native signal (SCSI UNIT ATTENTION / TEST UNIT READY for
|
||||
USB and ATAPI, PxSSTS for AHCI, namespace-change AER for NVMe) and
|
||||
consumed by the volume manager, which runs the SAME kill-retire-remount
|
||||
path it runs on channel death — one lifecycle, two triggers. The driver
|
||||
reports presence, never content; a pushed event carries no capability,
|
||||
which the kernel already guarantees. The device staying while its medium
|
||||
leaves is the one removable-media case the channel-death trigger cannot
|
||||
see; without this event a swapped SD card would be served with the old
|
||||
card's filesystem state.
|
||||
7. **The media-presence event** (the consuming half is BUILT; the
|
||||
transport-native signal stays future): the block protocol carries a pushed
|
||||
event — `medium_changed`, with present/absent and a change counter —
|
||||
produced today by the storage driver from a TEST UNIT READY poll (the
|
||||
transport's native signal — SCSI UNIT ATTENTION, PxSSTS for AHCI,
|
||||
namespace-change AER for NVMe — is the future refinement in place of the
|
||||
poll) and now **consumed** by the volume manager, which subscribes per
|
||||
device and runs the SAME kill-retire-remount path it runs on channel death.
|
||||
The driver reports presence, never content; a pushed event carries no
|
||||
capability, which the kernel already guarantees. The device staying while
|
||||
its medium leaves is the one removable-media case the channel-death trigger
|
||||
cannot see; without this event a swapped SD card would be served with the
|
||||
old card's filesystem state. A THIRD trigger closes the last gap — a
|
||||
storage driver that *crashes* while its device stays present: channel death
|
||||
there is invisible to presence polling, so the volume manager probes channel
|
||||
liveness (`geometry()`) each tick and reaps-then-rebuilds the volume on the
|
||||
restarted driver's fresh channel. One lifecycle, three triggers.
|
||||
8. **Volume identity, and the mount map as danos's fstab** (settled). The
|
||||
lesson is Linux's own history: fstab keyed on `/dev/sda1` for years and
|
||||
broke whenever a drive changed ports or enumeration order; `UUID=` entries
|
||||
exist because device-path identity failed. danos skips that era: the mount
|
||||
map (`volumes.csv` — configuration, read by the volume manager) keys on
|
||||
**content identity, never port or discovery order**. The prober reads
|
||||
identity off the medium, strongest first:
|
||||
1. GPT partition GUID — 128-bit, unique, stable for the volume's life;
|
||||
2. filesystem UUID (ext-family and most modern formats, in the superblock);
|
||||
**content identity, never port or discovery order**. Build status: rungs 1,
|
||||
3, and 4 (GPT partition GUID, FAT serial + label, MBR signature + index) are
|
||||
implemented (S1); rung 2 waits on a non-FAT engine. The `volumes.csv` map and
|
||||
the id-derived mount path are built (S2): a volume's mount path IS its content
|
||||
id (`/volumes/<id>`, e.g. `/volumes/fat-12345678`), or a `volumes.csv`
|
||||
override; the label is display metadata a `volumes` query returns, never the
|
||||
path. The ladder the prober reads off the medium, strongest first:
|
||||
1. GPT partition GUID — 128-bit, unique, stable for the volume's life — **built (S1)**;
|
||||
2. filesystem UUID (ext-family and most modern formats, in the superblock) *(planned)*;
|
||||
3. FAT volume serial + label — 32 bits, weak (dd-cloned sticks share it)
|
||||
but what real sticks carry;
|
||||
4. MBR disk signature + partition index;
|
||||
5. nothing — an anonymous volume: generated mount name, no persistence.
|
||||
but what real sticks carry — **built (S1)**;
|
||||
4. MBR disk signature + partition index — **built**; a bare FAT with no
|
||||
table takes index 0 over the whole device;
|
||||
5. nothing — an anonymous volume: generated mount name, no persistence *(planned)*.
|
||||
|
||||
Consequences, each mechanical once identity keys the map: **moving a drive
|
||||
to a different port changes nothing** — same identity, same mount point,
|
||||
whether USB port, hub depth, SATA port, or a stick that left as USB and
|
||||
returned in a SATA dock; **replug remounts at the same path** (the
|
||||
remount-after-return story is a map lookup); **the boot volume** is the
|
||||
recorded identity of the volume carrying `/system/configuration`, findable
|
||||
on any port; and **duplicate identity is a policy case, not a surprise** —
|
||||
two cloned sticks at once: first keeps the mapped name, second mounts
|
||||
suffixed and is logged loudly, never silently shadowed. Unknown identities
|
||||
returned in a SATA dock; **replug remounts at the same path** (the id-path is
|
||||
content-derived, so a volume returns to `/volumes/<id>` wherever it reappears;
|
||||
the re-adopt+remount code path is QEMU-proven by the driver-crash rebuild, but
|
||||
remount on a *physical* replug end-to-end is bench-pending, not QEMU-testable,
|
||||
because QEMU can't re-present the boot-controller device); **the boot volume** is the volume
|
||||
that resolves `/system/configuration` on its own media, findable on any port or
|
||||
partition; and **duplicate identity is a known S4 gap** — two cloned sticks
|
||||
share one content id, so today they collide on `/volumes/<id>` (the kernel
|
||||
remount-replaces; the last wins) and each boot-volume claim is logged loudly.
|
||||
Distinguishing them with a suffix is arbitration, deferred to S4. Unknown identities
|
||||
mount under a derived name (sanitized label, else generated) at
|
||||
`/volumes/<name>` — the hierarchy's documented home for attached media,
|
||||
which stands: `/system` is what danos IS; attached media is what it isn't.
|
||||
|
||||
@@ -89,14 +89,19 @@ comptime {
|
||||
}
|
||||
```
|
||||
|
||||
`maximum_domains = 64` and `maximum_devices = 64` agree today only by a sentence in a
|
||||
comment, and the agreement fails open. This is the clause with a live hole behind it,
|
||||
and the reason raising `maximum_devices` alone would be a privilege escalation rather
|
||||
than a fix.
|
||||
`maximum_domains = 64` and `maximum_devices = 64` once agreed only by a sentence in a
|
||||
comment, and that agreement failed open — the clause with the live hole behind it, where
|
||||
raising `maximum_devices` alone left every device id past the end of `iommu.confined`
|
||||
unconfined while `confineDevice` still reported success, a privilege escalation rather
|
||||
than a fix. The assert closed that: it held the two together while both stayed fixed, and
|
||||
when the device table was later made dynamic — no `maximum_devices` any more, only a
|
||||
per-registrar quota — that forced them apart, the assert having done its job. `confined`
|
||||
now grows to cover every id the broker mints, and `confineDevice` refuses when it cannot
|
||||
record a confinement rather than failing open.
|
||||
|
||||
## The worked bad case
|
||||
|
||||
`devices_broker.maximum_devices`, which had no comment at all:
|
||||
`devices_broker.maximum_devices`, which had no comment at all, before it was made dynamic:
|
||||
|
||||
```zig
|
||||
/// bound: device nodes for the whole machine — firmware-discovered plus registered
|
||||
|
||||
@@ -118,17 +118,22 @@ regions `init` frees. Keeping that boot-protocol knowledge on the loader side is
|
||||
deliberate — the kernel has no notion of "reclaimable" or of UEFI at all.
|
||||
|
||||
The one live piece in that memory is the boot stack the kernel starts on; the loader
|
||||
leaves the single region containing it `reserved`, so `init` won't hand it out. A
|
||||
later step will move task 0 onto a kernel-owned stack, freeing that last ~1 MiB
|
||||
region too (and giving user mode the clean stack it wants).
|
||||
leaves the single region containing it `reserved`, so `init` won't hand it out. The
|
||||
kernel is only on it for an instant, though — `_start`'s first instruction switches
|
||||
to a kernel-owned 64 KiB stack in `.bss` (that context becomes task 0). The region
|
||||
stays `reserved` because the loader's own `convertMemoryMap` was executing on that
|
||||
stack when it reclassified the RAM, and, like the map buffers below, nothing frees it
|
||||
yet.
|
||||
|
||||
## What's next (partly done since)
|
||||
|
||||
- **Contiguous allocation** — done: `allocContiguous` scans for a run of clear
|
||||
bits, with an optional physical ceiling for DMA (`dma_alloc` is its user), and
|
||||
`allocBelow` serves the SMP trampoline.
|
||||
- **A kernel stack for task 0** — still open: the boot processor's idle task runs
|
||||
on the boot stack to this day, so that region can't be freed.
|
||||
- **A kernel stack for task 0** — done: `_start`'s first instruction switches `rsp`
|
||||
to a kernel-owned 64 KiB stack in `.bss` (`bootstrap_stack`), and `scheduler.init`
|
||||
registers that running context as task 0 — the kernel is on the loader's boot
|
||||
stack for that one instruction and never again.
|
||||
- **Freeing the `reserved` `loader_data`** (the boot-time map buffers) — still
|
||||
open: the bitmap deliberately tracks those frames so they *can* be freed, but
|
||||
nothing frees them yet.
|
||||
|
||||
@@ -14,7 +14,7 @@ are mirrored to it explicitly (`system/kernel/kernel.zig`).
|
||||
## The pipeline
|
||||
|
||||
```
|
||||
process std.log ──▶ debug_write(level) ──▶ tagged kernel ring ──▶ logger service ──▶ /var/log/<boot-stamp>/<binary-path>.log
|
||||
process std.log ──▶ debug_write(level) ──▶ tagged kernel ring ──▶ logger service ──▶ /system/logs/<boot-stamp>/<binary-path>.log
|
||||
kernel log.print ─┘ │
|
||||
└▶ serial / 0xE9 sinks (QEMU, -Dserial)
|
||||
```
|
||||
@@ -49,12 +49,12 @@ kernel log.print ─┘ │
|
||||
|
||||
5. **Persist.** The **logger service** (`system/services/logger`) drains the
|
||||
ring every 250 ms and demultiplexes records into one file per source under
|
||||
`/var/log/<boot-stamp>/`, e.g.
|
||||
`/system/logs/<boot-stamp>/`, e.g.
|
||||
|
||||
```
|
||||
/var/log/2026-07-21T150434Z/kernel.log
|
||||
/var/log/2026-07-21T150434Z/system/services/fat.log
|
||||
/var/log/2026-07-21T150434Z/system/drivers/usb-storage.log
|
||||
/system/logs/2026-07-21T150434Z/kernel.log
|
||||
/system/logs/2026-07-21T150434Z/system/services/fat.log
|
||||
/system/logs/2026-07-21T150434Z/system/drivers/usb-storage.log
|
||||
```
|
||||
|
||||
The boot stamp is the RTC anchor from `klog_status` (FAT-safe: no colons; a
|
||||
@@ -95,6 +95,6 @@ written last so a reader only trusts a complete record).
|
||||
- A write-spamming process can evict other processes' unread records from the
|
||||
ring (a per-process quota is future work); the loss is at least visible via
|
||||
sequence gaps in every affected file.
|
||||
- `/var/log` files have no privacy until the VFS grows permissions.
|
||||
- `/system/logs` files have no privacy until the VFS grows permissions.
|
||||
- Records emitted after the logger's final shutdown drain reach serial and the
|
||||
ring but not the files.
|
||||
|
||||
@@ -14,7 +14,7 @@ process-manager server, and Fuchsia/seL4 control processes only through handles.
|
||||
|
||||
danos rules out `/proc` **as the primitive**: the path router lives in the
|
||||
kernel (`fs_resolve`), but what is mounted under a path is served by a
|
||||
user-process filesystem server (the way FAT serves `/mnt/usb`) — a `/proc`
|
||||
user-process filesystem server (the way FAT serves `/volumes/usb`) — a `/proc`
|
||||
would be one more such server, which would put a user process in the path of
|
||||
process control. If that server (or anything under it) hangs, nothing could be
|
||||
listed or killed, *including the hung server*. The control plane for processes
|
||||
@@ -117,9 +117,11 @@ the architecture layer calls up into `tick`.
|
||||
`process_exit_reason` (`process.exitReason`). This is the input to
|
||||
restart policy ([process-lifecycle.md](process-lifecycle.md)); an exit *code*
|
||||
for the clean case can still ride alongside later.
|
||||
- Enumerate writes through the caller's raw pointer under the bring-up trust
|
||||
model, like `device_enumerate` (an unmapped page is a self-DoS, not an
|
||||
isolation break).
|
||||
- ~~Enumerate writes through the caller's raw pointer under the bring-up trust
|
||||
model, like `device_enumerate`~~ Closed (8d4a7cf): both `process_enumerate`
|
||||
and `device_enumerate` describe a chunk into a kernel buffer and place it with
|
||||
`copyToUser`, which validates the range and resolves each page — an unmapped
|
||||
page returns `-EFAULT`, and the kernel never stores through the user pointer.
|
||||
|
||||
## Tests
|
||||
|
||||
|
||||
@@ -1,9 +1,9 @@
|
||||
# The protocol namespace
|
||||
|
||||
*Design, agreed 2026-07-31. Supersedes the `ServiceId` registry. P1–P3 of the
|
||||
*Design, agreed 2026-07-31. Supersedes the `ServiceId` registry. P1–P4 of the
|
||||
migration plan at the end have landed (the envelope, the registry and the
|
||||
`ServiceId` flag-day, and restriction stage one); P4 and P5 are the remaining
|
||||
work list.*
|
||||
`ServiceId` flag-day, restriction stage one, and the protocol rebase); P5
|
||||
(restriction stage two) is the remaining work.*
|
||||
|
||||
How a program finds, connects to, and is restricted from the things it talks to.
|
||||
Three ideas, kept deliberately separate:
|
||||
|
||||
@@ -23,8 +23,8 @@ INIT–SIPI–SIPI, brings each up into 64-bit long mode with its own descriptor
|
||||
LAPIC and timer, and drops it into the scheduler. Tasks run **genuinely in parallel** —
|
||||
the `smp` self-test confirms worker tasks executing on all four cores at once under
|
||||
QEMU `-smp 4`. Shared kernel state (scheduler queues, IPC) is serialised behind a big
|
||||
kernel lock. What's left is refinement, not first-light: per-core run queues, IPIs,
|
||||
and thread-to-core affinity (see [Implementation status](#implementation-status)).
|
||||
kernel lock. What's left is refinement, not first-light: per-core run queues and IPIs
|
||||
(see [Implementation status](#implementation-status)).
|
||||
|
||||
## The common microkernel instinct: don't share kernel state
|
||||
|
||||
@@ -238,10 +238,12 @@ next lands.
|
||||
- **Per-core run queues** — the Fiasco.OC direction, if the single global queue's lock
|
||||
contention ever bites. (Thread *affinity* already exists — see above; this is the
|
||||
further step of giving each core its own primary run queue for load distribution.)
|
||||
- **Fault recovery** — today a fault halts (only) the faulting core. Turning that into
|
||||
"kill the task, keep the core running" is the [resilience](resilience.md) track (it
|
||||
needs the task's lock/resource state handled), and for taking a core fully offline,
|
||||
its tasks migrated first.
|
||||
- **Fault recovery** — a ring-3 fault already **kills the faulting process and keeps the
|
||||
core (and the rest of the system) running**: the kernel trapped it on the task's own
|
||||
kernel stack, reclaims what the process held, and reschedules
|
||||
(`process.killCurrentProcess`; the `fault-recovery` test proves init keeps heartbeating
|
||||
through the kill) — the [resilience](resilience.md) track. What's still open here is
|
||||
taking a core fully **offline**, which additionally needs its tasks migrated off first.
|
||||
|
||||
## Further reading
|
||||
|
||||
|
||||
@@ -84,7 +84,7 @@ address space. Threads deliberately remove that boundary *within* a process:
|
||||
there is no isolation **between** threads.
|
||||
- Threads share fate — by contract: a fault in any thread, or a "kill the process"
|
||||
decision, takes down **all** of them, so restartability lives at the process level,
|
||||
not the thread level. (The kernel does not yet enforce this fan-out — see the
|
||||
not the thread level. (The kernel enforces this fan-out — see the
|
||||
Lifecycle note under
|
||||
[Interaction with the rest of the kernel](#interaction-with-the-rest-of-the-kernel).)
|
||||
- Shared mutable state reintroduces data races — the failure class the
|
||||
|
||||
@@ -36,9 +36,10 @@ and inside a VM alike; only the source behind it differs. The mechanism is in
|
||||
|
||||
So the timer hardware lives in the kernel, and there is **no `hpet` driver and no time
|
||||
server** to consume. (An earlier HPET driver existed only to *demonstrate* the driver
|
||||
model; that role now lives in [drivers.md](../device-driver-development/drivers.md), as documentation.) The one place
|
||||
a user-space time service *is* justified — **wall-clock / calendar time** — is discussed
|
||||
at the end; it is deliberately not built yet.
|
||||
model; that role now lives in [drivers.md](../device-driver-development/drivers.md), as documentation.) The one part
|
||||
left to user space — **calendar policy** over wall-clock time (time zones, formatting) —
|
||||
is discussed at the end; the wall-clock *seconds* it builds on are a kernel syscall
|
||||
(`wall_clock`), like the monotonic clock.
|
||||
|
||||
## The three system calls
|
||||
|
||||
@@ -95,15 +96,17 @@ The raw wrappers (`clock`, `sleepMillis`, `timerOnce`) and the ergonomic
|
||||
`Instant`/`Duration` layer both live in the `time` module
|
||||
(`library/kernel/time.zig`); the latter is what everyday code uses.
|
||||
|
||||
## Wall-clock time (not built)
|
||||
## Wall-clock time
|
||||
|
||||
Everything above is **monotonic**: elapsed time since boot, perfect for timeouts and
|
||||
measurement, useless for "what is the date?" Calendar time — a real-time clock, time
|
||||
zones, leap seconds — is genuinely a **user-space** concern, and it *is* the case a time
|
||||
service is for. It would be backed by an **RTC** driver (the CMOS real-time clock), not
|
||||
the HPET, and exposed as a `CLOCK_REALTIME`-style service alongside the monotonic
|
||||
syscall. It is deferred until something needs it; the monotonic clock the kernel already
|
||||
owns covers every current use.
|
||||
measurement, useless for "what is the date?" Calendar time needs a **real-time clock**.
|
||||
The kernel owns wall-clock *seconds* as mechanism, exactly like the monotonic clock: the
|
||||
`wall_clock` syscall (#33) returns Unix epoch seconds (UTC). The CMOS **RTC** is read
|
||||
once at boot and anchored to the monotonic clock (`system/kernel/wall-clock.zig`), so a
|
||||
query is a cheap arithmetic offset rather than a per-call CMOS poll; `time`'s
|
||||
`wallClock()` (`library/kernel/time.zig`) wraps it. Reading the hardware's value is not
|
||||
policy — time zones, leap seconds, calendars, and formatting layer on top in user space.
|
||||
It exists because the filesystem needs real timestamps (mtime).
|
||||
|
||||
## Verifying it
|
||||
|
||||
|
||||
@@ -123,12 +123,15 @@ The calls that return two values in `rax:rdx` today — `dma_alloc`
|
||||
`fs_resolve` (route tag + node token / backend handle) —
|
||||
become functions returning a two-`u64` struct. The System V ABI returns a
|
||||
16-byte struct in `rax:rdx`, so the stub is a plain `syscall; ret` — the
|
||||
C-ABI spelling of the existing convention, at zero cost. The one call that
|
||||
returns *three* values — `ipc_reply_wait` (receive_len in `rax`, badge in
|
||||
`rdx`, received capability in `r8`) — exceeds the two-register return: its
|
||||
function returns a three-`u64` struct, which the ABI passes via a hidden
|
||||
result pointer, so that one stub stores `rax`/`rdx`/`r8` through the pointer
|
||||
after the `syscall` — a few instructions rather than one.
|
||||
C-ABI spelling of the existing convention, at zero cost. The calls that
|
||||
return *three* values — `ipc_reply_wait` (receive_len in `rax`, badge in
|
||||
`rdx`, received capability in `r8`) and `dma_alloc` when the region is
|
||||
`shareable` (virtual in `rax`, physical in `rdx`, handle in `r8`) — exceed the
|
||||
two-register return: their functions return a three-`u64` struct, which the
|
||||
ABI passes via a hidden result pointer, so each stub stores `rax`/`rdx`/`r8`
|
||||
through the pointer after the `syscall` — a few instructions rather than one.
|
||||
(`ipc_call` likewise carries a received capability in `r8` alongside its `rax`
|
||||
result.)
|
||||
|
||||
Grouped as `abi.zig` groups them:
|
||||
|
||||
|
||||
@@ -143,8 +143,8 @@ Python shell uses it.
|
||||
The kernel/VFS cluster a shell forces (any shell, any language):
|
||||
|
||||
- **exec-of-path** — spawn an arbitrary VFS path, not a named ramdisk binary;
|
||||
- **argv/envp** — carried through spawn onto the child's entry stack (env from
|
||||
P1, argv new);
|
||||
- **argv/envp** — carried through spawn onto the child's entry stack (argv
|
||||
already built and tested via `spawnWithArguments`; envp new, from P1);
|
||||
- **numeric exit status** — extend the exit record beyond the categorical
|
||||
`ExitReason` (the gotcha the Zig roadmap flagged: `WEXITSTATUS` must be real);
|
||||
- **fd inheritance + pipes** — a kernel or service pipe (a character device by
|
||||
|
||||
@@ -0,0 +1,368 @@
|
||||
# Finishing the storage stack: the S1–S5 plan
|
||||
|
||||
*2026-08-09. Continues the volume-manager track (V0–V5, on main) from "one FAT
|
||||
volume" to "any filesystem, any number of volumes, identified by content,
|
||||
remounting where they belong, surviving a driver crash." Executes the settled
|
||||
design in
|
||||
[storage-architecture.md](file-system-development/storage-architecture.md) and
|
||||
[storage-design-rationale.md](file-system-development/storage-design-rationale.md).
|
||||
Track discipline as always: work on main; one commit per coherent step with
|
||||
`git commit -F` (no `-m`, no co-author trailer); every new test shown to FAIL
|
||||
against the old behavior; one QEMU suite at a time; CSV is configuration read by
|
||||
the volume manager (the policy), never itself policy; bounds discipline
|
||||
(`tools/check-bounds.py` gate); adversarial boundary review at each phase.*
|
||||
|
||||
## Where V0–V4 left it
|
||||
|
||||
The volume manager probes ONE storage device, parses its MBR (rung-4 identity =
|
||||
`(diskSignature<<8)|index`), spawns ONE FAT service confined to that partition's
|
||||
badge-scoped block range, supervises it, and unmounts it when the device is
|
||||
pulled. `partition.firstVolume` returns the FIRST partition; `openAnyStorage`
|
||||
adopts the FIRST device; `var volume: ?Volume` and `volume_id = 1` are singular;
|
||||
`filesystem_binary` and fat's mount prefixes are hardcoded; `medium_changed` is
|
||||
published by the driver but consumed by no one; remount-on-replug is
|
||||
bench-pending.
|
||||
|
||||
## The five phases and how they depend
|
||||
|
||||
```
|
||||
S1 identity ladder ──► S2 mount map ──► S3 multi-volume ──► S4 exFAT
|
||||
(needs S2+S3)
|
||||
S5 removal robustness ── independent; single-volume ── may land any time
|
||||
```
|
||||
|
||||
- **S1** grows the identity read off the medium (GPT GUID, FAT serial+label). It
|
||||
comes FIRST because a volume's mount name is now its identity (below), and the
|
||||
friendly form of that name is the FAT label / GPT name S1 parses.
|
||||
- **S2** moves the last policy out of hardcode into `volumes.csv` +
|
||||
`filesystems.csv`, and names each volume by its identity **id** (a GUID/key),
|
||||
keeping the label as separate, queryable display metadata — no port-name.
|
||||
- **S3** generalizes to N volumes across N devices.
|
||||
- **S4** adds exFAT — a COMPLETE second engine that proves the V1 harness
|
||||
extraction. Needs S2 (to route by signature) and S3 (to run a second volume).
|
||||
- **S5** closes the removal-lifecycle gaps. Independent of the rest; single-volume.
|
||||
|
||||
Recommended build order is S1 → S2 → S3 → S4 → S5. S5 may be pulled earlier.
|
||||
|
||||
---
|
||||
|
||||
## S1 — the identity ladder
|
||||
|
||||
**Goal.** Grow `system/services/volume-manager/partition.zig` from the single
|
||||
rung-4 identity into a ladder that reads the richest available content identity:
|
||||
GPT partition GUID (rung 1, 128-bit), FAT volume serial + label (rung 3), MBR
|
||||
signature + index (rung 4, kept), bare-FAT (kept, enriched to its serial). The
|
||||
`u64` identity becomes a small tagged struct `Identity{ rung, key: u128, label,
|
||||
has_label }` — `key` is the **id** (the path handle), `label` is the **display
|
||||
name** (the GPT 36-char partition name, or the FAT volume label), a separate
|
||||
field per the id/name split S2 relies on. Because GPT metadata is at LBA 1 and
|
||||
the entry array beyond it, and
|
||||
the FAT serial is in each partition's VBR, `firstVolume` stops taking one
|
||||
preloaded block-0 slice and takes a `SectorReader` (context + read-one-sector
|
||||
fn, mirroring the engine's `BlockDevice` vtable) — host-testable against a
|
||||
RAM-disk reader exactly as the four existing `partition.zig` tests are.
|
||||
|
||||
**Key touchpoints.** `partition.zig` (the `Rung`/`Identity`/`SectorReader` types;
|
||||
`gptFirstVolume`; `fatIdentity`; `firstVolume` control flow — GPT authoritative,
|
||||
else MBR walk skipping type-0xEE, else bare-FAT, each preferring the FAT serial
|
||||
over the disk signature); `volume-manager.zig` (`Volume.identity` type; a
|
||||
`ProbeReader` over the existing 512-byte bounce; the probe log prints
|
||||
`identity.key`); `build.zig` (add `b.dependency("volume-manager", .{})` to the
|
||||
package-test aggregation loop so the host fixtures run under root `zig build
|
||||
test`). Declared bound `gpt_entry_scan_maximum = 128` with the full bounds block;
|
||||
`sector_bytes`/`fat_label_bytes` named consts.
|
||||
|
||||
**Steps (commits).** (1) The SectorReader/Identity flag-day — pure refactor, no
|
||||
new behavior, all existing tests green. (2) GPT parsing (rung 1) — protective-MBR
|
||||
+ `EFI PART` signature + header CRC-32 + per-entry overflow-safe range
|
||||
validation (the confinement-safety invariant the driver's clamp rests on,
|
||||
extended to GPT). (3) FAT serial + label (rung 3), preferred over the disk
|
||||
signature; `Identity.eql`. (4) Test wiring, `check-bounds.py`, full suite, docs,
|
||||
memory.
|
||||
|
||||
**Discrimination.** Host: a GPT disk yields `rung==.gpt_guid` + the exact GUID
|
||||
key (old code walks the 0xEE protective entry as an ordinary partition); a GPT
|
||||
entry past the device is skipped, an out-of-device-only GPT returns null (the
|
||||
security-boundary guard); an invalid GPT header is not a volume; a bare FAT
|
||||
reports its real serial `0x12345678` not the rung-4 pseudo-signature; an
|
||||
MBR+FAT partition prefers the serial over the disk signature. On-image: the
|
||||
`volume-probe` QEMU regex tightens to `volume 0x0*12345678` — the boot image's
|
||||
real FAT32 serial reaches the running log.
|
||||
|
||||
**Top risks.** Adversarial GPT input from untrusted media (huge entry counts,
|
||||
bogus offsets, overflowing ranges) — mitigated by header CRC + size bounds +
|
||||
`gpt_entry_scan_maximum` + per-entry overflow-safe validation under boundary
|
||||
review. The `std.hash.crc` symbol in Zig 0.16 is unverified (fallback: a ~15-line
|
||||
reflected CRC-32, poly `0xEDB88320`, used by both parser and fixtures so they
|
||||
never drift onto magic constants).
|
||||
|
||||
---
|
||||
|
||||
## S2 — the mount map: volumes.csv + filesystems.csv
|
||||
|
||||
**Goal.** Move the last two pieces of storage policy out of hardcode into
|
||||
configuration read by the volume manager, and split a volume's **id** from its
|
||||
**label** (the database model: the id is the real, stable, unique key software
|
||||
uses; the label is a mutable display name). `filesystems.csv` (content signature
|
||||
→ filesystem binary) so the VM picks the binary from the probed signature;
|
||||
`volumes.csv` (id → mount prefix, danos's fstab) as the explicit override for a
|
||||
volume the user wants at a fixed path.
|
||||
|
||||
**The mount path IS the identity id, never a label or a port/role name.** A
|
||||
volume mounts at `/volumes/<id>` — the GPT partition GUID for a GPT volume, a
|
||||
`fat-<serial>` / `mbr-<sig>-<index>` form otherwise (exact rendering is an impl
|
||||
detail; the point is a stable, unique, content-derived string). Because the path
|
||||
is the id and never the label, two distinct volumes that happen to share a label
|
||||
(`Backup`, `UNTITLED`, unlabeled) get distinct paths automatically and never
|
||||
collide; only genuinely identical *ids* (dd-cloned media) hit the rationale's
|
||||
duplicate-identity policy (first mounts, second suffixed and logged).
|
||||
|
||||
**The label is display metadata, exposed by a protocol query, not the path.** S1
|
||||
reads it off the medium (FAT volume label, GPT partition name); a volume-manager
|
||||
`volumes` verb returns `{ id, mount_path, label }` per volume so a future
|
||||
shell/UI can show the friendly name — the id↔name split, like a table's primary
|
||||
key vs its display column. (`volumes.csv` may optionally carry a chosen label
|
||||
override alongside the path override, both keyed on id.) There is no
|
||||
`/volumes/usb` and no `/volumes/boot`; the boot volume is detected by content (it
|
||||
installs the `/system/configuration` + `/system/logs` rewrites) but is named by
|
||||
its id like any other.
|
||||
|
||||
Parsed with `library/csv` exactly as `device-registry` parses `devices.csv`. The
|
||||
VM hands the binary + volume-id + mount specs to fat at spawn (the argv channel
|
||||
V3b already uses for the volume id); fat retires `fat_mounts` and reads mounts
|
||||
from argv[2..].
|
||||
|
||||
**Key touchpoints.** New pure-logic modules `filesystem-map.zig` (parse +
|
||||
`match(signature)`) and `volume-map.zig` (parse + `mountsFor(identity)` +
|
||||
`derivedAnonymous`), mirroring `device-registry`, host-tested; `partition.zig`
|
||||
`Volume` gains a `signature`; `volume-manager.zig` loads both tables in
|
||||
`initialise`, resolves binary + mounts, spawns the chosen binary with the mount
|
||||
argv; `fat.zig` deletes `fat_mounts`, parses argv[2..] into a bounded
|
||||
`MountSpec` array; new `system/configuration/filesystems.csv` +
|
||||
`volumes.csv`; `build.zig` bundles them; `make-fat-image.py` writes a real 4-byte
|
||||
MBR disk signature so the boot volume's identity is a legible non-zero key.
|
||||
|
||||
**Steps (commits).** (1) partition emits a signature. (2) `filesystem-map`
|
||||
parser + host tests. (3) `volume-map` parser (id→prefix override) + the id-path
|
||||
deriver + the `volumes` label-query verb + host tests. (4) Ship the tables +
|
||||
VM integration **behind fat's still-hardcoded mounts** (behavior-preserving —
|
||||
parsers proven before consumption flips; full suite green). (5) fat consumes
|
||||
argv mounts; the VM mounts the boot volume at its id-path (`/volumes/fat-12345678`,
|
||||
from its serial); **migrate the fixtures + QEMU regexes off `/volumes/usb`**
|
||||
(fat-test, badge-scope-test, vfs-test, and the four cases) in the same commit;
|
||||
the `volume-identity-name` QEMU case (shown failing against HEAD~1); flip the
|
||||
docs' pending markers.
|
||||
|
||||
**Discrimination.** QEMU `volume-identity-name`: the boot volume mounts at its
|
||||
id-path (`/volumes/fat-12345678`, from its serial), a line the old hardcoded
|
||||
`/volumes/usb` never emits; the `volumes` query returns that id paired with the
|
||||
label `DANOS`. Host: two volumes sharing a label but not an id get distinct
|
||||
id-paths (the collision the label-as-name approach could not resolve stably); a
|
||||
`volumes.csv` override sends a mapped id to its chosen prefix; `match(.fat)`
|
||||
returns the configured binary; fat's argv parser makes installed mounts a
|
||||
function of argv.
|
||||
|
||||
**Top risks.** Step 5's blast radius — a parser/argv bug, OR the `/volumes/usb`→
|
||||
id-path migration missing a fixture/regex, breaks every fat-dependent case at
|
||||
once; mitigated by landing the VM half behind fat's hardcoded mounts first (step
|
||||
4) and making the name migration one atomic, complete sweep. The
|
||||
argv blob is 256 bytes (process.zig) — cap emitted mounts and refuse+log on
|
||||
overflow. `/system/logs` is now a `volumes.csv` concern: dropping the boot
|
||||
identity's rewrite rows silently stops log persistence — ship them by default
|
||||
and document the boot-identity contract in the CSV header.
|
||||
|
||||
---
|
||||
|
||||
## S3 — multi-volume
|
||||
|
||||
**Goal.** Generalize from `var volume: ?Volume` / `volume_id = 1` / first-device
|
||||
/ first-partition to a bounded table of volumes across a bounded table of
|
||||
devices. `partition.allVolumes` returns ALL partitions; the VM adopts EVERY
|
||||
mass-storage provider, probes each device's table, and for each partition spawns
|
||||
one FAT confined to that partition's range (the per-sender clamp is already
|
||||
built), with a distinct `/volumes/<name>` and its OWN backoff/crash-loop state.
|
||||
Removal is per-device. The **boot volume is identified by content** — the FAT
|
||||
process installs the `/system/configuration` + `/system/logs` rewrites only when
|
||||
its own volume resolves `/system/configuration` — so it works as the 2nd
|
||||
partition of the 2nd device just as the 1st of the 1st. (This clarifies the
|
||||
initrd relationship: the kernel already serves `/system/configuration` +
|
||||
binaries read-only from the initrd, which is what lets danos boot with NO volume
|
||||
mounted; a mounted boot volume only adds writable, persistent
|
||||
`/system/configuration` + `/system/logs` that shadow the initrd via
|
||||
longest-prefix match.)
|
||||
|
||||
**Key touchpoints.** `partition.zig` `firstVolume` → `allVolumes(block0,
|
||||
device_blocks, out) usize` (per-entry overflow-safe skip preserved);
|
||||
`volume-manager.zig` the core refactor — `Volume` absorbs the file-global
|
||||
supervision state as per-volume fields, a `StorageDevice` table owns each
|
||||
adopted device's channel once, `volumes[maximum_volumes]` replaces the singleton,
|
||||
a monotonic `next_volume_id`, `openAllStorage`/`adoptAndProbe`,
|
||||
`gatherPresentStorage` + per-device reconcile in `pollTick`, `onHello`/
|
||||
`onNotification` keyed across the table; `fat.zig` content-conditional boot
|
||||
mounts; `system/kernel/vfs.zig` raise `maximum_mounts` (8 → 16) with a refreshed
|
||||
bounds annotation; `make-fat-image.py` a partition-table mode; `build/images.zig`
|
||||
the test disk artifacts.
|
||||
|
||||
**Steps (commits).** (1) `partition.allVolumes` + two-partition host test. (2)
|
||||
Tables, behavior-preserving (still one device / one volume). (3) Multi-device +
|
||||
multi-partition. (4) Per-volume mount naming via argv — each volume by its
|
||||
id-path (from S2), one per volume, no port-name. (5) fat content-conditional
|
||||
boot mounts. (6) Raise `maximum_mounts`. (7) Partitioned-image tool. (8)
|
||||
`two-volume` QEMU case. (9) `boot-2nd-partition` case. (10) Adversarial review,
|
||||
full suite, docs, memory.
|
||||
|
||||
**Discrimination.** Host: an MBR with two partitions yields two volumes with
|
||||
distinct identities (old `firstVolume` returns one). QEMU `two-volume`: a
|
||||
two-partition second device yields two mount lines at two base_lbas (old
|
||||
`openAnyStorage` adopts only the first device). `boot-2nd-partition`: the boot
|
||||
volume works as partition 2 (old code confines fat to partition 1, whose
|
||||
`/system` rewrite backs empty space). Per-volume supervision: killing one
|
||||
volume's FAT restarts only that one (old module-scope supervision can't
|
||||
attribute an exit to one of two).
|
||||
|
||||
**Top risks.** OVMF booting an MBR ESP on partition 2 may be flaky in CI —
|
||||
fallback to a content-detection-ordering assertion + bench-verified boot (the
|
||||
track's existing precedent). N-client range reclamation in usb-storage
|
||||
(`maximum_ranges=64`) must reclaim each of N confined pids' ranges — the V2
|
||||
mechanism, previously exercised with one live client. Duplicate boot volumes:
|
||||
S3 supports exactly one and must log loudly if a second also resolves the boot
|
||||
markers (arbitration deferred to S4).
|
||||
|
||||
---
|
||||
|
||||
## S4 — exFAT: the second engine
|
||||
|
||||
**Goal.** A working exFAT filesystem as `system/services/exfat` that is nothing
|
||||
but an engine + a `main`, reusing `library/kernel/file-system-harness.zig`'s
|
||||
`Server(Engine)` wholesale — the reuse claim the architecture makes, now proven.
|
||||
The harness already owns vfs serving, the badge-scoped open-node table,
|
||||
create-on-open/O_TRUNC, mount registration, the exit sweep, bring-up retry,
|
||||
per-turn time stamping, and durable-on-close. S4 writes only the exFAT-specific
|
||||
bits — but **in full**: complete read AND write, directories, rename, and the
|
||||
real on-disk up-case table for correct case-folding. Not a read-first, minimal-
|
||||
write, or ASCII-only subset. The only limit that survives is the vfs protocol's
|
||||
u32 file-offset surface (a 4 GiB addressable-size cap that applies to FAT too),
|
||||
which is a separate vfs-protocol change, not an exFAT shortcut.
|
||||
|
||||
**Key touchpoints.** New `system/services/exfat/on-disk.zig` (the Main Boot
|
||||
Sector VBR + the five 32-byte directory-entry types as `align(1)` extern structs;
|
||||
`geometryOf` accepting only `"EXFAT "` + `0xAA55`; `setChecksum`, `nameHash`,
|
||||
and case-folding driven by the volume's **on-disk up-case table**); `engine.zig` (`FileSystem` behind the identical `BlockDevice`
|
||||
vtable with fat's exact method set; **allocation-bitmap** cluster authority — the
|
||||
deepest departure from FAT; read honoring `no_fat_chain` contiguous vs FAT-follow;
|
||||
File+Stream+FileName set assembly with recomputed set checksum); `exfat.zig` (the
|
||||
thin service, a near-clone of fat.zig); build wiring + `service("exfat")`;
|
||||
`tools/make-exfat-image.py` (pure stdlib, correct boot checksum, up-case table —
|
||||
**no committed .img**); the `filesystems.csv` EXFAT row (S2) + the `"EXFAT "`
|
||||
recognizer; a `exfat-test` fixture cloned from fat-test.
|
||||
|
||||
**Steps (commits).** (1) on-disk.zig byte layout. (2) engine read path. (3)
|
||||
engine write path (bitmap allocate/free, real 32-bit FAT chain with
|
||||
`no_fat_chain=0`, set-checksum recompute). (4) service + build wiring. (5)
|
||||
`make-exfat-image.py` + image assembly. (6) Routing: `filesystems.csv` +
|
||||
recognizer. (7) `exfat-test` fixture + **cross-engine discrimination** host test.
|
||||
(8) In-VM lifecycle drill (second removable device; mount/mutations/removal). (9)
|
||||
Bounds, docs, adversarial review, memory.
|
||||
|
||||
**Discrimination.** The named one: `fat.mount(exfat_img) == null` (fat reads
|
||||
bytes-per-sector at VBR offset 11 = exFAT's MustBeZero = 0 → reject) AND
|
||||
`exfat.mount(fat_img) == null`, each mounting its own as a control. Host: read
|
||||
across a cluster boundary on both a contiguous and a fragmented file; write
|
||||
across >1 cluster setting the bitmap bits (not the FAT) and a validating set
|
||||
checksum. QEMU: `exfat: mounted /volumes/exfat` + `exfat-test: ok`;
|
||||
`exfat-removal` yanks the exFAT stick mid-write while the FAT boot volume keeps
|
||||
serving.
|
||||
|
||||
**Top risks.** Allocation authority is the bitmap, not the FAT — allocating
|
||||
without setting the bit silently corrupts free space (highest-attention area).
|
||||
A directory-entry SET can straddle sector/cluster boundaries — scanDirectory,
|
||||
set-checksum, and updateStreamEntry must handle multi-sector sets. vfs offsets
|
||||
are u32 while exFAT DataLength is u64 — clamp and document (as fat does). The
|
||||
in-VM drill needs S3 (a non-boot exFAT volume beside the FAT boot volume); if S4
|
||||
landed before S3 the discrimination would rest on host tests until multi-volume
|
||||
exists.
|
||||
|
||||
---
|
||||
|
||||
## S5 — removal robustness
|
||||
|
||||
**Goal.** Close the three known gaps so every removal trigger is exercised
|
||||
end-to-end. (1) **Consume `medium_changed`** — the VM subscribes to the driver's
|
||||
already-published event so the "device stays, medium leaves" case (a card
|
||||
reader, an ejected removable) runs the same kill-retire-remount path as a pulled
|
||||
stick, closing the second of the "two triggers, one lifecycle" the architecture
|
||||
specifies. (2) **Storage-driver-crash rebuild** — a driver that dies while its
|
||||
device stays present is detected and the volume subtree rebuilt on the restarted
|
||||
driver's fresh channel, instead of leaving fat wedged on a dead channel (the V4
|
||||
review's open edge). (3) **QEMU-verified remount-on-replug** — the device-return
|
||||
half is proven, not merely asserted-unmount. Single-volume; independent of S1–S4.
|
||||
|
||||
**Key touchpoints.** `library/kernel/service.zig` an additive, behavior-neutral
|
||||
`on_buffered_message` callback so a buffered-message wake forwards its payload
|
||||
(no existing service sets it); `volume-manager.zig` subscribe on `bringUpVolume`
|
||||
success, `onMediumEvent` with change-count dedupe running a medium-teardown (with
|
||||
`encodeUnsubscribe` before close so the driver's 8-slot table doesn't leak), plus
|
||||
`channelAlive()` (a `geometry()` liveness probe) + `rebuildVolume()` used in
|
||||
`pollTick` and the child-exit path; `fat.zig` re-probe geometry on I/O failure
|
||||
and exit on channel death (device NAK keeps serving); `device-manager.zig` a
|
||||
`test-storage-restart` mode (mirroring `test-scanout-restart`) to kill usb-storage
|
||||
once, post-mount, as the discrimination trigger.
|
||||
|
||||
**Steps (commits).** (1) **Cheap decisive experiments first** (no commits): QMP-
|
||||
eject the boot medium and confirm `usb-storage: medium absent` fires under QEMU
|
||||
(the whole item-1 chain depends on it); and test whether a boot-controller
|
||||
`device_add` is re-presented (settles whether item 3 extends `volume-removal` or
|
||||
needs a second controller as H1 does). (2) Harness `on_buffered_message`
|
||||
(behavior-neutral). (3) VM consumes `medium_changed`. (4) `volume-medium-change`
|
||||
case (fails pre-step-3). (5) VM driver-crash rebuild. (6) fat observes dead
|
||||
channel and exits. (7) `volume-driver-restart` trigger + case. (8) `volume-replug`
|
||||
(second controller if needed). (9) Docs + the real-hardware bench protocol. (10)
|
||||
Full suite + memory.
|
||||
|
||||
**Discrimination.** `volume-medium-change`: an eject with the device left in the
|
||||
tree unmounts (old VM never subscribes → the event goes to no one → mount
|
||||
persists). `volume-driver-restart`: killing usb-storage post-mount while its
|
||||
child stays present triggers a rebuild and a SECOND mount + post-kill read (old
|
||||
`pollTick` only checks `isDevicePresent`, still true, and restarts fat against
|
||||
the stale channel → wedge/crash-loop). `volume-replug`: a device return on a
|
||||
second controller drives a remount (the existing case only ever sees the unmount
|
||||
half).
|
||||
|
||||
**Top risks.** QEMU medium-eject must make TEST UNIT READY report not-ready —
|
||||
step 1(a) validates this before any code. The op-16 overlap (`medium_changed` ==
|
||||
`hello` by number) is safe only because async events arrive as `isMessage`
|
||||
notifications and never reach `Serve.dispatch` — the intercept must run in the
|
||||
notification branch and never catch a synchronous hello. fat can't today
|
||||
distinguish EPEER from a device NAK (`CallError` swallows the errno) — the plan
|
||||
uses a geometry re-probe as the liveness oracle, which is correct but indirect.
|
||||
|
||||
---
|
||||
|
||||
## Decisions (settled)
|
||||
|
||||
Both flagged decisions are settled:
|
||||
|
||||
1. **A volume's path is its id; the label is display metadata (S1/S2).** The
|
||||
mount path is the identity id — the GPT GUID, else a `fat-<serial>` /
|
||||
`mbr-<sig>-<index>` form — a stable, unique, content-derived handle software
|
||||
uses. The label (FAT volume label / GPT partition name) is a mutable display
|
||||
name, NOT in the path; a volume-manager `volumes` verb returns
|
||||
`{ id, mount_path, label }` so a UI can show the friendly name (the database
|
||||
id/name split). Same-label-different-id volumes therefore never collide; only
|
||||
identical ids (dd-clones) hit first-wins-and-log. `volumes.csv` overrides the
|
||||
path (and optionally the label) for a chosen volume, keyed on id. Makes S1
|
||||
precede S2 and folds the `/volumes/usb` → id-path fixture + regex migration
|
||||
into S2 step 5.
|
||||
|
||||
2. **exFAT is implemented in full (S4).** A complete exFAT: full read and write,
|
||||
directories, rename, and the on-disk up-case table for correct case-folding —
|
||||
not a read-first or ASCII-only subset. The one remaining limit is the vfs
|
||||
protocol's u32 file-offset surface, which caps addressable file size at 4 GiB
|
||||
for ALL filesystems (FAT included); widening it to u64 is a separate vfs-
|
||||
protocol change, flagged but out of the exFAT engine's scope.
|
||||
|
||||
The ~26 smaller design-time questions are settled with the recommended default
|
||||
in the phase text (defer GPT entry-array CRC to correctness-only; GUID key =
|
||||
little-endian u128 pinned now; share the DOS date-time helper into a library
|
||||
module both engines import; a second removable usb-storage device for the exFAT
|
||||
drill; VM-poll `channelAlive()` as the load-bearing crash-detection guarantee).
|
||||
@@ -12,8 +12,10 @@ range confinement at the provider on one serving endpoint — the badge-scoped
|
||||
provider pattern the xHCI bus already uses, applied to blocks. The volume
|
||||
manager sets each filesystem process's range; the driver clamps and translates
|
||||
every transfer by the sender's kernel-stamped badge; filesystems address
|
||||
volume-relative LBAs from 0 and the FAT engine's `base_lba` is deleted rather
|
||||
than moved. Every other decision the phases below execute is recorded in the
|
||||
volume-relative LBAs from 0 (on the confined path the FAT engine's `base_lba`
|
||||
resolves to 0; the field and the engine's own MBR walk remain as now-inert
|
||||
legacy, the authoritative walk living in the volume manager's `partition.zig`).
|
||||
Every other decision the phases below execute is recorded in the
|
||||
rationale (decisions 1–8); nothing in this plan waits on a choice.
|
||||
|
||||
## V0 — `fs_unmount` ownership (the defect fix)
|
||||
|
||||
@@ -151,8 +151,9 @@ Its footprint was tiny: **five** call sites, all `unistd` file operations —
|
||||
(from the boot-log work) `init.zig` and `log-flush.zig`. `stdio.zig` was dead — nothing
|
||||
imported it. The plan — build `runtime.fs`, migrate those five to it, delete
|
||||
`library/posix/`, and drop the `posix` module from `build.zig`'s `addUserBinary` — has
|
||||
since been carried out: `library/` today holds only `mmio`, `runtime`, and
|
||||
`xkeyboard-config`.
|
||||
since been carried out: `library/` today holds `client`, `csv`, `device`,
|
||||
`kernel`, `protocol`, and `xkeyboard-config` (the `runtime` namespace was later
|
||||
reorganized into `library/kernel`, and `mmio` moved under `library/device`).
|
||||
|
||||
## Where danos stands: coverage vs. the gaps
|
||||
|
||||
|
||||
@@ -7,12 +7,25 @@
|
||||
//! `runtime.dma.alloc`), so whole sectors move without crossing the IPC size
|
||||
//! limit — the same handoff usb-storage uses toward the controller.
|
||||
|
||||
const std = @import("std");
|
||||
const envelope = @import("envelope");
|
||||
const ipc = @import("ipc");
|
||||
const block_protocol = @import("block-protocol");
|
||||
|
||||
const Protocol = block_protocol.Protocol;
|
||||
|
||||
/// The medium_changed event payload, re-exported so a consumer decodes it without
|
||||
/// reaching into the wire-format module.
|
||||
pub const MediumChanged = block_protocol.MediumChanged;
|
||||
|
||||
/// Decode a medium_changed event from a buffered-message payload a subscriber
|
||||
/// received (a `Received.isMessage` wake). Null if the bytes are too short to be
|
||||
/// one — a caller ignores anything that is not a well-formed event.
|
||||
pub fn decodeMediumChanged(payload: []const u8) ?MediumChanged {
|
||||
if (payload.len < envelope.prefix_size + @sizeOf(MediumChanged)) return null;
|
||||
return std.mem.bytesToValue(MediumChanged, payload[envelope.prefix_size..][0..@sizeOf(MediumChanged)]);
|
||||
}
|
||||
|
||||
pub const Geometry = struct { block_size: u32, block_count: u64 };
|
||||
|
||||
pub const Device = struct {
|
||||
@@ -88,6 +101,30 @@ pub const Device = struct {
|
||||
if (status.status != 0) return null;
|
||||
return reply[0..answer.len];
|
||||
}
|
||||
|
||||
/// Subscribe `subscriber` (an endpoint) to this device's medium_changed
|
||||
/// events: the reserved `subscribe` verb carries the subscriber's endpoint as
|
||||
/// the capability, and the driver then ipc.sends each medium transition to it.
|
||||
pub fn subscribeMedium(self: Device, subscriber: ipc.Handle) bool {
|
||||
var packet: [block_protocol.message_maximum]u8 = undefined;
|
||||
const framed = envelope.encodeSubscribe(0, &packet) orelse return false; // interest 0: every event (block has one)
|
||||
var reply: [block_protocol.message_maximum]u8 = undefined;
|
||||
const answer = ipc.callCap(self.endpoint, framed, &reply, subscriber) catch return false;
|
||||
const status = envelope.statusOf(reply[0..answer.len]) orelse return false;
|
||||
return status.status == 0;
|
||||
}
|
||||
|
||||
/// Unsubscribe from this device's medium_changed events. Call before closing
|
||||
/// the channel so the driver's bounded subscriber table frees the slot rather
|
||||
/// than holding a dead endpoint until an exit sweep notices.
|
||||
pub fn unsubscribeMedium(self: Device) bool {
|
||||
var packet: [block_protocol.message_maximum]u8 = undefined;
|
||||
const framed = envelope.encodeUnsubscribe(&packet) orelse return false;
|
||||
var reply: [block_protocol.message_maximum]u8 = undefined;
|
||||
const answer = ipc.callCap(self.endpoint, framed, &reply, null) catch return false;
|
||||
const status = envelope.statusOf(reply[0..answer.len]) orelse return false;
|
||||
return status.status == 0;
|
||||
}
|
||||
};
|
||||
|
||||
// There is deliberately no open-by-name here: `block` is not a registry name.
|
||||
|
||||
@@ -61,10 +61,14 @@ pub fn Server(comptime Engine: type) type {
|
||||
/// caller does the filesystem-specific bring-up (find the block
|
||||
/// device, set up DMA, mount the engine) and returns a `Volume`.
|
||||
bringUp: *const fn (endpoint: ipc.Handle) ?Volume,
|
||||
/// The vfs contract name to bind. A filesystem serving one volume
|
||||
/// binds "vfs" today; the volume-manager era hands each per-volume
|
||||
/// process its own establishment and this fades.
|
||||
service_name: ?[]const u8 = "vfs",
|
||||
/// A contract name to bind under /protocol, or null to bind none. In
|
||||
/// the volume-manager era every filesystem is a per-volume process and
|
||||
/// clients reach it through the kernel mount table — fs_resolve routes
|
||||
/// a path to its backing endpoint by prefix — so no filesystem binds a
|
||||
/// shared name. Two volumes would collide on one: the second's bind is
|
||||
/// refused and service.run would exit, so its volume never mounts. The
|
||||
/// endpoint still serves as the mount backend without a name.
|
||||
service_name: ?[]const u8 = null,
|
||||
};
|
||||
|
||||
// --- the harness's own state, one set per instantiation ---------------
|
||||
|
||||
@@ -67,6 +67,15 @@ pub const Callbacks = struct {
|
||||
/// A notification that is not a signal — a subscribed exit event, a bound
|
||||
/// IRQ, a timer landing. The raw badge; decode with the ipc helpers.
|
||||
on_notification: ?*const fn (badge: u64) void = null,
|
||||
/// A buffered async message (`Received.isMessage`): a pushed event from a
|
||||
/// provider this service subscribed to, its payload in the receive buffer.
|
||||
/// Unlike `on_message`, it never goes through the protocol dispatch — so an
|
||||
/// event whose reserved op number collides with one of this service's own
|
||||
/// verbs (a `block` `medium_changed` reaching the volume manager, whose own
|
||||
/// protocol numbers `hello` the same) is decoded by hand here, not
|
||||
/// mis-dispatched. Default null: the badge alone still reaches
|
||||
/// `on_notification`, exactly as before this callback existed.
|
||||
on_buffered_message: ?*const fn (message: []const u8) void = null,
|
||||
/// The reload signal. Default: ignored.
|
||||
on_reload: ?*const fn () void = null,
|
||||
/// The terminate signal, called before the loop returns. The clean exit is
|
||||
@@ -372,6 +381,13 @@ pub fn run(comptime maximum_message: usize, callbacks: Callbacks) void {
|
||||
if (got.isChildExit()) {
|
||||
if (callbacks.subscribers) |subscribers| subscribers.forget(got.childProcessId());
|
||||
}
|
||||
// A buffered async message (a pushed event) carries a payload; hand it
|
||||
// to the service that asked for it. The badge still reaches
|
||||
// on_notification below, so a coalesced timer/exit riding the same wake
|
||||
// is not lost — and a service without this callback is unchanged.
|
||||
if (got.isMessage()) {
|
||||
if (callbacks.on_buffered_message) |onBuffered| onBuffered(receive[0..got.len]);
|
||||
}
|
||||
if (callbacks.on_notification) |onNotification| onNotification(got.badge);
|
||||
continue;
|
||||
}
|
||||
|
||||
@@ -212,8 +212,9 @@ test "a tree report fits the push floor with the header folded in" {
|
||||
try std.testing.expectEqual(@as(usize, 48), @sizeOf(ChildAdded));
|
||||
try std.testing.expectEqual(envelope.post_maximum, Protocol.event_maximum);
|
||||
try std.testing.expect(Protocol.event_maximum <= envelope.post_maximum);
|
||||
// Ten records per enumerate reply — what the old count-header layout carried.
|
||||
try std.testing.expectEqual(@as(usize, 10), entries_per_reply);
|
||||
// Seven records per enumerate reply: (packet_maximum 256 - prefix 16) / 32.
|
||||
// (The old count-header layout carried ten; this asserts the current shape.)
|
||||
try std.testing.expectEqual(@as(usize, 7), entries_per_reply);
|
||||
}
|
||||
|
||||
test "the verb and event numbering, and the device id in the header" {
|
||||
|
||||
@@ -12,6 +12,7 @@
|
||||
//! "Establishment: two planes"). No channel in the reply means the volume is not
|
||||
//! ready yet — retryable, never a verdict.
|
||||
|
||||
const std = @import("std");
|
||||
const envelope = @import("envelope");
|
||||
|
||||
pub const version: u16 = 1;
|
||||
@@ -24,13 +25,91 @@ pub const Hello = extern struct {
|
||||
_padding: u16 = 0,
|
||||
};
|
||||
|
||||
/// A `volumes` query — no request fields; the reply's tail carries the volume's
|
||||
/// descriptor (`VolumeInfo`). The mechanism by which a shell or file manager
|
||||
/// reads a volume's display label: the mount path is its id (software's stable
|
||||
/// handle), the label is separate display metadata, the database id/name split.
|
||||
pub const Volumes = extern struct {
|
||||
_reserved: u32 = 0,
|
||||
};
|
||||
|
||||
/// The `volumes` reply: three length-prefixed strings packed into the reply tail
|
||||
/// — the volume's id (its mount path is /volumes/<id> unless overridden), its
|
||||
/// actual mount path, and its display label. `id` is what software keys on;
|
||||
/// `label` is what a UI shows.
|
||||
pub const VolumeInfo = struct {
|
||||
id: []const u8,
|
||||
mount_path: []const u8,
|
||||
label: []const u8,
|
||||
|
||||
const header_bytes = 6; // three u16 lengths, little-endian
|
||||
|
||||
/// Pack into `buf`, returning the used slice, or null if it does not fit.
|
||||
pub fn encode(self: VolumeInfo, buf: []u8) ?[]u8 {
|
||||
const total = header_bytes + self.id.len + self.mount_path.len + self.label.len;
|
||||
if (total > buf.len) return null;
|
||||
std.mem.writeInt(u16, buf[0..2], @intCast(self.id.len), .little);
|
||||
std.mem.writeInt(u16, buf[2..4], @intCast(self.mount_path.len), .little);
|
||||
std.mem.writeInt(u16, buf[4..6], @intCast(self.label.len), .little);
|
||||
var off: usize = header_bytes;
|
||||
@memcpy(buf[off..][0..self.id.len], self.id);
|
||||
off += self.id.len;
|
||||
@memcpy(buf[off..][0..self.mount_path.len], self.mount_path);
|
||||
off += self.mount_path.len;
|
||||
@memcpy(buf[off..][0..self.label.len], self.label);
|
||||
return buf[0..total];
|
||||
}
|
||||
|
||||
/// Decode a reply tail, or null if it is malformed (short or inconsistent).
|
||||
/// The returned slices point into `bytes`.
|
||||
pub fn decode(bytes: []const u8) ?VolumeInfo {
|
||||
if (bytes.len < header_bytes) return null;
|
||||
const id_len = std.mem.readInt(u16, bytes[0..2], .little);
|
||||
const path_len = std.mem.readInt(u16, bytes[2..4], .little);
|
||||
const label_len = std.mem.readInt(u16, bytes[4..6], .little);
|
||||
const total = header_bytes + @as(usize, id_len) + path_len + label_len;
|
||||
if (total > bytes.len) return null;
|
||||
var off: usize = header_bytes;
|
||||
const id = bytes[off..][0..id_len];
|
||||
off += id_len;
|
||||
const mount_path = bytes[off..][0..path_len];
|
||||
off += path_len;
|
||||
const label = bytes[off..][0..label_len];
|
||||
return .{ .id = id, .mount_path = mount_path, .label = label };
|
||||
}
|
||||
};
|
||||
|
||||
pub const Protocol = envelope.Define(.{
|
||||
.name = "volume-manager",
|
||||
.version = 1,
|
||||
.operations = &.{
|
||||
.{ .name = "hello", .request = Hello },
|
||||
.{ .name = "volumes", .request = Volumes },
|
||||
},
|
||||
});
|
||||
|
||||
pub const Operation = Protocol.Operation;
|
||||
pub const message_maximum: usize = Protocol.message_maximum;
|
||||
|
||||
// Named fixture sizes so the bounds gate (which flags literal array lengths)
|
||||
// stays quiet: test inputs, not runtime ceilings.
|
||||
const test_reply_bytes = 128;
|
||||
const test_tiny_bytes = 4;
|
||||
|
||||
test "VolumeInfo round-trips id, mount_path, and label" {
|
||||
var buf: [test_reply_bytes]u8 = undefined;
|
||||
const info = VolumeInfo{ .id = "fat-12345678", .mount_path = "/volumes/fat-12345678", .label = "DANOS" };
|
||||
const encoded = info.encode(&buf).?;
|
||||
const back = VolumeInfo.decode(encoded).?;
|
||||
try std.testing.expectEqualStrings("fat-12345678", back.id);
|
||||
try std.testing.expectEqualStrings("/volumes/fat-12345678", back.mount_path);
|
||||
try std.testing.expectEqualStrings("DANOS", back.label);
|
||||
}
|
||||
|
||||
test "VolumeInfo encode refuses a buffer that is too small; decode rejects a short tail" {
|
||||
var tiny: [test_tiny_bytes]u8 = undefined;
|
||||
const info = VolumeInfo{ .id = "fat-1", .mount_path = "/volumes/fat-1", .label = "" };
|
||||
try std.testing.expect(info.encode(&tiny) == null);
|
||||
try std.testing.expect(VolumeInfo.decode(&[_]u8{ 0, 0, 0 }) == null); // shorter than the header
|
||||
try std.testing.expect(VolumeInfo.decode(&[_]u8{ 0xFF, 0xFF, 0, 0, 0, 0 }) == null); // claims 65535 id bytes
|
||||
}
|
||||
|
||||
@@ -0,0 +1,8 @@
|
||||
# The filesystem map: a probed volume's content signature -> the service binary
|
||||
# that serves it (docs/file-system-development/storage-architecture.md). The
|
||||
# volume manager reads this (the policy); a signature no row matches goes
|
||||
# unserved, never guessed. Adding a filesystem adds a row.
|
||||
#
|
||||
# signature, binary
|
||||
fat, /system/services/fat
|
||||
exfat, /system/services/exfat
|
||||
|
@@ -79,6 +79,7 @@
|
||||
# 'kernel' as the supervisor. Nothing else changes: the binary must still match.
|
||||
/system/services/input, kernel, bind, input
|
||||
/system/services/device-manager, kernel, bind, device-manager
|
||||
/system/services/volume-manager, kernel, bind, volume-manager
|
||||
/system/services/fat, kernel, bind, vfs
|
||||
/system/services/display, kernel, bind, display
|
||||
/system/services/discovery, kernel, bind, power
|
||||
@@ -102,9 +103,14 @@
|
||||
# own endpoint (the mouse-listener thread opens /protocol/display like any other
|
||||
# client — threads share no handles), and the input stream that moves the cursor.
|
||||
/system/services/fat, /system/services/volume-manager, open, volume-manager
|
||||
# exfat reaches the volume manager the same way — the second engine, same lineage.
|
||||
/system/services/exfat, /system/services/volume-manager, open, volume-manager
|
||||
# The volume manager reaches the device manager to be routed to each storage
|
||||
# provider's block channel, then confines a filesystem to each volume.
|
||||
/system/services/volume-manager, /system/services/init, open, device-manager
|
||||
# ...and again under the kernel supervisor for the manual-tree drills (S5's
|
||||
# volume-driver-restart spawns the volume manager directly, not via init).
|
||||
/system/services/volume-manager, kernel, open, device-manager
|
||||
/system/services/display, /system/services/init, open, scanout
|
||||
/system/services/display, /system/services/init, open, display
|
||||
/system/services/display, /system/services/init, open, input
|
||||
|
||||
|
Can't render this file because it contains an unexpected character in line 12 and column 15.
|
@@ -0,0 +1,8 @@
|
||||
# The mount map (danos's fstab): a volume's content id -> a chosen mount prefix.
|
||||
# This is an OPTIONAL override, read by the volume manager. A volume with no row
|
||||
# mounts at its default /volumes/<id>, where <id> is the manager's rendered
|
||||
# content identity (e.g. fat-12345678, gpt-<guid>, mbr-<sig>-<index>) — stable,
|
||||
# unique, and never a port or a label. The label is display metadata, not here:
|
||||
# query it via the volume manager's `volumes` verb.
|
||||
#
|
||||
# id, mount_prefix
|
||||
|
@@ -227,6 +227,10 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void {
|
||||
usbStorageTest(boot_information);
|
||||
} else if (eql(case, "fat-mount")) {
|
||||
fatMountTest(boot_information);
|
||||
} else if (eql(case, "exfat-volume")) {
|
||||
exfatVolumeTest(boot_information);
|
||||
} else if (eql(case, "volume-driver-restart")) {
|
||||
volumeDriverRestartTest(boot_information);
|
||||
} else if (eql(case, "device-list")) {
|
||||
deviceListTest(boot_information);
|
||||
} else if (eql(case, "pci-scan")) {
|
||||
@@ -3036,6 +3040,32 @@ fn fatMountTest(boot_information: *const BootInformation) void {
|
||||
result();
|
||||
}
|
||||
|
||||
/// The exFAT mount chain (S4): boot the full tree, which brings up the USB storage
|
||||
/// chain. The harness attaches a SECOND device — a data-only exFAT volume — beside
|
||||
/// the FAT boot volume, so the volume manager spawns the exfat service for it
|
||||
/// (content-routed, its id-path /volumes/exfat-<serial>). Then spawn exfat-test,
|
||||
/// which reads the seeded file and mutates through the mount. The reuse of the
|
||||
/// shared harness by a second engine is proven end to end here.
|
||||
fn exfatVolumeTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: exfat-volume\n", .{});
|
||||
if (boot_information.initial_ramdisk_len == 0) {
|
||||
check("bootloader handed over the initial_ramdisk", false);
|
||||
result();
|
||||
return;
|
||||
}
|
||||
const ramdisk = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||
const rd = initial_ramdisk.Reader.init(ramdisk) orelse {
|
||||
check("initial_ramdisk image is valid", false);
|
||||
result();
|
||||
return;
|
||||
};
|
||||
process.setInitialRamdisk(ramdisk);
|
||||
const init_ok = if (process.spawnBundled("/system/services/init")) true else |_| false;
|
||||
check("init spawned (boots the tree, incl. the volume manager)", init_ok);
|
||||
check("exfat-test client spawned", spawnNamed(rd, "exfat-test"));
|
||||
result();
|
||||
}
|
||||
|
||||
/// Per-sender range confinement (V2a, docs/volume-manager-plan.md): the fixture
|
||||
/// acquires the block channel, confines ITSELF to a sub-range, and asserts it
|
||||
/// cannot read past that range or widen it. Boots init in REGISTRY-ONLY mode
|
||||
@@ -3657,6 +3687,44 @@ fn displayReattachTest(boot_information: *const BootInformation) void {
|
||||
while (true) scheduler.yield();
|
||||
}
|
||||
|
||||
/// Storage-driver-crash rebuild (S5): the device manager runs in
|
||||
/// "test-storage-restart" mode and kills the usb-storage driver once, a moment
|
||||
/// after its volume has mounted. The driver's device stays in the tree, so the
|
||||
/// volume manager's presence poll alone would miss the death and leave fat wedged
|
||||
/// on a dead channel; its channel-liveness probe must notice, reap the volume, and
|
||||
/// rebuild on the restarted driver's fresh channel — a SECOND mount of the same
|
||||
/// id-path is the proof. (A pre-S5 manager, checking only device presence, never
|
||||
/// reaps, so the second mount never appears.)
|
||||
fn volumeDriverRestartTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: volume-driver-restart\n", .{});
|
||||
if (boot_information.initial_ramdisk_len == 0) {
|
||||
check("bootloader handed over an initial_ramdisk", false);
|
||||
result();
|
||||
return;
|
||||
}
|
||||
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||
check("initial_ramdisk image is valid", false);
|
||||
result();
|
||||
return;
|
||||
};
|
||||
process.setInitialRamdisk(image);
|
||||
_ = spawnRegistry(rd);
|
||||
var manager: u32 = 0;
|
||||
var i: u32 = 0;
|
||||
while (i < rd.count) : (i += 1) {
|
||||
const item = rd.entry(i) orelse continue;
|
||||
if (!eql(initial_ramdisk.basename(item.name), "device-manager")) continue;
|
||||
manager = process.spawnProcessSupervised(item.blob, 4, &.{ item.name, "test-storage-restart" }, scheduler.currentId(), null) catch 0;
|
||||
break;
|
||||
}
|
||||
check("device-manager spawned (test-storage-restart mode)", manager != 0);
|
||||
check("volume-manager spawned", spawnNamed(rd, "volume-manager"));
|
||||
check("fat-test client spawned", spawnNamed(rd, "fat-test"));
|
||||
scheduler.setPriority(1); // below the tree, so it runs
|
||||
while (true) scheduler.yield();
|
||||
}
|
||||
|
||||
/// Process arguments, end to end: spawn args-echo bare (its argv[0] is the
|
||||
/// initial-ramdisk name). Instance 1 sees argc == 1 and respawns itself through
|
||||
/// `system_spawn` with the extra arguments "alpha beta-42" — the syscall argument
|
||||
|
||||
+25
-6
@@ -55,7 +55,17 @@ fn tokenIndex(t: u64) u64 {
|
||||
|
||||
// --- the mount table ---------------------------------------------------------
|
||||
|
||||
pub const maximum_mounts = 8;
|
||||
/// bound: prefixes mounted in the kernel VFS table at once
|
||||
/// decided-by: ours
|
||||
/// protects: the `mounts` table below
|
||||
/// at-limit: refuse - installMount returns false and mountBackend propagates it;
|
||||
/// the mounting filesystem's harness logs "could not mount <prefix>" and the
|
||||
/// mount simply does not exist (no silent success). Budget: the initrd's
|
||||
/// top-level dirs (/system, /test) plus one id-path mount per volume and the
|
||||
/// system volume's two FHS rewrites — a few over the volume manager's
|
||||
/// maximum_volumes (16); 32 leaves headroom.
|
||||
/// observed-by: the harness "file-system: could not mount <prefix>" ring line
|
||||
pub const maximum_mounts = 32;
|
||||
const maximum_prefix = 64;
|
||||
const maximum_rewrite = 32;
|
||||
|
||||
@@ -156,7 +166,10 @@ pub fn setInitialRamdisk(image: []const u8) void {
|
||||
for (directories[0..directory_count], 0..) |*d, index| {
|
||||
const parent = parentOf(d.slice());
|
||||
d.parent = directoryIndex(parent) orelse index;
|
||||
if (parent.len == 1) installMount(d.slice(), .kernel_initrd, null, "");
|
||||
// Boot-time install of one mount per top-level initrd dir (/system, /test):
|
||||
// provably few, far under maximum_mounts, so a full table here is
|
||||
// impossible — but discard the result explicitly rather than assume it.
|
||||
if (parent.len == 1) _ = installMount(d.slice(), .kernel_initrd, null, "");
|
||||
}
|
||||
}
|
||||
|
||||
@@ -167,7 +180,12 @@ fn directoryIndex(path: []const u8) ?usize {
|
||||
return null;
|
||||
}
|
||||
|
||||
fn installMount(prefix: []const u8, kind: MountKind, backend: ?*ipc.Endpoint, rewrite: []const u8) void {
|
||||
/// Install (or remount-replace) a prefix. Returns false when the table is full
|
||||
/// and no slot could be claimed — the caller must surface that, never report a
|
||||
/// dropped mount as success. A remount of an already-mounted prefix reuses its
|
||||
/// slot and always succeeds; a /protocol remount is refused-as-noop (returns
|
||||
/// true: the first mount stands, nothing is dropped).
|
||||
fn installMount(prefix: []const u8, kind: MountKind, backend: ?*ipc.Endpoint, rewrite: []const u8) bool {
|
||||
// Remount replaces: a restarted backend re-mounts its prefix.
|
||||
var slot: ?*Mount = null;
|
||||
for (&mounts) |*m| {
|
||||
@@ -176,19 +194,20 @@ fn installMount(prefix: []const u8, kind: MountKind, backend: ?*ipc.Endpoint, re
|
||||
// restarted FAT retakes /volumes/usb; letting it retake /protocol
|
||||
// would hand the whole naming layer to whoever asked second.
|
||||
// First mount wins, and init (PID 1) is always first.
|
||||
if (std.mem.eql(u8, prefix, protocol_root)) return;
|
||||
if (std.mem.eql(u8, prefix, protocol_root)) return true;
|
||||
if (m.backend) |old| ipc.dropRef(old);
|
||||
slot = m;
|
||||
break;
|
||||
}
|
||||
if (slot == null and !m.used) slot = m;
|
||||
}
|
||||
const m = slot orelse return;
|
||||
const m = slot orelse return false;
|
||||
m.* = .{ .used = true, .kind = kind, .backend = backend };
|
||||
@memcpy(m.prefix[0..prefix.len], prefix);
|
||||
m.prefix_len = prefix.len;
|
||||
@memcpy(m.rewrite[0..rewrite.len], rewrite);
|
||||
m.rewrite_len = rewrite.len;
|
||||
return true;
|
||||
}
|
||||
|
||||
// --- resolve -----------------------------------------------------------------
|
||||
@@ -406,7 +425,7 @@ pub fn mountBackend(prefix: []const u8, backend: *ipc.Endpoint, rewrite: []const
|
||||
if (!isInitrdCarveOut(prefix)) return false;
|
||||
}
|
||||
}
|
||||
installMount(prefix, .backend, backend, rewrite);
|
||||
if (!installMount(prefix, .backend, backend, rewrite)) return false; // table full
|
||||
for (&mounts) |*m| {
|
||||
if (m.used and std.mem.eql(u8, m.prefixSlice(), prefix)) m.owner = owner;
|
||||
}
|
||||
|
||||
@@ -170,6 +170,8 @@ var test_usb_killed = false;
|
||||
var test_pci_restart_mode = false;
|
||||
var test_scanout_restart_mode = false;
|
||||
var test_scanout_killed = false;
|
||||
var test_storage_restart_mode = false;
|
||||
var test_storage_killed = false;
|
||||
var test_kill_pid: u32 = 0;
|
||||
var test_kill_due_ns: u64 = 0;
|
||||
|
||||
@@ -600,6 +602,16 @@ fn onHello(_: void, invocation: Invocation(device_manager_protocol.Hello), _: An
|
||||
test_kill_due_ns = time.clock() + 1_500_000_000;
|
||||
_ = time.timerOnce(manager_endpoint, 1600);
|
||||
}
|
||||
// Storage-driver-crash drill (S5): once, a moment after usb-storage hellos —
|
||||
// long enough that its volume has mounted — kill it. The manager re-delegates
|
||||
// the still-present device to a restarted driver on a fresh channel; the volume
|
||||
// manager's channel-liveness probe must notice the dead channel and rebuild.
|
||||
if (test_storage_restart_mode and !test_storage_killed and std.mem.eql(u8, driver.name(), "/system/drivers/usb-storage")) {
|
||||
test_storage_killed = true;
|
||||
test_kill_pid = invocation.sender;
|
||||
test_kill_due_ns = time.clock() + 2_000_000_000; // after the ~0.6s mount
|
||||
_ = time.timerOnce(manager_endpoint, 2100);
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
@@ -763,6 +775,7 @@ pub fn main(init: process.Init) void {
|
||||
test_usb_restart_mode = std.mem.eql(u8, mode, "test-usb-restart");
|
||||
test_pci_restart_mode = std.mem.eql(u8, mode, "test-pci-restart");
|
||||
test_scanout_restart_mode = std.mem.eql(u8, mode, "test-scanout-restart");
|
||||
test_storage_restart_mode = std.mem.eql(u8, mode, "test-storage-restart");
|
||||
}
|
||||
service.run(device_manager_protocol.message_maximum, .{
|
||||
.service = "device-manager",
|
||||
|
||||
@@ -0,0 +1,34 @@
|
||||
//! The exfat service as a binary package (docs/build-packages-plan.md):
|
||||
//! this file names the binary and EXACTLY the modules its source imports —
|
||||
//! build-support resolves each name from the domains this zon declares.
|
||||
|
||||
const std = @import("std");
|
||||
const build_support = @import("build-support");
|
||||
|
||||
pub fn build(b: *std.Build) void {
|
||||
const exe = build_support.userBinary(b, .{
|
||||
.name = "exfat",
|
||||
.root_source_file = b.path("exfat.zig"),
|
||||
.imports = &.{
|
||||
"block", "channel", "envelope", "file-system-harness",
|
||||
"ipc", "logging", "memory", "process",
|
||||
"time", "volume-manager-protocol",
|
||||
},
|
||||
});
|
||||
b.installArtifact(exe);
|
||||
|
||||
// Standalone `zig build test`; the root aggregate depends on this step.
|
||||
const test_step = b.step("test", "Run the exfat unit tests");
|
||||
for ([_][]const u8{
|
||||
"on-disk.zig", // exFAT on-disk struct sizes + geometry + checksums
|
||||
"engine.zig", // exFAT read/write over a RAM-backed image
|
||||
}) |test_root| {
|
||||
const unit_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path(test_root),
|
||||
.target = b.resolveTargetQuery(.{}),
|
||||
}),
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(unit_tests).step);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,16 @@
|
||||
.{
|
||||
.name = .exfat,
|
||||
.version = "0.0.0",
|
||||
.fingerprint = 0x5eafdf02d20f93dd, // Changing this has security and trust implications.
|
||||
.minimum_zig_version = "0.16.0",
|
||||
.dependencies = .{
|
||||
// build-support supplies the shared recipe; kernel is implicit in
|
||||
// every binary (the root shim + link script live there). The rest
|
||||
// are exactly the homes of this binary's declared imports.
|
||||
.@"build-support" = .{ .path = "../../../build-support" },
|
||||
.kernel = .{ .path = "../../../library/kernel" },
|
||||
.device = .{ .path = "../../../library/device" },
|
||||
.protocol = .{ .path = "../../../library/protocol" },
|
||||
},
|
||||
.paths = .{""},
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,200 @@
|
||||
//! system/services/exfat — the exFAT filesystem service. Like fat.zig, this is
|
||||
//! only the format-specific half: it finds its block device, sets up the DMA
|
||||
//! bounce buffer, mounts the exFAT engine on it, and hands the mounted volume to
|
||||
//! the shared filesystem harness (library/kernel/file-system-harness), which owns
|
||||
//! everything else — vfs serving, the open-node table, mount registration, the
|
||||
//! exit sweep, durable-on-close. The engine (engine.zig) is the pure,
|
||||
//! host-testable format code; on-disk.zig its byte layout.
|
||||
//!
|
||||
//! This service is a near-clone of fat.zig: the second engine reuses the harness
|
||||
//! wholesale, which is the reuse the storage architecture promised
|
||||
//! (docs/file-system-development/storage-architecture.md). The block data path
|
||||
//! never crosses IPC: a DMA bounce buffer is handed to the block driver by
|
||||
//! physical address, and the engine copies sectors in and out.
|
||||
|
||||
const std = @import("std");
|
||||
const channel = @import("channel");
|
||||
const volume_manager_protocol = @import("volume-manager-protocol");
|
||||
const ipc = @import("ipc");
|
||||
const process = @import("process");
|
||||
const block = @import("block");
|
||||
const memory = @import("memory");
|
||||
const logging = @import("logging");
|
||||
const time = @import("time");
|
||||
const engine = @import("engine.zig");
|
||||
const envelope = @import("envelope");
|
||||
const harness = @import("file-system-harness");
|
||||
|
||||
/// The serving harness, specialized for the exFAT engine. One volume per process.
|
||||
const Harness = harness.Server(engine.FileSystem);
|
||||
|
||||
// The engine's BlockDevice, backed by the `.block` driver plus a DMA bounce
|
||||
// buffer the driver reads/writes by physical address.
|
||||
const IpcBlock = struct {
|
||||
device: block.Device,
|
||||
bounce: memory.DmaRegion, // engine.max_transfer_sectors * 512 bytes
|
||||
|
||||
fn readBlocks(context: *anyopaque, lba: u64, count: u32, buffer: []u8) bool {
|
||||
const self: *IpcBlock = @ptrCast(@alignCast(context));
|
||||
if (count == 0 or count > engine.max_transfer_sectors) return false;
|
||||
const len = count * 512;
|
||||
if (!self.device.read(lba, count, self.bounce.physical)) return false;
|
||||
const source: [*]const u8 = @ptrFromInt(self.bounce.virtual);
|
||||
@memcpy(buffer[0..len], source[0..len]);
|
||||
return true;
|
||||
}
|
||||
fn writeBlocks(context: *anyopaque, lba: u64, count: u32, buffer: []const u8) bool {
|
||||
const self: *IpcBlock = @ptrCast(@alignCast(context));
|
||||
if (count == 0 or count > engine.max_transfer_sectors) return false;
|
||||
const len = count * 512;
|
||||
const destination: [*]u8 = @ptrFromInt(self.bounce.virtual);
|
||||
@memcpy(destination[0..len], buffer[0..len]);
|
||||
if (!self.device.write(lba, count, self.bounce.physical)) return false;
|
||||
device_dirty = true; // a block reached the device; a close will flush it
|
||||
return true;
|
||||
}
|
||||
};
|
||||
|
||||
var ipc_block: IpcBlock = undefined;
|
||||
// Set whenever a block is written, cleared when the device cache is flushed on a
|
||||
// file close — so writes are committed to stable media before a power-off.
|
||||
var device_dirty: bool = false;
|
||||
var filesystem: engine.FileSystem = undefined;
|
||||
/// The volume this exFAT process serves, its id given as argv[1] by the volume
|
||||
/// manager that spawned it. The startup hello names it so the manager returns the
|
||||
/// right volume's channel.
|
||||
var my_volume_id: u64 = 0;
|
||||
|
||||
/// The volume's own mount path, handed in as argv[2] by the volume manager: the
|
||||
/// volume's content id-path (e.g. /volumes/exfat-12345678). Defaults to
|
||||
/// /volumes/exfat only for a bare launch with no argument; the manager always
|
||||
/// passes it. The slice points into the entry block, valid for the process life.
|
||||
var volume_mount_prefix: []const u8 = "/volumes/exfat";
|
||||
|
||||
/// The mounts this volume installs: its own root, plus — only if it is the boot
|
||||
/// volume (it resolves /system/configuration) — the two FHS rewrites, so the
|
||||
/// logger's /system/logs stays decoupled from which volume backs it. Boot-volume
|
||||
/// detection is by content, so it works no matter which volume carries /system.
|
||||
/// bound: mounts one volume installs (its root + the two boot rewrites)
|
||||
/// decided-by: ours
|
||||
/// protects: the mount_specs array
|
||||
/// at-limit: truncate - unreachable today (fixed at 3); more configured mounts
|
||||
/// would need this raised, a deliberate change
|
||||
/// observed-by: a mount silently missing from the harness's mount log
|
||||
const maximum_mounts_per_volume = 4;
|
||||
var mount_specs: [maximum_mounts_per_volume]harness.MountSpec = undefined;
|
||||
|
||||
/// Get this volume's block channel from the volume manager (establishment by
|
||||
/// lineage — `block` is not a registry name). The manager spawned this process,
|
||||
/// confined it to its partition, and answers the hello with the channel; the
|
||||
/// channel is range-confined to this process's badge. Null until the manager has
|
||||
/// the volume ready — this retries.
|
||||
fn acquireVolume() ?block.Device {
|
||||
var attempts: u32 = 0;
|
||||
const vm = while (attempts < 500) : (attempts += 1) {
|
||||
if (channel.openEndpoint("volume-manager")) |handle| break handle;
|
||||
time.sleepMillis(20);
|
||||
} else return null;
|
||||
|
||||
attempts = 0;
|
||||
while (attempts < 500) : (attempts += 1) {
|
||||
var packet: [volume_manager_protocol.message_maximum]u8 = undefined;
|
||||
const framed = volume_manager_protocol.Protocol.encodeRequest(.hello, my_volume_id, .{}, &.{}, &packet) orelse return null;
|
||||
var reply: [volume_manager_protocol.message_maximum]u8 = undefined;
|
||||
const answered = ipc.callCap(vm, framed, &reply, null) catch return null;
|
||||
const status = envelope.statusOf(reply[0..answered.len]) orelse return null;
|
||||
if (status.status != 0) {
|
||||
if (answered.cap) |stray| _ = ipc.close(stray);
|
||||
_ = logging.write("/system/services/exfat: volume manager refused the hello\n");
|
||||
return null;
|
||||
}
|
||||
if (answered.cap) |bus| return .{ .endpoint = bus };
|
||||
// Acked with no channel: the volume is not ready yet — retry.
|
||||
time.sleepMillis(20);
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Durable-on-close: commit the device write cache if any block reached it since
|
||||
/// the last flush. The harness calls this on every close; the dirty check keeps
|
||||
/// it cheap.
|
||||
fn flushIfDirty() void {
|
||||
if (device_dirty) {
|
||||
_ = ipc_block.device.flush();
|
||||
device_dirty = false;
|
||||
}
|
||||
}
|
||||
|
||||
/// exFAT bring-up: find the block device, set up DMA, mount the engine, and hand
|
||||
/// the volume to the harness — or null to retry on the harness's timer.
|
||||
fn exfatBringUp(endpoint: ipc.Handle) ?Harness.Volume {
|
||||
_ = endpoint;
|
||||
const device = acquireVolume() orelse return null;
|
||||
const geometry = device.geometry() orelse {
|
||||
_ = logging.write("/system/services/exfat: block geometry unavailable\n");
|
||||
return null;
|
||||
};
|
||||
// Shareable so the buffer's capability can be attached down the chain, making
|
||||
// its physical addresses reachable under an enforcing IOMMU. No-op otherwise.
|
||||
const bounce = memory.dmaAlloc(engine.max_transfer_sectors * 512, memory.dma_coherent | memory.dma_shareable) orelse return null;
|
||||
if (bounce.handle) |handle| {
|
||||
// Attach, detach, and attach again: the round trip exercises BOTH verbs of
|
||||
// the DMA-window lifecycle through the whole chain on every boot.
|
||||
if (!device.attach(handle)) {
|
||||
_ = logging.write("/system/services/exfat: could not attach the DMA bounce buffer\n");
|
||||
return null;
|
||||
}
|
||||
if (!device.detach(handle)) {
|
||||
_ = logging.write("/system/services/exfat: could not detach the DMA bounce buffer\n");
|
||||
return null;
|
||||
}
|
||||
if (!device.attach(handle)) {
|
||||
_ = logging.write("/system/services/exfat: could not re-attach the DMA bounce buffer\n");
|
||||
return null;
|
||||
}
|
||||
_ = ipc.close(handle); // the binding holds its own reference now
|
||||
}
|
||||
ipc_block = .{ .device = device, .bounce = bounce };
|
||||
|
||||
const block_device = engine.BlockDevice{
|
||||
.context = &ipc_block,
|
||||
.block_size = geometry.block_size,
|
||||
.block_count = geometry.block_count,
|
||||
.readBlocksFn = IpcBlock.readBlocks,
|
||||
.writeBlocksFn = IpcBlock.writeBlocks,
|
||||
};
|
||||
filesystem = engine.FileSystem.mount(block_device) orelse {
|
||||
_ = logging.write("/system/services/exfat: not an exFAT filesystem\n");
|
||||
return null;
|
||||
};
|
||||
std.log.info("mounted exFAT ({d} clusters, {d} sectors/cluster, serial 0x{x})", .{ filesystem.geometry.cluster_count, filesystem.geometry.sectors_per_cluster, filesystem.geometry.volume_serial_number });
|
||||
|
||||
// The volume mounts at its id-path (argv[2]). The boot/system volume — the one
|
||||
// carrying the /system tree — additionally installs the two FHS rewrites, by
|
||||
// CONTENT: it resolves /system/configuration on its own media. A data volume
|
||||
// mounts only at its id-path and never shadows the running system.
|
||||
mount_specs[0] = .{ .prefix = volume_mount_prefix };
|
||||
var mount_count: usize = 1;
|
||||
if (filesystem.resolve("/system/configuration") != null) {
|
||||
std.log.info("volume {d} carries the system tree; backing /system/configuration and /system/logs", .{my_volume_id});
|
||||
mount_specs[1] = .{ .prefix = "/system/configuration", .rewrite = "/system/configuration" };
|
||||
mount_specs[2] = .{ .prefix = "/system/logs", .rewrite = "/system/logs" };
|
||||
mount_count = 3;
|
||||
} else {
|
||||
std.log.info("volume {d} is a data volume; mounted at {s}", .{ my_volume_id, volume_mount_prefix });
|
||||
}
|
||||
return .{ .engine = &filesystem, .mounts = mount_specs[0..mount_count], .flush = flushIfDirty };
|
||||
}
|
||||
|
||||
pub fn main(init: process.Init) void {
|
||||
// The volume manager spawns this process with its volume id as argv[1] and the
|
||||
// volume's mount path (its id-path) as argv[2].
|
||||
if (init.arguments.get(1)) |id| {
|
||||
my_volume_id = std.fmt.parseInt(u64, id, 10) catch 0;
|
||||
}
|
||||
if (init.arguments.get(2)) |prefix| {
|
||||
volume_mount_prefix = prefix;
|
||||
}
|
||||
_ = logging.write("/system/services/exfat: starting, waiting for a block device\n");
|
||||
Harness.run(.{ .bringUp = exfatBringUp });
|
||||
}
|
||||
@@ -0,0 +1,475 @@
|
||||
//! The on-disk layout of an exFAT filesystem — the Main Boot Sector (VBR) and the
|
||||
//! six 32-byte directory-entry types — as `align(1)` extern structs that bit-cast
|
||||
//! straight out of a sector (multi-byte fields little-endian). Pure data, plus the
|
||||
//! three exFAT checksums (boot region, up-case table, directory-entry set), the
|
||||
//! name hash, and the packed timestamp <-> Unix-epoch conversion. Host-testable.
|
||||
//!
|
||||
//! exFAT departs from FAT in three ways this file encodes: geometry lives in a
|
||||
//! MustBeZero-guarded VBR (byte 11 is zero, which is exactly why the FAT prober
|
||||
//! rejects an exFAT volume — it reads a zero bytes-per-sector); a file is a SET of
|
||||
//! entries (a File entry, a Stream Extension, and one or more File Name entries)
|
||||
//! validated by a rotate-right checksum; and names are compared case-folded through
|
||||
//! the volume's own on-disk up-case table (the folding itself lives in the engine,
|
||||
//! which holds the loaded table; the hash it feeds is here).
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
// exFAT's fixed on-disk widths — byte and UTF-16-unit counts the FORMAT defines,
|
||||
// not ceilings danos chooses. Named so the wire-format structs carry no bare
|
||||
// literal lengths; the values are facts of the spec.
|
||||
pub const entry_bytes: usize = 32; // every directory entry
|
||||
const jump_boot_bytes = 3;
|
||||
const filesystem_name_bytes = 8; // "EXFAT "
|
||||
const must_be_zero_bytes = 53; // the FAT-BPB overlap the format holds zero
|
||||
const boot_code_bytes = 390;
|
||||
const volume_label_units = 11;
|
||||
|
||||
// --- the Main Boot Sector (VBR, sector 0) ------------------------------------
|
||||
|
||||
/// The exFAT Main Boot Sector. `must_be_zero` (offset 11..64) overlaps where a
|
||||
/// FAT BPB keeps bytes-per-sector/sectors-per-cluster/etc.; exFAT holds it zero,
|
||||
/// so a FAT prober reading a zero bytes-per-sector rejects the volume — the
|
||||
/// mutual-exclusion the two engines rely on.
|
||||
pub const MainBootSector = extern struct {
|
||||
jump_boot: [jump_boot_bytes]u8, // 0
|
||||
filesystem_name: [filesystem_name_bytes]u8, // 3 "EXFAT "
|
||||
must_be_zero: [must_be_zero_bytes]u8, // 11
|
||||
partition_offset: u64 align(1), // 64 sectors, informational
|
||||
volume_length: u64 align(1), // 72 sectors
|
||||
fat_offset: u32 align(1), // 80 sectors from volume start
|
||||
fat_length: u32 align(1), // 84 sectors, per FAT
|
||||
cluster_heap_offset: u32 align(1), // 88 sectors from volume start
|
||||
cluster_count: u32 align(1), // 92
|
||||
first_cluster_of_root: u32 align(1), // 96
|
||||
volume_serial_number: u32 align(1), // 100
|
||||
filesystem_revision: u16 align(1), // 104
|
||||
volume_flags: u16 align(1), // 106 (skipped by the boot checksum)
|
||||
bytes_per_sector_shift: u8, // 108 9..12
|
||||
sectors_per_cluster_shift: u8, // 109
|
||||
number_of_fats: u8, // 110 1 (2 for TexFAT)
|
||||
drive_select: u8, // 111
|
||||
percent_in_use: u8, // 112 (skipped by the boot checksum)
|
||||
reserved: [7]u8, // 113
|
||||
boot_code: [boot_code_bytes]u8, // 120
|
||||
boot_signature: u16 align(1), // 510 0xAA55
|
||||
};
|
||||
|
||||
// --- directory entries (32 bytes each) ---------------------------------------
|
||||
|
||||
/// Entry-type bytes. The high bit (0x80) is InUse: a type with it clear is not in
|
||||
/// use, and 0x00 ends the directory. Deleting an entry clears bit 7 (0x85 -> 0x05).
|
||||
pub const entry_type_allocation_bitmap: u8 = 0x81;
|
||||
pub const entry_type_upcase_table: u8 = 0x82;
|
||||
pub const entry_type_volume_label: u8 = 0x83;
|
||||
pub const entry_type_file: u8 = 0x85;
|
||||
pub const entry_type_stream_extension: u8 = 0xC0;
|
||||
pub const entry_type_file_name: u8 = 0xC1;
|
||||
pub const entry_type_in_use_bit: u8 = 0x80;
|
||||
pub const entry_type_end_of_directory: u8 = 0x00;
|
||||
|
||||
/// A raw 32-byte entry, for type dispatch before it is reinterpreted as a
|
||||
/// specific entry.
|
||||
pub const RawEntry = extern struct {
|
||||
entry_type: u8,
|
||||
data: [entry_bytes - 1]u8,
|
||||
|
||||
pub fn inUse(self: RawEntry) bool {
|
||||
return self.entry_type & entry_type_in_use_bit != 0;
|
||||
}
|
||||
pub fn isEnd(self: RawEntry) bool {
|
||||
return self.entry_type == entry_type_end_of_directory;
|
||||
}
|
||||
};
|
||||
|
||||
/// 0x81 — the Allocation Bitmap: one bit per cluster (cluster 2 = bit 0), the
|
||||
/// authority for which clusters are free. The deepest departure from FAT, where
|
||||
/// the chain itself was the authority.
|
||||
pub const AllocationBitmapEntry = extern struct {
|
||||
entry_type: u8, // 0 0x81
|
||||
bitmap_flags: u8, // 1
|
||||
reserved: [18]u8, // 2
|
||||
first_cluster: u32 align(1), // 20
|
||||
data_length: u64 align(1), // 24
|
||||
};
|
||||
|
||||
/// 0x82 — the Up-case Table: the on-disk case-fold map (code unit -> uppercase),
|
||||
/// referenced by cluster and validated by `table_checksum`.
|
||||
pub const UpcaseTableEntry = extern struct {
|
||||
entry_type: u8, // 0 0x82
|
||||
reserved1: [3]u8, // 1
|
||||
table_checksum: u32 align(1), // 4
|
||||
reserved2: [12]u8, // 8
|
||||
first_cluster: u32 align(1), // 20
|
||||
data_length: u64 align(1), // 24
|
||||
};
|
||||
|
||||
/// 0x83 — the Volume Label (up to 11 UTF-16 units).
|
||||
pub const VolumeLabelEntry = extern struct {
|
||||
entry_type: u8, // 0 0x83
|
||||
character_count: u8, // 1
|
||||
volume_label: [volume_label_units]u16 align(1), // 2
|
||||
reserved: [8]u8, // 24
|
||||
};
|
||||
|
||||
/// 0x85 — the File entry: the head of a set, carrying the attributes,
|
||||
/// timestamps, the secondary-entry count, and the set checksum.
|
||||
pub const FileEntry = extern struct {
|
||||
entry_type: u8, // 0 0x85
|
||||
secondary_count: u8, // 1 stream (1) + name entries
|
||||
set_checksum: u16 align(1), // 2 over the whole set, skipping these two bytes
|
||||
file_attributes: u16 align(1), // 4
|
||||
reserved1: u16 align(1), // 6
|
||||
create_timestamp: u32 align(1), // 8
|
||||
last_modified_timestamp: u32 align(1), // 12
|
||||
last_accessed_timestamp: u32 align(1), // 16
|
||||
create_10ms: u8, // 20
|
||||
last_modified_10ms: u8, // 21
|
||||
create_utc_offset: u8, // 22
|
||||
last_modified_utc_offset: u8, // 23
|
||||
last_accessed_utc_offset: u8, // 24
|
||||
reserved2: [7]u8, // 25
|
||||
};
|
||||
|
||||
/// 0xC0 — the Stream Extension: the second entry of every file set, carrying the
|
||||
/// name length + hash and the data location (first cluster, sizes, the
|
||||
/// no-FAT-chain flag).
|
||||
pub const StreamExtensionEntry = extern struct {
|
||||
entry_type: u8, // 0 0xC0
|
||||
general_secondary_flags: u8, // 1
|
||||
reserved1: u8, // 2
|
||||
name_length: u8, // 3 UTF-16 units
|
||||
name_hash: u16 align(1), // 4
|
||||
reserved2: u16 align(1), // 6
|
||||
valid_data_length: u64 align(1), // 8
|
||||
reserved3: u32 align(1), // 16
|
||||
first_cluster: u32 align(1), // 20
|
||||
data_length: u64 align(1), // 24
|
||||
};
|
||||
|
||||
/// 0xC1 — a File Name entry: 15 UTF-16 units of the name; a set carries
|
||||
/// ceil(name_length / 15) of them.
|
||||
pub const FileNameEntry = extern struct {
|
||||
entry_type: u8, // 0 0xC1
|
||||
general_secondary_flags: u8, // 1
|
||||
file_name: [name_units_per_entry]u16 align(1), // 2
|
||||
};
|
||||
|
||||
pub const name_units_per_entry: usize = 15;
|
||||
|
||||
// General secondary flags (Stream Extension + File Name entries).
|
||||
pub const secondary_flag_allocation_possible: u8 = 0x01;
|
||||
pub const secondary_flag_no_fat_chain: u8 = 0x02;
|
||||
|
||||
// File attributes (same bit assignments as FAT).
|
||||
pub const attribute_read_only: u16 = 0x0001;
|
||||
pub const attribute_hidden: u16 = 0x0002;
|
||||
pub const attribute_system: u16 = 0x0004;
|
||||
pub const attribute_directory: u16 = 0x0010;
|
||||
pub const attribute_archive: u16 = 0x0020;
|
||||
|
||||
// FAT special cluster values (exFAT's FAT is 32-bit; used only for a fragmented
|
||||
// chain, i.e. when no_fat_chain is clear).
|
||||
pub const first_data_cluster: u32 = 2;
|
||||
pub const end_of_chain: u32 = 0xFFFFFFFF;
|
||||
pub const bad_cluster: u32 = 0xFFFFFFF7;
|
||||
|
||||
pub const boot_signature_offset: usize = 510; // 0x55 0xAA
|
||||
|
||||
// --- geometry ----------------------------------------------------------------
|
||||
|
||||
pub const Geometry = struct {
|
||||
bytes_per_sector: u32,
|
||||
sectors_per_cluster: u32,
|
||||
cluster_count: u32,
|
||||
fat_offset_sectors: u32, // from volume start
|
||||
fat_length_sectors: u32,
|
||||
cluster_heap_offset_sectors: u32, // from volume start
|
||||
first_cluster_of_root: u32,
|
||||
volume_serial_number: u32,
|
||||
volume_length_sectors: u64,
|
||||
number_of_fats: u32,
|
||||
};
|
||||
|
||||
/// Derive the geometry from a Main Boot Sector. Returns null unless it is a
|
||||
/// plausible exFAT VBR: the "EXFAT " name, an all-zero MustBeZero region, the
|
||||
/// 0xAA55 signature, and sane shifts. Accepting ONLY these is what keeps exFAT and
|
||||
/// FAT from ever claiming each other's volumes.
|
||||
pub fn geometryOf(sector: []const u8) ?Geometry {
|
||||
if (sector.len < 512) return null;
|
||||
if (sector[boot_signature_offset] != 0x55 or sector[boot_signature_offset + 1] != 0xAA) return null;
|
||||
const vbr = std.mem.bytesToValue(MainBootSector, sector[0..@sizeOf(MainBootSector)]);
|
||||
if (!std.mem.eql(u8, &vbr.filesystem_name, "EXFAT ")) return null;
|
||||
for (vbr.must_be_zero) |byte| if (byte != 0) return null;
|
||||
if (vbr.bytes_per_sector_shift < 9 or vbr.bytes_per_sector_shift > 12) return null;
|
||||
// The exFAT spec caps a cluster at 2^25 bytes (32 MiB): bytes-per-sector-shift
|
||||
// plus sectors-per-cluster-shift must not exceed 25. Enforcing it here is also
|
||||
// what keeps the engine's u32 cluster-byte arithmetic (sectors_per_cluster *
|
||||
// 512) from overflowing on a crafted VBR off untrusted removable media.
|
||||
if (@as(u16, vbr.bytes_per_sector_shift) + vbr.sectors_per_cluster_shift > 25) return null;
|
||||
// cluster_count is capped at 0xFFFFFFF5 (the spec's ClusterCount maximum), so
|
||||
// cluster_count + first_data_cluster cannot overflow u32 in the bounds checks.
|
||||
if (vbr.number_of_fats == 0 or vbr.cluster_count == 0 or vbr.cluster_count > 0xFFFFFFF5) return null;
|
||||
if (vbr.first_cluster_of_root < first_data_cluster) return null;
|
||||
return .{
|
||||
.bytes_per_sector = @as(u32, 1) << @intCast(vbr.bytes_per_sector_shift),
|
||||
.sectors_per_cluster = @as(u32, 1) << @intCast(vbr.sectors_per_cluster_shift),
|
||||
.cluster_count = vbr.cluster_count,
|
||||
.fat_offset_sectors = vbr.fat_offset,
|
||||
.fat_length_sectors = vbr.fat_length,
|
||||
.cluster_heap_offset_sectors = vbr.cluster_heap_offset,
|
||||
.first_cluster_of_root = vbr.first_cluster_of_root,
|
||||
.volume_serial_number = vbr.volume_serial_number,
|
||||
.volume_length_sectors = vbr.volume_length,
|
||||
.number_of_fats = vbr.number_of_fats,
|
||||
};
|
||||
}
|
||||
|
||||
// --- checksums and the name hash ---------------------------------------------
|
||||
|
||||
/// The directory-entry-SET checksum (a File entry's `set_checksum`), a 16-bit
|
||||
/// rotate-right sum over every byte of the set, skipping the two checksum bytes
|
||||
/// themselves (offset 2..3 of the first entry). `entries` is the whole set:
|
||||
/// (secondary_count + 1) * 32 bytes.
|
||||
pub fn setChecksum(entries: []const u8) u16 {
|
||||
var checksum: u16 = 0;
|
||||
for (entries, 0..) |byte, i| {
|
||||
if (i == 2 or i == 3) continue;
|
||||
checksum = std.math.rotr(u16, checksum, 1) +% byte;
|
||||
}
|
||||
return checksum;
|
||||
}
|
||||
|
||||
/// The up-case-table checksum (an Up-case entry's `table_checksum`), a 32-bit
|
||||
/// rotate-right sum over the table's on-disk bytes.
|
||||
pub fn upcaseChecksum(table_bytes: []const u8) u32 {
|
||||
var checksum: u32 = 0;
|
||||
for (table_bytes) |byte| checksum = std.math.rotr(u32, checksum, 1) +% byte;
|
||||
return checksum;
|
||||
}
|
||||
|
||||
/// The boot-region checksum — the u32 the checksum sector repeats — a 32-bit
|
||||
/// rotate-right sum over the first eleven sectors, skipping VolumeFlags (offset
|
||||
/// 106..107) and PercentInUse (offset 112) of the first sector.
|
||||
pub fn bootChecksum(region: []const u8) u32 {
|
||||
var checksum: u32 = 0;
|
||||
for (region, 0..) |byte, i| {
|
||||
if (i == 106 or i == 107 or i == 112) continue;
|
||||
checksum = std.math.rotr(u32, checksum, 1) +% byte;
|
||||
}
|
||||
return checksum;
|
||||
}
|
||||
|
||||
/// The name hash a Stream entry carries: a 16-bit rotate-right sum over the
|
||||
/// UP-CASED name's bytes (low byte then high byte of each UTF-16 unit). The caller
|
||||
/// up-cases through the volume's table first; a mismatch lets a lookup reject a
|
||||
/// name without reading its File Name entries.
|
||||
pub fn nameHash(upcased: []const u16) u16 {
|
||||
var hash: u16 = 0;
|
||||
for (upcased) |unit| {
|
||||
hash = std.math.rotr(u16, hash, 1) +% @as(u8, @truncate(unit));
|
||||
hash = std.math.rotr(u16, hash, 1) +% @as(u8, @truncate(unit >> 8));
|
||||
}
|
||||
return hash;
|
||||
}
|
||||
|
||||
// --- timestamps --------------------------------------------------------------
|
||||
//
|
||||
// exFAT packs a timestamp into one u32: the high 16 bits are a DOS date
|
||||
// (year-1980 | month | day), the low 16 a DOS time (hour | minute | second/2).
|
||||
// Same field layout as FAT, so the epoch math matches; there is no timezone in
|
||||
// the packed value (a separate UTC-offset byte carries that, which danos leaves
|
||||
// zero = UTC).
|
||||
|
||||
fn isLeapYear(year: u32) bool {
|
||||
return (year % 4 == 0 and year % 100 != 0) or (year % 400 == 0);
|
||||
}
|
||||
|
||||
const days_in_month = [_]u8{ 31, 28, 31, 30, 31, 30, 31, 31, 30, 31, 30, 31 };
|
||||
|
||||
/// Convert a packed exFAT timestamp to Unix epoch seconds (UTC). 0 for unset.
|
||||
pub fn timestampToEpoch(timestamp: u32) u64 {
|
||||
if (timestamp == 0) return 0;
|
||||
const date: u32 = timestamp >> 16;
|
||||
const time: u32 = timestamp & 0xFFFF;
|
||||
const day: u32 = date & 0x1F;
|
||||
const month: u32 = (date >> 5) & 0x0F;
|
||||
const year: u32 = 1980 + (date >> 9);
|
||||
if (month < 1 or month > 12 or day < 1) return 0;
|
||||
const second: u32 = (time & 0x1F) * 2;
|
||||
const minute: u32 = (time >> 5) & 0x3F;
|
||||
const hour: u32 = (time >> 11) & 0x1F;
|
||||
|
||||
var days: u64 = 0;
|
||||
var y: u32 = 1970;
|
||||
while (y < year) : (y += 1) days += if (isLeapYear(y)) 366 else 365;
|
||||
var m: u32 = 1;
|
||||
while (m < month) : (m += 1) {
|
||||
days += days_in_month[m - 1];
|
||||
if (m == 2 and isLeapYear(year)) days += 1;
|
||||
}
|
||||
days += day - 1;
|
||||
return ((days * 24 + hour) * 60 + minute) * 60 + second;
|
||||
}
|
||||
|
||||
/// Convert Unix epoch seconds (UTC) to a packed exFAT timestamp. 0 for epoch 0 or
|
||||
/// any time before 1980 (unrepresentable).
|
||||
pub fn epochToTimestamp(epoch: u64) u32 {
|
||||
if (epoch == 0) return 0;
|
||||
var remaining = epoch;
|
||||
const second: u32 = @intCast(remaining % 60);
|
||||
remaining /= 60;
|
||||
const minute: u32 = @intCast(remaining % 60);
|
||||
remaining /= 60;
|
||||
const hour: u32 = @intCast(remaining % 24);
|
||||
remaining /= 24;
|
||||
var days: u32 = @intCast(remaining);
|
||||
|
||||
var year: u32 = 1970;
|
||||
while (true) {
|
||||
const year_days: u32 = if (isLeapYear(year)) 366 else 365;
|
||||
if (days < year_days) break;
|
||||
days -= year_days;
|
||||
year += 1;
|
||||
}
|
||||
if (year < 1980) return 0;
|
||||
var month: u32 = 1;
|
||||
while (true) {
|
||||
var month_days: u32 = days_in_month[month - 1];
|
||||
if (month == 2 and isLeapYear(year)) month_days += 1;
|
||||
if (days < month_days) break;
|
||||
days -= month_days;
|
||||
month += 1;
|
||||
}
|
||||
const day = days + 1;
|
||||
const date: u32 = ((year - 1980) << 9) | (month << 5) | day;
|
||||
const time: u32 = (hour << 11) | (minute << 5) | (second / 2);
|
||||
return (date << 16) | time;
|
||||
}
|
||||
|
||||
// --- tests -------------------------------------------------------------------
|
||||
|
||||
test "on-disk struct sizes match the specification" {
|
||||
try std.testing.expectEqual(@as(usize, 512), @sizeOf(MainBootSector));
|
||||
try std.testing.expectEqual(@as(usize, 32), @sizeOf(RawEntry));
|
||||
try std.testing.expectEqual(@as(usize, 32), @sizeOf(AllocationBitmapEntry));
|
||||
try std.testing.expectEqual(@as(usize, 32), @sizeOf(UpcaseTableEntry));
|
||||
try std.testing.expectEqual(@as(usize, 32), @sizeOf(VolumeLabelEntry));
|
||||
try std.testing.expectEqual(@as(usize, 32), @sizeOf(FileEntry));
|
||||
try std.testing.expectEqual(@as(usize, 32), @sizeOf(StreamExtensionEntry));
|
||||
try std.testing.expectEqual(@as(usize, 32), @sizeOf(FileNameEntry));
|
||||
}
|
||||
|
||||
test "MainBootSector field offsets" {
|
||||
try std.testing.expectEqual(@as(usize, 3), @offsetOf(MainBootSector, "filesystem_name"));
|
||||
try std.testing.expectEqual(@as(usize, 11), @offsetOf(MainBootSector, "must_be_zero"));
|
||||
try std.testing.expectEqual(@as(usize, 80), @offsetOf(MainBootSector, "fat_offset"));
|
||||
try std.testing.expectEqual(@as(usize, 88), @offsetOf(MainBootSector, "cluster_heap_offset"));
|
||||
try std.testing.expectEqual(@as(usize, 96), @offsetOf(MainBootSector, "first_cluster_of_root"));
|
||||
try std.testing.expectEqual(@as(usize, 106), @offsetOf(MainBootSector, "volume_flags"));
|
||||
try std.testing.expectEqual(@as(usize, 112), @offsetOf(MainBootSector, "percent_in_use"));
|
||||
try std.testing.expectEqual(@as(usize, 510), @offsetOf(MainBootSector, "boot_signature"));
|
||||
// The Stream Extension's data location must sit where the spec places it.
|
||||
try std.testing.expectEqual(@as(usize, 20), @offsetOf(StreamExtensionEntry, "first_cluster"));
|
||||
try std.testing.expectEqual(@as(usize, 24), @offsetOf(StreamExtensionEntry, "data_length"));
|
||||
}
|
||||
|
||||
test "geometryOf accepts exFAT and the MustBeZero guard rejects a FAT-shaped sector" {
|
||||
var sector = [_]u8{0} ** 512;
|
||||
@memcpy(sector[3..11], "EXFAT ");
|
||||
sector[510] = 0x55;
|
||||
sector[511] = 0xAA;
|
||||
// fat_offset=128, fat_length=64, cluster_heap_offset=256, cluster_count=1000,
|
||||
// root cluster=5, bytes/sector=512 (shift 9), sectors/cluster=8 (shift 3), 1 FAT.
|
||||
std.mem.writeInt(u32, sector[80..84], 128, .little);
|
||||
std.mem.writeInt(u32, sector[84..88], 64, .little);
|
||||
std.mem.writeInt(u32, sector[88..92], 256, .little);
|
||||
std.mem.writeInt(u32, sector[92..96], 1000, .little);
|
||||
std.mem.writeInt(u32, sector[96..100], 5, .little);
|
||||
sector[108] = 9; // bytes_per_sector_shift
|
||||
sector[109] = 3; // sectors_per_cluster_shift
|
||||
sector[110] = 1; // number_of_fats
|
||||
const geo = geometryOf(§or) orelse return error.ShouldParse;
|
||||
try std.testing.expectEqual(@as(u32, 512), geo.bytes_per_sector);
|
||||
try std.testing.expectEqual(@as(u32, 8), geo.sectors_per_cluster);
|
||||
try std.testing.expectEqual(@as(u32, 1000), geo.cluster_count);
|
||||
try std.testing.expectEqual(@as(u32, 5), geo.first_cluster_of_root);
|
||||
|
||||
// A non-zero byte in MustBeZero (where a FAT BPB keeps bytes-per-sector) is
|
||||
// rejected — the mutual exclusion between the engines.
|
||||
sector[11] = 0x02;
|
||||
try std.testing.expect(geometryOf(§or) == null);
|
||||
sector[11] = 0;
|
||||
// Wrong name is rejected too.
|
||||
sector[3] = 'F';
|
||||
try std.testing.expect(geometryOf(§or) == null);
|
||||
sector[3] = 'E';
|
||||
}
|
||||
|
||||
test "geometryOf rejects crafted VBRs that would overflow u32 cluster arithmetic" {
|
||||
var sector = [_]u8{0} ** 512;
|
||||
@memcpy(sector[3..11], "EXFAT ");
|
||||
sector[510] = 0x55;
|
||||
sector[511] = 0xAA;
|
||||
std.mem.writeInt(u32, sector[92..96], 1000, .little); // cluster_count
|
||||
std.mem.writeInt(u32, sector[96..100], 5, .little); // root cluster
|
||||
sector[108] = 9; // bytes_per_sector_shift
|
||||
sector[110] = 1; // number_of_fats
|
||||
// A cluster shift past the exFAT ceiling (9 + 17 = 26 > 25) would make
|
||||
// sectors_per_cluster * 512 overflow u32 — rejected.
|
||||
sector[109] = 17;
|
||||
try std.testing.expect(geometryOf(§or) == null);
|
||||
sector[109] = 3; // sane again
|
||||
try std.testing.expect(geometryOf(§or) != null);
|
||||
// cluster_count above the spec maximum (0xFFFFFFF5) would overflow
|
||||
// cluster_count + first_data_cluster in the bounds checks — rejected.
|
||||
std.mem.writeInt(u32, sector[92..96], 0xFFFFFFFF, .little);
|
||||
try std.testing.expect(geometryOf(§or) == null);
|
||||
}
|
||||
|
||||
test "set checksum skips its own two bytes and depends on the rest" {
|
||||
var set = [_]u8{0} ** 64; // a File entry + one secondary
|
||||
set[0] = entry_type_file;
|
||||
set[1] = 1;
|
||||
set[4] = 0x20; // an attribute byte
|
||||
set[40] = 0xAB; // a byte in the secondary entry
|
||||
const base = setChecksum(&set);
|
||||
// Changing the checksum field itself must NOT change the computed checksum.
|
||||
set[2] = 0xFF;
|
||||
set[3] = 0xEE;
|
||||
try std.testing.expectEqual(base, setChecksum(&set));
|
||||
// Changing any other byte MUST change it.
|
||||
set[4] = 0x21;
|
||||
try std.testing.expect(setChecksum(&set) != base);
|
||||
}
|
||||
|
||||
test "name hash is deterministic and order-sensitive" {
|
||||
const readme = [_]u16{ 'R', 'E', 'A', 'D', 'M', 'E' };
|
||||
const different = [_]u16{ 'E', 'R', 'A', 'D', 'M', 'E' };
|
||||
try std.testing.expectEqual(nameHash(&readme), nameHash(&readme));
|
||||
try std.testing.expect(nameHash(&readme) != nameHash(&different));
|
||||
}
|
||||
|
||||
test "boot checksum skips VolumeFlags and PercentInUse" {
|
||||
var region = [_]u8{0} ** 1536; // three 512-byte sectors is enough to exercise the skips
|
||||
region[64] = 0x11;
|
||||
const base = bootChecksum(®ion);
|
||||
for ([_]usize{ 106, 107, 112 }) |skipped| {
|
||||
var copy = region;
|
||||
copy[skipped] = 0xFF;
|
||||
try std.testing.expectEqual(base, bootChecksum(©));
|
||||
}
|
||||
var copy = region;
|
||||
copy[108] = 0xFF; // a non-skipped byte
|
||||
try std.testing.expect(bootChecksum(©) != base);
|
||||
}
|
||||
|
||||
test "exFAT timestamp <-> Unix epoch round trip" {
|
||||
for ([_]u64{ 1_577_836_800, 1_700_000_000, 1_262_304_000, 1_783_971_244 }) |epoch| {
|
||||
try std.testing.expectEqual(epoch, timestampToEpoch(epochToTimestamp(epoch)));
|
||||
}
|
||||
// 1577836800 is 2020-01-01 00:00:00 UTC.
|
||||
const stamp = epochToTimestamp(1_577_836_800);
|
||||
try std.testing.expectEqual(@as(u32, 2020), 1980 + (stamp >> 16 >> 9));
|
||||
try std.testing.expectEqual(@as(u64, 0), timestampToEpoch(0));
|
||||
try std.testing.expectEqual(@as(u32, 0), epochToTimestamp(0));
|
||||
}
|
||||
@@ -1657,3 +1657,21 @@ test "short-name checksum matches the reference vector" {
|
||||
const c = FileSystem.shortChecksum("REDAME TXT".*);
|
||||
try std.testing.expect(a != c);
|
||||
}
|
||||
|
||||
test "the FAT engine rejects an exFAT volume (mutual exclusion at mount)" {
|
||||
const allocator = std.testing.allocator;
|
||||
const bytes = try allocator.alloc(u8, 5000 * sector_size);
|
||||
defer allocator.free(bytes);
|
||||
@memset(bytes, 0);
|
||||
// An exFAT boot sector: the "EXFAT " name and 0x55AA, but MustBeZero (offset
|
||||
// 11, where a FAT BPB keeps bytes-per-sector) stays zero — so this engine's
|
||||
// geometryOf reads a zero bytes-per-sector and rejects it.
|
||||
@memcpy(bytes[3..11], "EXFAT ");
|
||||
bytes[on_disk.boot_signature_offset] = 0x55;
|
||||
bytes[on_disk.boot_signature_offset + 1] = 0xAA;
|
||||
var disk = RamDisk{ .bytes = bytes };
|
||||
try std.testing.expect(FileSystem.mount(disk.device()) == null);
|
||||
// Control: a real FAT16 mounts.
|
||||
formatFat16(bytes);
|
||||
try std.testing.expect(FileSystem.mount(disk.device()) != null);
|
||||
}
|
||||
|
||||
+42
-11
@@ -64,15 +64,24 @@ var filesystem: engine.FileSystem = undefined;
|
||||
/// the right volume's channel.
|
||||
var my_volume_id: u64 = 0;
|
||||
|
||||
/// The prefixes this volume installs: /volumes/usb from the volume root, plus
|
||||
/// the two hierarchy subtrees the boot volume carries (rewrite == prefix), so
|
||||
/// hierarchy paths (the logger's /system/logs) stay decoupled from which volume
|
||||
/// backs them.
|
||||
const fat_mounts = [_]harness.MountSpec{
|
||||
.{ .prefix = "/volumes/usb" },
|
||||
.{ .prefix = "/system/configuration", .rewrite = "/system/configuration" },
|
||||
.{ .prefix = "/system/logs", .rewrite = "/system/logs" },
|
||||
};
|
||||
/// The volume's own mount path, handed in as argv[2] by the volume manager: the
|
||||
/// volume's content id-path (e.g. /volumes/fat-12345678). Defaults to
|
||||
/// /volumes/usb only for a bare launch with no argument; the manager always
|
||||
/// passes it. The slice points into the entry block, valid for the process life.
|
||||
var volume_mount_prefix: []const u8 = "/volumes/usb";
|
||||
|
||||
/// The mounts this volume installs: its own root, plus — only if it is the boot
|
||||
/// volume (it resolves /system/configuration) — the two FHS rewrites, so the
|
||||
/// logger's /system/logs stays decoupled from which volume backs it. Boot-volume
|
||||
/// detection is by content, so it works no matter which volume carries /system.
|
||||
/// bound: mounts one volume installs (its root + the two boot rewrites)
|
||||
/// decided-by: ours
|
||||
/// protects: the mount_specs array
|
||||
/// at-limit: truncate - unreachable today (fixed at 3); more configured mounts
|
||||
/// would need this raised, a deliberate change
|
||||
/// observed-by: a mount silently missing from the harness's mount log
|
||||
const maximum_mounts_per_volume = 4;
|
||||
var mount_specs: [maximum_mounts_per_volume]harness.MountSpec = undefined;
|
||||
|
||||
/// Get this volume's block channel from the volume manager (establishment by
|
||||
/// lineage, communication.md "Establishment: two planes" — `block` is not a
|
||||
@@ -164,14 +173,36 @@ fn fatBringUp(endpoint: ipc.Handle) ?Harness.Volume {
|
||||
};
|
||||
std.log.info("mounted FAT ({s}, {d} clusters, partition lba {d})", .{ @tagName(filesystem.geometry.fat_type), filesystem.geometry.cluster_count, filesystem.base_lba });
|
||||
|
||||
return .{ .engine = &filesystem, .mounts = &fat_mounts, .flush = flushIfDirty };
|
||||
// Every volume mounts at its own id-path (argv[2]). The boot/system volume —
|
||||
// the one carrying the /system tree — ADDITIONALLY installs the two FHS
|
||||
// rewrites, so hierarchy paths (config reads, the logger's persistent
|
||||
// /system/logs) stay decoupled from which volume backs them. Detection is by
|
||||
// CONTENT, not spawn order: a volume is the system volume iff /system/
|
||||
// configuration resolves on its own media. A data volume has no /system, so it
|
||||
// mounts only at its id-path and never shadows the running system's config or
|
||||
// logs with a dead mount.
|
||||
mount_specs[0] = .{ .prefix = volume_mount_prefix };
|
||||
var mount_count: usize = 1;
|
||||
if (filesystem.resolve("/system/configuration") != null) {
|
||||
std.log.info("volume {d} carries the system tree; backing /system/configuration and /system/logs", .{my_volume_id});
|
||||
mount_specs[1] = .{ .prefix = "/system/configuration", .rewrite = "/system/configuration" };
|
||||
mount_specs[2] = .{ .prefix = "/system/logs", .rewrite = "/system/logs" };
|
||||
mount_count = 3;
|
||||
} else {
|
||||
std.log.info("volume {d} is a data volume; mounted at {s}", .{ my_volume_id, volume_mount_prefix });
|
||||
}
|
||||
return .{ .engine = &filesystem, .mounts = mount_specs[0..mount_count], .flush = flushIfDirty };
|
||||
}
|
||||
|
||||
pub fn main(init: process.Init) void {
|
||||
// The volume manager spawns this process with its volume id as argv[1].
|
||||
// The volume manager spawns this process with its volume id as argv[1] and
|
||||
// the volume's mount path (its id-path) as argv[2].
|
||||
if (init.arguments.get(1)) |id| {
|
||||
my_volume_id = std.fmt.parseInt(u64, id, 10) catch 0;
|
||||
}
|
||||
if (init.arguments.get(2)) |prefix| {
|
||||
volume_mount_prefix = prefix;
|
||||
}
|
||||
_ = logging.write("/system/services/fat: starting, waiting for a block device\n");
|
||||
Harness.run(.{ .bringUp = fatBringUp });
|
||||
}
|
||||
|
||||
@@ -10,21 +10,46 @@ pub fn build(b: *std.Build) void {
|
||||
.name = "volume-manager",
|
||||
.root_source_file = b.path("volume-manager.zig"),
|
||||
.imports = &.{
|
||||
"block", "channel", "device-manager-protocol", "driver",
|
||||
"envelope", "ipc", "logging", "memory",
|
||||
"process", "service", "time", "volume-manager-protocol",
|
||||
"block", "channel", "csv", "device-manager-protocol",
|
||||
"driver", "envelope", "file-system", "ipc",
|
||||
"logging", "memory", "process", "service",
|
||||
"time", "volume-manager-protocol",
|
||||
},
|
||||
});
|
||||
b.installArtifact(exe);
|
||||
|
||||
// Standalone `zig build test` for the partition parser; the root build keeps
|
||||
// its aggregate test step.
|
||||
const test_step = b.step("test", "Run the partition-parser unit tests");
|
||||
const tests = b.addTest(.{
|
||||
// Standalone `zig build test` for the parser + mount-map modules; the root
|
||||
// build keeps its aggregate test step.
|
||||
const csv = b.dependency("csv", .{});
|
||||
const test_step = b.step("test", "Run the partition parser + mount-map unit tests");
|
||||
|
||||
const partition_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("partition.zig"),
|
||||
.target = b.resolveTargetQuery(.{}),
|
||||
}),
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(tests).step);
|
||||
test_step.dependOn(&b.addRunArtifact(partition_tests).step);
|
||||
|
||||
// filesystem-map imports csv (and, by path, partition.zig), so its test
|
||||
// module needs csv wired.
|
||||
const filesystem_map_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("filesystem-map.zig"),
|
||||
.target = b.resolveTargetQuery(.{}),
|
||||
.imports = &.{.{ .name = "csv", .module = csv.module("csv") }},
|
||||
}),
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(filesystem_map_tests).step);
|
||||
|
||||
// volume-map imports csv (and, by path, partition.zig) for the id-path
|
||||
// deriver and the volumes.csv override parser.
|
||||
const volume_map_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("volume-map.zig"),
|
||||
.target = b.resolveTargetQuery(.{}),
|
||||
.imports = &.{.{ .name = "csv", .module = csv.module("csv") }},
|
||||
}),
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(volume_map_tests).step);
|
||||
}
|
||||
|
||||
@@ -11,6 +11,8 @@
|
||||
.kernel = .{ .path = "../../../library/kernel" },
|
||||
.device = .{ .path = "../../../library/device" },
|
||||
.protocol = .{ .path = "../../../library/protocol" },
|
||||
// csv parses filesystems.csv / volumes.csv, the mount-map configuration.
|
||||
.csv = .{ .path = "../../../library/csv" },
|
||||
},
|
||||
.paths = .{""},
|
||||
}
|
||||
|
||||
@@ -0,0 +1,121 @@
|
||||
//! filesystem-map — parse `/system/configuration/filesystems.csv` into
|
||||
//! content-signature → service-binary rules, and pick the binary for a probed
|
||||
//! volume's signature. The data-driven replacement for the volume manager's
|
||||
//! hardcoded `filesystem_binary` const: a signature no row matches goes unserved
|
||||
//! (logged), never guessed — the same discipline the device registry uses.
|
||||
//!
|
||||
//! Pure logic: no hardware, no syscalls, no allocator. The `binary` slice points
|
||||
//! into the CSV source, which the manager holds in a static buffer for the life
|
||||
//! of the process (zero-copy), so the source must outlive the rules.
|
||||
//!
|
||||
//! Format: one rule per line, two comma-separated fields, `#` comments (whole-
|
||||
//! line or trailing), blank lines ignored:
|
||||
//!
|
||||
//! signature, binary
|
||||
//!
|
||||
//! `signature` is a filesystem token (`fat`; `exfat` lands with S4); `binary` is
|
||||
//! a full ramdisk path.
|
||||
|
||||
const std = @import("std");
|
||||
const csv = @import("csv");
|
||||
const partition = @import("partition.zig");
|
||||
|
||||
/// One parsed row: a content signature and the service binary that serves it.
|
||||
pub const Rule = struct {
|
||||
kind: partition.FilesystemKind,
|
||||
binary: []const u8,
|
||||
};
|
||||
|
||||
/// How many rules landed, how many non-blank lines were malformed (for the
|
||||
/// manager to log), and whether there were more rules than the buffer could hold.
|
||||
pub const ParseResult = struct { count: usize, malformed: usize, truncated: bool };
|
||||
|
||||
const Line = union(enum) { rule: Rule, ignorable, malformed };
|
||||
|
||||
fn parseLine(line: []const u8) Line {
|
||||
const body = csv.stripComment(line);
|
||||
if (body.len == 0) return .ignorable;
|
||||
var it = csv.fields(body);
|
||||
const sig = it.next() orelse return .malformed;
|
||||
const binary = it.next() orelse return .malformed;
|
||||
if (it.next() != null) return .malformed; // too many columns
|
||||
if (binary.len == 0) return .malformed;
|
||||
const kind = partition.FilesystemKind.fromToken(sig);
|
||||
if (kind == .unknown) return .malformed; // an unrecognised signature token
|
||||
return .{ .rule = .{ .kind = kind, .binary = binary } };
|
||||
}
|
||||
|
||||
/// Parse a whole `filesystems.csv` into `out_rules`. The `binary` slices point
|
||||
/// into `source`, which must outlive them.
|
||||
pub fn parse(source: []const u8, out_rules: []Rule) ParseResult {
|
||||
var result: ParseResult = .{ .count = 0, .malformed = 0, .truncated = false };
|
||||
var lines = std.mem.splitScalar(u8, source, '\n');
|
||||
while (lines.next()) |line| {
|
||||
switch (parseLine(line)) {
|
||||
.ignorable => {},
|
||||
.malformed => result.malformed += 1,
|
||||
.rule => |rule| {
|
||||
if (result.count >= out_rules.len) {
|
||||
result.truncated = true;
|
||||
continue;
|
||||
}
|
||||
out_rules[result.count] = rule;
|
||||
result.count += 1;
|
||||
},
|
||||
}
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
/// The service binary for a probed volume's signature — the first matching row,
|
||||
/// or null (the volume goes unserved, like a device no registry row matches).
|
||||
pub fn match(rules: []const Rule, kind: partition.FilesystemKind) ?[]const u8 {
|
||||
for (rules) |rule| {
|
||||
if (rule.kind == kind) return rule.binary;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
// --- tests -------------------------------------------------------------------
|
||||
|
||||
const testing = std.testing;
|
||||
|
||||
// A fixture-sized rule buffer for the tests, named so the bounds gate (which
|
||||
// flags literal array lengths) stays quiet: this is a test input, not a runtime
|
||||
// ceiling — the real one is maximum_filesystem_rules in the volume manager.
|
||||
const test_rule_slots = 4;
|
||||
|
||||
test "a fat signature maps to its binary; an unmatched signature is null" {
|
||||
const text =
|
||||
\\# signature, binary
|
||||
\\fat, /system/services/fat
|
||||
;
|
||||
var rules: [test_rule_slots]Rule = undefined;
|
||||
const parsed = parse(text, &rules);
|
||||
try testing.expectEqual(@as(usize, 1), parsed.count);
|
||||
try testing.expectEqual(@as(usize, 0), parsed.malformed);
|
||||
try testing.expectEqualStrings("/system/services/fat", match(rules[0..parsed.count], .fat).?);
|
||||
try testing.expect(match(rules[0..parsed.count], .unknown) == null);
|
||||
}
|
||||
|
||||
test "the binary is chosen by content, not hardcoded" {
|
||||
// Point the fat row at a different binary and confirm that binary is chosen —
|
||||
// a constant could not satisfy this, which is the whole point of the map.
|
||||
const text = "fat, /system/services/other-fat\n";
|
||||
var rules: [test_rule_slots]Rule = undefined;
|
||||
const parsed = parse(text, &rules);
|
||||
try testing.expectEqualStrings("/system/services/other-fat", match(rules[0..parsed.count], .fat).?);
|
||||
}
|
||||
|
||||
test "malformed rows are counted, not bound" {
|
||||
const text =
|
||||
\\fat, /system/services/fat
|
||||
\\bogusfs, /system/services/x
|
||||
\\fat,
|
||||
\\fat, /a, /b
|
||||
;
|
||||
var rules: [test_rule_slots]Rule = undefined;
|
||||
const parsed = parse(text, &rules);
|
||||
try testing.expectEqual(@as(usize, 1), parsed.count); // only the first fat row
|
||||
try testing.expectEqual(@as(usize, 3), parsed.malformed); // bad token, empty binary, too many columns
|
||||
}
|
||||
@@ -1,56 +1,315 @@
|
||||
//! Partition-table parsing, the policy the storage architecture places above the
|
||||
//! block driver and below the filesystem (docs/file-system-development/
|
||||
//! storage-architecture.md): read block 0, decide what block sub-ranges are
|
||||
//! storage-architecture.md): read the medium, decide what block sub-ranges are
|
||||
//! volumes, and read each volume's content identity. The block DRIVER never does
|
||||
//! this — it clamps ranges it is told about; this is what tells it the numbers.
|
||||
//!
|
||||
//! Today: MBR (the four-entry table at offset 446) plus the bare-FAT case (a boot
|
||||
//! sector right at LBA 0). GPT is the next entry in the identity ladder and slots
|
||||
//! in here without touching anything above or below.
|
||||
//! Reads happen through a `SectorReader` (not one preloaded block-0 slice) so the
|
||||
//! parser can reach GPT metadata at LBA 1, the entry array beyond it, and each
|
||||
//! partition's VBR on demand. The identity it returns is a tagged `Identity`: the
|
||||
//! `key` is the id (the mount path is derived from it — a stable, unique,
|
||||
//! content-derived handle), and `label` is display metadata (the FAT volume label
|
||||
//! or the GPT partition name), never part of the id. Today's rung is MBR/bare-FAT;
|
||||
//! GPT (rung 1) and the FAT serial (rung 3) slot in without changing the shape.
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
/// One volume the parser found on the device: the block sub-range it occupies
|
||||
/// and a content identity stable for the volume's life (the mount map keys on
|
||||
/// it; the boot volume is recorded by it). `identity` is derived from the medium,
|
||||
/// never from a port — a moved drive keeps it.
|
||||
/// A single 512-byte sector's worth of bytes. The parser assumes 512-byte
|
||||
/// logical sectors (4Kn media is a separate concern, noted in the plan).
|
||||
pub const sector_bytes = 512;
|
||||
|
||||
/// bound: bytes of a volume's display label the parser records (a GPT partition
|
||||
/// name is 36 UTF-16 units; a FAT volume label is 11 bytes; 36 covers both)
|
||||
/// decided-by: hardware
|
||||
/// protects: the Identity.label buffer
|
||||
/// at-limit: degrade - a longer name is truncated to this many ASCII bytes
|
||||
/// observed-by: a volume whose displayed label is clipped
|
||||
pub const label_maximum = 36;
|
||||
|
||||
/// Which rung of the identity ladder produced this identity. The rung tags the
|
||||
/// `key` namespace so a FAT serial and an MBR signature that happen to share bits
|
||||
/// stay distinct, and it drives how the mount path is rendered from the id.
|
||||
pub const Rung = enum(u8) {
|
||||
gpt_guid = 1,
|
||||
filesystem_uuid = 2, // reserved: no engine reads a superblock UUID yet
|
||||
fat_serial = 3,
|
||||
mbr_index = 4,
|
||||
anonymous = 5,
|
||||
exfat_serial = 6, // exFAT's VolumeSerialNumber — content-strong like fat_serial
|
||||
};
|
||||
|
||||
/// A volume's content identity. `key` is the ID — the stable, unique handle the
|
||||
/// mount path is derived from and the mount map keys on. `label` is DISPLAY
|
||||
/// metadata (FAT volume label / GPT partition name), exposed to a UI but never
|
||||
/// part of the path; two volumes with the same label but different keys are
|
||||
/// different volumes. Derived from the medium, never from a port.
|
||||
pub const Identity = struct {
|
||||
rung: Rung,
|
||||
key: u128 = 0,
|
||||
label: [label_maximum]u8 = [_]u8{0} ** label_maximum,
|
||||
label_len: u8 = 0,
|
||||
|
||||
pub fn labelSlice(self: *const Identity) []const u8 {
|
||||
return self.label[0..self.label_len];
|
||||
}
|
||||
|
||||
/// Identity equality is the ID (rung + key) only — the label is display
|
||||
/// metadata and does not enter it. Same rung + same key means the same
|
||||
/// volume (the dd-cloned-media case the duplicate policy is for).
|
||||
pub fn eql(a: Identity, b: Identity) bool {
|
||||
return a.rung == b.rung and a.key == b.key;
|
||||
}
|
||||
};
|
||||
|
||||
/// Which filesystem a volume's content is — the key `filesystems.csv` maps to a
|
||||
/// service binary. FAT and exFAT are recognized by their VBRs; content that is
|
||||
/// neither falls back to `.fat`, the volume manager's historical hand-off.
|
||||
pub const FilesystemKind = enum {
|
||||
fat,
|
||||
exfat,
|
||||
unknown,
|
||||
|
||||
pub fn fromToken(token: []const u8) FilesystemKind {
|
||||
if (std.mem.eql(u8, token, "fat")) return .fat;
|
||||
if (std.mem.eql(u8, token, "exfat")) return .exfat;
|
||||
return .unknown;
|
||||
}
|
||||
};
|
||||
|
||||
/// One volume the parser found on the device: the block sub-range it occupies,
|
||||
/// its content identity, and which filesystem its content is (the signature the
|
||||
/// `filesystems.csv` map keys on to pick the service binary).
|
||||
pub const Volume = struct {
|
||||
base_lba: u64,
|
||||
block_count: u64,
|
||||
identity: u64,
|
||||
identity: Identity,
|
||||
signature: FilesystemKind = .fat,
|
||||
};
|
||||
|
||||
/// Read sectors on demand. `context` + `readFn` mirror the FAT engine's
|
||||
/// `BlockDevice` vtable; `readFn` returns false past the end of the device or on
|
||||
/// an I/O error, which the parser treats as "no volume".
|
||||
pub const SectorReader = struct {
|
||||
context: *anyopaque,
|
||||
readFn: *const fn (context: *anyopaque, lba: u64, buffer: *[sector_bytes]u8) bool,
|
||||
|
||||
pub fn read(self: SectorReader, lba: u64, buffer: *[sector_bytes]u8) bool {
|
||||
return self.readFn(self.context, lba, buffer);
|
||||
}
|
||||
};
|
||||
|
||||
/// The MBR disk signature (offset 440, 4 bytes LE) — a 32-bit id written at
|
||||
/// partition time. Weak (dd-cloned disks share it) but on the medium, and the
|
||||
/// simplest rung of the identity ladder; the fuller rungs (GPT partition GUID,
|
||||
/// FAT volume serial) refine `identityOf` without changing the shape.
|
||||
/// last rung of the identity ladder; the fuller rungs (GPT GUID, FAT serial)
|
||||
/// take precedence when present.
|
||||
fn diskSignature(block0: []const u8) u32 {
|
||||
if (block0.len < 444) return 0;
|
||||
return std.mem.readInt(u32, block0[440..444], .little);
|
||||
}
|
||||
|
||||
/// The identity of the volume at partition index `index`: the disk signature
|
||||
/// The rung-4 identity of the volume at partition `index`: the disk signature
|
||||
/// paired with the index, so two partitions of one disk stay distinct. For a
|
||||
/// bare FAT (no table) the index is 0.
|
||||
fn identityOf(block0: []const u8, index: u8) u64 {
|
||||
return (@as(u64, diskSignature(block0)) << 8) | index;
|
||||
/// bare FAT (no table) the index is 0. Carries no label.
|
||||
fn mbrIdentity(block0: []const u8, index: u8) Identity {
|
||||
return .{ .rung = .mbr_index, .key = (@as(u128, diskSignature(block0)) << 8) | index };
|
||||
}
|
||||
|
||||
/// Whether block 0 looks like a partition table (the 0x55AA boot signature). A
|
||||
/// bare FAT also carries it, so the caller distinguishes by whether any partition
|
||||
/// entry is non-empty.
|
||||
/// Whether a block looks like a boot sector / partition table (the 0x55AA boot
|
||||
/// signature). A bare FAT also carries it, so the caller distinguishes by whether
|
||||
/// any partition entry is non-empty.
|
||||
fn hasBootSignature(block0: []const u8) bool {
|
||||
return block0.len >= 512 and block0[510] == 0x55 and block0[511] == 0xAA;
|
||||
}
|
||||
|
||||
/// The first volume on a device whose block 0 is `block0` and whose whole-device
|
||||
/// size is `device_blocks`, or null if none is found. An MBR with a non-empty
|
||||
/// entry yields that partition's [start, size); otherwise a boot signature with
|
||||
/// no partitions is treated as a bare FAT spanning the whole device.
|
||||
pub fn firstVolume(block0: []const u8, device_blocks: u64) ?Volume {
|
||||
if (!hasBootSignature(block0)) return null;
|
||||
var index: u8 = 0;
|
||||
/// GPT header signature at LBA 1.
|
||||
const gpt_signature = "EFI PART";
|
||||
|
||||
/// bound: GPT partition entries scanned before the prober gives up
|
||||
/// decided-by: ours
|
||||
/// protects: the entry-array scan loop from an untrusted num_partition_entries
|
||||
/// at-limit: degrade - stop scanning; a device whose usable entry sits past the
|
||||
/// cap is treated as having no GPT volume (real tables carry <=128 entries)
|
||||
/// observed-by: the gpt-entry-past-device host test
|
||||
const gpt_entry_scan_maximum = 128;
|
||||
|
||||
/// Reflected CRC-32 (polynomial 0xEDB88320) — the ISO-HDLC variant GPT uses for
|
||||
/// its header checksum. Inlined so the parser and the host fixtures compute it
|
||||
/// the same way and never drift onto a magic constant.
|
||||
fn crc32(bytes: []const u8) u32 {
|
||||
var c: u32 = 0xFFFFFFFF;
|
||||
for (bytes) |b| {
|
||||
c ^= b;
|
||||
var k: u8 = 0;
|
||||
while (k < 8) : (k += 1) {
|
||||
c = if (c & 1 != 0) (c >> 1) ^ 0xEDB88320 else c >> 1;
|
||||
}
|
||||
}
|
||||
return c ^ 0xFFFFFFFF;
|
||||
}
|
||||
|
||||
/// A GPT disk carries a protective MBR: a boot-signed block 0 with a partition
|
||||
/// entry of type 0xEE. Its presence routes probing to the GPT (authoritative).
|
||||
fn isProtectiveMbr(block0: []const u8) bool {
|
||||
if (!hasBootSignature(block0)) return false;
|
||||
var index: usize = 0;
|
||||
while (index < 4) : (index += 1) {
|
||||
if (block0[446 + index * 16 + 4] == 0xEE) return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/// Copy the GPT partition name (36 UTF-16LE units, the 72 bytes at entry+56)
|
||||
/// into the identity's display label as ASCII, dropping non-ASCII units.
|
||||
fn setLabelFromUtf16(id: *Identity, name_bytes: []const u8) void {
|
||||
var out: usize = 0;
|
||||
var i: usize = 0;
|
||||
while (i + 1 < name_bytes.len and out < label_maximum) : (i += 2) {
|
||||
const unit = std.mem.readInt(u16, name_bytes[i..][0..2], .little);
|
||||
if (unit == 0) break;
|
||||
if (unit < 0x80) {
|
||||
id.label[out] = @intCast(unit);
|
||||
out += 1;
|
||||
}
|
||||
}
|
||||
id.label_len = @intCast(out);
|
||||
}
|
||||
|
||||
/// Append every valid GPT volume to `out` (up to `out.len`), returning the count
|
||||
/// (0 if LBA 1 is not a valid GPT header). The header CRC-32 and the per-entry
|
||||
/// overflow-safe range check are the confinement-safety guards the driver's clamp
|
||||
/// rests on — the invariant documented for MBR, extended to untrusted GPT
|
||||
/// metadata. The entry-array CRC is deferred (correctness-only; the range check
|
||||
/// carries safety).
|
||||
fn gptAllVolumes(reader: SectorReader, device_blocks: u64, out: []Volume) usize {
|
||||
var header: [sector_bytes]u8 = undefined;
|
||||
if (!reader.read(1, &header)) return 0;
|
||||
if (!std.mem.eql(u8, header[0..8], gpt_signature)) return 0;
|
||||
const header_size = std.mem.readInt(u32, header[12..16], .little);
|
||||
if (header_size < 92 or header_size > sector_bytes) return 0;
|
||||
const stored_crc = std.mem.readInt(u32, header[16..20], .little);
|
||||
var check: [sector_bytes]u8 = undefined;
|
||||
@memcpy(check[0..header_size], header[0..header_size]);
|
||||
@memset(check[16..20], 0);
|
||||
if (crc32(check[0..header_size]) != stored_crc) return 0;
|
||||
|
||||
const entry_lba = std.mem.readInt(u64, header[72..80], .little);
|
||||
const num_entries = std.mem.readInt(u32, header[80..84], .little);
|
||||
const entry_size = std.mem.readInt(u32, header[84..88], .little);
|
||||
if (entry_size != 128 and entry_size != 256 and entry_size != 512) return 0;
|
||||
if (entry_lba == 0 or entry_lba >= device_blocks) return 0;
|
||||
|
||||
const scan = @min(num_entries, gpt_entry_scan_maximum);
|
||||
var sector_buf: [sector_bytes]u8 = undefined;
|
||||
var loaded: u64 = std.math.maxInt(u64);
|
||||
var count: usize = 0;
|
||||
var i: u32 = 0;
|
||||
while (i < scan and count < out.len) : (i += 1) {
|
||||
const abs = @as(u64, i) * entry_size;
|
||||
const lba = entry_lba + abs / sector_bytes;
|
||||
const off = @as(usize, @intCast(abs % sector_bytes));
|
||||
if (lba != loaded) {
|
||||
if (!reader.read(lba, §or_buf)) break; // return what we have
|
||||
loaded = lba;
|
||||
}
|
||||
const entry = sector_buf[off..][0..128]; // the fields we read live in the first 128 bytes
|
||||
var type_nonzero = false;
|
||||
for (entry[0..16]) |b| {
|
||||
if (b != 0) {
|
||||
type_nonzero = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (!type_nonzero) continue;
|
||||
const start = std.mem.readInt(u64, entry[32..40], .little);
|
||||
const end = std.mem.readInt(u64, entry[40..48], .little); // inclusive last LBA
|
||||
// Untrusted range from removable media: overflow-safe validation. Reject a
|
||||
// partition that starts at 0, is reversed, or ends outside the device; only
|
||||
// then is start + count <= device_blocks guaranteed for the driver's clamp.
|
||||
if (start == 0 or end < start or end >= device_blocks) continue;
|
||||
var id = Identity{ .rung = .gpt_guid, .key = std.mem.readInt(u128, entry[16..32], .little) };
|
||||
setLabelFromUtf16(&id, entry[56..128]);
|
||||
out[count] = .{ .base_lba = start, .block_count = end - start + 1, .identity = id, .signature = signatureAt(reader, start) };
|
||||
count += 1;
|
||||
}
|
||||
return count;
|
||||
}
|
||||
|
||||
/// Trim trailing spaces (FAT labels are space-padded) and copy into the display
|
||||
/// label, clamped to label_maximum.
|
||||
fn setFatLabel(id: *Identity, label: []const u8) void {
|
||||
var end: usize = label.len;
|
||||
while (end > 0 and label[end - 1] == ' ') : (end -= 1) {}
|
||||
const n = @min(end, label_maximum);
|
||||
@memcpy(id.label[0..n], label[0..n]);
|
||||
id.label_len = @intCast(n);
|
||||
}
|
||||
|
||||
/// The FAT volume serial (BS_VolID) + label (BS_VolLab) read from the VBR at
|
||||
/// `start_lba` — rung 3, stronger than the MBR disk signature. Null if the
|
||||
/// sector is not an extended FAT boot record (no 0x55AA, or no 0x28/0x29
|
||||
/// extended boot signature). FAT32 is distinguished by fat_size_16 == 0; the
|
||||
/// serial and label live at different EBR offsets for FAT12/16 vs FAT32 (the
|
||||
/// offsets are cross-checked against system/services/fat/on-disk.zig).
|
||||
fn fatIdentity(reader: SectorReader, start_lba: u64) ?Identity {
|
||||
var vbr: [sector_bytes]u8 = undefined;
|
||||
if (!reader.read(start_lba, &vbr)) return null;
|
||||
if (vbr[510] != 0x55 or vbr[511] != 0xAA) return null;
|
||||
const is_fat32 = std.mem.readInt(u16, vbr[22..24], .little) == 0;
|
||||
const sig_off: usize = if (is_fat32) 66 else 38;
|
||||
if (vbr[sig_off] != 0x28 and vbr[sig_off] != 0x29) return null;
|
||||
const id_off: usize = if (is_fat32) 67 else 39;
|
||||
const label_off: usize = if (is_fat32) 71 else 43;
|
||||
var id = Identity{ .rung = .fat_serial, .key = std.mem.readInt(u32, vbr[id_off..][0..4], .little) };
|
||||
setFatLabel(&id, vbr[label_off..][0..11]);
|
||||
return id;
|
||||
}
|
||||
|
||||
/// The exFAT VolumeSerialNumber (offset 100) read from the Main Boot Sector at
|
||||
/// `start_lba` — its content identity, rung `exfat_serial`. Null unless the sector
|
||||
/// is an exFAT VBR (the "EXFAT " name at offset 3 + the 0x55AA signature; the
|
||||
/// name is where a FAT BPB keeps its OEM string, so the two never collide). The
|
||||
/// label lives in a root-directory entry, not the VBR, so it is left empty here.
|
||||
fn exfatIdentity(reader: SectorReader, start_lba: u64) ?Identity {
|
||||
var vbr: [sector_bytes]u8 = undefined;
|
||||
if (!reader.read(start_lba, &vbr)) return null;
|
||||
if (vbr[510] != 0x55 or vbr[511] != 0xAA) return null;
|
||||
if (!std.mem.eql(u8, vbr[3..11], "EXFAT ")) return null;
|
||||
return .{ .rung = .exfat_serial, .key = std.mem.readInt(u32, vbr[100..104], .little) };
|
||||
}
|
||||
|
||||
const Recognized = struct { identity: Identity, signature: FilesystemKind };
|
||||
|
||||
/// Recognize the filesystem at `start_lba` by its VBR: exFAT first (its serial and
|
||||
/// the `.exfat` signature), else FAT (its serial), else unknown content that keeps
|
||||
/// the MBR disk-signature identity and the historical `.fat` hand-off.
|
||||
fn recognize(reader: SectorReader, start_lba: u64, block0: *const [sector_bytes]u8, index: u8) Recognized {
|
||||
if (exfatIdentity(reader, start_lba)) |id| return .{ .identity = id, .signature = .exfat };
|
||||
if (fatIdentity(reader, start_lba)) |id| return .{ .identity = id, .signature = .fat };
|
||||
return .{ .identity = mbrIdentity(block0, index), .signature = .fat };
|
||||
}
|
||||
|
||||
/// The filesystem signature at `start_lba` when the identity is decided elsewhere
|
||||
/// (a GPT partition keeps its GUID identity but still needs its content's kind).
|
||||
fn signatureAt(reader: SectorReader, start_lba: u64) FilesystemKind {
|
||||
return if (exfatIdentity(reader, start_lba) != null) .exfat else .fat;
|
||||
}
|
||||
|
||||
/// Append every volume on the device `reader` addresses, whose whole-device size
|
||||
/// is `device_blocks`, to `out` (up to `out.len`), returning the count. A GPT
|
||||
/// disk (protective MBR) is enumerated by GPT, authoritatively — a zero count is
|
||||
/// final. Otherwise every fitting MBR entry is a volume; a boot signature with no
|
||||
/// partition entries is a bare FAT spanning the whole device. Each volume's
|
||||
/// [start, count) is validated overflow-safe (the confinement invariant the
|
||||
/// driver's clamp rests on), and each prefers its FAT serial identity over the
|
||||
/// disk signature.
|
||||
pub fn allVolumes(reader: SectorReader, device_blocks: u64, out: []Volume) usize {
|
||||
var block0: [sector_bytes]u8 = undefined;
|
||||
if (!reader.read(0, &block0)) return 0;
|
||||
if (!hasBootSignature(&block0)) return 0;
|
||||
if (isProtectiveMbr(&block0)) return gptAllVolumes(reader, device_blocks, out);
|
||||
var count: usize = 0;
|
||||
var index: u8 = 0;
|
||||
while (index < 4 and count < out.len) : (index += 1) {
|
||||
const entry = block0[446 + @as(usize, index) * 16 ..][0..16];
|
||||
const kind = entry[4];
|
||||
const start = std.mem.readInt(u32, entry[8..12], .little);
|
||||
@@ -63,12 +322,46 @@ pub fn firstVolume(block0: []const u8, device_blocks: u64) ?Volume {
|
||||
// device (usb-storage.zig resolveTransfer), which only holds because the
|
||||
// range handed down is validated here. The subtraction cannot overflow.
|
||||
if (start > device_blocks or device_blocks - start < size) continue;
|
||||
return .{ .base_lba = start, .block_count = size, .identity = identityOf(block0, index) };
|
||||
const found = recognize(reader, start, &block0, index);
|
||||
out[count] = .{ .base_lba = start, .block_count = size, .identity = found.identity, .signature = found.signature };
|
||||
count += 1;
|
||||
}
|
||||
// No partition entries: a bare FAT spanning the device.
|
||||
return .{ .base_lba = 0, .block_count = device_blocks, .identity = identityOf(block0, 0) };
|
||||
if (count == 0 and out.len > 0) {
|
||||
// No partition entries: a bare FAT or exFAT spanning the device.
|
||||
const found = recognize(reader, 0, &block0, 0);
|
||||
out[0] = .{ .base_lba = 0, .block_count = device_blocks, .identity = found.identity, .signature = found.signature };
|
||||
return 1;
|
||||
}
|
||||
return count;
|
||||
}
|
||||
|
||||
/// firstVolume is allVolumes into a one-element buffer.
|
||||
const one_volume_slot = 1;
|
||||
|
||||
/// The first volume on the device, or null — the single-volume case of
|
||||
/// `allVolumes`, kept for callers that want just one.
|
||||
pub fn firstVolume(reader: SectorReader, device_blocks: u64) ?Volume {
|
||||
var one: [one_volume_slot]Volume = undefined;
|
||||
return if (allVolumes(reader, device_blocks, &one) > 0) one[0] else null;
|
||||
}
|
||||
|
||||
/// A read-only RAM disk over a byte slice of sectors, for the host tests.
|
||||
const RamDisk = struct {
|
||||
sectors: []const u8,
|
||||
|
||||
fn readFn(context: *anyopaque, lba: u64, buffer: *[sector_bytes]u8) bool {
|
||||
const self: *const RamDisk = @ptrCast(@alignCast(context));
|
||||
const off = lba * sector_bytes;
|
||||
if (off + sector_bytes > self.sectors.len) return false;
|
||||
@memcpy(buffer, self.sectors[off..][0..sector_bytes]);
|
||||
return true;
|
||||
}
|
||||
|
||||
fn reader(self: *const RamDisk) SectorReader {
|
||||
return .{ .context = @constCast(self), .readFn = readFn };
|
||||
}
|
||||
};
|
||||
|
||||
test "an MBR with one partition yields its range and a distinct identity" {
|
||||
var block0 = [_]u8{0} ** 512;
|
||||
block0[510] = 0x55;
|
||||
@@ -78,24 +371,53 @@ test "an MBR with one partition yields its range and a distinct identity" {
|
||||
block0[446 + 4] = 0x0c;
|
||||
std.mem.writeInt(u32, block0[446 + 8 ..][0..4], 2048, .little);
|
||||
std.mem.writeInt(u32, block0[446 + 12 ..][0..4], 100000, .little);
|
||||
const v = firstVolume(&block0, 200000).?;
|
||||
const disk = RamDisk{ .sectors = &block0 };
|
||||
const v = firstVolume(disk.reader(), 200000).?;
|
||||
try std.testing.expectEqual(@as(u64, 2048), v.base_lba);
|
||||
try std.testing.expectEqual(@as(u64, 100000), v.block_count);
|
||||
try std.testing.expectEqual((@as(u64, 0xDEADBEEF) << 8) | 0, v.identity);
|
||||
try std.testing.expectEqual(Rung.mbr_index, v.identity.rung);
|
||||
try std.testing.expectEqual((@as(u128, 0xDEADBEEF) << 8) | 0, v.identity.key);
|
||||
}
|
||||
|
||||
test "a boot signature with no partitions is a bare FAT over the whole device" {
|
||||
var block0 = [_]u8{0} ** 512;
|
||||
block0[510] = 0x55;
|
||||
block0[511] = 0xAA;
|
||||
const v = firstVolume(&block0, 65536).?;
|
||||
const disk = RamDisk{ .sectors = &block0 };
|
||||
const v = firstVolume(disk.reader(), 65536).?;
|
||||
try std.testing.expectEqual(@as(u64, 0), v.base_lba);
|
||||
try std.testing.expectEqual(@as(u64, 65536), v.block_count);
|
||||
}
|
||||
|
||||
test "no boot signature is no volume" {
|
||||
const block0 = [_]u8{0} ** 512;
|
||||
try std.testing.expect(firstVolume(&block0, 65536) == null);
|
||||
const disk = RamDisk{ .sectors = &block0 };
|
||||
try std.testing.expect(firstVolume(disk.reader(), 65536) == null);
|
||||
}
|
||||
test "a bare exFAT volume is recognized by its VBR, with its serial as the id" {
|
||||
var block0 = [_]u8{0} ** 512;
|
||||
@memcpy(block0[3..11], "EXFAT ");
|
||||
block0[510] = 0x55;
|
||||
block0[511] = 0xAA;
|
||||
std.mem.writeInt(u32, block0[100..104], 0xDA7A0001, .little); // VolumeSerialNumber
|
||||
const disk = RamDisk{ .sectors = &block0 };
|
||||
const v = firstVolume(disk.reader(), 65536).?;
|
||||
try std.testing.expectEqual(FilesystemKind.exfat, v.signature);
|
||||
try std.testing.expectEqual(Rung.exfat_serial, v.identity.rung);
|
||||
try std.testing.expectEqual(@as(u128, 0xDA7A0001), v.identity.key);
|
||||
}
|
||||
test "a FAT VBR is recognized as fat, not exfat — the signatures never collide" {
|
||||
var block0 = [_]u8{0} ** 512;
|
||||
@memcpy(block0[3..11], "MSWIN4.1"); // a FAT OEM name, not "EXFAT "
|
||||
block0[510] = 0x55;
|
||||
block0[511] = 0xAA;
|
||||
std.mem.writeInt(u16, block0[22..24], 16, .little); // fat_size_16 != 0 -> FAT16 shape
|
||||
block0[38] = 0x29; // extended boot signature
|
||||
std.mem.writeInt(u32, block0[39..43], 0x12345678, .little); // volume id
|
||||
const disk = RamDisk{ .sectors = &block0 };
|
||||
const v = firstVolume(disk.reader(), 65536).?;
|
||||
try std.testing.expectEqual(FilesystemKind.fat, v.signature);
|
||||
try std.testing.expectEqual(Rung.fat_serial, v.identity.rung);
|
||||
}
|
||||
|
||||
test "a partition that runs past the device is skipped, not trusted" {
|
||||
@@ -110,7 +432,189 @@ test "a partition that runs past the device is skipped, not trusted" {
|
||||
block0[462 + 4] = 0x0c;
|
||||
std.mem.writeInt(u32, block0[462 + 8 ..][0..4], 2048, .little);
|
||||
std.mem.writeInt(u32, block0[462 + 12 ..][0..4], 1000, .little);
|
||||
const v = firstVolume(&block0, 200000).?;
|
||||
const disk = RamDisk{ .sectors = &block0 };
|
||||
const v = firstVolume(disk.reader(), 200000).?;
|
||||
try std.testing.expectEqual(@as(u64, 2048), v.base_lba); // the fitting one, not the overflowing one
|
||||
try std.testing.expectEqual(@as(u64, 1000), v.block_count);
|
||||
}
|
||||
|
||||
/// A single 128-byte GPT partition entry for the tests.
|
||||
fn gptEntry(type_nonzero: bool, unique_guid: u128, start: u64, end: u64) [128]u8 {
|
||||
var e = [_]u8{0} ** 128;
|
||||
if (type_nonzero) e[0] = 0x01; // any non-zero byte makes the type GUID non-zero
|
||||
std.mem.writeInt(u128, e[16..32], unique_guid, .little);
|
||||
std.mem.writeInt(u64, e[32..40], start, .little);
|
||||
std.mem.writeInt(u64, e[40..48], end, .little);
|
||||
return e;
|
||||
}
|
||||
|
||||
/// Lay out a disk with `entry_size`-spaced GPT entries: protective MBR (LBA 0),
|
||||
/// GPT header with a correct CRC (LBA 1), the entry array (LBA 2+).
|
||||
fn buildGptDiskSized(disk: []u8, entries: []const [128]u8, entry_size: u32) void {
|
||||
@memset(disk, 0);
|
||||
disk[510] = 0x55;
|
||||
disk[511] = 0xAA;
|
||||
disk[446 + 4] = 0xEE; // protective entry type
|
||||
std.mem.writeInt(u32, disk[446 + 8 ..][0..4], 1, .little);
|
||||
std.mem.writeInt(u32, disk[446 + 12 ..][0..4], 0xFFFFFFFF, .little);
|
||||
const h = disk[sector_bytes..][0..sector_bytes];
|
||||
@memcpy(h[0..8], gpt_signature);
|
||||
std.mem.writeInt(u32, h[12..16], 92, .little); // header_size
|
||||
std.mem.writeInt(u64, h[72..80], 2, .little); // partition_entry_lba
|
||||
std.mem.writeInt(u32, h[80..84], @intCast(entries.len), .little);
|
||||
std.mem.writeInt(u32, h[84..88], entry_size, .little); // size_of_partition_entry
|
||||
@memset(h[16..20], 0);
|
||||
std.mem.writeInt(u32, h[16..20], crc32(h[0..92]), .little);
|
||||
const step: usize = @intCast(entry_size);
|
||||
var i: usize = 0;
|
||||
while (i < entries.len) : (i += 1) {
|
||||
const abs = 2 * sector_bytes + i * step;
|
||||
@memcpy(disk[abs..][0..128], &entries[i]);
|
||||
}
|
||||
}
|
||||
|
||||
/// The common 128-byte-entry case.
|
||||
fn buildGptDisk(disk: []u8, entries: []const [128]u8) void {
|
||||
buildGptDiskSized(disk, entries, 128);
|
||||
}
|
||||
|
||||
test "a GPT disk yields the partition GUID as the identity id" {
|
||||
var disk = [_]u8{0} ** (4 * sector_bytes);
|
||||
const guid: u128 = 0x112233445566778899AABBCCDDEEFF00;
|
||||
const entries = [_][128]u8{gptEntry(true, guid, 2048, 4095)};
|
||||
buildGptDisk(&disk, &entries);
|
||||
const rd = RamDisk{ .sectors = &disk };
|
||||
const v = firstVolume(rd.reader(), 200000).?;
|
||||
try std.testing.expectEqual(@as(u64, 2048), v.base_lba);
|
||||
try std.testing.expectEqual(@as(u64, 2048), v.block_count); // 4095 - 2048 + 1
|
||||
try std.testing.expectEqual(Rung.gpt_guid, v.identity.rung);
|
||||
try std.testing.expectEqual(guid, v.identity.key);
|
||||
}
|
||||
|
||||
test "a GPT entry past the device is skipped; an all-out-of-range table is no volume" {
|
||||
var disk = [_]u8{0} ** (4 * sector_bytes);
|
||||
const entries = [_][128]u8{
|
||||
gptEntry(true, 0xAAA, 2048, 999999), // ends past a 200000-block device
|
||||
gptEntry(true, 0xBBB, 4096, 8191), // fits
|
||||
};
|
||||
buildGptDisk(&disk, &entries);
|
||||
const rd = RamDisk{ .sectors = &disk };
|
||||
const v = firstVolume(rd.reader(), 200000).?;
|
||||
try std.testing.expectEqual(@as(u64, 4096), v.base_lba); // the fitting one, not the overflowing one
|
||||
try std.testing.expectEqual(@as(u128, 0xBBB), v.identity.key);
|
||||
|
||||
var solo_disk = [_]u8{0} ** (4 * sector_bytes);
|
||||
const solo = [_][128]u8{gptEntry(true, 0xAAA, 2048, 999999)};
|
||||
buildGptDisk(&solo_disk, &solo);
|
||||
const rd2 = RamDisk{ .sectors = &solo_disk };
|
||||
try std.testing.expect(firstVolume(rd2.reader(), 200000) == null);
|
||||
}
|
||||
|
||||
test "a protective MBR with a broken GPT header is not a volume" {
|
||||
var disk = [_]u8{0} ** (4 * sector_bytes);
|
||||
const entries = [_][128]u8{gptEntry(true, 0xCCC, 2048, 4095)};
|
||||
buildGptDisk(&disk, &entries);
|
||||
disk[sector_bytes] = 'X'; // wreck the 'EFI PART' signature
|
||||
const rd = RamDisk{ .sectors = &disk };
|
||||
try std.testing.expect(firstVolume(rd.reader(), 200000) == null);
|
||||
|
||||
var bad_crc = [_]u8{0} ** (4 * sector_bytes);
|
||||
buildGptDisk(&bad_crc, &entries);
|
||||
bad_crc[sector_bytes + 16] ^= 0xFF; // corrupt a header-CRC byte
|
||||
const rd2 = RamDisk{ .sectors = &bad_crc };
|
||||
try std.testing.expect(firstVolume(rd2.reader(), 200000) == null);
|
||||
}
|
||||
|
||||
test "a bare FAT32 reports its volume serial and label as the identity" {
|
||||
var block0 = [_]u8{0} ** 512;
|
||||
block0[510] = 0x55;
|
||||
block0[511] = 0xAA;
|
||||
std.mem.writeInt(u16, block0[22..24], 0, .little); // fat_size_16 == 0 → FAT32
|
||||
block0[66] = 0x29; // FAT32 extended boot signature
|
||||
std.mem.writeInt(u32, block0[67..71], 0x12345678, .little); // BS_VolID
|
||||
@memcpy(block0[71..82], "DANOS "); // BS_VolLab, space-padded to 11
|
||||
const disk = RamDisk{ .sectors = &block0 };
|
||||
const v = firstVolume(disk.reader(), 65536).?;
|
||||
try std.testing.expectEqual(@as(u64, 0), v.base_lba);
|
||||
try std.testing.expectEqual(Rung.fat_serial, v.identity.rung);
|
||||
try std.testing.expectEqual(@as(u128, 0x12345678), v.identity.key);
|
||||
try std.testing.expectEqualStrings("DANOS", v.identity.labelSlice());
|
||||
}
|
||||
|
||||
test "an MBR FAT partition prefers the volume serial; a non-FAT partition keeps rung 4" {
|
||||
var disk = [_]u8{0} ** (3 * 512);
|
||||
disk[510] = 0x55;
|
||||
disk[511] = 0xAA;
|
||||
std.mem.writeInt(u32, disk[440..444], 0xDEADBEEF, .little);
|
||||
disk[446 + 4] = 0x0c; // FAT32-LBA partition
|
||||
std.mem.writeInt(u32, disk[446 + 8 ..][0..4], 1, .little); // start LBA 1
|
||||
std.mem.writeInt(u32, disk[446 + 12 ..][0..4], 2, .little); // size 2
|
||||
const vbr = disk[512..][0..512]; // a FAT16 VBR at the partition start
|
||||
vbr[510] = 0x55;
|
||||
vbr[511] = 0xAA;
|
||||
std.mem.writeInt(u16, vbr[22..24], 0x0080, .little); // fat_size_16 != 0 → FAT16
|
||||
vbr[38] = 0x29; // FAT12/16 extended boot signature
|
||||
std.mem.writeInt(u32, vbr[39..43], 0xCAFEBABE, .little);
|
||||
@memcpy(vbr[43..54], "MYVOL ");
|
||||
const rd = RamDisk{ .sectors = &disk };
|
||||
const v = firstVolume(rd.reader(), 200000).?;
|
||||
try std.testing.expectEqual(@as(u64, 1), v.base_lba);
|
||||
try std.testing.expectEqual(Rung.fat_serial, v.identity.rung);
|
||||
try std.testing.expectEqual(@as(u128, 0xCAFEBABE), v.identity.key);
|
||||
try std.testing.expectEqualStrings("MYVOL", v.identity.labelSlice());
|
||||
|
||||
// A partition whose VBR is not an extended FAT falls back to the rung-4 id.
|
||||
var plain = [_]u8{0} ** (3 * 512);
|
||||
@memcpy(plain[0..512], disk[0..512]); // same MBR; LBA 1 left blank
|
||||
const rd2 = RamDisk{ .sectors = &plain };
|
||||
const v2 = firstVolume(rd2.reader(), 200000).?;
|
||||
try std.testing.expectEqual(Rung.mbr_index, v2.identity.rung);
|
||||
try std.testing.expectEqual((@as(u128, 0xDEADBEEF) << 8) | 0, v2.identity.key);
|
||||
}
|
||||
|
||||
test "GPT with 256-byte entries reads the non-128 offset arithmetic correctly" {
|
||||
// With entry_size 256, entry 1 lands at offset 256 of the same sector (LBA 2).
|
||||
// Put the only valid entry at index 1 so the off = (i*entry_size) % 512 path
|
||||
// (256, not 0) is exercised — the sharp edge the 128-byte tests never hit.
|
||||
var disk = [_]u8{0} ** (5 * sector_bytes);
|
||||
const entries = [_][128]u8{
|
||||
gptEntry(false, 0, 0, 0), // index 0: unused (type GUID zero)
|
||||
gptEntry(true, 0xF00D, 4096, 8191), // index 1: at offset 256
|
||||
};
|
||||
buildGptDiskSized(&disk, &entries, 256);
|
||||
const rd = RamDisk{ .sectors = &disk };
|
||||
const v = firstVolume(rd.reader(), 200000).?;
|
||||
try std.testing.expectEqual(@as(u64, 4096), v.base_lba);
|
||||
try std.testing.expectEqual(Rung.gpt_guid, v.identity.rung);
|
||||
try std.testing.expectEqual(@as(u128, 0xF00D), v.identity.key);
|
||||
}
|
||||
|
||||
// A fixture-sized volume buffer for the multi-volume tests, named so the bounds
|
||||
// gate (which flags literal array lengths) stays quiet: a test input.
|
||||
const test_volume_slots = 4;
|
||||
|
||||
test "allVolumes returns every fitting MBR partition with distinct identities" {
|
||||
var block0 = [_]u8{0} ** 512;
|
||||
block0[510] = 0x55;
|
||||
block0[511] = 0xAA;
|
||||
std.mem.writeInt(u32, block0[440..444], 0xDEADBEEF, .little);
|
||||
// partition 0: start 2048, size 1000
|
||||
block0[446 + 4] = 0x0c;
|
||||
std.mem.writeInt(u32, block0[446 + 8 ..][0..4], 2048, .little);
|
||||
std.mem.writeInt(u32, block0[446 + 12 ..][0..4], 1000, .little);
|
||||
// partition 1: start 4096, size 2000
|
||||
block0[462 + 4] = 0x0c;
|
||||
std.mem.writeInt(u32, block0[462 + 8 ..][0..4], 4096, .little);
|
||||
std.mem.writeInt(u32, block0[462 + 12 ..][0..4], 2000, .little);
|
||||
const disk = RamDisk{ .sectors = &block0 };
|
||||
var vols: [test_volume_slots]Volume = undefined;
|
||||
const n = allVolumes(disk.reader(), 200000, &vols);
|
||||
try std.testing.expectEqual(@as(usize, 2), n); // both partitions, not just the first
|
||||
try std.testing.expectEqual(@as(u64, 2048), vols[0].base_lba);
|
||||
try std.testing.expectEqual(@as(u64, 4096), vols[1].base_lba);
|
||||
// distinct rung-4 identities (no FAT VBR at those LBAs): index 0 vs 1.
|
||||
try std.testing.expectEqual((@as(u128, 0xDEADBEEF) << 8) | 0, vols[0].identity.key);
|
||||
try std.testing.expectEqual((@as(u128, 0xDEADBEEF) << 8) | 1, vols[1].identity.key);
|
||||
// firstVolume (the 1-buffer case) still returns just the first.
|
||||
try std.testing.expectEqual(@as(u64, 2048), firstVolume(disk.reader(), 200000).?.base_lba);
|
||||
}
|
||||
|
||||
@@ -8,10 +8,12 @@
|
||||
//! supervises the filesystems it spawns, exactly as the device manager
|
||||
//! supervises drivers.
|
||||
//!
|
||||
//! This increment (V3b) is the flip: the FAT service stops acquiring its own
|
||||
//! volume and is spawned here instead, confined to its partition, and handed
|
||||
//! its channel over the volume-manager protocol. Single volume for now; the
|
||||
//! mount map (volumes.csv) and multi-volume land next.
|
||||
//! The manager holds a table of adopted storage DEVICES and a table of the
|
||||
//! VOLUMES on them: it adopts every storage device the device-manager tree
|
||||
//! carries, probes each one's whole partition table, and spawns one filesystem
|
||||
//! process per volume — each confined to its partition's badge-scoped block
|
||||
//! range, each supervised with its own budget. A device leaving the tree takes
|
||||
//! its volumes with it.
|
||||
|
||||
const std = @import("std");
|
||||
const channel = @import("channel");
|
||||
@@ -26,60 +28,146 @@ const process = @import("process");
|
||||
const service = @import("service");
|
||||
const time = @import("time");
|
||||
const envelope = @import("envelope");
|
||||
const fs = @import("file-system");
|
||||
const partition = @import("partition.zig");
|
||||
const filesystem_map = @import("filesystem-map.zig");
|
||||
const volume_map = @import("volume-map.zig");
|
||||
|
||||
const Serve = volume_manager_protocol.Protocol.Provider(void);
|
||||
const Invocation = envelope.Invocation;
|
||||
const Answer = envelope.Answer;
|
||||
|
||||
/// The single volume this increment handles: its provider channel, its block
|
||||
/// sub-range, its identity, the id it is addressed by, and the filesystem
|
||||
/// process serving it (0 until spawned; reset on death for respawn).
|
||||
const Volume = struct {
|
||||
storage: block.Device,
|
||||
storage_device_id: u64, // the device-manager id this volume's provider serves
|
||||
base_lba: u64,
|
||||
block_count: u64,
|
||||
identity: u64,
|
||||
id: u64,
|
||||
filesystem_pid: u32 = 0,
|
||||
/// One adopted storage device: the block channel to its provider (opened once and
|
||||
/// shared — refcounted per confined filesystem via the hello reply) and the
|
||||
/// device-manager id it serves. A device leaving the tree takes its volumes.
|
||||
const StorageDevice = struct {
|
||||
used: bool = false,
|
||||
device_id: u64 = 0,
|
||||
channel: block.Device = undefined,
|
||||
};
|
||||
|
||||
/// The filesystem binary a probed volume is served by. The signature->binary
|
||||
/// map (filesystems.csv) lands with the identity ladder; for now every FAT-shaped
|
||||
/// volume gets the FAT service.
|
||||
const filesystem_binary = "/system/services/fat";
|
||||
const volume_id: u64 = 1;
|
||||
/// One volume: which device serves it, its block sub-range, its content
|
||||
/// identity, the id it is addressed by, the service binary + mount path it was
|
||||
/// spawned with, the filesystem process serving it, and its own supervision
|
||||
/// budget (so one volume's crash loop never touches another's).
|
||||
const Volume = struct {
|
||||
used: bool = false,
|
||||
device_id: u64 = 0,
|
||||
base_lba: u64 = 0,
|
||||
block_count: u64 = 0,
|
||||
identity: partition.Identity = .{ .rung = .anonymous },
|
||||
id: u64 = 0,
|
||||
binary: []const u8 = "",
|
||||
mount_prefix: []const u8 = "",
|
||||
filesystem_pid: u32 = 0,
|
||||
// Per-volume supervision, mirroring the device manager's: a clean exit is not
|
||||
// restarted, a fault restarts with backoff, a fast crash loop gives up.
|
||||
restarts: u32 = 0,
|
||||
spawn_ns: u64 = 0,
|
||||
failed: bool = false,
|
||||
restart_pending: bool = false,
|
||||
restart_due_ns: u64 = 0,
|
||||
};
|
||||
|
||||
// The mount map, read from configuration at boot (the policy home, storage-
|
||||
// architecture.md): filesystems.csv (content signature -> service binary) and
|
||||
// volumes.csv (an optional id -> mount-prefix override). The sources are held
|
||||
// for the process life so the parsed rules' slices into them stay valid.
|
||||
/// bound: bytes of filesystems.csv / volumes.csv the manager reads
|
||||
/// decided-by: ours
|
||||
/// protects: the config source buffers below
|
||||
/// at-limit: truncate - a longer file is cut; a row split by the cut is malformed
|
||||
/// observed-by: the per-file "malformed/truncated" log line
|
||||
const config_source_bytes = 2048;
|
||||
var filesystems_source: [config_source_bytes]u8 = undefined;
|
||||
var volumes_source: [config_source_bytes]u8 = undefined;
|
||||
/// bound: filesystem-map rules held (one per content signature)
|
||||
/// decided-by: ours
|
||||
/// protects: the filesystem_rules table
|
||||
/// at-limit: truncate - extra rows are dropped and the "truncated" note logged
|
||||
/// observed-by: the "truncated" log line
|
||||
const maximum_filesystem_rules = 8;
|
||||
/// bound: volumes.csv override rows held (one per pinned volume id)
|
||||
/// decided-by: ours
|
||||
/// protects: the volume_rules table
|
||||
/// at-limit: truncate - extra rows are dropped and the "truncated" note logged
|
||||
/// observed-by: the "truncated" log line
|
||||
const maximum_volume_rules = 64;
|
||||
var filesystem_rules: [maximum_filesystem_rules]filesystem_map.Rule = undefined;
|
||||
var filesystem_rule_count: usize = 0;
|
||||
var volume_rules: [maximum_volume_rules]volume_map.Override = undefined;
|
||||
var volume_rule_count: usize = 0;
|
||||
/// bound: bytes of a composed /volumes/<id> mount path
|
||||
/// decided-by: ours
|
||||
/// protects: the per-volume mount_prefix buffers below
|
||||
/// at-limit: truncate - bufPrint fails; the volume mounts at a fallback path (logged)
|
||||
/// observed-by: the fallback path in the log
|
||||
const mount_path_maximum = 64;
|
||||
|
||||
/// bound: volumes the manager serves at once
|
||||
/// decided-by: ours
|
||||
/// protects: the volumes table and its per-volume mount-path buffers
|
||||
/// at-limit: truncate - a further partition is left unserved and logged (real
|
||||
/// machines carry a handful of volumes, far under this)
|
||||
/// observed-by: the "volume table full" log line
|
||||
const maximum_volumes = 16;
|
||||
/// bound: storage devices the manager adopts at once
|
||||
/// decided-by: ours
|
||||
/// protects: the devices table
|
||||
/// at-limit: truncate - a further device is left unadopted and logged
|
||||
/// observed-by: the "device table full" log line
|
||||
const maximum_devices = 8;
|
||||
var devices = [_]StorageDevice{.{}} ** maximum_devices;
|
||||
var volumes = [_]Volume{.{}} ** maximum_volumes;
|
||||
/// Each volume's composed default mount path lives in its slot's buffer; a
|
||||
/// volumes.csv override is used in place (a slice into volumes_source, no buffer).
|
||||
var mount_prefix_bufs: [maximum_volumes][mount_path_maximum]u8 = undefined;
|
||||
var next_volume_id: u64 = 1; // monotonic — never reused, so a stale id can't address the wrong child
|
||||
|
||||
var service_endpoint: ipc.Handle = 0;
|
||||
var manager_handle: ?ipc.Handle = null;
|
||||
var bounce: memory.DmaRegion = undefined;
|
||||
var bounce_ready = false;
|
||||
/// The currently-mounted volume, or null while no storage is present. The whole
|
||||
/// removal lifecycle is this field going null and back: the poll sees the
|
||||
/// storage provider leave the device tree (a pulled stick), kills the filesystem
|
||||
/// and clears this; when it returns, the poll re-acquires and re-mounts.
|
||||
var volume: ?Volume = null;
|
||||
var logged_no_volume = false;
|
||||
/// How often the poll checks whether the storage provider is present. Fast
|
||||
/// enough that an unplug unmounts promptly; the poll is a bare device-manager
|
||||
/// enumerate, no channel work, so it is cheap to run continuously.
|
||||
/// How often the poll checks device presence and fires due restarts. Fast enough
|
||||
/// that an unplug unmounts promptly; the poll is a bare device-manager enumerate,
|
||||
/// no channel work, so it is cheap to run continuously.
|
||||
const poll_interval_ms = 500;
|
||||
|
||||
// Filesystem supervision, mirroring the device manager's (device-manager.zig):
|
||||
// a clean exit is not restarted, a fault restarts with backoff, and a fast
|
||||
// crash loop gives up rather than spinning. Without this a faulting filesystem
|
||||
// respawns in a zero-delay loop.
|
||||
// Filesystem supervision, mirroring the device manager's (device-manager.zig).
|
||||
const fast_death_ns: u64 = 2_000_000_000;
|
||||
const crash_loop_cap: u32 = 3;
|
||||
const backoff_base_ms: u64 = 300;
|
||||
var fs_restarts: u32 = 0;
|
||||
var fs_spawn_ns: u64 = 0;
|
||||
var fs_failed = false;
|
||||
/// A fat restart is due at `restart_due_ns`; the poll loop performs it once the
|
||||
/// backoff has elapsed (one timer, folded into the poll — no second timer).
|
||||
var restart_pending = false;
|
||||
var restart_due_ns: u64 = 0;
|
||||
/// bytes to format a u64 volume id as decimal (20 digits fit)
|
||||
const id_decimal_bytes = 24;
|
||||
|
||||
// --- table lookups -----------------------------------------------------------
|
||||
|
||||
fn deviceById(id: u64) ?*StorageDevice {
|
||||
for (&devices) |*d| if (d.used and d.device_id == id) return d;
|
||||
return null;
|
||||
}
|
||||
fn claimDevice() ?*StorageDevice {
|
||||
for (&devices) |*d| if (!d.used) return d;
|
||||
return null;
|
||||
}
|
||||
fn volumeById(id: u64) ?*Volume {
|
||||
for (&volumes) |*v| if (v.used and v.id == id) return v;
|
||||
return null;
|
||||
}
|
||||
fn volumeByPid(pid: u32) ?*Volume {
|
||||
for (&volumes) |*v| if (v.used and v.filesystem_pid == pid) return v;
|
||||
return null;
|
||||
}
|
||||
fn firstUsedVolume() ?*Volume {
|
||||
for (&volumes) |*v| if (v.used) return v;
|
||||
return null;
|
||||
}
|
||||
fn claimVolumeIndex() ?usize {
|
||||
for (&volumes, 0..) |*v, i| if (!v.used) return i;
|
||||
return null;
|
||||
}
|
||||
|
||||
// --- device-manager plumbing -------------------------------------------------
|
||||
|
||||
fn deviceManager() ?ipc.Handle {
|
||||
if (manager_handle) |h| return h;
|
||||
@@ -90,13 +178,12 @@ fn deviceManager() ?ipc.Handle {
|
||||
|
||||
const OpenedStorage = struct { device_id: u64, device: block.Device };
|
||||
|
||||
/// The first mass-storage provider whose block channel actually opens, with its
|
||||
/// device id. A device-manager tree can carry more than one entry of the
|
||||
/// mass-storage identity — a phantom that no driver is bound to answers a
|
||||
/// consumer hello with NO channel — so this tries each and takes the first that
|
||||
/// yields a channel, exactly as a filesystem's own acquisition loop does.
|
||||
/// Called only when there is no volume (an insertion), so the hellos it makes
|
||||
/// are not per-poll churn.
|
||||
/// The first mass-storage provider whose block channel opens and is NOT already
|
||||
/// adopted, with its device id. A device-manager tree can carry more than one
|
||||
/// entry of the mass-storage identity — a phantom that no driver is bound to
|
||||
/// answers a consumer hello with NO channel — so this tries each and takes the
|
||||
/// first that yields a channel. Skips already-adopted devices so a re-poll does
|
||||
/// not re-open a device it already serves.
|
||||
fn openAnyStorage() ?OpenedStorage {
|
||||
const manager = deviceManager() orelse return null;
|
||||
const Entry = device_manager_protocol.ChildEntry;
|
||||
@@ -116,6 +203,7 @@ fn openAnyStorage() ?OpenedStorage {
|
||||
const entry = std.mem.bytesToValue(Entry, tail[index * @sizeOf(Entry) ..][0..@sizeOf(Entry)]);
|
||||
if (entry.device_id == device_manager_protocol.no_device) continue;
|
||||
if ((entry.identity >> 16) & 0xff != 0x08 or (entry.identity >> 8) & 0xff != 0x06) continue;
|
||||
if (deviceById(entry.device_id) != null) continue; // already adopted
|
||||
const exchanged = driver.helloOn(manager, .consumer, entry.device_id, null, true) orelse continue;
|
||||
const provider = exchanged.channel orelse continue; // a phantom / not-yet-bound entry
|
||||
return .{ .device_id = entry.device_id, .device = .{ .endpoint = provider } };
|
||||
@@ -126,7 +214,7 @@ fn openAnyStorage() ?OpenedStorage {
|
||||
|
||||
/// Whether `device_id` is still in the device-manager tree — a bare enumerate,
|
||||
/// no consumer-hello, so it is cheap to call every poll. This is how removal is
|
||||
/// detected: the specific device the mounted volume sits on disappears.
|
||||
/// detected: the specific device a mounted volume sits on disappears.
|
||||
fn isDevicePresent(device_id: u64) bool {
|
||||
const manager = deviceManager() orelse return false;
|
||||
const Entry = device_manager_protocol.ChildEntry;
|
||||
@@ -150,118 +238,205 @@ fn isDevicePresent(device_id: u64) bool {
|
||||
}
|
||||
}
|
||||
|
||||
/// Spawn the filesystem for `v`, confine it to the volume's range, and record
|
||||
/// its pid. The confinement is defined for the fresh pid BEFORE the filesystem
|
||||
/// runs, so its first read is already bounded; the volume manager is the
|
||||
/// confinement controller (it defines the first range on the device).
|
||||
// --- lifecycle ---------------------------------------------------------------
|
||||
|
||||
/// Spawn the filesystem for `v`, confine it to the volume's range on its device's
|
||||
/// channel, and record its pid. The confinement is defined for the fresh pid
|
||||
/// BEFORE the filesystem runs, so its first read is already bounded; the volume
|
||||
/// manager is the confinement controller (it defines the first range on the
|
||||
/// device).
|
||||
fn spawnFilesystem(v: *Volume) void {
|
||||
if (fs_failed) return;
|
||||
const pid = process.spawnSupervised(filesystem_binary, &.{"1"}, service_endpoint) orelse {
|
||||
if (v.failed) return;
|
||||
const dev = deviceById(v.device_id) orelse return; // its device left — poll will clean up
|
||||
var id_str_buf: [id_decimal_bytes]u8 = undefined;
|
||||
const id_str = std.fmt.bufPrint(&id_str_buf, "{d}", .{v.id}) catch "1";
|
||||
const pid = process.spawnSupervised(v.binary, &.{ id_str, v.mount_prefix }, service_endpoint) orelse {
|
||||
_ = logging.write("volume-manager: could not spawn the filesystem; retrying\n");
|
||||
armRestart();
|
||||
armRestart(v);
|
||||
return;
|
||||
};
|
||||
if (!v.storage.defineRange(pid, v.base_lba, v.block_count)) {
|
||||
if (!dev.channel.defineRange(pid, v.base_lba, v.block_count)) {
|
||||
_ = logging.write("volume-manager: could not confine the filesystem to its volume; retrying\n");
|
||||
_ = process.kill(pid);
|
||||
armRestart();
|
||||
armRestart(v);
|
||||
return;
|
||||
}
|
||||
v.filesystem_pid = pid;
|
||||
fs_spawn_ns = time.clock();
|
||||
std.log.info("volume 0x{x} -> {s} (pid {d}), lba {d}, {d} blocks", .{ v.identity, filesystem_binary, pid, v.base_lba, v.block_count });
|
||||
v.spawn_ns = time.clock();
|
||||
std.log.info("volume 0x{x} -> {s} (pid {d}), lba {d}, {d} blocks", .{ v.identity.key, v.binary, pid, v.base_lba, v.block_count });
|
||||
}
|
||||
|
||||
/// Schedule a fat restart after backoff; the poll loop performs it once due.
|
||||
fn armRestart() void {
|
||||
const delay = if (fs_restarts == 0) backoff_base_ms else backoff_base_ms << @intCast(@min(fs_restarts - 1, 5));
|
||||
restart_due_ns = time.clock() + delay * 1_000_000;
|
||||
restart_pending = true;
|
||||
/// Schedule a restart for `v` after backoff; the poll loop performs it once due.
|
||||
fn armRestart(v: *Volume) void {
|
||||
const delay = if (v.restarts == 0) backoff_base_ms else backoff_base_ms << @intCast(@min(v.restarts - 1, 5));
|
||||
v.restart_due_ns = time.clock() + delay * 1_000_000;
|
||||
v.restart_pending = true;
|
||||
}
|
||||
|
||||
/// A storage provider just appeared: open its channel, read block 0, parse the
|
||||
/// volume, and spawn its filesystem. On any failure the channel is closed (so a
|
||||
/// present-but-unreadable device does not leak a handle every poll) and `volume`
|
||||
/// stays null — the next poll retries. A fresh medium gets a fresh supervision
|
||||
/// budget.
|
||||
fn bringUpVolume() void {
|
||||
/// Compose a volume's mount path (its id-path `/volumes/<id>`, or a volumes.csv
|
||||
/// override) into its slot's buffer, and return the slice.
|
||||
fn composeMountPrefix(slot: usize, identity: partition.Identity) []const u8 {
|
||||
var id_buf: [volume_map.id_maximum]u8 = undefined;
|
||||
const id = volume_map.idString(identity, &id_buf);
|
||||
return volume_map.overrideFor(volume_rules[0..volume_rule_count], id) orelse
|
||||
(std.fmt.bufPrint(&mount_prefix_bufs[slot], "/volumes/{s}", .{id}) catch "/volumes/unknown");
|
||||
}
|
||||
|
||||
/// Adopt the next present, not-yet-adopted storage device: take its channel,
|
||||
/// probe its whole partition table, and spawn a filesystem per volume it carries.
|
||||
/// Returns true when it consumed a device (so the caller can loop to adopt every
|
||||
/// present device in one tick), false when none remain or the device table is full.
|
||||
///
|
||||
/// A device is adopted exactly once and kept until it leaves the tree — even when
|
||||
/// it carries no volume we can serve, or its geometry cannot be read. Keeping the
|
||||
/// empty/unreadable device adopted (rather than dropping and re-probing) is what
|
||||
/// lets openAnyStorage advance PAST it to the devices behind it; dropping it would
|
||||
/// make openAnyStorage hand back the same unservable device every tick and starve
|
||||
/// the rest. A genuine removal frees the slot (removeDevice); a re-insert gets a
|
||||
/// fresh device id and is probed anew.
|
||||
fn bringUpVolume() bool {
|
||||
if (!bounce_ready) {
|
||||
bounce = memory.dmaAlloc(512, memory.dma_coherent | memory.dma_shareable) orelse return;
|
||||
bounce = memory.dmaAlloc(512, memory.dma_coherent | memory.dma_shareable) orelse return false;
|
||||
bounce_ready = true;
|
||||
}
|
||||
const opened = openAnyStorage() orelse return;
|
||||
const opened = openAnyStorage() orelse return false;
|
||||
const dev = claimDevice() orelse {
|
||||
_ = logging.write("volume-manager: device table full; a storage device is left unadopted\n");
|
||||
_ = ipc.close(opened.device.endpoint);
|
||||
return false;
|
||||
};
|
||||
dev.* = .{ .used = true, .device_id = opened.device_id, .channel = opened.device };
|
||||
// Consume this device's medium_changed events (the second of the removal
|
||||
// lifecycle's two triggers: the device stays in the tree while its medium
|
||||
// leaves — a card reader, an eject). Best effort: a provider that never
|
||||
// publishes the event simply never wakes us, and device-pull is still caught
|
||||
// by the presence poll.
|
||||
_ = opened.device.subscribeMedium(service_endpoint);
|
||||
const device = opened.device;
|
||||
// Attach the read buffer to THIS device (a no-op without an enforcing IOMMU).
|
||||
// The handle is kept, not closed, so it can be re-attached to the next
|
||||
// device after a replug.
|
||||
// The handle is kept, not closed, so it can be re-attached after a replug. A
|
||||
// failed attach or geometry read leaves the device adopted but empty — we just
|
||||
// cannot read it, and the slot still watches it for removal.
|
||||
if (bounce.handle) |handle| {
|
||||
if (!device.attach(handle)) {
|
||||
_ = ipc.close(device.endpoint);
|
||||
return;
|
||||
_ = logging.write("volume-manager: could not attach the read buffer to a storage device; no volume served\n");
|
||||
return true;
|
||||
}
|
||||
}
|
||||
const geometry = device.geometry() orelse {
|
||||
_ = ipc.close(device.endpoint);
|
||||
return;
|
||||
_ = logging.write("volume-manager: could not read a storage device's geometry; no volume served\n");
|
||||
return true;
|
||||
};
|
||||
if (!device.read(0, 1, bounce.physical)) {
|
||||
_ = ipc.close(device.endpoint);
|
||||
return;
|
||||
const ProbeReader = struct {
|
||||
device: block.Device,
|
||||
fn readSector(context: *anyopaque, lba: u64, buffer: *[partition.sector_bytes]u8) bool {
|
||||
const self: *@This() = @ptrCast(@alignCast(context));
|
||||
if (!self.device.read(lba, 1, bounce.physical)) return false;
|
||||
const src: [*]const u8 = @ptrFromInt(bounce.virtual);
|
||||
@memcpy(buffer, src[0..partition.sector_bytes]);
|
||||
return true;
|
||||
}
|
||||
const sector: [*]const u8 = @ptrFromInt(bounce.virtual);
|
||||
const found = partition.firstVolume(sector[0..512], geometry.block_count) orelse {
|
||||
if (!logged_no_volume) {
|
||||
_ = logging.write("volume-manager: storage present but no recognizable volume\n");
|
||||
logged_no_volume = true;
|
||||
}
|
||||
_ = ipc.close(device.endpoint);
|
||||
return;
|
||||
};
|
||||
logged_no_volume = false;
|
||||
fs_restarts = 0;
|
||||
fs_failed = false;
|
||||
restart_pending = false;
|
||||
volume = .{ .storage = device, .storage_device_id = opened.device_id, .base_lba = found.base_lba, .block_count = found.block_count, .identity = found.identity, .id = volume_id };
|
||||
spawnFilesystem(&volume.?);
|
||||
var probe = ProbeReader{ .device = device };
|
||||
const reader = partition.SectorReader{ .context = &probe, .readFn = ProbeReader.readSector };
|
||||
var found: [maximum_volumes]partition.Volume = undefined;
|
||||
const n = partition.allVolumes(reader, geometry.block_count, found[0..]);
|
||||
if (n == 0) {
|
||||
std.log.info("device {d} present but carries no recognizable volume", .{dev.device_id});
|
||||
return true;
|
||||
}
|
||||
for (found[0..n]) |fv| {
|
||||
// Pick the service binary from the volume's content signature. A signature
|
||||
// no filesystems.csv row serves goes unserved (logged), like an unbound
|
||||
// device — the manager does not guess.
|
||||
const binary = filesystem_map.match(filesystem_rules[0..filesystem_rule_count], fv.signature) orelse {
|
||||
_ = logging.write("volume-manager: no filesystem serves this volume's content; unserved\n");
|
||||
continue;
|
||||
};
|
||||
const slot = claimVolumeIndex() orelse {
|
||||
_ = logging.write("volume-manager: volume table full; a volume is left unserved\n");
|
||||
break;
|
||||
};
|
||||
volumes[slot] = .{
|
||||
.used = true,
|
||||
.device_id = dev.device_id,
|
||||
.base_lba = fv.base_lba,
|
||||
.block_count = fv.block_count,
|
||||
.identity = fv.identity,
|
||||
.id = next_volume_id,
|
||||
.binary = binary,
|
||||
.mount_prefix = composeMountPrefix(slot, fv.identity),
|
||||
};
|
||||
next_volume_id += 1;
|
||||
spawnFilesystem(&volumes[slot]);
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
/// The storage provider left the device tree (a pulled stick): kill the
|
||||
/// filesystem so its mounts are retired. Retirement is lazy, not an eager
|
||||
/// death-time sweep — killing the process marks the filesystem's backend
|
||||
/// endpoint dead, and the VFS router drops each mount that endpoint backed on
|
||||
/// the next path resolution under it (that resolve frees the slot and returns
|
||||
/// not_found). Then drop the now-dead channel and clear the volume; the next
|
||||
/// poll that sees storage return re-mounts.
|
||||
fn removeVolume() void {
|
||||
const v = volume orelse return;
|
||||
/// Close a device's channel and free its slot. No volumes are touched (the caller
|
||||
/// ensures none remain, or there never were any).
|
||||
fn dropDevice(dev: *StorageDevice) void {
|
||||
// Free the driver's subscriber slot before the channel closes. On a still-live
|
||||
// channel (a medium eject) this frees the slot; on a dead one (a device pull)
|
||||
// the call fails fast and the exit sweep frees it anyway.
|
||||
_ = dev.channel.unsubscribeMedium();
|
||||
_ = ipc.close(dev.channel.endpoint);
|
||||
dev.* = .{};
|
||||
}
|
||||
|
||||
/// Retire one volume: kill its filesystem so its mounts are retired. Retirement
|
||||
/// is lazy, not an eager death-time sweep — killing the process marks the
|
||||
/// filesystem's backend endpoint dead, and the VFS router drops each mount that
|
||||
/// endpoint backed on the next path resolution under it (that resolve frees the
|
||||
/// slot and returns not_found). Then free the volume slot.
|
||||
fn removeVolumeState(v: *Volume) void {
|
||||
std.log.info("storage for volume {d} removed; unmounting", .{v.id});
|
||||
if (v.filesystem_pid != 0) _ = process.kill(v.filesystem_pid);
|
||||
_ = ipc.close(v.storage.endpoint);
|
||||
volume = null;
|
||||
restart_pending = false;
|
||||
fs_restarts = 0;
|
||||
fs_failed = false;
|
||||
v.* = .{};
|
||||
}
|
||||
|
||||
/// One poll tick. Removal is checked FIRST and supersedes a pending restart: if
|
||||
/// the device is gone there is nothing to restart fat onto, and respawning it
|
||||
/// against the dead channel would just churn until the crash cap. Only once the
|
||||
/// device is confirmed present does a due restart fire.
|
||||
/// A storage device left the tree (a pulled stick): retire every volume it served
|
||||
/// and drop its channel. One removal path, whether the device is pulled cleanly
|
||||
/// or vanishes.
|
||||
fn removeDevice(dev: *StorageDevice) void {
|
||||
for (&volumes) |*v| {
|
||||
if (v.used and v.device_id == dev.device_id) removeVolumeState(v);
|
||||
}
|
||||
dropDevice(dev);
|
||||
}
|
||||
|
||||
/// Whether a device's block channel still answers — a geometry() probe. A storage
|
||||
/// driver that DIED while its device stays in the tree (it crashed; the device
|
||||
/// manager will re-delegate the device to a restarted driver on a FRESH channel)
|
||||
/// leaves a dead channel here, even though isDevicePresent still reports the device
|
||||
/// present. geometry() on the dead endpoint fails fast, so this catches the crash
|
||||
/// that presence-polling alone cannot — the V4 review's open edge.
|
||||
fn channelAlive(dev: *StorageDevice) bool {
|
||||
return dev.channel.geometry() != null;
|
||||
}
|
||||
|
||||
/// One poll tick. Device removal is reconciled FIRST and supersedes a pending
|
||||
/// restart: a volume whose device left (a pull) OR whose driver died on a channel
|
||||
/// that no longer answers is retired before its restart could fire, so nothing
|
||||
/// respawns against a dead channel. Dropping the device frees its slot, so the
|
||||
/// adopt loop below re-adopts the still-present device on the restarted driver's
|
||||
/// fresh channel — the rebuild. Then due restarts fire for present volumes.
|
||||
fn pollTick() void {
|
||||
if (volume) |v| {
|
||||
// Serving: watch for the specific device leaving (a pulled stick).
|
||||
if (!isDevicePresent(v.storage_device_id)) {
|
||||
removeVolume();
|
||||
return;
|
||||
for (&devices) |*dev| {
|
||||
if (dev.used and (!isDevicePresent(dev.device_id) or !channelAlive(dev))) removeDevice(dev);
|
||||
}
|
||||
if (restart_pending and time.clock() >= restart_due_ns) {
|
||||
restart_pending = false;
|
||||
spawnFilesystem(&volume.?);
|
||||
for (&volumes) |*v| {
|
||||
if (v.used and v.restart_pending and time.clock() >= v.restart_due_ns) {
|
||||
v.restart_pending = false;
|
||||
spawnFilesystem(v);
|
||||
}
|
||||
} else {
|
||||
// Idle: try to bring a present storage device up.
|
||||
bringUpVolume();
|
||||
}
|
||||
// Adopt every present, not-yet-adopted storage device. Each call consumes at
|
||||
// most one device (openAnyStorage skips the adopted), so the loop terminates
|
||||
// once none remain; the maximum_devices guard is insurance against a logic
|
||||
// slip, never the normal exit.
|
||||
var adopted: usize = 0;
|
||||
while (adopted < maximum_devices and bringUpVolume()) : (adopted += 1) {}
|
||||
}
|
||||
|
||||
/// A filesystem announces itself for the volume it was spawned to serve. Reply
|
||||
@@ -269,29 +444,79 @@ fn pollTick() void {
|
||||
/// badge) as the call's returned capability. No channel means the volume is not
|
||||
/// ready — the filesystem retries.
|
||||
fn onHello(_: void, invocation: Invocation(volume_manager_protocol.Hello), _: Answer(void)) isize {
|
||||
const v = volume orelse return 0; // not probed yet — retryable, no cap
|
||||
if (invocation.target != v.id) return 0; // unknown volume — retryable
|
||||
const v = volumeById(invocation.target) orelse return 0; // not probed yet — retryable, no cap
|
||||
if (invocation.sender != v.filesystem_pid) {
|
||||
// Not the filesystem we spawned for this volume. Refuse: only the
|
||||
// confined filesystem gets the channel.
|
||||
// Not the filesystem we spawned for this volume. Refuse: only the confined
|
||||
// filesystem gets the channel.
|
||||
std.log.info("refused hello for volume {d} from process {d}", .{ invocation.target, invocation.sender });
|
||||
return -envelope.EPERM;
|
||||
}
|
||||
service.replyWithCapability(v.storage.endpoint);
|
||||
const dev = deviceById(v.device_id) orelse return 0; // its device left — retryable
|
||||
service.replyWithCapability(dev.channel.endpoint);
|
||||
std.log.info("handed volume {d} to pid {d}", .{ v.id, invocation.sender });
|
||||
return 0;
|
||||
}
|
||||
|
||||
const handlers = Serve.Handlers{ .hello = onHello };
|
||||
/// Answer a `volumes` query with a mounted volume's descriptor — its id (its
|
||||
/// mount path is /volumes/<id> unless overridden), its actual mount path, and its
|
||||
/// display label. Software keys on the id; a UI shows the label. Returns the first
|
||||
/// mounted volume for now; a full enumerate is a later refinement. Empty reply
|
||||
/// means no volume is mounted.
|
||||
fn onVolumes(_: void, _: Invocation(volume_manager_protocol.Volumes), answer: Answer(void)) isize {
|
||||
const v = firstUsedVolume() orelse return 0;
|
||||
var id_buf: [volume_map.id_maximum]u8 = undefined;
|
||||
const info = volume_manager_protocol.VolumeInfo{
|
||||
.id = volume_map.idString(v.identity, &id_buf),
|
||||
.mount_path = v.mount_prefix,
|
||||
.label = v.identity.labelSlice(),
|
||||
};
|
||||
const encoded = info.encode(answer.tail()) orelse return 0;
|
||||
return @intCast(encoded.len);
|
||||
}
|
||||
|
||||
const handlers = Serve.Handlers{ .hello = onHello, .volumes = onVolumes };
|
||||
|
||||
fn onMessage(message: []const u8, out: []u8, sender: u32, arrived: *ipc.Arrival) usize {
|
||||
// No verb takes a capability up, so the turn closes whatever arrives.
|
||||
return Serve.dispatch({}, handlers, message, sender, arrived.peek(), out);
|
||||
}
|
||||
|
||||
/// Read a config file into `buf`, returning the byte count (0 if missing).
|
||||
fn readConfig(path: []const u8, buf: []u8) usize {
|
||||
var file = fs.open(path, .{}) orelse {
|
||||
std.log.info("volume-manager: {s} missing", .{path});
|
||||
return 0;
|
||||
};
|
||||
defer file.close();
|
||||
var used: usize = 0;
|
||||
while (used < buf.len) {
|
||||
const nn = file.read(buf[used..]) orelse break;
|
||||
if (nn == 0) break;
|
||||
used += nn;
|
||||
}
|
||||
return used;
|
||||
}
|
||||
|
||||
/// Load the mount map from configuration once at boot (mirrors the device
|
||||
/// manager's registry load). A missing or empty filesystems.csv means no volume
|
||||
/// is served; volumes.csv is optional — no rows means every volume takes its
|
||||
/// default /volumes/<id> path.
|
||||
fn loadTables() void {
|
||||
const fs_used = readConfig("/system/configuration/filesystems.csv", &filesystems_source);
|
||||
const fr = filesystem_map.parse(filesystems_source[0..fs_used], &filesystem_rules);
|
||||
filesystem_rule_count = fr.count;
|
||||
if (fr.malformed != 0 or fr.truncated) std.log.info("filesystems.csv: {d} malformed, truncated={}", .{ fr.malformed, fr.truncated });
|
||||
|
||||
const vol_used = readConfig("/system/configuration/volumes.csv", &volumes_source);
|
||||
const vr = volume_map.parse(volumes_source[0..vol_used], &volume_rules);
|
||||
volume_rule_count = vr.count;
|
||||
if (vr.malformed != 0 or vr.truncated) std.log.info("volumes.csv: {d} malformed, truncated={}", .{ vr.malformed, vr.truncated });
|
||||
}
|
||||
|
||||
fn initialise(endpoint: ipc.Handle) bool {
|
||||
service_endpoint = endpoint;
|
||||
_ = logging.write("volume-manager: starting, waiting for a storage device\n");
|
||||
loadTables();
|
||||
_ = process.subscribeExits(endpoint);
|
||||
pollTick();
|
||||
_ = time.timerOnce(endpoint, poll_interval_ms); // the poll runs for the life of the boot
|
||||
@@ -311,23 +536,59 @@ fn onNotification(badge: u64) void {
|
||||
// reclaimed by the driver on the same death; the respawn confines afresh.
|
||||
if (got.isChildExit()) {
|
||||
const dead = got.childProcessId();
|
||||
const v = &(volume orelse return);
|
||||
if (v.filesystem_pid != dead) return;
|
||||
const v = volumeByPid(dead) orelse return;
|
||||
v.filesystem_pid = 0;
|
||||
const reason = process.exitReason(dead) orelse .fault;
|
||||
if (reason == .exited) {
|
||||
std.log.info("filesystem for volume {d} exited cleanly; not restarting", .{v.id});
|
||||
return;
|
||||
}
|
||||
const alive = time.clock() -| fs_spawn_ns;
|
||||
fs_restarts = if (alive < fast_death_ns) fs_restarts + 1 else 1;
|
||||
if (fs_restarts >= crash_loop_cap) {
|
||||
fs_failed = true;
|
||||
const alive = time.clock() -| v.spawn_ns;
|
||||
v.restarts = if (alive < fast_death_ns) v.restarts + 1 else 1;
|
||||
if (v.restarts >= crash_loop_cap) {
|
||||
v.failed = true;
|
||||
std.log.info("filesystem for volume {d} is failing repeatedly; giving up", .{v.id});
|
||||
return;
|
||||
}
|
||||
std.log.info("filesystem for volume {d} died ({s}); restarting", .{ v.id, @tagName(reason) });
|
||||
armRestart();
|
||||
armRestart(v);
|
||||
}
|
||||
}
|
||||
|
||||
fn anyVolumeOn(device_id: u64) bool {
|
||||
for (&volumes) |*v| if (v.used and v.device_id == device_id) return true;
|
||||
return false;
|
||||
}
|
||||
|
||||
/// A storage device published `medium_changed` — the second removal trigger: the
|
||||
/// device stays in the tree while its medium leaves or returns (a card reader, an
|
||||
/// eject). This arrives as a buffered async message, NOT a protocol request, so it
|
||||
/// never reaches `Serve.dispatch` (its event op number collides with the manager's
|
||||
/// own `hello`); it is decoded here by hand. Single-volume scope: the event names
|
||||
/// no device, so `absent` retires every adopted device (its volumes unmount and
|
||||
/// the poll re-adopts the still-present device with its now-empty medium), and
|
||||
/// `present` frees any empty adopted device so the poll re-probes and remounts it.
|
||||
///
|
||||
/// We act on every edge and do NOT dedup on `change_count`. The driver publishes
|
||||
/// exactly once per transition, each with a unique monotonic count, so a count is
|
||||
/// never legitimately re-sent within one subscription — an equality dedup could
|
||||
/// only ever fire spuriously, and it did: `change_count` restarts at 0 in each
|
||||
/// driver instance (usb-storage.zig), so a global "last count" carried across a
|
||||
/// driver restart (S5's own crash-rebuild) mistook the fresh instance's first
|
||||
/// edge for a re-delivery and dropped a real eject, wedging a mount over absent
|
||||
/// media. Both branches are idempotent (a freed device stops matching `dev.used`)
|
||||
/// and the poll reconciles, so reacting to each genuine edge is safe.
|
||||
fn onMediumEvent(payload: []const u8) void {
|
||||
const event = block.decodeMediumChanged(payload) orelse return;
|
||||
if (event.present == 0) {
|
||||
std.log.info("medium left a storage device; unmounting its volume(s)", .{});
|
||||
for (&devices) |*dev| {
|
||||
if (dev.used) removeDevice(dev);
|
||||
}
|
||||
} else {
|
||||
for (&devices) |*dev| {
|
||||
if (dev.used and !anyVolumeOn(dev.device_id)) removeDevice(dev);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -338,5 +599,6 @@ pub fn main(init: process.Init) void {
|
||||
.init = initialise,
|
||||
.on_message = onMessage,
|
||||
.on_notification = onNotification,
|
||||
.on_buffered_message = onMediumEvent,
|
||||
});
|
||||
}
|
||||
|
||||
@@ -0,0 +1,139 @@
|
||||
//! volume-map — render a volume's identity into its stable mount id-string, and
|
||||
//! parse `/system/configuration/volumes.csv` (danos's fstab) into optional
|
||||
//! id → mount-prefix overrides. This is where the id/label split becomes the
|
||||
//! path: a volume's mount point is derived from its content identity (the id),
|
||||
//! never from a port or a label. Two distinct volumes that share a label get
|
||||
//! distinct id-strings automatically; only identical ids (dd-cloned media) can
|
||||
//! collide, which is the narrow case the manager's duplicate policy is for.
|
||||
//!
|
||||
//! `volumes.csv` is an OPTIONAL override: a row `id, mount_prefix` pins a volume
|
||||
//! (by its id-string) to a chosen path. A volume with no row takes its default
|
||||
//! `/volumes/<id>`. The label is display metadata, exposed by the manager's
|
||||
//! `volumes` query, and never appears here.
|
||||
//!
|
||||
//! Pure logic: no syscalls, no allocator. Override slices point into the CSV
|
||||
//! source, which the manager holds in a static buffer for the process life.
|
||||
|
||||
const std = @import("std");
|
||||
const csv = @import("csv");
|
||||
const partition = @import("partition.zig");
|
||||
|
||||
/// bound: bytes of the longest volume id-string the deriver renders
|
||||
/// decided-by: ours
|
||||
/// protects: the caller's id-string buffer
|
||||
/// at-limit: truncate - bufPrint fails and idString returns ""; the volume goes
|
||||
/// unnamed and the manager logs it rather than mounting at an empty path
|
||||
/// observed-by: a volume with an empty id in the `volumes` query / the log
|
||||
pub const id_maximum = 40; // "gpt-" (4) or "uuid-" (5) + 32 hex fits in 40
|
||||
|
||||
/// One parsed override row: a volume id-string and the mount prefix it pins to.
|
||||
pub const Override = struct { id: []const u8, prefix: []const u8 };
|
||||
|
||||
/// How many overrides landed, how many non-blank lines were malformed, and
|
||||
/// whether there were more rows than the buffer could hold.
|
||||
pub const ParseResult = struct { count: usize, malformed: usize, truncated: bool };
|
||||
|
||||
/// Render a volume's identity into its id-string — the content-derived, unique,
|
||||
/// order-independent token whose default mount path is `/volumes/<id>`. The rung
|
||||
/// tags the scheme so ids never collide across rungs; the key is the content id,
|
||||
/// so a moved drive keeps its id (and thus its path).
|
||||
pub fn idString(identity: partition.Identity, buf: []u8) []const u8 {
|
||||
return switch (identity.rung) {
|
||||
.gpt_guid => std.fmt.bufPrint(buf, "gpt-{x:0>32}", .{identity.key}) catch "",
|
||||
.filesystem_uuid => std.fmt.bufPrint(buf, "uuid-{x:0>32}", .{identity.key}) catch "",
|
||||
.fat_serial => std.fmt.bufPrint(buf, "fat-{x:0>8}", .{@as(u32, @truncate(identity.key))}) catch "",
|
||||
.exfat_serial => std.fmt.bufPrint(buf, "exfat-{x:0>8}", .{@as(u32, @truncate(identity.key))}) catch "",
|
||||
.mbr_index => std.fmt.bufPrint(buf, "mbr-{x}-{d}", .{
|
||||
@as(u32, @truncate(identity.key >> 8)),
|
||||
@as(u8, @truncate(identity.key & 0xff)),
|
||||
}) catch "",
|
||||
.anonymous => std.fmt.bufPrint(buf, "anon-{x}", .{identity.key}) catch "",
|
||||
};
|
||||
}
|
||||
|
||||
const Line = union(enum) { override: Override, ignorable, malformed };
|
||||
|
||||
fn parseLine(line: []const u8) Line {
|
||||
const body = csv.stripComment(line);
|
||||
if (body.len == 0) return .ignorable;
|
||||
var it = csv.fields(body);
|
||||
const id = it.next() orelse return .malformed;
|
||||
const prefix = it.next() orelse return .malformed;
|
||||
if (it.next() != null) return .malformed; // too many columns
|
||||
if (id.len == 0 or prefix.len == 0) return .malformed;
|
||||
return .{ .override = .{ .id = id, .prefix = prefix } };
|
||||
}
|
||||
|
||||
/// Parse a whole `volumes.csv` into `out_rules`. The slices point into `source`,
|
||||
/// which must outlive them.
|
||||
pub fn parse(source: []const u8, out_rules: []Override) ParseResult {
|
||||
var result: ParseResult = .{ .count = 0, .malformed = 0, .truncated = false };
|
||||
var lines = std.mem.splitScalar(u8, source, '\n');
|
||||
while (lines.next()) |line| {
|
||||
switch (parseLine(line)) {
|
||||
.ignorable => {},
|
||||
.malformed => result.malformed += 1,
|
||||
.override => |ov| {
|
||||
if (result.count >= out_rules.len) {
|
||||
result.truncated = true;
|
||||
continue;
|
||||
}
|
||||
out_rules[result.count] = ov;
|
||||
result.count += 1;
|
||||
},
|
||||
}
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
/// The override mount prefix for a volume whose id-string is `id`, or null (the
|
||||
/// volume takes its default `/volumes/<id>` path). First matching row wins.
|
||||
pub fn overrideFor(rules: []const Override, id: []const u8) ?[]const u8 {
|
||||
for (rules) |rule| {
|
||||
if (std.mem.eql(u8, rule.id, id)) return rule.prefix;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
// --- tests -------------------------------------------------------------------
|
||||
|
||||
const testing = std.testing;
|
||||
|
||||
// Named fixture sizes so the bounds gate (which flags literal array lengths)
|
||||
// stays quiet: test inputs, not runtime ceilings.
|
||||
const test_override_slots = 4;
|
||||
|
||||
test "idString renders each rung's id token" {
|
||||
var buf: [id_maximum]u8 = undefined;
|
||||
try testing.expectEqualStrings("fat-12345678", idString(.{ .rung = .fat_serial, .key = 0x12345678 }, &buf));
|
||||
try testing.expectEqualStrings("exfat-da7a0001", idString(.{ .rung = .exfat_serial, .key = 0xDA7A0001 }, &buf));
|
||||
try testing.expectEqualStrings("mbr-deadbeef-1", idString(.{ .rung = .mbr_index, .key = (@as(u128, 0xDEADBEEF) << 8) | 1 }, &buf));
|
||||
const guid: u128 = 0x00112233445566778899AABBCCDDEEFF;
|
||||
try testing.expectEqualStrings("gpt-00112233445566778899aabbccddeeff", idString(.{ .rung = .gpt_guid, .key = guid }, &buf));
|
||||
}
|
||||
|
||||
test "overrideFor returns the mapped prefix, else null" {
|
||||
const text =
|
||||
\\# id, mount_prefix
|
||||
\\fat-12345678, /mnt/boot
|
||||
;
|
||||
var rules: [test_override_slots]Override = undefined;
|
||||
const parsed = parse(text, &rules);
|
||||
try testing.expectEqual(@as(usize, 1), parsed.count);
|
||||
try testing.expectEqual(@as(usize, 0), parsed.malformed);
|
||||
try testing.expectEqualStrings("/mnt/boot", overrideFor(rules[0..parsed.count], "fat-12345678").?);
|
||||
try testing.expect(overrideFor(rules[0..parsed.count], "fat-99999999") == null);
|
||||
}
|
||||
|
||||
test "malformed volume rows are counted, not bound" {
|
||||
const text =
|
||||
\\fat-1, /mnt/a
|
||||
\\onlyonecolumn
|
||||
\\fat-2,
|
||||
\\fat-3, /a, /b
|
||||
;
|
||||
var rules: [test_override_slots]Override = undefined;
|
||||
const parsed = parse(text, &rules);
|
||||
try testing.expectEqual(@as(usize, 1), parsed.count); // only the first valid row
|
||||
try testing.expectEqual(@as(usize, 3), parsed.malformed); // one column, empty prefix, too many columns
|
||||
}
|
||||
+146
-8
@@ -179,7 +179,7 @@ CASES = [
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"qemu_extra": ["-device", "intel-iommu,intremap=off"],
|
||||
"expect": r"(?s)(?=.*/system/kernel: iommu online)(?=.*fat: mounted /volumes/usb)(?=.*fat-test: ok)",
|
||||
"expect": r"(?s)(?=.*/system/kernel: iommu online)(?=.*fat: mounted /volumes/fat-12345678)(?=.*fat-test: ok)",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL|DANOS-IOMMU-FAULT"},
|
||||
# DMA + MSI under translation: interrupt-IN reports arrive through translated DMA and
|
||||
# the xHC's MSI/MSI-X still delivers (the 0xFEE00000 interrupt window bypasses second-
|
||||
@@ -215,7 +215,7 @@ CASES = [
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"qemu_extra": ["-device", "amd-iommu,dma-remap=on,intremap=off"],
|
||||
"expect": r"(?s)(?=.*iommu online \(AMD-Vi\))(?=.*fat: mounted /volumes/usb)(?=.*fat-test: ok)",
|
||||
"expect": r"(?s)(?=.*iommu online \(AMD-Vi\))(?=.*fat: mounted /volumes/fat-12345678)(?=.*fat-test: ok)",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL|DANOS-IOMMU-FAULT"},
|
||||
# Port I/O grants: a claimed device's io_port resource lets a driver read/write its
|
||||
# ports (PS/2 status 0x64), gated by the claim; out-of-range/unclaimed is refused.
|
||||
@@ -749,13 +749,26 @@ CASES = [
|
||||
"expect": r"usb-storage: ready[\s\S]*usb-storage: block 0 signature 0x55aa",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# FAT mount end to end: the fat server mounts the boot usb-storage device (the
|
||||
# FAT32 image) into the VFS at /volumes/usb. A fat-test client then lists and reads
|
||||
# FAT32 image) into the VFS at /volumes/fat-12345678. A fat-test client then lists and reads
|
||||
# through the mount — proof of the whole stack: block device -> FAT parse ->
|
||||
# VFS routing -> file read.
|
||||
{"name": "fat-mount",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"expect": r"fat: mounted /volumes/usb[\s\S]*fat-test: ok",
|
||||
"expect": r"fat: mounted /volumes/fat-12345678[\s\S]*fat-test: ok",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# The id-path naming (S2, storage-stack-plan.md). The boot volume mounts at
|
||||
# its CONTENT-derived id-path (/volumes/fat-12345678, from the FAT32 serial
|
||||
# 0x12345678) — never a port name — and keeps its FHS rewrites so /system/logs
|
||||
# persistence still rides the volume. Discrimination: before S2's flip fat
|
||||
# hardcoded /volumes/usb, so the id-path mount line never appears. (Making the
|
||||
# rewrites content-conditional on which volume carries the system is S3.)
|
||||
{"name": "volume-identity-name",
|
||||
"build_case": "fat-mount",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"expect": r"(?s)fat: mounted /volumes/fat-12345678"
|
||||
r"[\s\S]*fat: mounted /system/logs",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# The removal lifecycle (V4, docs/volume-manager-plan.md): pull the boot stick
|
||||
# mid-run. device_del the usb-storage device -> the bus reports the port empty
|
||||
@@ -780,9 +793,45 @@ CASES = [
|
||||
"qmp_sequence": [
|
||||
{"delay": 8, "command": "device_del", "arguments": {"id": "bootstorage"}},
|
||||
],
|
||||
"expect": r"(?s)fat: mounted /volumes/usb"
|
||||
"expect": r"(?s)fat: mounted /volumes/fat-12345678"
|
||||
r"[\s\S]*volume-manager: storage for volume \d+ removed; unmounting",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# S5 medium_changed: the SECOND removal trigger. QMP-eject the MEDIUM (the
|
||||
# block backend, not the device) — the usb-storage device stays in the tree,
|
||||
# but its TEST UNIT READY poll reports not-ready and publishes medium_changed
|
||||
# (absent). The volume manager, now a subscriber, runs the same unmount path as
|
||||
# a device pull. Discrimination: before S5 the manager never subscribed, so the
|
||||
# event reached no one and the mount persisted (device-presence polling cannot
|
||||
# see a medium leave while the device stays). One lifecycle, two triggers.
|
||||
{"name": "volume-medium-change",
|
||||
"build_case": "fat-mount",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"qmp_sequence": [
|
||||
{"delay": 8, "command": "eject", "arguments": {"device": "bootusb", "force": True}},
|
||||
],
|
||||
"expect": r"(?s)fat: mounted /volumes/fat-12345678"
|
||||
r"[\s\S]*usb-storage: medium absent"
|
||||
r"[\s\S]*volume-manager: medium left a storage device; unmounting"
|
||||
r"[\s\S]*volume-manager: storage for volume \d+ removed; unmounting",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# S5 storage-driver-crash rebuild. The device manager (test-storage-restart
|
||||
# mode) kills usb-storage once, ~2s in — after its volume mounted. The device
|
||||
# stays in the tree, so device-presence polling alone would leave fat wedged on
|
||||
# the dead channel; the volume manager's channel-liveness probe (a geometry()
|
||||
# that fails on the dead endpoint) must notice, reap the volume, and rebuild on
|
||||
# the restarted driver's fresh channel — a SECOND mount of the same id-path.
|
||||
# Discrimination: a pre-S5 manager checks only isDevicePresent (still true), so
|
||||
# it never reaps and the second mount never appears (it would restart fat on
|
||||
# the stale channel and crash-loop).
|
||||
{"name": "volume-driver-restart",
|
||||
"build_case": "volume-driver-restart",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"expect": r"(?s)fat: mounted /volumes/fat-12345678"
|
||||
r"[\s\S]*volume-manager: storage for volume \d+ removed; unmounting"
|
||||
r"[\s\S]*fat: mounted /volumes/fat-12345678",
|
||||
"fail": r"failing repeatedly; giving up|\[FAIL\]|DANOS-TEST-RESULT: FAIL"},
|
||||
# Volume-manager discovery + probe (V3a, docs/volume-manager-plan.md). Reuses
|
||||
# the fat-mount kernel build (the default boot now spawns the volume manager
|
||||
# from init.csv). It acquires the mass-storage block channel through the
|
||||
@@ -795,11 +844,72 @@ CASES = [
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
# The volume manager probes the partition table, then confines a filesystem
|
||||
# to the volume and hands it over — one log line naming the volume's range,
|
||||
# its identity, and the filesystem it spawned for it.
|
||||
"expect": r"volume-manager: volume 0x[0-9a-f]+ -> \S+ \(pid \d+\), lba \d+, \d+ blocks"
|
||||
# to the volume and hands it over. Since S1 rung 3 the identity is the boot
|
||||
# image's real FAT32 serial (0x12345678, from make-fat-image.py) — before
|
||||
# rung 3 it was the rung-4 pseudo-signature read from VBR offset 440 (~0x0),
|
||||
# so this tightened value is an on-image witness that the enriched identity
|
||||
# reaches the running log, not just the host tests.
|
||||
"expect": r"volume-manager: volume 0x0*12345678 -> \S+ \(pid \d+\), lba \d+, \d+ blocks"
|
||||
r"[\s\S]*volume-manager: handed volume \d+ to pid \d+",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# S3 multi-volume: a SECOND usb-storage device (a generated data volume, serial
|
||||
# da7a0001, an empty FAT with no /system) plugged in beside the boot volume.
|
||||
# Proves the volume manager adopts BOTH devices and spawns a confined fat per
|
||||
# volume, each mounted at its own CONTENT id-path (/volumes/fat-<serial>); and
|
||||
# that boot-volume detection is by content — only the volume that carries
|
||||
# /system backs /system/configuration, while the data volume mounts at its
|
||||
# id-path alone. Against the pre-S3 one-device/one-volume manager the data
|
||||
# volume never mounts, so the da7a0001 lookaheads fail (toggle-demonstrated by
|
||||
# checking out the step-2 volume-manager.zig).
|
||||
{"name": "two-volumes",
|
||||
"build_case": "fat-mount",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"data_volume": {"serial": "DA7A0001", "label": "DATAVOL", "size_mib": 64},
|
||||
"expect": r"(?s)(?=.*volume-manager: volume 0x0*12345678 -> )"
|
||||
r"(?=.*volume-manager: volume 0x0*da7a0001 -> )"
|
||||
r"(?=.*fat: mounted /volumes/fat-12345678)"
|
||||
r"(?=.*fat: mounted /volumes/fat-da7a0001)"
|
||||
r"(?=.*carries the system tree)"
|
||||
r"(?=.*data volume; mounted at /volumes/fat-da7a0001)",
|
||||
"fail": r"data volume; mounted at /volumes/fat-12345678|DANOS-TEST-RESULT: FAIL"},
|
||||
# S3 shared-channel multi-volume: ONE usb-storage device carrying an MBR with
|
||||
# TWO FAT partitions (da7a0001 at lba 2048, da7a0002 at lba 83968). allVolumes
|
||||
# walks the table and the manager spawns a confined fat per partition on the
|
||||
# SAME block channel, each clamped to its own LBA range (usb-storage's
|
||||
# per-badge range table) — the path a pair of single-volume sticks (the
|
||||
# two-volumes case) does NOT exercise. The two mount lines sit at two DISTINCT
|
||||
# non-zero base_lbas on one device. Against the pre-uncap allVolumes (S3 step
|
||||
# 2, capped to one partition) only da7a0001 mounts, so the da7a0002 lookaheads
|
||||
# fail.
|
||||
{"name": "partitioned-volume",
|
||||
"build_case": "fat-mount",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"data_volume": {"partitions": [{"serial": "DA7A0001", "size_mib": 40},
|
||||
{"serial": "DA7A0002", "size_mib": 40}]},
|
||||
"expect": r"(?s)(?=.*volume 0x0*da7a0001 -> \S+ \(pid \d+\), lba 2048, )"
|
||||
r"(?=.*volume 0x0*da7a0002 -> \S+ \(pid \d+\), lba 83968, )"
|
||||
r"(?=.*fat: mounted /volumes/fat-da7a0001)"
|
||||
r"(?=.*fat: mounted /volumes/fat-da7a0002)",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# S4 second engine: a bare exFAT data device (serial e0fa0001) attached beside
|
||||
# the FAT boot volume. The volume manager content-routes it to the exFAT
|
||||
# service (not fat), which mounts it at its id-path /volumes/exfat-e0fa0001;
|
||||
# the exfat-test client then reads the seeded HELLO.TXT and mutates through the
|
||||
# mount (mkdir/write/rename/read/remove). Proves the second engine reuses the
|
||||
# shared harness end to end. Fails against pre-S4 (no exfat binary, csv row, or
|
||||
# VBR recognizer — the device would go to fat, which rejects the exFAT VBR).
|
||||
{"name": "exfat-volume",
|
||||
"build_case": "exfat-volume",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"data_volume": {"exfat": True, "serial": "E0FA0001", "size_mib": 48},
|
||||
"expect": r"(?s)(?=.*volume 0x0*e0fa0001 -> /system/services/exfat )"
|
||||
r"(?=.*exfat: mounted /volumes/exfat-e0fa0001)"
|
||||
r"(?=.*exfat-test: read HELLO.TXT ok)"
|
||||
r"(?=.*exfat-test: ok)",
|
||||
"fail": r"exfat-test: FAILED|DANOS-TEST-RESULT: FAIL"},
|
||||
# Phase 2b: mkdir/unlink through the mount. Reuses the fat-mount build — the
|
||||
# fat-test client, after listing, makes a directory, writes+reads a file inside
|
||||
# it, then removes the file, exercising the whole VFS -> fat mutation path.
|
||||
@@ -1406,6 +1516,34 @@ def run_case(arch, case):
|
||||
cmd[cmd.index("-m") + 1] = case["mem"]
|
||||
if case.get("qemu_extra"): # extra qemu args, e.g. -device intel-iommu for the IOMMU case
|
||||
cmd += case["qemu_extra"]
|
||||
# A multi-volume case attaches a second usb-storage device backed by a freshly
|
||||
# GENERATED data volume: a distinct-serial FAT32 with no /system tree, so the
|
||||
# volume manager mounts it at its own id-path and the fat process marks it a
|
||||
# data volume (never a system volume). Regenerated per run — no image is
|
||||
# committed to the tree (the user keeps the boot files copyable, not baked in).
|
||||
if case.get("data_volume"):
|
||||
dv = case["data_volume"]
|
||||
data_img = os.path.join(WORK, "data-volume.img")
|
||||
if dv.get("partitions"):
|
||||
# One device, an MBR with several FAT partitions: several volumes share
|
||||
# ONE block channel, each confined to its own LBA range.
|
||||
gen = [sys.executable, os.path.join(REPO, "tools", "make-partitioned-image.py"), data_img]
|
||||
for part in dv["partitions"]:
|
||||
gen += [part["serial"], str(part.get("size_mib", 40))]
|
||||
elif dv.get("exfat"):
|
||||
# One device, a bare exFAT volume — the second engine's medium.
|
||||
gen = [sys.executable, os.path.join(REPO, "tools", "make-exfat-image.py"),
|
||||
"--serial", dv["serial"], data_img, str(dv.get("size_mib", 48))]
|
||||
else:
|
||||
# One device, one bare FAT volume.
|
||||
gen = [sys.executable, os.path.join(REPO, "tools", "make-fat-image.py"),
|
||||
"--serial", dv["serial"], "--label", dv.get("label", "DATAVOL"),
|
||||
data_img, str(dv.get("size_mib", 64))]
|
||||
subprocess.run(gen, check=True, stdout=subprocess.DEVNULL)
|
||||
cmd += [
|
||||
"-drive", f"if=none,id=datausb,format=raw,file={data_img}",
|
||||
"-device", "usb-storage,bus=xhci.0,port=4,drive=datausb,removable=on,id=datastorage",
|
||||
]
|
||||
# A QMP control socket, always present (additive): how a case's `qmp_after`
|
||||
# hook injects host-side events into the guest mid-run. Kept under a short temp
|
||||
# dir, not WORK: a unix socket path is capped at ~104 bytes (sun_path), and a
|
||||
|
||||
@@ -38,7 +38,7 @@ const time = @import("time");
|
||||
/// A scratch file on the volume, so the node the intruder tries to write through
|
||||
/// is one nothing else reads. (A foreign write that *succeeded* would prove the
|
||||
/// bug — it must not also damage the boot volume proving it.)
|
||||
const held_path = "/volumes/usb/BADGE.TXT";
|
||||
const held_path = "/volumes/fat-12345678/BADGE.TXT";
|
||||
const held_contents = "held";
|
||||
|
||||
fn line(comptime format: []const u8, arguments: anytype) void {
|
||||
@@ -46,12 +46,12 @@ fn line(comptime format: []const u8, arguments: anytype) void {
|
||||
_ = logging.write(std.fmt.bufPrint(&buffer, format, arguments) catch return);
|
||||
}
|
||||
|
||||
/// The fat server mounts /volumes/usb only after the whole USB storage chain is
|
||||
/// The fat server mounts /volumes/fat-12345678 only after the whole USB storage chain is
|
||||
/// up, and both instances race it.
|
||||
fn waitForVolume() bool {
|
||||
var tries: u32 = 0;
|
||||
while (tries < 1400) : (tries += 1) {
|
||||
if (fs.openDirectory("/volumes/usb")) |opened| {
|
||||
if (fs.openDirectory("/volumes/fat-12345678")) |opened| {
|
||||
var directory = opened;
|
||||
directory.close();
|
||||
return true;
|
||||
@@ -79,7 +79,7 @@ pub fn main(init: process.Init) void {
|
||||
|
||||
fn own() void {
|
||||
if (!waitForVolume()) {
|
||||
_ = logging.write("badge-scope-test: FAILED (/volumes/usb never became available)\n");
|
||||
_ = logging.write("badge-scope-test: FAILED (/volumes/fat-12345678 never became available)\n");
|
||||
return;
|
||||
}
|
||||
var held = fs.open(held_path, .{ .create = true, .truncate = true }) orelse {
|
||||
@@ -142,7 +142,7 @@ fn own() void {
|
||||
|
||||
fn intrude(foreign_node: u64, foreign_layer: u32) void {
|
||||
if (!waitForVolume()) {
|
||||
_ = logging.write("badge-scope-test: FAILED (/volumes/usb never became available)\n");
|
||||
_ = logging.write("badge-scope-test: FAILED (/volumes/fat-12345678 never became available)\n");
|
||||
return;
|
||||
}
|
||||
const node_verdict = probeNode(foreign_node);
|
||||
|
||||
@@ -0,0 +1,15 @@
|
||||
//! The exfat-test test fixture as a binary package (docs/build-packages-plan.md):
|
||||
//! this file names the binary and EXACTLY the modules its source imports —
|
||||
//! build-support resolves each name from the domains this zon declares.
|
||||
|
||||
const std = @import("std");
|
||||
const build_support = @import("build-support");
|
||||
|
||||
pub fn build(b: *std.Build) void {
|
||||
const exe = build_support.userBinary(b, .{
|
||||
.name = "exfat-test",
|
||||
.root_source_file = b.path("exfat-test.zig"),
|
||||
.imports = &.{ "file-system", "logging", "process", "time" },
|
||||
});
|
||||
b.installArtifact(exe);
|
||||
}
|
||||
@@ -0,0 +1,14 @@
|
||||
.{
|
||||
.name = .exfat_test,
|
||||
.version = "0.0.0",
|
||||
.fingerprint = 0x77b19e3f7ee43ce3, // Changing this has security and trust implications.
|
||||
.minimum_zig_version = "0.16.0",
|
||||
.dependencies = .{
|
||||
// build-support supplies the shared recipe; kernel is implicit in
|
||||
// every binary (the root shim + link script live there). The rest
|
||||
// are exactly the homes of this binary's declared imports.
|
||||
.@"build-support" = .{ .path = "../../../../build-support" },
|
||||
.kernel = .{ .path = "../../../../library/kernel" },
|
||||
},
|
||||
.paths = .{""},
|
||||
}
|
||||
@@ -0,0 +1,88 @@
|
||||
//! test/system/services/exfat-test — a client that proves the exFAT mount end to
|
||||
//! end: it waits for the exfat server to mount the volume at /volumes/exfat-
|
||||
//! e0fa0001, reads the seeded HELLO.TXT off it, and exercises mkdir / write /
|
||||
//! read / rename / remove through the VFS (which routes the id-path to the exfat
|
||||
//! backend). Shipped in the initial_ramdisk; the `exfat-volume` QEMU case spawns
|
||||
//! it alongside init with an exFAT data device attached beside the FAT boot volume.
|
||||
|
||||
const std = @import("std");
|
||||
const fs = @import("file-system");
|
||||
const process = @import("process");
|
||||
const time = @import("time");
|
||||
const logging = @import("logging");
|
||||
|
||||
// The exFAT data device's content id-path — its VolumeSerialNumber is 0xE0FA0001
|
||||
// (the `exfat-volume` case passes --serial E0FA0001 to make-exfat-image.py).
|
||||
const mount = "/volumes/exfat-e0fa0001";
|
||||
|
||||
fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||
var line: [128]u8 = undefined;
|
||||
_ = logging.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
||||
}
|
||||
|
||||
pub fn main(init: process.Init) void {
|
||||
_ = init;
|
||||
|
||||
// Wait for the exfat server to bring up the USB storage chain and mount.
|
||||
var opened: ?fs.Directory = null;
|
||||
var tries: u32 = 0;
|
||||
while (opened == null and tries < 1400) : (tries += 1) {
|
||||
opened = fs.openDirectory(mount);
|
||||
if (opened == null) time.sleepMillis(50);
|
||||
}
|
||||
var dir = opened orelse {
|
||||
_ = logging.write("exfat-test: " ++ mount ++ " never became available\n");
|
||||
return;
|
||||
};
|
||||
var count: u32 = 0;
|
||||
var entry: fs.Entry = .{};
|
||||
while (dir.next(&entry)) {
|
||||
writeLine("exfat-test: entry '{s}' size={d}\n", .{ entry.name(), entry.size });
|
||||
count += 1;
|
||||
if (count > 32) break;
|
||||
}
|
||||
dir.close();
|
||||
|
||||
// Read the seeded HELLO.TXT (make-exfat-image.py writes "exfat hello danos\n").
|
||||
var read_ok = false;
|
||||
if (fs.open(mount ++ "/HELLO.TXT", .{})) |opened_file| {
|
||||
var file = opened_file;
|
||||
var buf: [32]u8 = undefined;
|
||||
const n = file.read(&buf) orelse 0;
|
||||
file.close();
|
||||
read_ok = std.mem.startsWith(u8, buf[0..n], "exfat hello danos");
|
||||
}
|
||||
if (read_ok) _ = logging.write("exfat-test: read HELLO.TXT ok\n");
|
||||
|
||||
// Mutation through the mount: mkdir, create + write, rename, read back, remove
|
||||
// — proof the write path reaches the engine over a real device.
|
||||
var mut_ok = false;
|
||||
if (fs.makeDirectory(mount ++ "/TESTDIR")) {
|
||||
var wrote = false;
|
||||
if (fs.open(mount ++ "/TESTDIR/W.TXT", .{ .create = true, .truncate = true })) |created| {
|
||||
var f = created;
|
||||
wrote = (f.writeAll("exfat-mutation-ok") orelse 0) == "exfat-mutation-ok".len;
|
||||
f.close();
|
||||
}
|
||||
const renamed = fs.rename(mount ++ "/TESTDIR/W.TXT", mount ++ "/TESTDIR/R.TXT");
|
||||
var readback = false;
|
||||
if (fs.open(mount ++ "/TESTDIR/R.TXT", .{})) |reopened| {
|
||||
var f = reopened;
|
||||
var buf: [32]u8 = undefined;
|
||||
const got = f.read(&buf) orelse 0;
|
||||
f.close();
|
||||
readback = std.mem.eql(u8, buf[0..got], "exfat-mutation-ok");
|
||||
}
|
||||
const removed = fs.remove(mount ++ "/TESTDIR/R.TXT");
|
||||
mut_ok = wrote and renamed and readback and removed;
|
||||
}
|
||||
if (mut_ok) _ = logging.write("exfat-test: mutations ok\n");
|
||||
|
||||
if (read_ok and mut_ok) {
|
||||
while (true) {
|
||||
_ = logging.write("exfat-test: ok\n");
|
||||
time.sleepMillis(1000);
|
||||
}
|
||||
}
|
||||
writeLine("exfat-test: FAILED (read={} mutations={})\n", .{ read_ok, mut_ok });
|
||||
}
|
||||
@@ -1,6 +1,6 @@
|
||||
//! test/system/services/fat-test — a client that proves the FAT mount end to end:
|
||||
//! it waits for the fat server to mount the USB volume at /volumes/usb, lists the
|
||||
//! root directory through the VFS (which routes /volumes/usb to the fat backend), and
|
||||
//! it waits for the fat server to mount the USB volume at /volumes/fat-12345678, lists the
|
||||
//! root directory through the VFS (which routes /volumes/fat-12345678 to the fat backend), and
|
||||
//! reads a known file off it. Shipped in the initial_ramdisk; the `fat-mount`
|
||||
//! kernel test spawns it alongside init.
|
||||
|
||||
@@ -18,16 +18,16 @@ fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||
pub fn main(init: process.Init) void {
|
||||
_ = init;
|
||||
|
||||
// Wait for /volumes/usb to be mounted — the fat server races us at boot (it must
|
||||
// Wait for /volumes/fat-12345678 to be mounted — the fat server races us at boot (it must
|
||||
// bring up the whole USB storage chain first).
|
||||
var opened: ?fs.Directory = null;
|
||||
var tries: u32 = 0;
|
||||
while (opened == null and tries < 1400) : (tries += 1) {
|
||||
opened = fs.openDirectory("/volumes/usb");
|
||||
opened = fs.openDirectory("/volumes/fat-12345678");
|
||||
if (opened == null) time.sleepMillis(50);
|
||||
}
|
||||
var dir = opened orelse {
|
||||
_ = logging.write("fat-test: /volumes/usb never became available\n");
|
||||
_ = logging.write("fat-test: /volumes/fat-12345678 never became available\n");
|
||||
return;
|
||||
};
|
||||
|
||||
@@ -43,56 +43,56 @@ pub fn main(init: process.Init) void {
|
||||
|
||||
// Read a known file off the boot volume through the mount (best effort): the
|
||||
// kernel image is an ELF, so its first bytes are the ELF magic.
|
||||
if (fs.open("/volumes/usb/system/kernel", .{})) |opened_file| {
|
||||
if (fs.open("/volumes/fat-12345678/system/kernel", .{})) |opened_file| {
|
||||
var file = opened_file;
|
||||
var magic: [4]u8 = undefined;
|
||||
const n = file.read(&magic) orelse 0;
|
||||
file.close();
|
||||
if (n == 4 and magic[0] == 0x7F and magic[1] == 'E' and magic[2] == 'L' and magic[3] == 'F') {
|
||||
_ = logging.write("fat-test: read /volumes/usb/system/kernel ELF magic ok\n");
|
||||
_ = logging.write("fat-test: read /volumes/fat-12345678/system/kernel ELF magic ok\n");
|
||||
} else {
|
||||
writeLine("fat-test: /volumes/usb/system/kernel read {d} bytes (not ELF magic)\n", .{n});
|
||||
writeLine("fat-test: /volumes/fat-12345678/system/kernel read {d} bytes (not ELF magic)\n", .{n});
|
||||
}
|
||||
}
|
||||
|
||||
// Exercise directory + file mutation through the mount: mkdir, create a file
|
||||
// inside it, read it back, then remove it — proof mkdir/unlink reach the engine.
|
||||
if (fs.makeDirectory("/volumes/usb/TESTDIR")) {
|
||||
if (fs.makeDirectory("/volumes/fat-12345678/TESTDIR")) {
|
||||
var wrote = false;
|
||||
if (fs.open("/volumes/usb/TESTDIR/HELLO.TXT", .{ .create = true, .truncate = true })) |created| {
|
||||
if (fs.open("/volumes/fat-12345678/TESTDIR/HELLO.TXT", .{ .create = true, .truncate = true })) |created| {
|
||||
var f = created;
|
||||
wrote = (f.writeAll("mutation-ok") orelse 0) == "mutation-ok".len;
|
||||
f.close();
|
||||
}
|
||||
// The created file carries a real modification time (stamped from the RTC).
|
||||
var mtime_ok = false;
|
||||
if (fs.attributes("/volumes/usb/TESTDIR/HELLO.TXT")) |attrs| {
|
||||
if (fs.attributes("/volumes/fat-12345678/TESTDIR/HELLO.TXT")) |attrs| {
|
||||
writeLine("fat-test: mtime {d}\n", .{attrs.mtime});
|
||||
mtime_ok = attrs.mtime > 1_577_836_800; // after 2020-01-01
|
||||
}
|
||||
if (mtime_ok) _ = logging.write("fat-test: mtime ok\n");
|
||||
|
||||
// Rename it, then read from the new name and confirm the old name is gone.
|
||||
const renamed = fs.rename("/volumes/usb/TESTDIR/HELLO.TXT", "/volumes/usb/TESTDIR/RENAMED.TXT");
|
||||
const old_gone = !fs.exists("/volumes/usb/TESTDIR/HELLO.TXT");
|
||||
const renamed = fs.rename("/volumes/fat-12345678/TESTDIR/HELLO.TXT", "/volumes/fat-12345678/TESTDIR/RENAMED.TXT");
|
||||
const old_gone = !fs.exists("/volumes/fat-12345678/TESTDIR/HELLO.TXT");
|
||||
if (renamed and old_gone) _ = logging.write("fat-test: rename ok\n");
|
||||
var readback = false;
|
||||
if (fs.open("/volumes/usb/TESTDIR/RENAMED.TXT", .{})) |reopened| {
|
||||
if (fs.open("/volumes/fat-12345678/TESTDIR/RENAMED.TXT", .{})) |reopened| {
|
||||
var f = reopened;
|
||||
var buf: [16]u8 = undefined;
|
||||
const got = f.read(&buf) orelse 0;
|
||||
f.close();
|
||||
readback = std.mem.eql(u8, buf[0..got], "mutation-ok");
|
||||
}
|
||||
const removed = fs.remove("/volumes/usb/TESTDIR/RENAMED.TXT");
|
||||
const gone = !fs.exists("/volumes/usb/TESTDIR/RENAMED.TXT");
|
||||
const removed = fs.remove("/volumes/fat-12345678/TESTDIR/RENAMED.TXT");
|
||||
const gone = !fs.exists("/volumes/fat-12345678/TESTDIR/RENAMED.TXT");
|
||||
if (wrote and mtime_ok and renamed and old_gone and readback and removed and gone) {
|
||||
_ = logging.write("fat-test: mutations ok\n");
|
||||
} else {
|
||||
writeLine("fat-test: mutations FAILED (wrote={} mtime={} renamed={} oldgone={} read={} removed={} gone={})\n", .{ wrote, mtime_ok, renamed, old_gone, readback, removed, gone });
|
||||
}
|
||||
} else {
|
||||
_ = logging.write("fat-test: mkdir /volumes/usb/TESTDIR failed\n");
|
||||
_ = logging.write("fat-test: mkdir /volumes/fat-12345678/TESTDIR failed\n");
|
||||
}
|
||||
|
||||
if (count > 0) {
|
||||
|
||||
@@ -77,7 +77,7 @@ fn park() void {
|
||||
var parked: ?fs.File = null;
|
||||
var tries: u32 = 0;
|
||||
while (parked == null and tries < 1000) : (tries += 1) {
|
||||
parked = fs.open("/volumes/usb/parked", .{ .create = true });
|
||||
parked = fs.open("/volumes/fat-12345678/parked", .{ .create = true });
|
||||
if (parked == null) time.sleepMillis(20);
|
||||
}
|
||||
if (parked == null) {
|
||||
@@ -92,16 +92,16 @@ fn park() void {
|
||||
// "parked" marker, which fails the vfs-client-death case: before the
|
||||
// ownership gate existed, any process could unmount any prefix, and this
|
||||
// fixture would have deleted the volume out from under the whole boot.
|
||||
if (fs.fsUnmount("/volumes/usb")) {
|
||||
if (fs.fsUnmount("/volumes/fat-12345678")) {
|
||||
_ = logging.write("vfstest: foreign unmount was ALLOWED\n");
|
||||
return;
|
||||
}
|
||||
if (fs.open("/volumes/usb/parked", .{})) |resolved| {
|
||||
if (fs.open("/volumes/fat-12345678/parked", .{})) |resolved| {
|
||||
var verification = resolved;
|
||||
verification.close(); // the park below must be the client's ONLY open
|
||||
// handle — the kernel test string-matches "released 1 handle(s)".
|
||||
} else {
|
||||
_ = logging.write("vfstest: /volumes/usb gone after refused unmount\n");
|
||||
_ = logging.write("vfstest: /volumes/fat-12345678 gone after refused unmount\n");
|
||||
return;
|
||||
}
|
||||
_ = logging.write("vfstest: foreign unmount refused\n");
|
||||
|
||||
@@ -194,7 +194,6 @@ system/kernel/process.zig:write_buffer
|
||||
system/kernel/scheduler.zig:ipc_maximum_handles
|
||||
system/kernel/scheduler.zig:maximum_space_mappings
|
||||
system/kernel/vfs.zig:maximum_directories
|
||||
system/kernel/vfs.zig:maximum_mounts
|
||||
system/kernel/vfs.zig:maximum_prefix
|
||||
system/kernel/vfs.zig:maximum_rewrite
|
||||
system/services/acpi/acpi.zig:blocks
|
||||
|
||||
@@ -0,0 +1,307 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Format a real exFAT image from scratch — the danos exFAT test volume.
|
||||
|
||||
Pure Python 3 standard library (no mkfs.exfat / mtools). It writes a valid exFAT
|
||||
filesystem — a Main Boot Sector + its boot-region checksum + a backup region, the
|
||||
32-bit FAT, an allocation bitmap, an up-case table (with its checksum), and a root
|
||||
directory whose entry sets a real exFAT reader (and the danos exfat engine) mount
|
||||
and walk. Mirrors tools/make-fat-image.py in spirit.
|
||||
|
||||
make-exfat-image.py [--serial <hex>] [--label <name>] <out.img> <size-MiB>
|
||||
make-exfat-image.py --verify <out.img>
|
||||
|
||||
The image seeds one file, HELLO.TXT, so a mount can be proven by reading it.
|
||||
"""
|
||||
|
||||
import struct
|
||||
import sys
|
||||
|
||||
SECTOR = 512
|
||||
UPCASE_UNITS = 256 # a-z -> A-Z, the rest identity; covers ASCII names
|
||||
|
||||
|
||||
def align_up(value, to):
|
||||
return (value + to - 1) // to * to
|
||||
|
||||
|
||||
def rotr16(v):
|
||||
return ((v >> 1) | (v << 15)) & 0xFFFF
|
||||
|
||||
|
||||
def rotr32(v):
|
||||
return ((v >> 1) | (v << 31)) & 0xFFFFFFFF
|
||||
|
||||
|
||||
def boot_checksum(region):
|
||||
"""32-bit rotate-right sum over the boot region, skipping VolumeFlags
|
||||
(106,107) and PercentInUse (112) of the first sector."""
|
||||
checksum = 0
|
||||
for i, byte in enumerate(region):
|
||||
if i in (106, 107, 112):
|
||||
continue
|
||||
checksum = (rotr32(checksum) + byte) & 0xFFFFFFFF
|
||||
return checksum
|
||||
|
||||
|
||||
def upcase_checksum(table_bytes):
|
||||
checksum = 0
|
||||
for byte in table_bytes:
|
||||
checksum = (rotr32(checksum) + byte) & 0xFFFFFFFF
|
||||
return checksum
|
||||
|
||||
|
||||
def set_checksum(entries):
|
||||
"""16-bit rotate-right sum over a directory-entry set, skipping its own two
|
||||
checksum bytes (offset 2..3 of the first entry)."""
|
||||
checksum = 0
|
||||
for i, byte in enumerate(entries):
|
||||
if i in (2, 3):
|
||||
continue
|
||||
checksum = (rotr16(checksum) + byte) & 0xFFFF
|
||||
return checksum
|
||||
|
||||
|
||||
def name_hash(upcased_units):
|
||||
h = 0
|
||||
for unit in upcased_units:
|
||||
h = (rotr16(h) + (unit & 0xFF)) & 0xFFFF
|
||||
h = (rotr16(h) + (unit >> 8)) & 0xFFFF
|
||||
return h
|
||||
|
||||
|
||||
def ascii_upper(unit):
|
||||
return unit - ord("a") + ord("A") if ord("a") <= unit <= ord("z") else unit
|
||||
|
||||
|
||||
def solve_geometry(total_sectors, spc):
|
||||
"""Solve for cluster count / FAT length / heap offset that fit. The FAT sits
|
||||
after the main + backup boot regions (24 sectors)."""
|
||||
fat_offset = 24
|
||||
fat_length = 1
|
||||
while True:
|
||||
heap_offset = align_up(fat_offset + fat_length, spc)
|
||||
cluster_count = (total_sectors - heap_offset) // spc
|
||||
needed = ((cluster_count + 2) * 4 + SECTOR - 1) // SECTOR
|
||||
if needed <= fat_length:
|
||||
return cluster_count, fat_offset, fat_length, heap_offset
|
||||
fat_length = needed
|
||||
|
||||
|
||||
class ExfatImage:
|
||||
def __init__(self, size_mib, volume_id=0x1234ABCD, label="DANOS"):
|
||||
self.total_sectors = size_mib * 1024 * 1024 // SECTOR
|
||||
self.spc = 8 # 4 KiB clusters
|
||||
self.volume_id = volume_id & 0xFFFFFFFF
|
||||
self.label = label
|
||||
self.cluster_count, self.fat_offset, self.fat_length, self.heap_offset = solve_geometry(self.total_sectors, self.spc)
|
||||
if self.cluster_count < 16:
|
||||
sys.exit(f"error: image too small for exFAT ({self.cluster_count} clusters)")
|
||||
self.cluster_bytes = self.spc * SECTOR
|
||||
# Layout: the allocation bitmap (as many clusters as it needs — one per
|
||||
# 8*cluster_bytes clusters of the volume), then the up-case table, the root
|
||||
# directory, and the seeded file. A single-cluster bitmap (small volumes,
|
||||
# e.g. the 48 MiB fixture) puts root at cluster 4, as before.
|
||||
self.bitmap_bytes = (self.cluster_count + 7) // 8
|
||||
self.bitmap_clusters = (self.bitmap_bytes + self.cluster_bytes - 1) // self.cluster_bytes
|
||||
self.bitmap_cluster = 2
|
||||
self.upcase_cluster = self.bitmap_cluster + self.bitmap_clusters
|
||||
self.root_cluster = self.upcase_cluster + 1
|
||||
self.hello_cluster = self.root_cluster + 1
|
||||
self.image = bytearray(self.total_sectors * SECTOR)
|
||||
|
||||
def cluster_offset(self, cluster):
|
||||
return (self.heap_offset + (cluster - 2) * self.spc) * SECTOR
|
||||
|
||||
def set_fat(self, cluster, value):
|
||||
struct.pack_into("<I", self.image, self.fat_offset * SECTOR + cluster * 4, value)
|
||||
|
||||
def mark_allocated(self, cluster):
|
||||
bit = cluster - 2
|
||||
pos = self.cluster_offset(2) + bit // 8
|
||||
self.image[pos] |= 1 << (bit % 8)
|
||||
|
||||
def main_boot_sector(self):
|
||||
sector = bytearray(SECTOR)
|
||||
sector[0:3] = b"\xEB\x76\x90" # jump boot
|
||||
sector[3:11] = b"EXFAT " # filesystem name
|
||||
# 11..64 MustBeZero (already zero)
|
||||
struct.pack_into("<Q", sector, 72, self.total_sectors) # volume length
|
||||
struct.pack_into("<I", sector, 80, self.fat_offset) # fat offset
|
||||
struct.pack_into("<I", sector, 84, self.fat_length) # fat length
|
||||
struct.pack_into("<I", sector, 88, self.heap_offset) # cluster heap offset
|
||||
struct.pack_into("<I", sector, 92, self.cluster_count) # cluster count
|
||||
struct.pack_into("<I", sector, 96, self.root_cluster) # first cluster of root
|
||||
struct.pack_into("<I", sector, 100, self.volume_id) # volume serial number
|
||||
struct.pack_into("<H", sector, 104, 0x0100) # filesystem revision 1.0
|
||||
sector[108] = 9 # bytes per sector shift (512)
|
||||
sector[109] = self.spc.bit_length() - 1 # sectors per cluster shift
|
||||
sector[110] = 1 # number of FATs
|
||||
sector[111] = 0x80 # drive select
|
||||
sector[112] = 0xFF # percent in use (unknown)
|
||||
sector[510] = 0x55
|
||||
sector[511] = 0xAA
|
||||
return sector
|
||||
|
||||
def build(self):
|
||||
# Main boot region (sectors 0..11): VBR, eight extended boot sectors, OEM
|
||||
# parameters, reserved, then the checksum sector.
|
||||
vbr = self.main_boot_sector()
|
||||
self.image[0:SECTOR] = vbr
|
||||
for s in range(1, 9): # extended boot sectors carry the 0xAA550000 signature
|
||||
struct.pack_into("<I", self.image, s * SECTOR + 508, 0xAA550000)
|
||||
# sectors 9 (OEM) and 10 (reserved) stay zero
|
||||
region = bytes(self.image[0 : 11 * SECTOR])
|
||||
checksum = boot_checksum(region)
|
||||
for i in range(SECTOR // 4):
|
||||
struct.pack_into("<I", self.image, 11 * SECTOR + i * 4, checksum)
|
||||
# Backup boot region (sectors 12..23) is a copy of 0..11.
|
||||
self.image[12 * SECTOR : 24 * SECTOR] = self.image[0 : 12 * SECTOR]
|
||||
|
||||
# FAT: reserved entries, then a single-cluster chain per metadata object,
|
||||
# except the bitmap which spans self.bitmap_clusters (a real FAT chain).
|
||||
self.set_fat(0, 0xFFFFFFF8)
|
||||
self.set_fat(1, 0xFFFFFFFF)
|
||||
used = []
|
||||
for i in range(self.bitmap_clusters):
|
||||
cluster = self.bitmap_cluster + i
|
||||
self.set_fat(cluster, 0xFFFFFFFF if i == self.bitmap_clusters - 1 else cluster + 1)
|
||||
used.append(cluster)
|
||||
for cluster in (self.upcase_cluster, self.root_cluster, self.hello_cluster):
|
||||
self.set_fat(cluster, 0xFFFFFFFF)
|
||||
used.append(cluster)
|
||||
|
||||
# Allocation bitmap: every metadata/file cluster in use.
|
||||
for cluster in used:
|
||||
self.mark_allocated(cluster)
|
||||
|
||||
# Up-case table: 256 explicit units, a-z -> A-Z.
|
||||
upcase = bytearray(UPCASE_UNITS * 2)
|
||||
for i in range(UPCASE_UNITS):
|
||||
struct.pack_into("<H", upcase, i * 2, ascii_upper(i))
|
||||
off = self.cluster_offset(self.upcase_cluster)
|
||||
self.image[off : off + len(upcase)] = upcase
|
||||
table_checksum = upcase_checksum(upcase)
|
||||
|
||||
# Seed file HELLO.TXT (contiguous, one cluster).
|
||||
content = b"exfat hello danos\n"
|
||||
off = self.cluster_offset(self.hello_cluster)
|
||||
self.image[off : off + len(content)] = content
|
||||
|
||||
# Root directory: bitmap, up-case, volume label, HELLO set.
|
||||
root = self.cluster_offset(self.root_cluster)
|
||||
# 0x81 Allocation Bitmap
|
||||
struct.pack_into("<BBB", self.image, root, 0x81, 0, 0)
|
||||
struct.pack_into("<I", self.image, root + 20, self.bitmap_cluster)
|
||||
struct.pack_into("<Q", self.image, root + 24, self.bitmap_bytes)
|
||||
# 0x82 Up-case Table
|
||||
struct.pack_into("<B", self.image, root + 32, 0x82)
|
||||
struct.pack_into("<I", self.image, root + 32 + 4, table_checksum)
|
||||
struct.pack_into("<I", self.image, root + 32 + 20, self.upcase_cluster)
|
||||
struct.pack_into("<Q", self.image, root + 32 + 24, UPCASE_UNITS * 2)
|
||||
# 0x83 Volume Label
|
||||
label_units = [ord(c) for c in self.label[:11]]
|
||||
struct.pack_into("<BB", self.image, root + 64, 0x83, len(label_units))
|
||||
for i, u in enumerate(label_units):
|
||||
struct.pack_into("<H", self.image, root + 64 + 2 + i * 2, u)
|
||||
# HELLO.TXT set: File (0x85) + Stream (0xC0) + Name (0xC1)
|
||||
name = "HELLO.TXT"
|
||||
self.write_file_set(root + 96, name, first_cluster=self.hello_cluster, length=len(content))
|
||||
|
||||
def write_file_set(self, offset, name, first_cluster, length):
|
||||
entries = bytearray(32 * 3)
|
||||
# File entry
|
||||
entries[0] = 0x85
|
||||
entries[1] = 2 # stream + one name entry
|
||||
struct.pack_into("<H", entries, 4, 0x20) # attributes: archive
|
||||
# Stream entry
|
||||
entries[32 + 0] = 0xC0
|
||||
entries[32 + 1] = 0x01 | 0x02 # allocation possible + no FAT chain (contiguous)
|
||||
entries[32 + 3] = len(name)
|
||||
upname = [ascii_upper(ord(c)) for c in name]
|
||||
struct.pack_into("<H", entries, 32 + 4, name_hash(upname))
|
||||
struct.pack_into("<Q", entries, 32 + 8, length) # valid data length
|
||||
struct.pack_into("<I", entries, 32 + 20, first_cluster)
|
||||
struct.pack_into("<Q", entries, 32 + 24, length) # data length
|
||||
# File Name entry
|
||||
entries[64 + 0] = 0xC1
|
||||
for i, c in enumerate(name):
|
||||
struct.pack_into("<H", entries, 64 + 2 + i * 2, ord(c))
|
||||
struct.pack_into("<H", entries, 2, set_checksum(entries))
|
||||
self.image[offset : offset + len(entries)] = entries
|
||||
|
||||
def serialize(self):
|
||||
self.build()
|
||||
return bytes(self.image)
|
||||
|
||||
|
||||
def verify(path):
|
||||
with open(path, "rb") as handle:
|
||||
data = handle.read()
|
||||
if len(data) < 512 or data[510] != 0x55 or data[511] != 0xAA:
|
||||
sys.exit("verify: missing 0x55AA boot signature")
|
||||
if data[3:11] != b"EXFAT ":
|
||||
sys.exit("verify: not an exFAT boot sector")
|
||||
if any(data[11:64]):
|
||||
sys.exit("verify: MustBeZero region is not zero")
|
||||
fat_offset = struct.unpack_from("<I", data, 80)[0]
|
||||
heap_offset = struct.unpack_from("<I", data, 88)[0]
|
||||
cluster_count = struct.unpack_from("<I", data, 92)[0]
|
||||
root_cluster = struct.unpack_from("<I", data, 96)[0]
|
||||
spc = 1 << data[109]
|
||||
# Boot checksum sector 11 must match a fresh checksum over sectors 0..10.
|
||||
expected = boot_checksum(data[0 : 11 * SECTOR])
|
||||
got = struct.unpack_from("<I", data, 11 * SECTOR)[0]
|
||||
if expected != got:
|
||||
sys.exit(f"verify: boot checksum mismatch (0x{got:08X} != 0x{expected:08X})")
|
||||
# Walk the root directory for the HELLO.TXT set and check its checksum.
|
||||
root = (heap_offset + (root_cluster - 2) * spc) * SECTOR
|
||||
found = False
|
||||
for i in range(spc * SECTOR // 32):
|
||||
entry = root + i * 32
|
||||
if data[entry] == 0x00:
|
||||
break
|
||||
if data[entry] == 0x85:
|
||||
secondary = data[entry + 1]
|
||||
total = (secondary + 1) * 32
|
||||
stored = struct.unpack_from("<H", data, entry + 2)[0]
|
||||
if set_checksum(data[entry : entry + total]) != stored:
|
||||
sys.exit("verify: a file set checksum is wrong")
|
||||
found = True
|
||||
if not found:
|
||||
sys.exit("verify: no file set in the root directory")
|
||||
print(f"make-exfat-image: {path} OK "
|
||||
f"({cluster_count} clusters of {spc * SECTOR} bytes, fat@{fat_offset}, heap@{heap_offset})")
|
||||
|
||||
|
||||
def main(argv):
|
||||
if len(argv) == 3 and argv[1] == "--verify":
|
||||
verify(argv[2])
|
||||
return 0
|
||||
argv = list(argv)
|
||||
volume_id = 0x1234ABCD
|
||||
label = "DANOS"
|
||||
i = 1
|
||||
while i < len(argv):
|
||||
if argv[i] == "--serial" and i + 1 < len(argv):
|
||||
volume_id = int(argv[i + 1], 16)
|
||||
del argv[i : i + 2]
|
||||
elif argv[i] == "--label" and i + 1 < len(argv):
|
||||
label = argv[i + 1]
|
||||
del argv[i : i + 2]
|
||||
else:
|
||||
i += 1
|
||||
if len(argv) != 3:
|
||||
sys.exit("usage: make-exfat-image.py [--serial <hex>] [--label <name>] <out.img> <size-MiB>\n"
|
||||
" make-exfat-image.py --verify <out.img>")
|
||||
out_path = argv[1]
|
||||
size_mib = int(argv[2])
|
||||
image = ExfatImage(size_mib, volume_id, label)
|
||||
with open(out_path, "wb") as handle:
|
||||
handle.write(image.serialize())
|
||||
print(f"make-exfat-image: wrote {out_path} "
|
||||
f"({size_mib} MiB exFAT, {image.cluster_count} clusters, serial 0x{image.volume_id:08X})")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main(sys.argv))
|
||||
+31
-7
@@ -47,8 +47,14 @@ def fat32_geometry(total_sectors):
|
||||
|
||||
|
||||
class Fat32Image:
|
||||
def __init__(self, total_sectors):
|
||||
def __init__(self, total_sectors, volume_id=0x12345678, label="DANOS"):
|
||||
self.total_sectors = total_sectors
|
||||
# The FAT volume serial (its content identity — the /volumes/fat-<id>
|
||||
# mount path danos derives from it) and the display label. A second image
|
||||
# needs a distinct serial so its id-path does not collide with the boot
|
||||
# volume's.
|
||||
self.volume_id = volume_id & 0xFFFFFFFF
|
||||
self.label = label
|
||||
self.fat_size, self.cluster_count = fat32_geometry(total_sectors)
|
||||
if self.cluster_count < 65525:
|
||||
sys.exit(f"error: image too small for FAT32 ({self.cluster_count} clusters "
|
||||
@@ -146,8 +152,8 @@ class Fat32Image:
|
||||
0x80, # drive number
|
||||
0, # reserved
|
||||
0x29, # extended boot signature
|
||||
0x12345678, # volume id
|
||||
b"DANOS ", # volume label
|
||||
self.volume_id, # volume id
|
||||
self.label.encode("ascii", "replace")[:11].ljust(11, b" "), # volume label
|
||||
b"FAT32 ", # filesystem type
|
||||
)
|
||||
sector[510] = 0x55
|
||||
@@ -276,9 +282,9 @@ def build_tree(pairs):
|
||||
return root
|
||||
|
||||
|
||||
def build(out_path, size_mib, pairs):
|
||||
def build(out_path, size_mib, pairs, volume_id=0x12345678, label="DANOS"):
|
||||
total_sectors = size_mib * 1024 * 1024 // SECTOR
|
||||
image = Fat32Image(total_sectors)
|
||||
image = Fat32Image(total_sectors, volume_id, label)
|
||||
tree = build_tree(pairs)
|
||||
write_directory(image, 2, tree, 0, True)
|
||||
with open(out_path, "wb") as handle:
|
||||
@@ -350,14 +356,32 @@ def main(argv):
|
||||
if len(argv) == 3 and argv[1] == "--verify":
|
||||
verify(argv[2])
|
||||
return 0
|
||||
# Optional flags ahead of the positionals: --serial <hex> sets the FAT volume
|
||||
# id (the /volumes/fat-<id> content identity), --label <name> its display
|
||||
# label. A second FAT image passes a distinct --serial so its id-path cannot
|
||||
# collide with the boot volume's.
|
||||
argv = list(argv)
|
||||
volume_id = 0x12345678
|
||||
label = "DANOS"
|
||||
i = 1
|
||||
while i < len(argv):
|
||||
if argv[i] == "--serial" and i + 1 < len(argv):
|
||||
volume_id = int(argv[i + 1], 16)
|
||||
del argv[i:i + 2]
|
||||
elif argv[i] == "--label" and i + 1 < len(argv):
|
||||
label = argv[i + 1]
|
||||
del argv[i:i + 2]
|
||||
else:
|
||||
i += 1
|
||||
if len(argv) < 3 or (len(argv) - 3) % 2 != 0:
|
||||
sys.exit("usage: make-fat-image.py <out.img> <size-MiB> [<dest> <host>]...\n"
|
||||
sys.exit("usage: make-fat-image.py [--serial <hex>] [--label <name>] "
|
||||
"<out.img> <size-MiB> [<dest> <host>]...\n"
|
||||
" make-fat-image.py --verify <out.img>")
|
||||
out_path = argv[1]
|
||||
size_mib = int(argv[2])
|
||||
rest = argv[3:]
|
||||
pairs = [(rest[i], rest[i + 1]) for i in range(0, len(rest), 2)]
|
||||
build(out_path, size_mib, pairs)
|
||||
build(out_path, size_mib, pairs, volume_id, label)
|
||||
return 0
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,88 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Assemble an MBR-partitioned disk image from N FAT32 partitions — the danos
|
||||
multi-volume test disk.
|
||||
|
||||
Pure Python 3 stdlib (no mtools / parted). Each partition is a real FAT32
|
||||
filesystem produced by make-fat-image.py, laid out behind a classic MBR so the
|
||||
danos partition prober (partition.allVolumes) walks the table and the volume
|
||||
manager spawns one confined filesystem per partition — several volumes sharing
|
||||
ONE block channel, each clamped to its own LBA range. That shared-channel,
|
||||
per-partition path is what a single stick with two partitions exercises and a
|
||||
pair of single-volume sticks does not.
|
||||
|
||||
make-partitioned-image.py <out.img> [<serial-hex> <size-MiB>]...
|
||||
|
||||
Each partition is an empty FAT32 with the given volume serial (its /volumes/
|
||||
fat-<serial> content id). Partitions are 1-MiB aligned; the MBR marks each
|
||||
type 0x0C (FAT32 LBA). At most four (an MBR holds four primaries).
|
||||
"""
|
||||
|
||||
import os
|
||||
import struct
|
||||
import subprocess
|
||||
import sys
|
||||
import tempfile
|
||||
|
||||
SECTOR = 512
|
||||
ALIGN = 2048 # sectors (1 MiB) — standard partition alignment, and the gap the MBR sits in
|
||||
MBR_TYPE_FAT32_LBA = 0x0C
|
||||
MAX_PRIMARY_PARTITIONS = 4
|
||||
HERE = os.path.dirname(os.path.abspath(__file__))
|
||||
|
||||
|
||||
def align_up(sectors, to=ALIGN):
|
||||
return (sectors + to - 1) // to * to
|
||||
|
||||
|
||||
def main(argv):
|
||||
if len(argv) < 4 or (len(argv) - 2) % 2 != 0:
|
||||
sys.exit("usage: make-partitioned-image.py <out.img> [<serial-hex> <size-MiB>]...")
|
||||
out_path = argv[1]
|
||||
specs = [(argv[i], int(argv[i + 1])) for i in range(2, len(argv), 2)]
|
||||
if len(specs) > MAX_PRIMARY_PARTITIONS:
|
||||
sys.exit(f"error: an MBR holds at most {MAX_PRIMARY_PARTITIONS} primary partitions")
|
||||
|
||||
# Generate each partition's FAT32 image, then place it at its aligned start.
|
||||
partitions = [] # (start_sector, sector_count, bytes)
|
||||
cursor = ALIGN # leave the first 1 MiB for the MBR + alignment gap
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
for idx, (serial, size_mib) in enumerate(specs):
|
||||
part_path = os.path.join(tmp, f"p{idx}.img")
|
||||
subprocess.run(
|
||||
[sys.executable, os.path.join(HERE, "make-fat-image.py"),
|
||||
"--serial", serial, "--label", f"DATA{idx}",
|
||||
part_path, str(size_mib)],
|
||||
check=True, stdout=subprocess.DEVNULL)
|
||||
with open(part_path, "rb") as handle:
|
||||
data = handle.read()
|
||||
count = len(data) // SECTOR
|
||||
partitions.append((cursor, count, data))
|
||||
cursor = align_up(cursor + count)
|
||||
|
||||
total_sectors = cursor
|
||||
disk = bytearray(total_sectors * SECTOR)
|
||||
# The MBR: a disk signature, one partition entry per FAT partition, 0x55AA.
|
||||
# No boot code (this disk is data, never booted); danos's mount() sees the
|
||||
# signature but no BPB at LBA 0 and takes the MBR-walk path.
|
||||
struct.pack_into("<I", disk, 440, 0x0D05DA05) # arbitrary but fixed disk signature
|
||||
for idx, (start, count, data) in enumerate(partitions):
|
||||
entry = 446 + idx * 16
|
||||
disk[entry + 0] = 0x00 # not bootable
|
||||
disk[entry + 1:entry + 4] = b"\xFE\xFF\xFF" # CHS start (LBA-aware tools ignore)
|
||||
disk[entry + 4] = MBR_TYPE_FAT32_LBA
|
||||
disk[entry + 5:entry + 8] = b"\xFE\xFF\xFF" # CHS end
|
||||
struct.pack_into("<I", disk, entry + 8, start) # start LBA
|
||||
struct.pack_into("<I", disk, entry + 12, count) # sector count
|
||||
disk[start * SECTOR:start * SECTOR + len(data)] = data
|
||||
disk[510] = 0x55
|
||||
disk[511] = 0xAA
|
||||
|
||||
with open(out_path, "wb") as handle:
|
||||
handle.write(disk)
|
||||
print(f"make-partitioned-image: wrote {out_path} "
|
||||
f"({total_sectors * SECTOR // (1024 * 1024)} MiB, {len(partitions)} partitions)")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main(sys.argv))
|
||||
Reference in New Issue
Block a user