Compare commits
78
Commits
3c9f454398
...
main
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
4037e746aa | ||
|
|
4dfb5012c0 | ||
|
|
061eb7c004 | ||
|
|
5dc966838a | ||
|
|
700452dc4e | ||
|
|
0faa0fd21b | ||
|
|
c47215821c | ||
|
|
6ddb08091d | ||
|
|
77b64229c2 | ||
|
|
90906bcefe | ||
|
|
2c2745e9e5 | ||
|
|
2a6d604577 | ||
|
|
e81a4e6f1d | ||
|
|
e240341bfb | ||
|
|
62eb2a748a | ||
|
|
56bd2e7678 | ||
|
|
0b25cd2c94 | ||
|
|
d59279422e | ||
|
|
bf9f8560c6 | ||
|
|
da7dcce64e | ||
|
|
9750db14da | ||
|
|
b2a5a0a3c6 | ||
|
|
7efe7b72d8 | ||
|
|
d4b544d66b | ||
|
|
6d4992ae02 | ||
|
|
df61693065 | ||
|
|
f1e79d0eeb | ||
|
|
167e9c7a9e | ||
|
|
5bfdb75e12 | ||
|
|
3fb8a9b96f | ||
|
|
ea8ccf65d0 | ||
|
|
aca3d5855a | ||
|
|
5637d0e5fc | ||
|
|
48ab12e262 | ||
|
|
020e31bc8f | ||
|
|
c81120ef0f | ||
|
|
4f9196c03e | ||
|
|
eff95416d0 | ||
|
|
addd264880 | ||
|
|
fca41b351e | ||
|
|
bf0595763e | ||
|
|
8216be991d | ||
|
|
68e65803eb | ||
|
|
af47d41989 | ||
|
|
5e89b111cf | ||
|
|
e3ec9fa668 | ||
|
|
9e67a74232 | ||
|
|
b9058fe020 | ||
|
|
7c6ed2ca09 | ||
|
|
a67a7015bf | ||
|
|
301bdcaf5b | ||
|
|
d56b1b81c0 | ||
|
|
bc67771bfd | ||
|
|
89d4592777 | ||
|
|
7af65697cc | ||
|
|
73fbbd3922 | ||
|
|
c37402891a | ||
|
|
f1bdce25e0 | ||
|
|
7ea54a84e5 | ||
|
|
d63a008148 | ||
|
|
b59f981c58 | ||
|
|
60b41c0e82 | ||
|
|
451abba000 | ||
|
|
9da63e81e2 | ||
|
|
ada251150a | ||
|
|
728b436d0f | ||
|
|
092817ba2e | ||
|
|
a44b397bed | ||
|
|
fa54ef6915 | ||
|
|
fae616fa5a | ||
|
|
5a8a2a4d7e | ||
|
|
d4f8dc51b9 | ||
|
|
36ca3f98b3 | ||
|
|
c3604429d4 | ||
|
|
77e7001878 | ||
|
|
a6a3402d92 | ||
|
|
d565a6b845 | ||
|
|
4a2df587bb |
@@ -122,9 +122,11 @@ fn driverArtifact(comptime package: []const u8, comptime artifact: []const u8) S
|
||||
/// (docs/build-packages-plan.md).
|
||||
const production_ship = [_]ShipRow{
|
||||
service("fat"),
|
||||
service("exfat"),
|
||||
service("display"),
|
||||
service("display-demo"),
|
||||
service("device-manager"),
|
||||
service("volume-manager"),
|
||||
service("input"),
|
||||
service("logger"),
|
||||
driver("pci-bus"),
|
||||
@@ -316,6 +318,11 @@ pub fn build(b: *std.Build) void {
|
||||
// out of the same read-only initrd, before it spawns anything — the registrar
|
||||
// has to know its policy before the first provider asks.
|
||||
bundled_list.append(b.allocator, .{ .path = "system/configuration/protocol.csv", .binary = b.path("system/configuration/protocol.csv") }) catch @panic("OOM");
|
||||
// The storage mount map (docs/file-system-development/storage-architecture.md):
|
||||
// filesystems.csv (content signature -> service binary) and volumes.csv (the
|
||||
// optional id -> mount-prefix override), both read by the volume manager.
|
||||
bundled_list.append(b.allocator, .{ .path = "system/configuration/filesystems.csv", .binary = b.path("system/configuration/filesystems.csv") }) catch @panic("OOM");
|
||||
bundled_list.append(b.allocator, .{ .path = "system/configuration/volumes.csv", .binary = b.path("system/configuration/volumes.csv") }) catch @panic("OOM");
|
||||
// A no-option build assumes neither -Dtest-case nor -Ddiagnose: it ships the
|
||||
// production set only. The userspace test fixtures under /test join in only
|
||||
// for a test build — which the QEMU harness signals by passing
|
||||
@@ -326,6 +333,7 @@ pub fn build(b: *std.Build) void {
|
||||
if (test_case != null) for ([_][]const u8{
|
||||
"vfs-test", // the user-space VFS round-trip client
|
||||
"fat-test",
|
||||
"exfat-test", // the exFAT mount round-trip client (S4)
|
||||
"badge-scope-test", // the guessable-id probe: a second process names the first's node and layer
|
||||
"shared-memory-server",
|
||||
"shared-memory-client",
|
||||
@@ -343,6 +351,7 @@ pub fn build(b: *std.Build) void {
|
||||
"protocol-denied-test", // restriction stage one: an ungranted open answers as absence
|
||||
"protocol-conformance-test", // the reserved verbs, asked of every provider the boot bound
|
||||
"device-authority-test", // the attacker: a process handed no device, asserting what it cannot do
|
||||
"block-range-test", // confines itself to a block sub-range, then proves it cannot cross or widen it
|
||||
}) |fixture| {
|
||||
const package = b.lazyDependency(fixture, .{}) orelse
|
||||
@panic("a test fixture package is missing under test/system/services");
|
||||
@@ -440,6 +449,8 @@ pub fn build(b: *std.Build) void {
|
||||
csv_library,
|
||||
xkeyboard_config_library,
|
||||
b.dependency("fat", .{}),
|
||||
b.dependency("exfat", .{}),
|
||||
b.dependency("volume-manager", .{}),
|
||||
b.dependency("display", .{}),
|
||||
b.dependency("ps2-bus", .{}),
|
||||
b.dependency("usb-hid", .{}),
|
||||
|
||||
@@ -46,9 +46,11 @@
|
||||
.@"pci-bus" = .{ .path = "system/drivers/pci-bus" },
|
||||
.init = .{ .path = "system/services/init" },
|
||||
.fat = .{ .path = "system/services/fat" },
|
||||
.exfat = .{ .path = "system/services/exfat" },
|
||||
.display = .{ .path = "system/services/display" },
|
||||
.@"display-demo" = .{ .path = "system/services/display-demo" },
|
||||
.@"device-manager" = .{ .path = "system/services/device-manager" },
|
||||
.@"volume-manager" = .{ .path = "system/services/volume-manager" },
|
||||
.input = .{ .path = "system/services/input" },
|
||||
.logger = .{ .path = "system/services/logger" },
|
||||
// The discovery pair and the /test fixtures are lazy: only what a
|
||||
@@ -63,6 +65,7 @@
|
||||
.@"virtio-gpu" = .{ .path = "system/drivers/virtio-gpu" },
|
||||
.@"vfs-test" = .{ .path = "test/system/services/vfs-test", .lazy = true },
|
||||
.@"fat-test" = .{ .path = "test/system/services/fat-test", .lazy = true },
|
||||
.@"exfat-test" = .{ .path = "test/system/services/exfat-test", .lazy = true },
|
||||
.@"badge-scope-test" = .{ .path = "test/system/services/badge-scope-test", .lazy = true },
|
||||
.@"shared-memory-server" = .{ .path = "test/system/services/shared-memory-server", .lazy = true },
|
||||
.@"shared-memory-client" = .{ .path = "test/system/services/shared-memory-client", .lazy = true },
|
||||
@@ -80,6 +83,7 @@
|
||||
.@"protocol-denied-test" = .{ .path = "test/system/services/protocol-denied-test", .lazy = true },
|
||||
.@"protocol-conformance-test" = .{ .path = "test/system/services/protocol-conformance-test", .lazy = true },
|
||||
.@"device-authority-test" = .{ .path = "test/system/services/device-authority-test", .lazy = true },
|
||||
.@"block-range-test" = .{ .path = "test/system/services/block-range-test", .lazy = true },
|
||||
// See `zig fetch --save <url>` for a command-line interface for adding dependencies.
|
||||
//.example = .{
|
||||
// // When updating this field to a new URL, be sure to delete the corresponding
|
||||
|
||||
+5
-1
@@ -51,7 +51,11 @@ rather than restate it. Roughly in the order things happen at runtime:
|
||||
14. **[vfs-protocol.md](file-system-development/vfs-protocol.md) — the VFS wire protocol.** The language-neutral
|
||||
byte-level spec of the file protocol spoken over IPC: request/reply headers,
|
||||
the operation table, mount routing, and the append-only evolution rules — the
|
||||
first IPC protocol documented as public ABI.
|
||||
first IPC protocol documented as public ABI. Its architectural frame is
|
||||
**[storage-architecture.md](file-system-development/storage-architecture.md) — the storage stack**:
|
||||
the layers from application to hardware, the three kinds of boundary
|
||||
(protocol, library, control-plane), how to add a filesystem, and the
|
||||
per-layer responsibilities when removable media is yanked and returned.
|
||||
15. **[drivers.md](device-driver-development/drivers.md) — writing a driver.** The payoff: a driver is an
|
||||
ordinary ring-3 process that claims a device, maps its registers, and **sleeps
|
||||
until its hardware interrupts it**. The claim is the capability; `irq_ack` is the
|
||||
|
||||
@@ -136,7 +136,7 @@ xHCI match already uses); USB children carry the (class, subclass, protocol) tri
|
||||
from usb-ids.zig — each bus's native language, decoded by the shared ids modules.
|
||||
(Since the registry landed, `child_added` also carries a `bus` discriminator and
|
||||
the numeric `vendor`/`device`/`subsystem` ids the finer match levels need —
|
||||
see [/etc/devices.csv](devices-csv.md).)
|
||||
see [/system/configuration/devices.csv](devices-csv.md).)
|
||||
|
||||
## Supervision and restart
|
||||
|
||||
@@ -227,7 +227,7 @@ published exit events, signals + `process`). On top of those:
|
||||
class triple alone, so a virtio-gpu could only be matched as a generic display
|
||||
function and the driver had to re-confirm its `1AF4:1050` identity from config
|
||||
space after being spawned. The manifest the earlier note anticipated landed as a
|
||||
human-readable registry: **[/etc/devices.csv](devices-csv.md)**, parsed by the
|
||||
human-readable registry: **[/system/configuration/devices.csv](devices-csv.md)**, parsed by the
|
||||
pure `device-registry` module and read by the manager at boot. A row binds a
|
||||
driver to a device by any of base / subclass / prog-IF / vendor / device /
|
||||
subsystem / `_HID`, most-specific match winning; it is authoritative (no
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
# /etc/devices.csv — the device registry
|
||||
# /system/configuration/devices.csv — the device registry
|
||||
|
||||
**Status: built (2026-07-26).** The device manager reads `/etc/devices.csv` at
|
||||
**Status: built (2026-07-26).** The device manager reads `/system/configuration/devices.csv` at
|
||||
boot and binds every device a bus driver reports to the driver the registry
|
||||
names. It replaces the three hand-written `switch` tables that used to live in
|
||||
the manager (`pciDriverForIdentity`, `hidDriverFor`, `usbDriverForIdentity`) —
|
||||
@@ -73,15 +73,15 @@ the first, so the shadowed rule is visible rather than silently dropped.
|
||||
|
||||
There is no compiled-in default table behind the registry. A device that no row
|
||||
matches goes **unbound** and is logged; the manager never guesses. A missing or
|
||||
empty `/etc/devices.csv` therefore means nothing matches — which is loud at boot,
|
||||
empty `/system/configuration/devices.csv` therefore means nothing matches — which is loud at boot,
|
||||
not a silent half-working system.
|
||||
|
||||
## How the manager reads it
|
||||
|
||||
`/etc/devices.csv` is bundled into the initial ramdisk (`build.zig`'s `bundled`
|
||||
list). The kernel serves the initrd's `/etc` tree directly — the `fat` service is
|
||||
spawned *after* the device manager and is irrelevant to `/etc` — so the manager
|
||||
reads the file with a plain `fs.open("/etc/devices.csv")` + `read`, with no
|
||||
`/system/configuration/devices.csv` is bundled into the initial ramdisk (`build.zig`'s `bundled`
|
||||
list). The kernel serves the initrd's `/system/configuration` tree directly — the `fat` service is
|
||||
spawned *after* the device manager and is irrelevant to `/system/configuration` — so the manager
|
||||
reads the file with a plain `fs.open("/system/configuration/devices.csv")` + `read`, with no
|
||||
filesystem service running and no boot-ordering dependency. It parses the bytes
|
||||
once in `initialise`, before any bus driver can report a device to match.
|
||||
|
||||
@@ -105,6 +105,6 @@ in different namespaces — against the right `bus` column.
|
||||
`build_support.programModule`). Then bundle it at `/system/drivers/<name>`:
|
||||
one dependency + one bundled entry in the root `build.zig`, one line in the
|
||||
root `build.zig.zon`.
|
||||
2. Add a row to `etc/devices.csv` naming the identity it binds and its full path.
|
||||
2. Add a row to `system/configuration/devices.csv` naming the identity it binds and its full path.
|
||||
|
||||
No device-manager change is required — the registry is the seam.
|
||||
|
||||
@@ -193,12 +193,14 @@ class driver, the device manager, or the kernel may share them freely.
|
||||
a PS/2 or 16550 driver possible; the low-rate legacy hardware that needs it is fine with
|
||||
a syscall per access. `io_port` resources were recorded by discovery and ignored — now
|
||||
they're used.
|
||||
- **M16 (detection)** — the IOMMU is now *found*: discovery parses the ACPI DMAR table,
|
||||
maps the first VT-d unit, and reads its version + capabilities (`iommu_present` in the
|
||||
platform info). This is detection only — **no translation domains are programmed, so
|
||||
DMA is still unprotected** (the caveat below). Enforcement lands with the first DMA
|
||||
driver, which is what there is to protect and test against. Proven in the `iommu` test,
|
||||
booted with an emulated `intel-iommu`.
|
||||
- **M16 (enforcement)** — the IOMMU is now *found and used*: discovery parses the ACPI
|
||||
DMAR table, maps the first VT-d unit, and reads its version + capabilities
|
||||
(`iommu_present` in the platform info), and **`device_claim` programs a private
|
||||
per-device translation domain** for the claimed function — rolling the claim back with
|
||||
`-ECONFINE` if it cannot confine it — so `dma_alloc` buffers are bound into that domain
|
||||
and torn down at process death. DMA is protected on any IOMMU-equipped machine; the
|
||||
system fails open only when there is no IOMMU at all. Proven in the `iommu` test, booted
|
||||
with an emulated `intel-iommu`.
|
||||
- **`system_spawn`** — a user-space supervisor starts a driver:
|
||||
`system_spawn(name, arguments)` loads a binary bundled in the initial-ramdisk as a
|
||||
fresh ring-3 process; `name` becomes the child's argv[0] and the optional
|
||||
@@ -369,25 +371,29 @@ capability walk (MSI, MSI-X, PCIe extended caps) without any new syscall.
|
||||
Note QEMU's HPET reports `Tn_FSB_INT_DEL_CAP = 0` — no MSI — so an HPET timer could never
|
||||
exercise this path. The first MSI driver will be the first PCI driver.
|
||||
|
||||
## M16 — the IOMMU, and the honest caveat ◑ detection done, enforcement pending
|
||||
## M16 — the IOMMU, and the honest caveat ✅ done
|
||||
|
||||
*The IOMMU is now detected (DMAR parsed, VT-d unit mapped and read — see the `iommu`
|
||||
test), but **enforcement is not built**: no translation domains are programmed, so the
|
||||
caveat below still holds in full. Detection can't be taken further usefully until there
|
||||
is a DMA driver to protect and QEMU's `intel-iommu` to test the protection against —
|
||||
building the per-device domains alongside that first driver is both the natural order
|
||||
and the only way to verify them. The rest of this section is the original caveat.*
|
||||
test) **and enforced**: `device_claim` programs a private per-device VT-d/AMD-Vi
|
||||
translation domain for the claimed function and rolls the claim back with `-ECONFINE` if
|
||||
it cannot confine it, `dma_alloc` buffers are bound into that domain and torn down at
|
||||
process death, and the machine fails open only when it has no IOMMU at all. So the caveat
|
||||
below no longer holds except on IOMMU-less hardware. The rest of this section is the
|
||||
original caveat, kept for the reasoning.*
|
||||
|
||||
Everything above is capability-gated at the *CPU*. None of it is gated at the *device*.
|
||||
A driver that can program a bus-mastering engine can make that device write to any
|
||||
physical address, because page tables sit between the CPU and RAM, not between a device
|
||||
and RAM. Until VT-d/DMAR (or SMMU on ARM) is programmed from the DMAR table, **`device_claim`
|
||||
on any DMA-capable device is equivalent to granting ring 0.**
|
||||
Everything above is capability-gated at the *CPU*. CPU page tables alone do not gate the
|
||||
*device*: a driver that can program a bus-mastering engine could make that device write to
|
||||
any physical address, because those page tables sit between the CPU and RAM, not between a
|
||||
device and RAM. That is exactly what the IOMMU closes. Now that VT-d/DMAR (and AMD-Vi; SMMU
|
||||
on ARM) is programmed, **`device_claim` confines the function into a private translation
|
||||
domain** and rolls the claim back with `-ECONFINE` if it cannot — so a claimed DMA-capable
|
||||
device is no longer equivalent to granting ring 0.
|
||||
|
||||
This does not make the model useless — it's the same position Linux is in with the
|
||||
IOMMU off, and every other guarantee (crash isolation, restart, no shared address
|
||||
space) still holds. But "user-space drivers are memory-safe" is not true yet, and the
|
||||
gap should be named rather than implied.
|
||||
This puts the model ahead of Linux-with-the-IOMMU-off: with an IOMMU present,
|
||||
"user-space drivers are memory-safe" now holds, alongside every other guarantee (crash
|
||||
isolation, restart, no shared address space). The one remaining gap — a machine with no
|
||||
IOMMU at all, where the system deliberately fails open — should be named rather than
|
||||
implied.
|
||||
|
||||
## Ordering
|
||||
|
||||
|
||||
@@ -32,7 +32,7 @@ kernel ──spawns──► init (PID 1) ──spawns──► device-manag
|
||||
spawns only init, the service supervisor: the driver supervisor: enumerates
|
||||
publishes the starts the system /system/devices, matches each device
|
||||
initial-ramdisk services (device-manager, to a driver, and system_spawn's it
|
||||
so user space can fat, logger, ...). Its
|
||||
so user space can volume-manager, logger, ...). Its
|
||||
system_spawn from it list is init policy.
|
||||
```
|
||||
|
||||
@@ -43,7 +43,7 @@ the optional arguments its argv[1..], on a SysV entry stack, see sysv.md). Every
|
||||
|
||||
- **init** ([system/services/init](system/services/init/init.zig)) is the **service
|
||||
supervisor**. It spawns the system services danos brings up at boot — today `input`,
|
||||
the `device-manager`, `fat`, `display`, `display-demo`, and the `logger` — from a
|
||||
the `device-manager`, `volume-manager`, `display`, `display-demo`, and the `logger` — from a
|
||||
small list. Drivers are deliberately *not* its job. (An earlier draft listed a `vfs`
|
||||
service here; that service is retired — the router moved into the kernel as
|
||||
`fs_resolve`.)
|
||||
@@ -66,9 +66,10 @@ the optional arguments its argv[1..], on a SysV entry stack, see sysv.md). Every
|
||||
|
||||
So "how is a driver discovered and configured" has two halves: **discovery** is the
|
||||
kernel's device table, read by anyone; **configuration** is two user-space policies —
|
||||
init's service list and the device-manager's match table. Both are hardcoded in their
|
||||
respective programs today; the natural next step is to move them into `/etc` (see the
|
||||
milestone notes in [driver-model.md](driver-model.md)). `system_spawn` is currently
|
||||
init's service list and the device-manager's match table. Both are now data, not code —
|
||||
init reads `/system/configuration/init.csv` and the device manager reads
|
||||
`/system/configuration/devices.csv`, each parsed at startup (the compiled-in switch
|
||||
tables are gone). `system_spawn` is currently
|
||||
ungated — any process may spawn any bundled binary — because there is no spawn
|
||||
capability yet.
|
||||
|
||||
@@ -301,12 +302,13 @@ process releases its claims and IRQ/MSI bindings — `releaseAllOwnedBy`,
|
||||
- **Page granularity.** `mmio_map` rounds to 4 KiB. Two devices sharing a page means
|
||||
granting one grants the other. A `device_register`ed child's *resource* can be narrower
|
||||
than a page, but its *mapping* can't.
|
||||
- **DMA is not contained.** A driver that can program a bus-mastering device can make
|
||||
that device write to *any* physical address — page tables don't sit between a device
|
||||
and RAM; an IOMMU does. The IOMMU is now *detected* (M16), but no translation domains
|
||||
are programmed, so `device_claim` on a DMA-capable device is still effectively
|
||||
equivalent to granting ring 0. This is the largest gap between the design's promise and
|
||||
what it delivers; enforcement lands with the first DMA driver.
|
||||
- **DMA is contained — except on a machine with no IOMMU.** A driver that can program a
|
||||
bus-mastering device could make that device write to *any* physical address — page
|
||||
tables don't sit between a device and RAM; an IOMMU does. `device_claim` now confines
|
||||
each claimed PCI function into its own VT-d/AMD-Vi translation domain and rolls the
|
||||
claim back with `-ECONFINE` if it can't (`system/kernel/process.zig`); DMA buffers are
|
||||
bound into that domain and torn down at process death. The residual gap is fail-open:
|
||||
where the machine exposes **no IOMMU at all**, a DMA-capable claim still reaches RAM.
|
||||
- **No voluntary `dev_release`.** A *live* driver can't drop a claim — only exit
|
||||
releases it (any path out of a process runs `releaseAllOwnedBy`) — so handing a
|
||||
device between running drivers still means exiting.
|
||||
@@ -365,10 +367,9 @@ $ python3 test/qemu_test.py device-manager acpi-ps2 pci-scan containment irqfree
|
||||
## What's next (not done here)
|
||||
|
||||
The big driver-model pieces — capability passing (class drivers), DMA + barriers, MSI,
|
||||
and IOMMU detection — are **now done** ([driver-model.md](driver-model.md), M13–M16), as
|
||||
is **port I/O** (`io_read`/`io_write`, the claim-gated syscalls that make a PS/2 or 16550
|
||||
driver possible). What's left is IOMMU *enforcement* (per-device domains — it waits on
|
||||
the first DMA driver to protect and test against) and these smaller items:
|
||||
and per-device IOMMU confinement — are **now done** ([driver-model.md](driver-model.md),
|
||||
M13–M16), as is **port I/O** (`io_read`/`io_write`, the claim-gated syscalls that make a
|
||||
PS/2 or 16550 driver possible). What's left is a handful of smaller items:
|
||||
|
||||
- **Releasing a claim** — half done. The kernel now drops *all* of a dead driver's
|
||||
claims on every path out of a process (`releaseAllOwnedBy`, called from process
|
||||
|
||||
@@ -162,7 +162,7 @@ Without the boot-tree row the binary never reaches the image and the
|
||||
device-manager has nothing to spawn. (The package also builds standalone:
|
||||
`cd system/drivers/intel-uhd-graphics-750 && zig build`.)
|
||||
|
||||
## 3. Add the match rule to `etc/devices.csv`
|
||||
## 3. Add the match rule to `system/configuration/devices.csv`
|
||||
|
||||
One row: bus, class triplet, vendor/device, driver path. **Copy the class
|
||||
triplet from the pci-bus boot log line, not from another row** — for the iGPU
|
||||
@@ -195,9 +195,9 @@ mapping — with zero risk to the hardware.
|
||||
## 5. Verify the plumbing
|
||||
|
||||
- `zig build test` still passes.
|
||||
- On the image: `/var/log/<boot-stamp>/system/services/device-manager.log`
|
||||
- On the image: `/system/logs/<boot-stamp>/system/services/device-manager.log`
|
||||
shows `spawned <name> for device <N>`, and
|
||||
`/var/log/<boot-stamp>/system/drivers/<name>.log` holds the resource list and
|
||||
`/system/logs/<boot-stamp>/system/drivers/<name>.log` holds the resource list and
|
||||
your first read.
|
||||
- If the driver did not spawn, diagnose in this order: binary on the image
|
||||
(step 2) → CSV row matches the log line exactly (step 3) → path identical in
|
||||
|
||||
@@ -32,8 +32,9 @@ Read-only and writable halves of `/system`: the program subtrees (`kernel`,
|
||||
initrd-backed or synthetic — while `configuration` and `logs` are mutable
|
||||
machine state served by the boot-volume FAT backend. The kernel's
|
||||
reserved-prefix rule (no mount may shadow `/system`, `/test`, or `/protocol`)
|
||||
needs a carve-out for exactly these two writable subtrees; that lands with the
|
||||
path migration below.
|
||||
carves out exactly these two writable subtrees (`initrd_carve_outs` in
|
||||
`system/kernel/vfs.zig`), so the boot volume mounts them while every other
|
||||
`/system` path stays initrd-served.
|
||||
|
||||
Deliberately not defined yet: a temporary-files location and per-application
|
||||
mutable storage. Both belong to the `/applications` design and will be
|
||||
@@ -74,10 +75,10 @@ expect, mapped onto the real tree; the tree itself stays danos-native.
|
||||
|
||||
## Migration
|
||||
|
||||
The tree above is the specification; some code still writes the unix paths it
|
||||
replaced. The flag-day converting them:
|
||||
The tree above is the specification, and the code writes it. A flag-day already
|
||||
converted the unix paths it replaced:
|
||||
|
||||
| Today (in code) | Becomes | Where |
|
||||
| Was | Now | Where |
|
||||
|------------------------------------------|-------------------------------------------|-----------------------------------------------------------------|
|
||||
| `/etc/init.csv` | `/system/configuration/init.csv` | `system/services/init/init.zig` |
|
||||
| `/etc/devices.csv` | `/system/configuration/devices.csv` | `system/services/device-manager/device-manager.zig` |
|
||||
@@ -85,6 +86,6 @@ replaced. The flag-day converting them:
|
||||
| `/mnt/usb` | `/volumes/usb` | `system/services/fat/fat.zig`, the fat/vfs tests |
|
||||
| `ServiceId` lookup | resolve + open under `/protocol` | every service and client; [protocol-namespace.md](../os-development/protocol-namespace.md) |
|
||||
|
||||
The boot-image builder and the on-volume directory layout move in the same
|
||||
The boot-image builder and the on-volume directory layout moved in the same
|
||||
change, so a freshly written image and the paths the services expect never
|
||||
disagree.
|
||||
|
||||
@@ -0,0 +1,300 @@
|
||||
# The storage architecture: layers, boundaries, responsibilities
|
||||
|
||||
> **Status:** the layered model below is the settled design
|
||||
> ([storage-design-rationale.md](storage-design-rationale.md) records how it was
|
||||
> reached, and [volume-manager-plan.md](../volume-manager-plan.md) how it was
|
||||
> built). **Built** (V0–V4 + the storage-stack S1/S2): the data path, the driver
|
||||
> range confinement (per-sender clamp + the confinement gate), the `medium_changed`
|
||||
> presence event, the volume manager itself — it probes the partition table,
|
||||
> confines each filesystem to its partition, spawns one filesystem per volume, and
|
||||
> supervises it — the removal half of the lifecycle (a pulled stick unmounts), the
|
||||
> identity ladder (GPT GUID + name, FAT serial + label, MBR), and the mount map:
|
||||
> `filesystems.csv` (signature → binary) + `volumes.csv` (identity → optional
|
||||
> override), a volume's mount path IS its content id (`/volumes/<id>`), with the
|
||||
> label as display metadata a `volumes` query returns. Multi-volume is **built**:
|
||||
> the manager adopts every storage device, probes each device's whole partition
|
||||
> table, and spawns one range-confined FAT per volume — several volumes across
|
||||
> several devices, or several partitions sharing one device's channel — each at
|
||||
> its own `/volumes/<id>` path with its own supervision. The boot volume is
|
||||
> identified by **content** (a volume backs `/system/configuration` + `/system/logs`
|
||||
> only when it resolves `/system/configuration` on its own media), so it works as
|
||||
> any partition of any device. exFAT is **built** as a second engine
|
||||
> (`system/services/exfat`): full read + write, directories, rename, and on-disk
|
||||
> up-case folding, reusing `library/kernel/file-system-harness` wholesale — the
|
||||
> reuse claim, proven — and a volume routes to fat or exfat by its VBR, at an
|
||||
> `exfat-<serial>` id-path. Removal is robust to all three triggers now: a
|
||||
> pulled device (presence polling), a medium that leaves while its device stays
|
||||
> (the volume manager CONSUMES `medium_changed`), and a storage driver that
|
||||
> crashes while its device stays present (a channel-liveness `geometry()` probe
|
||||
> reaps the volume and rebuilds it on the restarted driver's fresh channel). The
|
||||
> re-adopt-and-remount path is QEMU-proven by the driver-crash rebuild; a physical
|
||||
> unplug/replug exercises the same path but is bench-pending (QEMU cannot
|
||||
> re-present a usb-storage `device_add`). **Still pending**: the `filesystem UUID`
|
||||
> rung (ext-family superblocks, which need such an engine); and arbitration when
|
||||
> two volumes both resolve the boot markers (S3 mounts both and logs each claim;
|
||||
> picking one is deferred). A few
|
||||
> markers below are left where a duty is still pending.
|
||||
|
||||
## The model
|
||||
|
||||
One pattern, applied twice: an application is a client of a service over a
|
||||
protocol; a service is a client of a driver over a protocol.
|
||||
|
||||
```
|
||||
application
|
||||
│ vfs protocol (routed by the kernel mount table)
|
||||
▼
|
||||
filesystem service ── one process per volume
|
||||
│ inside: [file-protocol shell] ── Zig API ── [fs engine]
|
||||
│ block protocol (channel received at spawn)
|
||||
▼
|
||||
storage driver ── one process per device (usb-storage per stick)
|
||||
│ usb-transfer protocol (channel received via hello, by lineage)
|
||||
▼
|
||||
bus driver ── one process per controller (usb-xhci-bus)
|
||||
│ hardware
|
||||
```
|
||||
|
||||
Three kinds of boundary, deliberately different:
|
||||
|
||||
- **Protocol boundaries** sit where processes meet (vfs, block, usb-transfer).
|
||||
Each side is independently restartable; channels are established by
|
||||
capability handoff, never by registry names (communication.md
|
||||
"Establishment: two planes"). Steady-state file I/O crosses exactly two —
|
||||
vfs and block.
|
||||
- **Library boundaries** sit where concerns meet inside one process. The
|
||||
filesystem service's engine (on-disk format logic, host-testable, behind
|
||||
the four-function `BlockDevice` vtable) and shell (IPC serving, open-node
|
||||
table, mount registration) talk through a Zig API. No protocol between
|
||||
them: they share fate regardless — a corrupt engine corrupts the answers
|
||||
either way — so a channel there would add a hop per file operation and buy
|
||||
nothing. The seam exists for compile-time testability and reuse; the
|
||||
*process* is the restart unit.
|
||||
- **Control-plane relationships** sit beside the data path, never on it. Two
|
||||
supervisors, one per layer: the **device manager** wires and revives the
|
||||
device layers (bus and storage drivers — devices only); the **volume
|
||||
manager** *(built)* wires and revives the volume layer (filesystem
|
||||
services). Neither touches steady-state I/O.
|
||||
|
||||
## Who does what
|
||||
|
||||
**Bus driver** (usb-xhci-bus, per controller): hardware only. Enumerates,
|
||||
reports children to the device manager, serves the transfer contract to its
|
||||
own children's class drivers, routed by lineage.
|
||||
|
||||
**Storage driver** (usb-storage, per device; later nvme, ahci): hardware →
|
||||
blocks. Speaks its transport upward from its device; serves the block
|
||||
protocol (geometry, read, write, flush, attach, detach — and the pushed
|
||||
`medium_changed` presence event, published today from a slow TEST UNIT READY
|
||||
poll; *planned*: translating it from the transport's native signal instead of
|
||||
polling). **Content-blind, permanently**:
|
||||
it reads block 0 only as a bring-up self-check and parses nothing — no MBR, no
|
||||
GPT, no filesystem magic, ever. It has one content-blind mechanism *(built)*:
|
||||
**named sub-ranges** — "serve blocks [a, b) as a channel of the
|
||||
same block contract". It clamps and offsets; it never knows the numbers came
|
||||
from a partition table. The clamp lives here and nowhere else because a
|
||||
channel must carry exactly the authority it grants: handing a filesystem the
|
||||
whole disk plus a polite base offset would let a buggy or compromised
|
||||
filesystem scribble the neighboring partition — the same authority-overshoot
|
||||
the device-authority track eliminated for MMIO and DMA.
|
||||
|
||||
**Volume manager** *(built; `system/services/volume-manager`)*: the policy home
|
||||
of the volume layer, one service, supervised by init. It watches the device
|
||||
manager's tree for a storage provider; when one appears it consumer-hellos for
|
||||
the block channel, reads the partition table and the first blocks itself
|
||||
(**it** is the prober), defines the volume's sub-range on the driver, spawns the
|
||||
matching filesystem service confined to that range, and supervises it (backoff,
|
||||
crash-loop cap). *(Built)*: it picks the filesystem binary from
|
||||
`filesystems.csv` by the volume's content signature, and mounts the volume at its
|
||||
content id (`/volumes/<id>`) — or a `volumes.csv` override. The label is display
|
||||
metadata the `volumes` query returns, never the path. Those tables are CSV
|
||||
configuration, read by it (the policy), enforced by nobody else:
|
||||
|
||||
- `filesystems.csv` *(built)* — content signature → filesystem binary. Adding
|
||||
a filesystem adds a row.
|
||||
- `volumes.csv` *(built)* — the mount map, danos's fstab: an OPTIONAL **volume
|
||||
identity → mount prefix** override (a volume with no row mounts at its default
|
||||
`/volumes/<id>`), keyed on content identity and never on port, path, or arrival
|
||||
order (the lesson of Linux's `/dev/sda1`-era fstab, which broke on every
|
||||
port move until `UUID=` replaced it). Identity is read off the medium by
|
||||
the prober, strongest first: GPT partition GUID → filesystem UUID → FAT
|
||||
serial + label → MBR signature + partition index → anonymous (generated
|
||||
name, no persistence). Same identity, same mount point: a drive moved to
|
||||
another port — USB to another hub, SATA to another bay, even a stick
|
||||
returning in a different dock — lands exactly where it was. Duplicate
|
||||
identity (cloned sticks, together) is policy: first keeps the name, the
|
||||
second mounts suffixed and is logged loudly. The boot volume is the
|
||||
recorded identity of the volume carrying `/system/configuration` and
|
||||
`/system/logs`, findable on any port. Every volume's default mount is
|
||||
`/volumes/<id>` — its rendered content identity.
|
||||
|
||||
**Filesystem service** (the FAT service today; one process per volume): the
|
||||
proven unit — block-client + engine + file-protocol provider in one binary. It
|
||||
receives its block channel at spawn; it never discovers devices. It registers
|
||||
its own mounts with the kernel; its write cache lives inside the process, so a
|
||||
write error is observed by the code that owns the volume and surfaces on the
|
||||
owning channel (the anti-fsyncgate rule — never a system-wide dirty pool).
|
||||
*(Built:)* fat receives its mount path as `argv[2]` from the volume manager (the
|
||||
volume's id-path, e.g. `/volumes/fat-12345678`) and mounts its root there. It
|
||||
installs the two `/system` hierarchy rewrites (`/system/configuration`,
|
||||
`/system/logs`) only when it is the boot volume — decided by **content**: it
|
||||
resolves `/system/configuration` on its own media at mount, so a data volume
|
||||
mounts at its id-path alone and never shadows the running system. It no longer
|
||||
self-acquires a volume — the V3b flip made it receive its volume id and block
|
||||
channel from the volume manager, consistent with "it never discovers devices"
|
||||
above. Because several volumes now serve at once, no filesystem binds a shared
|
||||
service name; clients reach each through the kernel mount table (`fs_resolve`
|
||||
routes by prefix to the backing endpoint).
|
||||
|
||||
**Kernel** (mechanism only): the mount table routes paths to backend
|
||||
endpoints — resolve and redirect, never data. Remount-replace is the restart
|
||||
story; a dead backend's slot is swept lazily on the next resolution.
|
||||
`fs_unmount` is ownership-gated *(built, V0)* — only the mounting task may
|
||||
unmount its own prefix; anyone else is refused (`EPERM`). No dead-owner
|
||||
exception: a dead owner's slot is swept lazily by resolution, and restart goes
|
||||
through remount-replace, never through a stranger's unmount.
|
||||
|
||||
## Adding a filesystem
|
||||
|
||||
1. **Write the engine**: pure Zig, no IPC, behind the `BlockDevice` vtable
|
||||
(four functions). On-disk layout in `align(1)` extern structs. Host-test it
|
||||
with `zig build test` fixtures — the FAT engine
|
||||
([engine.zig](../../system/services/fat/engine.zig),
|
||||
[on-disk.zig](../../system/services/fat/on-disk.zig)) is the template, and
|
||||
its ~500 lines of host tests the standard.
|
||||
2. **Reuse the shell**: the filesystem harness — establishment, the
|
||||
badge-scoped open-node table, the nine vfs-protocol handlers, mount
|
||||
registration, the removal path — is shared code, not per-filesystem code.
|
||||
*(Done: extracted from fat's original 434-line shell into
|
||||
`library/kernel/file-system-harness.zig`; fat imports it and instantiates
|
||||
`harness.Server(engine.FileSystem)`.)* An engine plus a `main` wiring it into the
|
||||
harness is a complete filesystem service.
|
||||
3. **Add the configuration row**: one line in `filesystems.csv` mapping the
|
||||
on-disk signature to the binary. No other component changes: the volume
|
||||
manager probes, matches, spawns; the vfs protocol is already
|
||||
backend-neutral (a second provider serves it in production today — init's
|
||||
synthetic registry backend).
|
||||
4. **Prove removal**: extend the removable-media suite (below) for the new
|
||||
filesystem — surprise-yank during writes must lose only what was
|
||||
unflushed, said honestly, with the volume consistent enough to remount.
|
||||
|
||||
The engine never sees: partitions (it receives a volume-shaped block channel;
|
||||
base offsets are the driver's clamp), device discovery (the channel arrives at
|
||||
spawn), mount policy (prefixes are handed to it), or other volumes (one
|
||||
process, one volume).
|
||||
|
||||
## Removable media: responsibilities on removal, per layer
|
||||
|
||||
The design rule, learned from what Linux cannot do: there is exactly **one**
|
||||
surprise-removal path — kill the filesystem process, retire its mounts,
|
||||
respawn on return. No half-alive states, no `remount-ro`, no mounts that
|
||||
error forever (Plan 9's dead-server wart).
|
||||
|
||||
The path folds **three triggers into one lifecycle**: the *device* leaving (a
|
||||
pulled stick — presence polling); the *medium* leaving while the device stays
|
||||
(an SD card pulled from its reader, an ATAPI tray opened, a USB card reader);
|
||||
and a storage *driver crashing* while its device stays in the tree. The second
|
||||
trigger is the pushed `medium_changed` event on the block protocol, published
|
||||
from a TEST UNIT READY poll — the volume manager now **consumes** it (subscribed
|
||||
per device), running the same kill-retire path and re-probing on medium return,
|
||||
so a swapped card is never served with the previous card's filesystem state. The
|
||||
third is caught by a channel-liveness `geometry()` probe: presence polling alone
|
||||
sees the device still present, but the channel is dead, so the manager reaps the
|
||||
volume and rebuilds it on the restarted driver's fresh channel. Still *planned*
|
||||
is translating the transport's native signal (SCSI UNIT ATTENTION, AHCI PxSSTS,
|
||||
NVMe namespace-change AER) in place of the presence poll.
|
||||
|
||||
| Layer | Observes | Must do | Guarantees |
|
||||
|---|---|---|---|
|
||||
| Bus driver | port/hub status change | tear down the device's slots (children first, recursively — built, hot-plug matrix), report `child_removed` per interface | the device tree is honest within one reconcile tick |
|
||||
| Device manager | `child_removed` / reporter death | prune the child; **reap the bound driver** (built) — the storage driver for that stick dies now, not never | no zombie storage processes; re-report rebinds |
|
||||
| Storage driver | its own death (it IS the removed device's driver) | nothing — dying is its removal handling; DMA/IOMMU/claims release mechanically at death | in-flight transfers fail visibly to callers, never hang |
|
||||
| Volume manager *(built)* | a device leaving the tree (poll), a `medium_changed` event, or a dead channel under a still-present device (a crashed driver — `geometry()` liveness probe) | kill that volume's filesystem service (its mounts retire), then re-adopt + remount on return or on the restarted driver's fresh channel | one removal path for all three triggers; mounts never dangle; the manager never serves from behind a dead channel |
|
||||
| Filesystem service | its block channel dies (`EPEER`) mid-operation, or it is killed by the volume manager | if it observes the death first: flush nothing (the medium is gone), answer in-flight requests with errors, exit; dirty write-back data is **lost and said to be lost** | the unflushed write-back window is dropped on a surprise yank — danos writes no on-disk dirty/clean-shutdown marker today; the process never serves from behind a dead channel |
|
||||
| Kernel | backend endpoint death | lazy mount-slot sweep on next resolve (built); ownership-gated `fs_unmount` (built, V0) | resolution under a dead mount is `not_found`, not a hang |
|
||||
| Application | `not_found` / error on paths under the vanished mount | its own error handling — the contract is honest absence, identical to the path never existing | no operation blocks forever on removed media |
|
||||
|
||||
**On return** the same table runs upward in reverse: the bus re-enumerates and
|
||||
re-reports; the device manager respawns the storage driver (built); the volume
|
||||
manager re-probes — same content, same volume identity — respawns the
|
||||
filesystem service, and remounts at the same prefix; applications see the
|
||||
subtree reappear. A different stick in the same port is a *different volume*
|
||||
(identity is content, not port) and mounts wherever its identity says —
|
||||
possibly nowhere but `/volumes/…`.
|
||||
|
||||
### The boot volume: identity is what makes yanking it survivable
|
||||
|
||||
The boot volume is the sharpest instance of the return story, because the
|
||||
system's own log persistence rides it (`/system/logs`), and because nothing
|
||||
else about the running system depends on it at all — every binary and the
|
||||
boot-time configuration live in the initial ramdisk. Its lifecycle,
|
||||
end to end:
|
||||
|
||||
- **At first mount** the volume manager records the identity of the volume
|
||||
carrying `/system/configuration` and `/system/logs` — THE boot volume,
|
||||
from then on a fact about content, not about a port.
|
||||
- **While it is absent**, the system runs on. The kernel log ring keeps
|
||||
accumulating — it is the buffer that makes the absence survivable — and
|
||||
the logger keeps draining it; only *persistence* pauses. File operations
|
||||
under the retired mounts fail honestly (`not_found`). The ring is bounded,
|
||||
so a long absence overwrites its oldest entries: that window is the data
|
||||
loss, and it must be *said* — the logger marks the gap in the file when
|
||||
persistence resumes, never splicing the stream silently.
|
||||
- **On return — any port, any hub, even a different transport** — the
|
||||
prober reads the same identity, the mount map answers with the same
|
||||
prefixes, the filesystem service is respawned, and `/system/logs` is the
|
||||
same tree it was: the logger resumes appending into the SAME
|
||||
`<boot-stamp>` directory, per-binary files continuing where they left
|
||||
off (plus the gap marker if the ring wrapped).
|
||||
- **What never comes back** is the write-back window lost at the yank —
|
||||
the dirty-honesty rule, unchanged; the boot volume gets no exemption.
|
||||
|
||||
This is also the requirement that shapes the logger: it treats the log tree
|
||||
as a volume that comes and goes — failed flushes are retried on the same
|
||||
patient cadence the fat service already uses for storage that arrives late,
|
||||
never abandoned after the first `not_found`.
|
||||
|
||||
What is lost on a surprise yank is exactly the write-back window of the
|
||||
filesystem service, no more: the engine owns its cache, so the blast radius of
|
||||
a yank is one volume's unflushed writes, which are simply lost — danos records
|
||||
no on-disk dirty/clean-shutdown bit yet. A
|
||||
filesystem format with better crash honesty (journaling, copy-on-write — the
|
||||
lesson of QNX's Power-Safe) narrows that window further and slots in as
|
||||
implementation N+1 through the table above, changing nothing else.
|
||||
|
||||
## How the lifecycle is enforced
|
||||
|
||||
Responsibility tables are convention; convention is worthless (the bounds
|
||||
audit's lesson). The volume lifecycle is enforced with the same three levers
|
||||
the device lifecycle already uses, mapped one-to-one:
|
||||
|
||||
1. **A filesystem cannot acquire — it can only be given.** Filesystem
|
||||
binaries hold NO establishment grants: no `open device-manager`, no
|
||||
registry name to look up. The only block channel a filesystem process ever
|
||||
has is the one handed to it at spawn by the volume manager. Serving the
|
||||
wrong volume, a second volume, or a self-discovered volume is not
|
||||
forbidden but *impossible* — the same way a driver cannot claim hardware
|
||||
it was not delegated (device-authority.md).
|
||||
2. **The volume manager supervises with teeth.** Mount-within-deadline or be
|
||||
stopped (a filesystem wedged on a corrupt volume is killed and crash-loop
|
||||
capped; the volume is marked bad, not retried forever). On removal the
|
||||
manager KILLS the filesystem process and retires its mounts — the polite
|
||||
observe-`EPEER`-and-exit path is an optimization; the kill is the
|
||||
guarantee, because a process cannot be trusted to observe its own
|
||||
obsolescence (the reap argument, proven on the device tree).
|
||||
3. **The shared harness is how every filesystem inherits the lifecycle by
|
||||
construction.** The harness — not the engine — owns the state machine:
|
||||
establishment at spawn, mount registration, the flush-on-close hook (an
|
||||
in-memory device-dirty check that commits the device write cache on close),
|
||||
error-out-and-exit on channel death. The engine sits behind the
|
||||
four-function vtable and never sees a channel; it cannot opt out of the
|
||||
lifecycle for the same reason it cannot find a device. This is why the
|
||||
harness is extracted BEFORE the second engine is written.
|
||||
|
||||
Beneath all three, two kernel mechanisms: ownership-gated `fs_unmount` (only
|
||||
the mounting endpoint's holder may unmount), and the lazy dead-endpoint mount
|
||||
sweep as the backstop nothing can disable. And above them, the check: a
|
||||
**lifecycle conformance drill** — mount, serve, yank mid-write, verify honest
|
||||
loss, replug, remount — parameterized over filesystem implementations, so
|
||||
"danos supports filesystem X" MEANS "X passes the drill through the harness",
|
||||
exactly as provider conformance means passing the reserved-verb suite.
|
||||
@@ -0,0 +1,267 @@
|
||||
# The storage design rationale: why the stack is shaped this way
|
||||
|
||||
*2026-08-09. The design record behind
|
||||
[storage-architecture.md](storage-architecture.md): the survey, the
|
||||
trade-offs, and the decisions with their reasons — kept so future changes
|
||||
argue against the evidence rather than rediscovering it. The questions that
|
||||
drove it, verbatim: should the block protocol be separate from the VFS? how
|
||||
do channels work with these block devices? how should we wire up different
|
||||
filesystems? Grounded in how Minix 3, QNX Neutrino, Fuchsia, Redox, Plan 9,
|
||||
and Linux each answered the same questions, and in exactly where our own
|
||||
seams sat. The layering rule it serves: drivers are the lowest level
|
||||
(hardware only); VFS and the filesystems are higher layers; the protocol
|
||||
layer routes between them.*
|
||||
|
||||
## What the survey says, compressed
|
||||
|
||||
**Everyone separates block from VFS.** Even Plan 9, which unifies the *protocol*
|
||||
(a disk is a file tree), keeps the layers: kernel `sd` serves raw ranges,
|
||||
user-space filesystem servers consume them and serve files. The two poles are
|
||||
Minix 3 (every layer a process: one VFS server, one filesystem server *per
|
||||
mounted volume*, driver processes below — driver crashes proven survivable by
|
||||
fault-injection, tens of thousands of injected faults) and QNX (filesystems as
|
||||
DLLs inside the disk-driver process — fastest path, but one corrupt volume takes
|
||||
every volume behind that controller down). Fuchsia is the modern capability-
|
||||
native reference: a tiny block contract that every layer speaks, control channel
|
||||
separate from a data fast-path with pre-registered buffers, filesystems as
|
||||
separate processes launched by a storage-policy component (`fshost`) that probes
|
||||
content and hands each filesystem its block channel at startup. The filesystem
|
||||
never discovers devices.
|
||||
|
||||
**Partition tables are parsed in user space, below the filesystem, above the
|
||||
raw device — never in the kernel, never in the filesystem engine.** Fuchsia
|
||||
tried partitions-as-drivers and formally reversed it (RFC-0257). Plan 9 has the
|
||||
cleanest split of all: the kernel offers only a *mechanism* — "create a named
|
||||
sub-range of this disk" — and a user-space prober parses the table and issues
|
||||
those commands. QNX (`io-blk`) and Minix (driver-side shared library) agree on
|
||||
placement: the block layer multiplexes one physical block object into several
|
||||
logical block objects **of the same contract**.
|
||||
|
||||
**The failure lessons are unanimous.** Linux's shared page cache across the
|
||||
filesystem boundary produced fsyncgate (write errors observed by the wrong
|
||||
process, dirty pages marked clean); the lesson is to keep write caching *inside*
|
||||
the filesystem process, so an error surfaces on the channel that owns the
|
||||
volume. Linux's `errors=remount-ro`/lazy-unmount half-alive states exist because
|
||||
a monolith cannot kill and respawn a filesystem; Plan 9 leaves dead servers'
|
||||
mounts in the namespace erroring forever. A restartable-parts system should have
|
||||
exactly one surprise-removal path: kill the filesystem process, retire its
|
||||
mounts, respawn on return. QNX's deepest lesson: after years of removable-media
|
||||
pain they concluded detection isn't enough and built a copy-on-write filesystem
|
||||
(Power-Safe) — *format-level* crash honesty. And QNX's `mcd` (a standalone
|
||||
policy daemon with declarative insertion/removal rules) plus Fuchsia's `fshost`
|
||||
agree that media policy is one dedicated component, not code scattered through
|
||||
filesystems.
|
||||
|
||||
## Where we already stand
|
||||
|
||||
Closer than expected. The block protocol is 61 lines, five verbs, and its
|
||||
docblock already reserves `Header.target` for volumes ("targets = volumes,
|
||||
`enumerate` lists them; nothing else about the protocol changes"). usb-storage
|
||||
is content-blind — it reads block 0 only as a bring-up self-check and parses
|
||||
nothing. The vfs protocol is already served by a second provider in production
|
||||
(init's registry serves it synthetically), so a second filesystem is *not* a
|
||||
protocol problem. The FAT service is already internally split: a host-testable
|
||||
engine (1659 + 310 lines) behind a four-function BlockDevice vtable, and a
|
||||
434-line IPC shell. The kernel mount table already has the right restart
|
||||
semantics (remount-replace; lazy dead-endpoint sweep on resolve).
|
||||
|
||||
The misplacements, all small: the MBR walk lives *inside the FAT engine*
|
||||
(`engine.zig:196-212`) — every future engine would duplicate it; fat hardcodes
|
||||
its acquisition policy ("first mass-storage child, mount at /volumes/usb") —
|
||||
that is policy in a filesystem binary; `fs_unmount` is gated by nothing (any
|
||||
process may unmount any prefix — latent now, an obvious cross-tenant hole once
|
||||
mounts multiply); and fat's `mounted` flag is one-way — no path back after its
|
||||
block channel dies.
|
||||
|
||||
## The proposed shape
|
||||
|
||||
Three layers, matching the stated rule, every boundary a channel:
|
||||
|
||||
**Block (driver layer, mechanism).** usb-storage stays hardware→blocks and
|
||||
content-blind. It gains one mechanism, Plan 9's: *named sub-ranges*. A
|
||||
`define_range`-style op creates a logical block object (a partition) clamped
|
||||
and offset onto the disk; sub-ranges are `target` ids on the same endpoint, or
|
||||
separate endpoints handed out per range — either way they speak the **same
|
||||
block contract** (Fuchsia's closed-under-layering rule), so a filesystem cannot
|
||||
tell whole-disk from partition. The driver never parses a table; it clamps
|
||||
ranges it is told about. Offsets translate at definition time, so the data path
|
||||
stays one hop (Fuchsia's session-mapping trick, for free).
|
||||
|
||||
**Volumes (service layer, policy).** One new service — the *volume manager*
|
||||
(fshost/mcd-shaped, init-supervised; the device manager stays devices-only). It
|
||||
subscribes to the device manager's `child_added`/`child_removed`, consumer-
|
||||
hellos for each new mass-storage provider's block channel, reads the partition
|
||||
table and the first blocks itself (the prober is policy), consults
|
||||
configuration — `filesystems.csv`: content signature → filesystem binary;
|
||||
`mounts.csv` or similar: volume identity → mount prefix — defines sub-ranges on
|
||||
the driver, spawns **one filesystem process per volume**, hands it its block
|
||||
channel at startup, and supervises it with the reap-and-rebuild idiom the
|
||||
device manager proved: provider dies or medium leaves → kill the filesystem
|
||||
process, retire its mounts; medium returns → re-probe, respawn, remount. The
|
||||
boot volume is chosen by *content* (which volume carries /system/configuration
|
||||
and /system/logs), closing the two-sticks question honestly.
|
||||
|
||||
**Filesystems (per volume, one process).** The proven unit everywhere from
|
||||
Plan 9's `dossrv` to Minix to Fuchsia: block-client + engine + file-protocol
|
||||
provider in one binary, one process per volume (9front practice; per-volume
|
||||
fault isolation is what our supervision makes cheap). fat's shell became a
|
||||
shared *filesystem harness* library (`library/kernel/file-system-harness`), and
|
||||
the second engine — **exFAT**, `system/services/exfat` — now reuses it wholesale:
|
||||
the reuse this design promised, proven. exfat is nothing but the exFAT engine +
|
||||
a near-clone of fat's thin service, full read + write + directories + rename +
|
||||
on-disk up-case folding, differing only in the format it wraps. A partition walk
|
||||
lives in the volume manager (`partition.zig`), which recognizes fat vs exFAT by
|
||||
VBR and routes each to its engine; write caching stays inside the process (the
|
||||
anti-fsyncgate rule). Each mounts its prefixes into the kernel mount table
|
||||
itself. Two surface limits are shared across both engines and are the vfs
|
||||
layer's, not an exFAT shortcut: file offsets are u32 (a 4 GiB addressable cap),
|
||||
and file names are ASCII bytes (a non-ASCII unit becomes `?`) — teaching the vfs
|
||||
name layer UTF-8 is a separate cross-cutting change.
|
||||
|
||||
**Kernel: two small changes only.** `fs_unmount` gains ownership (only the
|
||||
mounting endpoint's holder may unmount — possession-is-capability, consistent
|
||||
with everything else), and the 8-slot mount table gets a declared bound or
|
||||
growth once volumes multiply. The mount table stays the router; per-process
|
||||
namespaces (Plan 9's extra) remain separable future work.
|
||||
|
||||
## Transport generality: NVMe and SATA against this design
|
||||
|
||||
The layering was chosen transport-agnostic on purpose (the 9front lesson:
|
||||
filesystem services cannot tell a USB stick from an ATA disk); NVMe and AHCI
|
||||
are the test of that claim, and they fit — with three named pressure points.
|
||||
|
||||
**What transfers untouched:** the block protocol (nothing USB in it), the
|
||||
volume manager, partitions/ranges, filesystems, mounts, and the whole
|
||||
enforcement stack — delegation, IOMMU confinement (these are first-party PCI
|
||||
DMA masters, so confinement applies even more directly than USB),
|
||||
reap-and-rebuild, the hot-plug lifecycle. Integration cost per transport is
|
||||
what the architecture promises: one driver binary plus one `devices.csv` row.
|
||||
|
||||
**NVMe** shortens the stack — no bus/class split; the driver IS the
|
||||
controller driver, one hop fewer than USB. Its structural novelty,
|
||||
**namespaces** (hardware-native multiple volumes behind one controller), is
|
||||
exactly the case the block protocol reserved on day one — still reserved, not
|
||||
implemented (pressure point 1 below): decision 4 settles per-volume addressing
|
||||
as per-sender confinement on a single endpoint, and names endpoint-per-volume
|
||||
only as an unbuilt future refactor. **AHCI** is a shape choice, not a problem: an
|
||||
HBA fronts up to 32 ports plus port multipliers — structurally a bus — so
|
||||
either mirror USB (ahci-bus + a per-port disk driver: maximum restart
|
||||
granularity, the proven shape) or mirror NVMe (one driver per controller, one
|
||||
block channel per port). Leaning per-port processes for consistency with the
|
||||
matrix-proven shape; genuinely open.
|
||||
|
||||
**The pressure points, honestly:**
|
||||
|
||||
1. **Multi-volume is built; multi-namespace-per-provider is untried.** The
|
||||
volume manager adopts every device and spawns one range-confined FAT per
|
||||
partition — several volumes across several devices, or several partitions
|
||||
sharing one device's channel, both proven on USB. What is untried is a single
|
||||
provider exposing several volumes as *namespaces* (NVMe): the endpoint and
|
||||
per-badge range machinery generalizes, but no such driver exists yet to
|
||||
exercise it.
|
||||
2. **The current transport will bottleneck NVMe.** Synchronous call/reply,
|
||||
one operation in flight, one bounce buffer — fine for a USB2 stick,
|
||||
forfeits an NVMe drive's queue depth and per-queue MSI-X. Correctness
|
||||
needs nothing; performance is gated on the shm-ring data plane
|
||||
communication.md already names (Fuchsia's FIFO+VMO is the precedent).
|
||||
NVMe is not worth building before that milestone.
|
||||
3. **Media lifecycle ≠ device lifecycle.** ATAPI trays and SD card readers
|
||||
(USB ones too, today) keep the DEVICE present while the MEDIUM comes and
|
||||
goes — a removal trigger our channel-death lifecycle does not carry. The
|
||||
fix is decision 7 below. (Related small item: fat's bounce sizing
|
||||
hardcodes 512-byte sectors; 4Kn drives need the geometry honored
|
||||
throughout.)
|
||||
|
||||
## The decisions on the table
|
||||
|
||||
1. **Partitions as driver-side sub-ranges** (mechanism in usb-storage, parsing
|
||||
in the volume manager) — versus a separate partition *process* re-serving
|
||||
block (Fuchsia's storage-host). Sub-ranges are ~20 lines of clamp in the
|
||||
driver and keep the data path one hop; a separate process is purer layering
|
||||
at the cost of a hop or session plumbing. Recommendation: sub-ranges.
|
||||
2. **The volume manager as a new service** owning probe, spawn, supervision,
|
||||
and mount policy — with `filesystems.csv` and the mount map as
|
||||
configuration. Recommendation: yes; it is the missing policy home that fat
|
||||
is currently squatting in.
|
||||
3. **One filesystem process per volume** (fat's binary becomes "the FAT
|
||||
implementation", spawned per FAT volume). Recommendation: yes — it extends
|
||||
recompile-and-restart-live to filesystems and isolates corrupt media.
|
||||
4. **Sub-range addressing — SETTLED as per-sender confinement at the
|
||||
provider, one serving endpoint.** The deciding argument is precedent: the
|
||||
xHCI bus already serves every class driver on one endpoint with authority
|
||||
scoped by the kernel-stamped badge (the per-client device-token table) —
|
||||
that IS danos's provider pattern, and per-badge range confinement is the
|
||||
same pattern applied to blocks. The volume manager sets each filesystem
|
||||
process's range on the driver; the driver clamps AND translates every
|
||||
transfer by the sender's range, so filesystems address volume-relative
|
||||
LBAs from 0. On the confined path the FAT engine's `base_lba` therefore
|
||||
resolves to 0 on every access; the field and the engine's own MBR walk
|
||||
remain in `engine.zig` as now-inert legacy code (the authoritative partition
|
||||
walk lives in the volume manager's `partition.zig`), not yet deleted.
|
||||
The enforcement point (the clamp at the provider, never in the consumer)
|
||||
is what carries the security property; endpoint-per-volume would deliver
|
||||
the same property only by inventing a multi-endpoint harness the pattern
|
||||
does not need. It stays available as a future refactor if a multi-endpoint
|
||||
harness ever exists for other reasons; the wire contract is identical
|
||||
either way.
|
||||
5. **`fs_unmount` ownership** — a defect fix more than a decision.
|
||||
6. **Later, kept open**: the shm-ring data plane (communication.md already
|
||||
names it as the 256-byte ceiling's unlock — Fuchsia's FIFO+VMO is the
|
||||
precedent); format-level crash honesty (a Power-Safe-style journaling or COW
|
||||
filesystem) once danos outgrows FAT; per-process namespaces.
|
||||
7. **The media-presence event** (the consuming half is BUILT; the
|
||||
transport-native signal stays future): the block protocol carries a pushed
|
||||
event — `medium_changed`, with present/absent and a change counter —
|
||||
produced today by the storage driver from a TEST UNIT READY poll (the
|
||||
transport's native signal — SCSI UNIT ATTENTION, PxSSTS for AHCI,
|
||||
namespace-change AER for NVMe — is the future refinement in place of the
|
||||
poll) and now **consumed** by the volume manager, which subscribes per
|
||||
device and runs the SAME kill-retire-remount path it runs on channel death.
|
||||
The driver reports presence, never content; a pushed event carries no
|
||||
capability, which the kernel already guarantees. The device staying while
|
||||
its medium leaves is the one removable-media case the channel-death trigger
|
||||
cannot see; without this event a swapped SD card would be served with the
|
||||
old card's filesystem state. A THIRD trigger closes the last gap — a
|
||||
storage driver that *crashes* while its device stays present: channel death
|
||||
there is invisible to presence polling, so the volume manager probes channel
|
||||
liveness (`geometry()`) each tick and reaps-then-rebuilds the volume on the
|
||||
restarted driver's fresh channel. One lifecycle, three triggers.
|
||||
8. **Volume identity, and the mount map as danos's fstab** (settled). The
|
||||
lesson is Linux's own history: fstab keyed on `/dev/sda1` for years and
|
||||
broke whenever a drive changed ports or enumeration order; `UUID=` entries
|
||||
exist because device-path identity failed. danos skips that era: the mount
|
||||
map (`volumes.csv` — configuration, read by the volume manager) keys on
|
||||
**content identity, never port or discovery order**. Build status: rungs 1,
|
||||
3, and 4 (GPT partition GUID, FAT serial + label, MBR signature + index) are
|
||||
implemented (S1); rung 2 waits on a non-FAT engine. The `volumes.csv` map and
|
||||
the id-derived mount path are built (S2): a volume's mount path IS its content
|
||||
id (`/volumes/<id>`, e.g. `/volumes/fat-12345678`), or a `volumes.csv`
|
||||
override; the label is display metadata a `volumes` query returns, never the
|
||||
path. The ladder the prober reads off the medium, strongest first:
|
||||
1. GPT partition GUID — 128-bit, unique, stable for the volume's life — **built (S1)**;
|
||||
2. filesystem UUID (ext-family and most modern formats, in the superblock) *(planned)*;
|
||||
3. FAT volume serial + label — 32 bits, weak (dd-cloned sticks share it)
|
||||
but what real sticks carry — **built (S1)**;
|
||||
4. MBR disk signature + partition index — **built**; a bare FAT with no
|
||||
table takes index 0 over the whole device;
|
||||
5. nothing — an anonymous volume: generated mount name, no persistence *(planned)*.
|
||||
|
||||
Consequences, each mechanical once identity keys the map: **moving a drive
|
||||
to a different port changes nothing** — same identity, same mount point,
|
||||
whether USB port, hub depth, SATA port, or a stick that left as USB and
|
||||
returned in a SATA dock; **replug remounts at the same path** (the id-path is
|
||||
content-derived, so a volume returns to `/volumes/<id>` wherever it reappears;
|
||||
the re-adopt+remount code path is QEMU-proven by the driver-crash rebuild, but
|
||||
remount on a *physical* replug end-to-end is bench-pending, not QEMU-testable,
|
||||
because QEMU can't re-present the boot-controller device); **the boot volume** is the volume
|
||||
that resolves `/system/configuration` on its own media, findable on any port or
|
||||
partition; and **duplicate identity is a known S4 gap** — two cloned sticks
|
||||
share one content id, so today they collide on `/volumes/<id>` (the kernel
|
||||
remount-replaces; the last wins) and each boot-volume claim is logged loudly.
|
||||
Distinguishing them with a suffix is arbitration, deferred to S4. Unknown identities
|
||||
mount under a derived name (sanitized label, else generated) at
|
||||
`/volumes/<name>` — the hierarchy's documented home for attached media,
|
||||
which stands: `/system` is what danos IS; attached media is what it isn't.
|
||||
danos never needs to MINT identifiers to detect volumes — detection only
|
||||
reads — until it grows formatting, which brings the entropy question and
|
||||
is deliberately out of scope here.
|
||||
@@ -0,0 +1,54 @@
|
||||
# The hot-plug matrix: unplug anything, replug anywhere
|
||||
|
||||
*Status: COMPLETE, 2026-08-09, same day — H1..H5 all green, full suite 124/124.
|
||||
The predicted cascade bug was real and found by H2: `tearDownPort` did not recurse
|
||||
into a hub yanked from a root port, so its subtree stayed live against vanished
|
||||
hardware and the replugged hub found its port occupied; fixed by mirroring
|
||||
`tearDownHubDevice`'s children-first recursion (77e7001). Every other cell passed
|
||||
against the establishment-track machinery unchanged.*
|
||||
|
||||
*2026-08-09. The requirement, verbatim: "i want to be able to unplug devices and
|
||||
plug them back in, in any order and on any hub." The machinery exists end to end —
|
||||
per-device driver processes, cascading teardown (`tearDownHubDevice` recurses a
|
||||
departing hub's subtree, children first), reap-on-removal in the manager, idempotent
|
||||
per-port registration identity, lineage rebinding — but only one cell of the matrix
|
||||
is verified: a leaf keyboard behind a hub, unplugged and replugged on the same port
|
||||
(`usb-hub-unplug`). This plan verifies the rest and fixes what verification flushes
|
||||
out. The cascade path has never executed; expect it to carry at least one bug.*
|
||||
|
||||
**Out of scope, stated up front:** unplugging the BOOT stick (the drivers handle it;
|
||||
fat holds a dead block channel until M21 remount gives it re-acquisition), and
|
||||
real-hardware root-port timing (the Ryzen bench is the acceptance run for that, as
|
||||
always — QEMU proves the logic, not the silicon).
|
||||
|
||||
## The discrimination lever
|
||||
|
||||
Every case's rebind tail (— reaping → delegated → second-generation `ok`) is
|
||||
impossible without the manager's reap: stashing the reap out of
|
||||
`onChildRemoved`/`pruneChildrenOf` makes every matrix case fail at the dedupe wall.
|
||||
That is the standing discrimination check for the whole matrix — run it once per
|
||||
case shape, not per commit.
|
||||
|
||||
## The matrix
|
||||
|
||||
| Case | Topology | Event sequence | What it proves |
|
||||
|---|---|---|---|
|
||||
| H1 `usb-root-replug` | keyboard on a 2nd controller's ROOT port | del @8s, add @14s (same port) | the `tearDownPort` path + reap + rebind; root-port changes arrive via the 250 ms reconcile tick (QEMU raises no root-port events) |
|
||||
| H2 `usb-hub-yank` | hub on 2nd controller; keyboard AND mouse behind it | del the HUB @8s; re-add hub @14s, kbd @17s, mouse @18s | the recursive cascade: one event removes the subtree, EVERY bound driver is reaped, and the rebuilt hub re-binds both |
|
||||
| H3 `usb-hub-nested-yank` | hub → hub → keyboard | del the OUTER hub @8s; re-add all three @14–18s | the cascade recursion depth ≥ 2, and a nested rebuild |
|
||||
| H4 `usb-replug-moved` | keyboard behind hub port 1.1 | del @8s; add on port 1.2 @14s | a replug on a DIFFERENT port is simply a new device: new port identity, new id, fresh match — no stale state ties a driver to the old port |
|
||||
| H5 `usb-replug-cycles` | keyboard behind hub | del/add three times (6 hooks) | no state leaks across generations: slots, the bus's open table, driver entries (reaped entries must be reusable) |
|
||||
|
||||
Ordered expects follow the `usb-hub-unplug` shape: the boot devices' `ok` lines all
|
||||
precede the first unplug, so a tail of `removed → reaping → delegated → ok` can only
|
||||
be satisfied by the generation the sequence created. H2/H3 require one `reaping` per
|
||||
bound child; H5 requires three.
|
||||
|
||||
## Order and discipline
|
||||
|
||||
H1 → H2 → H3 → H4 → H5, one commit per case (plus its fix if it finds a bug — the
|
||||
case and the fix land together, the case having failed first). The standing
|
||||
discrimination run once for H1 and once for H2's shape. Full suite green at the end,
|
||||
docs touched where behavior was corrected, memory updated. Fixes stay within the
|
||||
settled design: teardown order (children first), reap-on-removal, identity per port
|
||||
— anything design-shaped that surfaces stops the loop and asks.
|
||||
@@ -89,14 +89,19 @@ comptime {
|
||||
}
|
||||
```
|
||||
|
||||
`maximum_domains = 64` and `maximum_devices = 64` agree today only by a sentence in a
|
||||
comment, and the agreement fails open. This is the clause with a live hole behind it,
|
||||
and the reason raising `maximum_devices` alone would be a privilege escalation rather
|
||||
than a fix.
|
||||
`maximum_domains = 64` and `maximum_devices = 64` once agreed only by a sentence in a
|
||||
comment, and that agreement failed open — the clause with the live hole behind it, where
|
||||
raising `maximum_devices` alone left every device id past the end of `iommu.confined`
|
||||
unconfined while `confineDevice` still reported success, a privilege escalation rather
|
||||
than a fix. The assert closed that: it held the two together while both stayed fixed, and
|
||||
when the device table was later made dynamic — no `maximum_devices` any more, only a
|
||||
per-registrar quota — that forced them apart, the assert having done its job. `confined`
|
||||
now grows to cover every id the broker mints, and `confineDevice` refuses when it cannot
|
||||
record a confinement rather than failing open.
|
||||
|
||||
## The worked bad case
|
||||
|
||||
`devices_broker.maximum_devices`, which had no comment at all:
|
||||
`devices_broker.maximum_devices`, which had no comment at all, before it was made dynamic:
|
||||
|
||||
```zig
|
||||
/// bound: device nodes for the whole machine — firmware-discovered plus registered
|
||||
|
||||
@@ -118,17 +118,22 @@ regions `init` frees. Keeping that boot-protocol knowledge on the loader side is
|
||||
deliberate — the kernel has no notion of "reclaimable" or of UEFI at all.
|
||||
|
||||
The one live piece in that memory is the boot stack the kernel starts on; the loader
|
||||
leaves the single region containing it `reserved`, so `init` won't hand it out. A
|
||||
later step will move task 0 onto a kernel-owned stack, freeing that last ~1 MiB
|
||||
region too (and giving user mode the clean stack it wants).
|
||||
leaves the single region containing it `reserved`, so `init` won't hand it out. The
|
||||
kernel is only on it for an instant, though — `_start`'s first instruction switches
|
||||
to a kernel-owned 64 KiB stack in `.bss` (that context becomes task 0). The region
|
||||
stays `reserved` because the loader's own `convertMemoryMap` was executing on that
|
||||
stack when it reclassified the RAM, and, like the map buffers below, nothing frees it
|
||||
yet.
|
||||
|
||||
## What's next (partly done since)
|
||||
|
||||
- **Contiguous allocation** — done: `allocContiguous` scans for a run of clear
|
||||
bits, with an optional physical ceiling for DMA (`dma_alloc` is its user), and
|
||||
`allocBelow` serves the SMP trampoline.
|
||||
- **A kernel stack for task 0** — still open: the boot processor's idle task runs
|
||||
on the boot stack to this day, so that region can't be freed.
|
||||
- **A kernel stack for task 0** — done: `_start`'s first instruction switches `rsp`
|
||||
to a kernel-owned 64 KiB stack in `.bss` (`bootstrap_stack`), and `scheduler.init`
|
||||
registers that running context as task 0 — the kernel is on the loader's boot
|
||||
stack for that one instruction and never again.
|
||||
- **Freeing the `reserved` `loader_data`** (the boot-time map buffers) — still
|
||||
open: the bitmap deliberately tracks those frames so they *can* be freed, but
|
||||
nothing frees them yet.
|
||||
|
||||
@@ -14,7 +14,7 @@ are mirrored to it explicitly (`system/kernel/kernel.zig`).
|
||||
## The pipeline
|
||||
|
||||
```
|
||||
process std.log ──▶ debug_write(level) ──▶ tagged kernel ring ──▶ logger service ──▶ /var/log/<boot-stamp>/<binary-path>.log
|
||||
process std.log ──▶ debug_write(level) ──▶ tagged kernel ring ──▶ logger service ──▶ /system/logs/<boot-stamp>/<binary-path>.log
|
||||
kernel log.print ─┘ │
|
||||
└▶ serial / 0xE9 sinks (QEMU, -Dserial)
|
||||
```
|
||||
@@ -49,12 +49,12 @@ kernel log.print ─┘ │
|
||||
|
||||
5. **Persist.** The **logger service** (`system/services/logger`) drains the
|
||||
ring every 250 ms and demultiplexes records into one file per source under
|
||||
`/var/log/<boot-stamp>/`, e.g.
|
||||
`/system/logs/<boot-stamp>/`, e.g.
|
||||
|
||||
```
|
||||
/var/log/2026-07-21T150434Z/kernel.log
|
||||
/var/log/2026-07-21T150434Z/system/services/fat.log
|
||||
/var/log/2026-07-21T150434Z/system/drivers/usb-storage.log
|
||||
/system/logs/2026-07-21T150434Z/kernel.log
|
||||
/system/logs/2026-07-21T150434Z/system/services/fat.log
|
||||
/system/logs/2026-07-21T150434Z/system/drivers/usb-storage.log
|
||||
```
|
||||
|
||||
The boot stamp is the RTC anchor from `klog_status` (FAT-safe: no colons; a
|
||||
@@ -95,6 +95,6 @@ written last so a reader only trusts a complete record).
|
||||
- A write-spamming process can evict other processes' unread records from the
|
||||
ring (a per-process quota is future work); the loss is at least visible via
|
||||
sequence gaps in every affected file.
|
||||
- `/var/log` files have no privacy until the VFS grows permissions.
|
||||
- `/system/logs` files have no privacy until the VFS grows permissions.
|
||||
- Records emitted after the logger's final shutdown drain reach serial and the
|
||||
ring but not the files.
|
||||
|
||||
@@ -14,7 +14,7 @@ process-manager server, and Fuchsia/seL4 control processes only through handles.
|
||||
|
||||
danos rules out `/proc` **as the primitive**: the path router lives in the
|
||||
kernel (`fs_resolve`), but what is mounted under a path is served by a
|
||||
user-process filesystem server (the way FAT serves `/mnt/usb`) — a `/proc`
|
||||
user-process filesystem server (the way FAT serves `/volumes/usb`) — a `/proc`
|
||||
would be one more such server, which would put a user process in the path of
|
||||
process control. If that server (or anything under it) hangs, nothing could be
|
||||
listed or killed, *including the hung server*. The control plane for processes
|
||||
@@ -117,9 +117,11 @@ the architecture layer calls up into `tick`.
|
||||
`process_exit_reason` (`process.exitReason`). This is the input to
|
||||
restart policy ([process-lifecycle.md](process-lifecycle.md)); an exit *code*
|
||||
for the clean case can still ride alongside later.
|
||||
- Enumerate writes through the caller's raw pointer under the bring-up trust
|
||||
model, like `device_enumerate` (an unmapped page is a self-DoS, not an
|
||||
isolation break).
|
||||
- ~~Enumerate writes through the caller's raw pointer under the bring-up trust
|
||||
model, like `device_enumerate`~~ Closed (8d4a7cf): both `process_enumerate`
|
||||
and `device_enumerate` describe a chunk into a kernel buffer and place it with
|
||||
`copyToUser`, which validates the range and resolves each page — an unmapped
|
||||
page returns `-EFAULT`, and the kernel never stores through the user pointer.
|
||||
|
||||
## Tests
|
||||
|
||||
|
||||
@@ -1,9 +1,9 @@
|
||||
# The protocol namespace
|
||||
|
||||
*Design, agreed 2026-07-31. Supersedes the `ServiceId` registry. P1–P3 of the
|
||||
*Design, agreed 2026-07-31. Supersedes the `ServiceId` registry. P1–P4 of the
|
||||
migration plan at the end have landed (the envelope, the registry and the
|
||||
`ServiceId` flag-day, and restriction stage one); P4 and P5 are the remaining
|
||||
work list.*
|
||||
`ServiceId` flag-day, restriction stage one, and the protocol rebase); P5
|
||||
(restriction stage two) is the remaining work.*
|
||||
|
||||
How a program finds, connects to, and is restricted from the things it talks to.
|
||||
Three ideas, kept deliberately separate:
|
||||
|
||||
@@ -23,8 +23,8 @@ INIT–SIPI–SIPI, brings each up into 64-bit long mode with its own descriptor
|
||||
LAPIC and timer, and drops it into the scheduler. Tasks run **genuinely in parallel** —
|
||||
the `smp` self-test confirms worker tasks executing on all four cores at once under
|
||||
QEMU `-smp 4`. Shared kernel state (scheduler queues, IPC) is serialised behind a big
|
||||
kernel lock. What's left is refinement, not first-light: per-core run queues, IPIs,
|
||||
and thread-to-core affinity (see [Implementation status](#implementation-status)).
|
||||
kernel lock. What's left is refinement, not first-light: per-core run queues and IPIs
|
||||
(see [Implementation status](#implementation-status)).
|
||||
|
||||
## The common microkernel instinct: don't share kernel state
|
||||
|
||||
@@ -238,10 +238,12 @@ next lands.
|
||||
- **Per-core run queues** — the Fiasco.OC direction, if the single global queue's lock
|
||||
contention ever bites. (Thread *affinity* already exists — see above; this is the
|
||||
further step of giving each core its own primary run queue for load distribution.)
|
||||
- **Fault recovery** — today a fault halts (only) the faulting core. Turning that into
|
||||
"kill the task, keep the core running" is the [resilience](resilience.md) track (it
|
||||
needs the task's lock/resource state handled), and for taking a core fully offline,
|
||||
its tasks migrated first.
|
||||
- **Fault recovery** — a ring-3 fault already **kills the faulting process and keeps the
|
||||
core (and the rest of the system) running**: the kernel trapped it on the task's own
|
||||
kernel stack, reclaims what the process held, and reschedules
|
||||
(`process.killCurrentProcess`; the `fault-recovery` test proves init keeps heartbeating
|
||||
through the kill) — the [resilience](resilience.md) track. What's still open here is
|
||||
taking a core fully **offline**, which additionally needs its tasks migrated off first.
|
||||
|
||||
## Further reading
|
||||
|
||||
|
||||
@@ -84,7 +84,7 @@ address space. Threads deliberately remove that boundary *within* a process:
|
||||
there is no isolation **between** threads.
|
||||
- Threads share fate — by contract: a fault in any thread, or a "kill the process"
|
||||
decision, takes down **all** of them, so restartability lives at the process level,
|
||||
not the thread level. (The kernel does not yet enforce this fan-out — see the
|
||||
not the thread level. (The kernel enforces this fan-out — see the
|
||||
Lifecycle note under
|
||||
[Interaction with the rest of the kernel](#interaction-with-the-rest-of-the-kernel).)
|
||||
- Shared mutable state reintroduces data races — the failure class the
|
||||
|
||||
@@ -36,9 +36,10 @@ and inside a VM alike; only the source behind it differs. The mechanism is in
|
||||
|
||||
So the timer hardware lives in the kernel, and there is **no `hpet` driver and no time
|
||||
server** to consume. (An earlier HPET driver existed only to *demonstrate* the driver
|
||||
model; that role now lives in [drivers.md](../device-driver-development/drivers.md), as documentation.) The one place
|
||||
a user-space time service *is* justified — **wall-clock / calendar time** — is discussed
|
||||
at the end; it is deliberately not built yet.
|
||||
model; that role now lives in [drivers.md](../device-driver-development/drivers.md), as documentation.) The one part
|
||||
left to user space — **calendar policy** over wall-clock time (time zones, formatting) —
|
||||
is discussed at the end; the wall-clock *seconds* it builds on are a kernel syscall
|
||||
(`wall_clock`), like the monotonic clock.
|
||||
|
||||
## The three system calls
|
||||
|
||||
@@ -95,15 +96,17 @@ The raw wrappers (`clock`, `sleepMillis`, `timerOnce`) and the ergonomic
|
||||
`Instant`/`Duration` layer both live in the `time` module
|
||||
(`library/kernel/time.zig`); the latter is what everyday code uses.
|
||||
|
||||
## Wall-clock time (not built)
|
||||
## Wall-clock time
|
||||
|
||||
Everything above is **monotonic**: elapsed time since boot, perfect for timeouts and
|
||||
measurement, useless for "what is the date?" Calendar time — a real-time clock, time
|
||||
zones, leap seconds — is genuinely a **user-space** concern, and it *is* the case a time
|
||||
service is for. It would be backed by an **RTC** driver (the CMOS real-time clock), not
|
||||
the HPET, and exposed as a `CLOCK_REALTIME`-style service alongside the monotonic
|
||||
syscall. It is deferred until something needs it; the monotonic clock the kernel already
|
||||
owns covers every current use.
|
||||
measurement, useless for "what is the date?" Calendar time needs a **real-time clock**.
|
||||
The kernel owns wall-clock *seconds* as mechanism, exactly like the monotonic clock: the
|
||||
`wall_clock` syscall (#33) returns Unix epoch seconds (UTC). The CMOS **RTC** is read
|
||||
once at boot and anchored to the monotonic clock (`system/kernel/wall-clock.zig`), so a
|
||||
query is a cheap arithmetic offset rather than a per-call CMOS poll; `time`'s
|
||||
`wallClock()` (`library/kernel/time.zig`) wraps it. Reading the hardware's value is not
|
||||
policy — time zones, leap seconds, calendars, and formatting layer on top in user space.
|
||||
It exists because the filesystem needs real timestamps (mtime).
|
||||
|
||||
## Verifying it
|
||||
|
||||
|
||||
@@ -123,12 +123,15 @@ The calls that return two values in `rax:rdx` today — `dma_alloc`
|
||||
`fs_resolve` (route tag + node token / backend handle) —
|
||||
become functions returning a two-`u64` struct. The System V ABI returns a
|
||||
16-byte struct in `rax:rdx`, so the stub is a plain `syscall; ret` — the
|
||||
C-ABI spelling of the existing convention, at zero cost. The one call that
|
||||
returns *three* values — `ipc_reply_wait` (receive_len in `rax`, badge in
|
||||
`rdx`, received capability in `r8`) — exceeds the two-register return: its
|
||||
function returns a three-`u64` struct, which the ABI passes via a hidden
|
||||
result pointer, so that one stub stores `rax`/`rdx`/`r8` through the pointer
|
||||
after the `syscall` — a few instructions rather than one.
|
||||
C-ABI spelling of the existing convention, at zero cost. The calls that
|
||||
return *three* values — `ipc_reply_wait` (receive_len in `rax`, badge in
|
||||
`rdx`, received capability in `r8`) and `dma_alloc` when the region is
|
||||
`shareable` (virtual in `rax`, physical in `rdx`, handle in `r8`) — exceed the
|
||||
two-register return: their functions return a three-`u64` struct, which the
|
||||
ABI passes via a hidden result pointer, so each stub stores `rax`/`rdx`/`r8`
|
||||
through the pointer after the `syscall` — a few instructions rather than one.
|
||||
(`ipc_call` likewise carries a received capability in `r8` alongside its `rax`
|
||||
result.)
|
||||
|
||||
Grouped as `abi.zig` groups them:
|
||||
|
||||
|
||||
@@ -143,8 +143,8 @@ Python shell uses it.
|
||||
The kernel/VFS cluster a shell forces (any shell, any language):
|
||||
|
||||
- **exec-of-path** — spawn an arbitrary VFS path, not a named ramdisk binary;
|
||||
- **argv/envp** — carried through spawn onto the child's entry stack (env from
|
||||
P1, argv new);
|
||||
- **argv/envp** — carried through spawn onto the child's entry stack (argv
|
||||
already built and tested via `spawnWithArguments`; envp new, from P1);
|
||||
- **numeric exit status** — extend the exit record beyond the categorical
|
||||
`ExitReason` (the gotcha the Zig roadmap flagged: `WEXITSTATUS` must be real);
|
||||
- **fd inheritance + pipes** — a kernel or service pipe (a character device by
|
||||
|
||||
@@ -0,0 +1,368 @@
|
||||
# Finishing the storage stack: the S1–S5 plan
|
||||
|
||||
*2026-08-09. Continues the volume-manager track (V0–V5, on main) from "one FAT
|
||||
volume" to "any filesystem, any number of volumes, identified by content,
|
||||
remounting where they belong, surviving a driver crash." Executes the settled
|
||||
design in
|
||||
[storage-architecture.md](file-system-development/storage-architecture.md) and
|
||||
[storage-design-rationale.md](file-system-development/storage-design-rationale.md).
|
||||
Track discipline as always: work on main; one commit per coherent step with
|
||||
`git commit -F` (no `-m`, no co-author trailer); every new test shown to FAIL
|
||||
against the old behavior; one QEMU suite at a time; CSV is configuration read by
|
||||
the volume manager (the policy), never itself policy; bounds discipline
|
||||
(`tools/check-bounds.py` gate); adversarial boundary review at each phase.*
|
||||
|
||||
## Where V0–V4 left it
|
||||
|
||||
The volume manager probes ONE storage device, parses its MBR (rung-4 identity =
|
||||
`(diskSignature<<8)|index`), spawns ONE FAT service confined to that partition's
|
||||
badge-scoped block range, supervises it, and unmounts it when the device is
|
||||
pulled. `partition.firstVolume` returns the FIRST partition; `openAnyStorage`
|
||||
adopts the FIRST device; `var volume: ?Volume` and `volume_id = 1` are singular;
|
||||
`filesystem_binary` and fat's mount prefixes are hardcoded; `medium_changed` is
|
||||
published by the driver but consumed by no one; remount-on-replug is
|
||||
bench-pending.
|
||||
|
||||
## The five phases and how they depend
|
||||
|
||||
```
|
||||
S1 identity ladder ──► S2 mount map ──► S3 multi-volume ──► S4 exFAT
|
||||
(needs S2+S3)
|
||||
S5 removal robustness ── independent; single-volume ── may land any time
|
||||
```
|
||||
|
||||
- **S1** grows the identity read off the medium (GPT GUID, FAT serial+label). It
|
||||
comes FIRST because a volume's mount name is now its identity (below), and the
|
||||
friendly form of that name is the FAT label / GPT name S1 parses.
|
||||
- **S2** moves the last policy out of hardcode into `volumes.csv` +
|
||||
`filesystems.csv`, and names each volume by its identity **id** (a GUID/key),
|
||||
keeping the label as separate, queryable display metadata — no port-name.
|
||||
- **S3** generalizes to N volumes across N devices.
|
||||
- **S4** adds exFAT — a COMPLETE second engine that proves the V1 harness
|
||||
extraction. Needs S2 (to route by signature) and S3 (to run a second volume).
|
||||
- **S5** closes the removal-lifecycle gaps. Independent of the rest; single-volume.
|
||||
|
||||
Recommended build order is S1 → S2 → S3 → S4 → S5. S5 may be pulled earlier.
|
||||
|
||||
---
|
||||
|
||||
## S1 — the identity ladder
|
||||
|
||||
**Goal.** Grow `system/services/volume-manager/partition.zig` from the single
|
||||
rung-4 identity into a ladder that reads the richest available content identity:
|
||||
GPT partition GUID (rung 1, 128-bit), FAT volume serial + label (rung 3), MBR
|
||||
signature + index (rung 4, kept), bare-FAT (kept, enriched to its serial). The
|
||||
`u64` identity becomes a small tagged struct `Identity{ rung, key: u128, label,
|
||||
has_label }` — `key` is the **id** (the path handle), `label` is the **display
|
||||
name** (the GPT 36-char partition name, or the FAT volume label), a separate
|
||||
field per the id/name split S2 relies on. Because GPT metadata is at LBA 1 and
|
||||
the entry array beyond it, and
|
||||
the FAT serial is in each partition's VBR, `firstVolume` stops taking one
|
||||
preloaded block-0 slice and takes a `SectorReader` (context + read-one-sector
|
||||
fn, mirroring the engine's `BlockDevice` vtable) — host-testable against a
|
||||
RAM-disk reader exactly as the four existing `partition.zig` tests are.
|
||||
|
||||
**Key touchpoints.** `partition.zig` (the `Rung`/`Identity`/`SectorReader` types;
|
||||
`gptFirstVolume`; `fatIdentity`; `firstVolume` control flow — GPT authoritative,
|
||||
else MBR walk skipping type-0xEE, else bare-FAT, each preferring the FAT serial
|
||||
over the disk signature); `volume-manager.zig` (`Volume.identity` type; a
|
||||
`ProbeReader` over the existing 512-byte bounce; the probe log prints
|
||||
`identity.key`); `build.zig` (add `b.dependency("volume-manager", .{})` to the
|
||||
package-test aggregation loop so the host fixtures run under root `zig build
|
||||
test`). Declared bound `gpt_entry_scan_maximum = 128` with the full bounds block;
|
||||
`sector_bytes`/`fat_label_bytes` named consts.
|
||||
|
||||
**Steps (commits).** (1) The SectorReader/Identity flag-day — pure refactor, no
|
||||
new behavior, all existing tests green. (2) GPT parsing (rung 1) — protective-MBR
|
||||
+ `EFI PART` signature + header CRC-32 + per-entry overflow-safe range
|
||||
validation (the confinement-safety invariant the driver's clamp rests on,
|
||||
extended to GPT). (3) FAT serial + label (rung 3), preferred over the disk
|
||||
signature; `Identity.eql`. (4) Test wiring, `check-bounds.py`, full suite, docs,
|
||||
memory.
|
||||
|
||||
**Discrimination.** Host: a GPT disk yields `rung==.gpt_guid` + the exact GUID
|
||||
key (old code walks the 0xEE protective entry as an ordinary partition); a GPT
|
||||
entry past the device is skipped, an out-of-device-only GPT returns null (the
|
||||
security-boundary guard); an invalid GPT header is not a volume; a bare FAT
|
||||
reports its real serial `0x12345678` not the rung-4 pseudo-signature; an
|
||||
MBR+FAT partition prefers the serial over the disk signature. On-image: the
|
||||
`volume-probe` QEMU regex tightens to `volume 0x0*12345678` — the boot image's
|
||||
real FAT32 serial reaches the running log.
|
||||
|
||||
**Top risks.** Adversarial GPT input from untrusted media (huge entry counts,
|
||||
bogus offsets, overflowing ranges) — mitigated by header CRC + size bounds +
|
||||
`gpt_entry_scan_maximum` + per-entry overflow-safe validation under boundary
|
||||
review. The `std.hash.crc` symbol in Zig 0.16 is unverified (fallback: a ~15-line
|
||||
reflected CRC-32, poly `0xEDB88320`, used by both parser and fixtures so they
|
||||
never drift onto magic constants).
|
||||
|
||||
---
|
||||
|
||||
## S2 — the mount map: volumes.csv + filesystems.csv
|
||||
|
||||
**Goal.** Move the last two pieces of storage policy out of hardcode into
|
||||
configuration read by the volume manager, and split a volume's **id** from its
|
||||
**label** (the database model: the id is the real, stable, unique key software
|
||||
uses; the label is a mutable display name). `filesystems.csv` (content signature
|
||||
→ filesystem binary) so the VM picks the binary from the probed signature;
|
||||
`volumes.csv` (id → mount prefix, danos's fstab) as the explicit override for a
|
||||
volume the user wants at a fixed path.
|
||||
|
||||
**The mount path IS the identity id, never a label or a port/role name.** A
|
||||
volume mounts at `/volumes/<id>` — the GPT partition GUID for a GPT volume, a
|
||||
`fat-<serial>` / `mbr-<sig>-<index>` form otherwise (exact rendering is an impl
|
||||
detail; the point is a stable, unique, content-derived string). Because the path
|
||||
is the id and never the label, two distinct volumes that happen to share a label
|
||||
(`Backup`, `UNTITLED`, unlabeled) get distinct paths automatically and never
|
||||
collide; only genuinely identical *ids* (dd-cloned media) hit the rationale's
|
||||
duplicate-identity policy (first mounts, second suffixed and logged).
|
||||
|
||||
**The label is display metadata, exposed by a protocol query, not the path.** S1
|
||||
reads it off the medium (FAT volume label, GPT partition name); a volume-manager
|
||||
`volumes` verb returns `{ id, mount_path, label }` per volume so a future
|
||||
shell/UI can show the friendly name — the id↔name split, like a table's primary
|
||||
key vs its display column. (`volumes.csv` may optionally carry a chosen label
|
||||
override alongside the path override, both keyed on id.) There is no
|
||||
`/volumes/usb` and no `/volumes/boot`; the boot volume is detected by content (it
|
||||
installs the `/system/configuration` + `/system/logs` rewrites) but is named by
|
||||
its id like any other.
|
||||
|
||||
Parsed with `library/csv` exactly as `device-registry` parses `devices.csv`. The
|
||||
VM hands the binary + volume-id + mount specs to fat at spawn (the argv channel
|
||||
V3b already uses for the volume id); fat retires `fat_mounts` and reads mounts
|
||||
from argv[2..].
|
||||
|
||||
**Key touchpoints.** New pure-logic modules `filesystem-map.zig` (parse +
|
||||
`match(signature)`) and `volume-map.zig` (parse + `mountsFor(identity)` +
|
||||
`derivedAnonymous`), mirroring `device-registry`, host-tested; `partition.zig`
|
||||
`Volume` gains a `signature`; `volume-manager.zig` loads both tables in
|
||||
`initialise`, resolves binary + mounts, spawns the chosen binary with the mount
|
||||
argv; `fat.zig` deletes `fat_mounts`, parses argv[2..] into a bounded
|
||||
`MountSpec` array; new `system/configuration/filesystems.csv` +
|
||||
`volumes.csv`; `build.zig` bundles them; `make-fat-image.py` writes a real 4-byte
|
||||
MBR disk signature so the boot volume's identity is a legible non-zero key.
|
||||
|
||||
**Steps (commits).** (1) partition emits a signature. (2) `filesystem-map`
|
||||
parser + host tests. (3) `volume-map` parser (id→prefix override) + the id-path
|
||||
deriver + the `volumes` label-query verb + host tests. (4) Ship the tables +
|
||||
VM integration **behind fat's still-hardcoded mounts** (behavior-preserving —
|
||||
parsers proven before consumption flips; full suite green). (5) fat consumes
|
||||
argv mounts; the VM mounts the boot volume at its id-path (`/volumes/fat-12345678`,
|
||||
from its serial); **migrate the fixtures + QEMU regexes off `/volumes/usb`**
|
||||
(fat-test, badge-scope-test, vfs-test, and the four cases) in the same commit;
|
||||
the `volume-identity-name` QEMU case (shown failing against HEAD~1); flip the
|
||||
docs' pending markers.
|
||||
|
||||
**Discrimination.** QEMU `volume-identity-name`: the boot volume mounts at its
|
||||
id-path (`/volumes/fat-12345678`, from its serial), a line the old hardcoded
|
||||
`/volumes/usb` never emits; the `volumes` query returns that id paired with the
|
||||
label `DANOS`. Host: two volumes sharing a label but not an id get distinct
|
||||
id-paths (the collision the label-as-name approach could not resolve stably); a
|
||||
`volumes.csv` override sends a mapped id to its chosen prefix; `match(.fat)`
|
||||
returns the configured binary; fat's argv parser makes installed mounts a
|
||||
function of argv.
|
||||
|
||||
**Top risks.** Step 5's blast radius — a parser/argv bug, OR the `/volumes/usb`→
|
||||
id-path migration missing a fixture/regex, breaks every fat-dependent case at
|
||||
once; mitigated by landing the VM half behind fat's hardcoded mounts first (step
|
||||
4) and making the name migration one atomic, complete sweep. The
|
||||
argv blob is 256 bytes (process.zig) — cap emitted mounts and refuse+log on
|
||||
overflow. `/system/logs` is now a `volumes.csv` concern: dropping the boot
|
||||
identity's rewrite rows silently stops log persistence — ship them by default
|
||||
and document the boot-identity contract in the CSV header.
|
||||
|
||||
---
|
||||
|
||||
## S3 — multi-volume
|
||||
|
||||
**Goal.** Generalize from `var volume: ?Volume` / `volume_id = 1` / first-device
|
||||
/ first-partition to a bounded table of volumes across a bounded table of
|
||||
devices. `partition.allVolumes` returns ALL partitions; the VM adopts EVERY
|
||||
mass-storage provider, probes each device's table, and for each partition spawns
|
||||
one FAT confined to that partition's range (the per-sender clamp is already
|
||||
built), with a distinct `/volumes/<name>` and its OWN backoff/crash-loop state.
|
||||
Removal is per-device. The **boot volume is identified by content** — the FAT
|
||||
process installs the `/system/configuration` + `/system/logs` rewrites only when
|
||||
its own volume resolves `/system/configuration` — so it works as the 2nd
|
||||
partition of the 2nd device just as the 1st of the 1st. (This clarifies the
|
||||
initrd relationship: the kernel already serves `/system/configuration` +
|
||||
binaries read-only from the initrd, which is what lets danos boot with NO volume
|
||||
mounted; a mounted boot volume only adds writable, persistent
|
||||
`/system/configuration` + `/system/logs` that shadow the initrd via
|
||||
longest-prefix match.)
|
||||
|
||||
**Key touchpoints.** `partition.zig` `firstVolume` → `allVolumes(block0,
|
||||
device_blocks, out) usize` (per-entry overflow-safe skip preserved);
|
||||
`volume-manager.zig` the core refactor — `Volume` absorbs the file-global
|
||||
supervision state as per-volume fields, a `StorageDevice` table owns each
|
||||
adopted device's channel once, `volumes[maximum_volumes]` replaces the singleton,
|
||||
a monotonic `next_volume_id`, `openAllStorage`/`adoptAndProbe`,
|
||||
`gatherPresentStorage` + per-device reconcile in `pollTick`, `onHello`/
|
||||
`onNotification` keyed across the table; `fat.zig` content-conditional boot
|
||||
mounts; `system/kernel/vfs.zig` raise `maximum_mounts` (8 → 16) with a refreshed
|
||||
bounds annotation; `make-fat-image.py` a partition-table mode; `build/images.zig`
|
||||
the test disk artifacts.
|
||||
|
||||
**Steps (commits).** (1) `partition.allVolumes` + two-partition host test. (2)
|
||||
Tables, behavior-preserving (still one device / one volume). (3) Multi-device +
|
||||
multi-partition. (4) Per-volume mount naming via argv — each volume by its
|
||||
id-path (from S2), one per volume, no port-name. (5) fat content-conditional
|
||||
boot mounts. (6) Raise `maximum_mounts`. (7) Partitioned-image tool. (8)
|
||||
`two-volume` QEMU case. (9) `boot-2nd-partition` case. (10) Adversarial review,
|
||||
full suite, docs, memory.
|
||||
|
||||
**Discrimination.** Host: an MBR with two partitions yields two volumes with
|
||||
distinct identities (old `firstVolume` returns one). QEMU `two-volume`: a
|
||||
two-partition second device yields two mount lines at two base_lbas (old
|
||||
`openAnyStorage` adopts only the first device). `boot-2nd-partition`: the boot
|
||||
volume works as partition 2 (old code confines fat to partition 1, whose
|
||||
`/system` rewrite backs empty space). Per-volume supervision: killing one
|
||||
volume's FAT restarts only that one (old module-scope supervision can't
|
||||
attribute an exit to one of two).
|
||||
|
||||
**Top risks.** OVMF booting an MBR ESP on partition 2 may be flaky in CI —
|
||||
fallback to a content-detection-ordering assertion + bench-verified boot (the
|
||||
track's existing precedent). N-client range reclamation in usb-storage
|
||||
(`maximum_ranges=64`) must reclaim each of N confined pids' ranges — the V2
|
||||
mechanism, previously exercised with one live client. Duplicate boot volumes:
|
||||
S3 supports exactly one and must log loudly if a second also resolves the boot
|
||||
markers (arbitration deferred to S4).
|
||||
|
||||
---
|
||||
|
||||
## S4 — exFAT: the second engine
|
||||
|
||||
**Goal.** A working exFAT filesystem as `system/services/exfat` that is nothing
|
||||
but an engine + a `main`, reusing `library/kernel/file-system-harness.zig`'s
|
||||
`Server(Engine)` wholesale — the reuse claim the architecture makes, now proven.
|
||||
The harness already owns vfs serving, the badge-scoped open-node table,
|
||||
create-on-open/O_TRUNC, mount registration, the exit sweep, bring-up retry,
|
||||
per-turn time stamping, and durable-on-close. S4 writes only the exFAT-specific
|
||||
bits — but **in full**: complete read AND write, directories, rename, and the
|
||||
real on-disk up-case table for correct case-folding. Not a read-first, minimal-
|
||||
write, or ASCII-only subset. The only limit that survives is the vfs protocol's
|
||||
u32 file-offset surface (a 4 GiB addressable-size cap that applies to FAT too),
|
||||
which is a separate vfs-protocol change, not an exFAT shortcut.
|
||||
|
||||
**Key touchpoints.** New `system/services/exfat/on-disk.zig` (the Main Boot
|
||||
Sector VBR + the five 32-byte directory-entry types as `align(1)` extern structs;
|
||||
`geometryOf` accepting only `"EXFAT "` + `0xAA55`; `setChecksum`, `nameHash`,
|
||||
and case-folding driven by the volume's **on-disk up-case table**); `engine.zig` (`FileSystem` behind the identical `BlockDevice`
|
||||
vtable with fat's exact method set; **allocation-bitmap** cluster authority — the
|
||||
deepest departure from FAT; read honoring `no_fat_chain` contiguous vs FAT-follow;
|
||||
File+Stream+FileName set assembly with recomputed set checksum); `exfat.zig` (the
|
||||
thin service, a near-clone of fat.zig); build wiring + `service("exfat")`;
|
||||
`tools/make-exfat-image.py` (pure stdlib, correct boot checksum, up-case table —
|
||||
**no committed .img**); the `filesystems.csv` EXFAT row (S2) + the `"EXFAT "`
|
||||
recognizer; a `exfat-test` fixture cloned from fat-test.
|
||||
|
||||
**Steps (commits).** (1) on-disk.zig byte layout. (2) engine read path. (3)
|
||||
engine write path (bitmap allocate/free, real 32-bit FAT chain with
|
||||
`no_fat_chain=0`, set-checksum recompute). (4) service + build wiring. (5)
|
||||
`make-exfat-image.py` + image assembly. (6) Routing: `filesystems.csv` +
|
||||
recognizer. (7) `exfat-test` fixture + **cross-engine discrimination** host test.
|
||||
(8) In-VM lifecycle drill (second removable device; mount/mutations/removal). (9)
|
||||
Bounds, docs, adversarial review, memory.
|
||||
|
||||
**Discrimination.** The named one: `fat.mount(exfat_img) == null` (fat reads
|
||||
bytes-per-sector at VBR offset 11 = exFAT's MustBeZero = 0 → reject) AND
|
||||
`exfat.mount(fat_img) == null`, each mounting its own as a control. Host: read
|
||||
across a cluster boundary on both a contiguous and a fragmented file; write
|
||||
across >1 cluster setting the bitmap bits (not the FAT) and a validating set
|
||||
checksum. QEMU: `exfat: mounted /volumes/exfat` + `exfat-test: ok`;
|
||||
`exfat-removal` yanks the exFAT stick mid-write while the FAT boot volume keeps
|
||||
serving.
|
||||
|
||||
**Top risks.** Allocation authority is the bitmap, not the FAT — allocating
|
||||
without setting the bit silently corrupts free space (highest-attention area).
|
||||
A directory-entry SET can straddle sector/cluster boundaries — scanDirectory,
|
||||
set-checksum, and updateStreamEntry must handle multi-sector sets. vfs offsets
|
||||
are u32 while exFAT DataLength is u64 — clamp and document (as fat does). The
|
||||
in-VM drill needs S3 (a non-boot exFAT volume beside the FAT boot volume); if S4
|
||||
landed before S3 the discrimination would rest on host tests until multi-volume
|
||||
exists.
|
||||
|
||||
---
|
||||
|
||||
## S5 — removal robustness
|
||||
|
||||
**Goal.** Close the three known gaps so every removal trigger is exercised
|
||||
end-to-end. (1) **Consume `medium_changed`** — the VM subscribes to the driver's
|
||||
already-published event so the "device stays, medium leaves" case (a card
|
||||
reader, an ejected removable) runs the same kill-retire-remount path as a pulled
|
||||
stick, closing the second of the "two triggers, one lifecycle" the architecture
|
||||
specifies. (2) **Storage-driver-crash rebuild** — a driver that dies while its
|
||||
device stays present is detected and the volume subtree rebuilt on the restarted
|
||||
driver's fresh channel, instead of leaving fat wedged on a dead channel (the V4
|
||||
review's open edge). (3) **QEMU-verified remount-on-replug** — the device-return
|
||||
half is proven, not merely asserted-unmount. Single-volume; independent of S1–S4.
|
||||
|
||||
**Key touchpoints.** `library/kernel/service.zig` an additive, behavior-neutral
|
||||
`on_buffered_message` callback so a buffered-message wake forwards its payload
|
||||
(no existing service sets it); `volume-manager.zig` subscribe on `bringUpVolume`
|
||||
success, `onMediumEvent` with change-count dedupe running a medium-teardown (with
|
||||
`encodeUnsubscribe` before close so the driver's 8-slot table doesn't leak), plus
|
||||
`channelAlive()` (a `geometry()` liveness probe) + `rebuildVolume()` used in
|
||||
`pollTick` and the child-exit path; `fat.zig` re-probe geometry on I/O failure
|
||||
and exit on channel death (device NAK keeps serving); `device-manager.zig` a
|
||||
`test-storage-restart` mode (mirroring `test-scanout-restart`) to kill usb-storage
|
||||
once, post-mount, as the discrimination trigger.
|
||||
|
||||
**Steps (commits).** (1) **Cheap decisive experiments first** (no commits): QMP-
|
||||
eject the boot medium and confirm `usb-storage: medium absent` fires under QEMU
|
||||
(the whole item-1 chain depends on it); and test whether a boot-controller
|
||||
`device_add` is re-presented (settles whether item 3 extends `volume-removal` or
|
||||
needs a second controller as H1 does). (2) Harness `on_buffered_message`
|
||||
(behavior-neutral). (3) VM consumes `medium_changed`. (4) `volume-medium-change`
|
||||
case (fails pre-step-3). (5) VM driver-crash rebuild. (6) fat observes dead
|
||||
channel and exits. (7) `volume-driver-restart` trigger + case. (8) `volume-replug`
|
||||
(second controller if needed). (9) Docs + the real-hardware bench protocol. (10)
|
||||
Full suite + memory.
|
||||
|
||||
**Discrimination.** `volume-medium-change`: an eject with the device left in the
|
||||
tree unmounts (old VM never subscribes → the event goes to no one → mount
|
||||
persists). `volume-driver-restart`: killing usb-storage post-mount while its
|
||||
child stays present triggers a rebuild and a SECOND mount + post-kill read (old
|
||||
`pollTick` only checks `isDevicePresent`, still true, and restarts fat against
|
||||
the stale channel → wedge/crash-loop). `volume-replug`: a device return on a
|
||||
second controller drives a remount (the existing case only ever sees the unmount
|
||||
half).
|
||||
|
||||
**Top risks.** QEMU medium-eject must make TEST UNIT READY report not-ready —
|
||||
step 1(a) validates this before any code. The op-16 overlap (`medium_changed` ==
|
||||
`hello` by number) is safe only because async events arrive as `isMessage`
|
||||
notifications and never reach `Serve.dispatch` — the intercept must run in the
|
||||
notification branch and never catch a synchronous hello. fat can't today
|
||||
distinguish EPEER from a device NAK (`CallError` swallows the errno) — the plan
|
||||
uses a geometry re-probe as the liveness oracle, which is correct but indirect.
|
||||
|
||||
---
|
||||
|
||||
## Decisions (settled)
|
||||
|
||||
Both flagged decisions are settled:
|
||||
|
||||
1. **A volume's path is its id; the label is display metadata (S1/S2).** The
|
||||
mount path is the identity id — the GPT GUID, else a `fat-<serial>` /
|
||||
`mbr-<sig>-<index>` form — a stable, unique, content-derived handle software
|
||||
uses. The label (FAT volume label / GPT partition name) is a mutable display
|
||||
name, NOT in the path; a volume-manager `volumes` verb returns
|
||||
`{ id, mount_path, label }` so a UI can show the friendly name (the database
|
||||
id/name split). Same-label-different-id volumes therefore never collide; only
|
||||
identical ids (dd-clones) hit first-wins-and-log. `volumes.csv` overrides the
|
||||
path (and optionally the label) for a chosen volume, keyed on id. Makes S1
|
||||
precede S2 and folds the `/volumes/usb` → id-path fixture + regex migration
|
||||
into S2 step 5.
|
||||
|
||||
2. **exFAT is implemented in full (S4).** A complete exFAT: full read and write,
|
||||
directories, rename, and the on-disk up-case table for correct case-folding —
|
||||
not a read-first or ASCII-only subset. The one remaining limit is the vfs
|
||||
protocol's u32 file-offset surface, which caps addressable file size at 4 GiB
|
||||
for ALL filesystems (FAT included); widening it to u64 is a separate vfs-
|
||||
protocol change, flagged but out of the exFAT engine's scope.
|
||||
|
||||
The ~26 smaller design-time questions are settled with the recommended default
|
||||
in the phase text (defer GPT entry-array CRC to correctness-only; GUID key =
|
||||
little-endian u128 pinned now; share the DOS date-time helper into a library
|
||||
module both engines import; a second removable usb-storage device for the exFAT
|
||||
drill; VM-poll `channelAlive()` as the load-bearing crash-detection guarantee).
|
||||
@@ -0,0 +1,126 @@
|
||||
# The volume manager: the plan
|
||||
|
||||
*2026-08-09. Executes the settled design in
|
||||
[storage-architecture.md](file-system-development/storage-architecture.md) and
|
||||
[storage-design-rationale.md](file-system-development/storage-design-rationale.md)
|
||||
(decisions 1–8). Track discipline as always: one commit per coherent step, suite
|
||||
green at phase boundaries, every new test shown to fail against the old
|
||||
behavior, one QEMU suite at a time, work on main.*
|
||||
|
||||
**No open decisions.** Decision 4 is settled in the rationale as per-sender
|
||||
range confinement at the provider on one serving endpoint — the badge-scoped
|
||||
provider pattern the xHCI bus already uses, applied to blocks. The volume
|
||||
manager sets each filesystem process's range; the driver clamps and translates
|
||||
every transfer by the sender's kernel-stamped badge; filesystems address
|
||||
volume-relative LBAs from 0 (on the confined path the FAT engine's `base_lba`
|
||||
resolves to 0; the field and the engine's own MBR walk remain as now-inert
|
||||
legacy, the authoritative walk living in the volume manager's `partition.zig`).
|
||||
Every other decision the phases below execute is recorded in the
|
||||
rationale (decisions 1–8); nothing in this plan waits on a choice.
|
||||
|
||||
## V0 — `fs_unmount` ownership (the defect fix)
|
||||
|
||||
The kernel records the mounting process on each mount slot; `fs_unmount` is
|
||||
refused for any caller but the owner (the `/protocol` special case stays).
|
||||
Death cleanup is unaffected (the lazy sweep is not an unmount). Discrimination:
|
||||
a fixture unmounting a prefix it does not own must be refused — fails against
|
||||
today's kernel, which lets any process unmount anything.
|
||||
|
||||
## V1 — the filesystem harness (extraction, no behavior change)
|
||||
|
||||
fat's 434-line shell becomes `library/file-system/harness` (name per
|
||||
convention): establishment, the badge-scoped open-node table, the nine
|
||||
vfs-protocol handlers, mount registration, bring-up/teardown. fat becomes
|
||||
engine + on-disk + a thin main wiring the harness. Suite green is the gate;
|
||||
nothing observable changes. This lands FIRST so every later phase touches the
|
||||
harness once, not fat and the harness both.
|
||||
|
||||
## V2 — the driver mechanism: ranges, confinement, presence
|
||||
|
||||
- Block protocol additions (appended; numbering holds): `define_range`
|
||||
(volume-manager-only in practice — see grants), and the pushed
|
||||
`medium_changed` event (present/absent + change counter).
|
||||
- usb-storage: per-badge range table (clamp + translate per sender), range
|
||||
definitions from the volume manager, and presence: a slow idle-time
|
||||
TEST UNIT READY poll plus sense-key inspection on failed transfers, emitting
|
||||
`medium_changed` on transitions. QEMU test lever: `eject` /
|
||||
`blockdev-remove-medium` against a `removable=on` usb-storage device; if
|
||||
QEMU's model refuses, the fallback drill is device_del/add of the whole
|
||||
stick (the H-matrix already proves that path) and presence gets its real
|
||||
test on the bench with a card reader.
|
||||
- Discrimination: a fixture transferring outside its assigned range must be
|
||||
refused; fails against a driver without the clamp.
|
||||
|
||||
**Sequencing (as executed).** V2a landed the range clamp + its discrimination
|
||||
fixture (block-range) — the security mechanism is testable in isolation. V2b
|
||||
adds `medium_changed` and makes usb-storage a publisher that emits it on
|
||||
presence transitions, but its END-TO-END test (eject → medium_changed →
|
||||
unmount/remount) lands in V4 with the real consumer, the volume manager —
|
||||
rather than a throwaway subscriber fixture V3 would immediately replace. Same
|
||||
work, no duplicated scaffolding.
|
||||
|
||||
## V3 — the volume manager service
|
||||
|
||||
New binary `system/services/volume-manager`, spawned by init, serving the
|
||||
(genuinely singular) name `volume-manager`. Duties, all moved OUT of fat:
|
||||
|
||||
- subscribe to the device manager; consumer-hello each storage provider for
|
||||
its block channel;
|
||||
- probe: partition table walk (MBR now, GPT next — the walk LEAVES the FAT
|
||||
engine) and content identity (the decision-8 ladder: GPT GUID → fs UUID →
|
||||
FAT serial+label → MBR signature+index → anonymous);
|
||||
- configuration: `filesystems.csv` (signature → filesystem binary) and
|
||||
`volumes.csv` (identity → mount prefix; the fstab). Boot volume identity
|
||||
recorded at first sight of `/system/configuration`;
|
||||
- define ranges on the driver; spawn one filesystem process per volume
|
||||
(argv: volume id); answer each filesystem's startup hello with its volume
|
||||
channel (the reply-capability path, same as the device manager's);
|
||||
- supervise: hello/mount deadline, crash-loop cap, reap on removal.
|
||||
|
||||
fat sheds `acquireVolume` and its device-manager grant; filesystem binaries
|
||||
get `open volume-manager` only — a filesystem cannot acquire, only be given.
|
||||
Grants move with the code in the same commits.
|
||||
|
||||
**Sequencing (as executed).** V3a (discovery+probe) and V3b (the flip: spawn +
|
||||
confine + hand over the channel, single volume) landed the core. The V3c items —
|
||||
the `volumes.csv` mount map, the fuller identity ladder (FAT serial, GPT GUID),
|
||||
and multi-volume spawning — mainly serve the MULTI-volume drills (two-partitions,
|
||||
clone-policy). The user's goal is the single-volume boot-stick removal lifecycle,
|
||||
so V4's removal lifecycle runs next on the single volume, and the multi-volume
|
||||
work + its drills become a documented follow-on (V3c/multi-volume). fat keeps its
|
||||
hardcoded mount prefixes until the mount map lands.
|
||||
|
||||
## V4 — the removal lifecycle, end to end
|
||||
|
||||
Two triggers, one path: storage-channel death and `medium_changed(absent)`
|
||||
both drive kill-the-filesystem-process + retire-its-mounts; return (device
|
||||
re-report or `medium_changed(present)`) drives re-probe → respawn → remount
|
||||
at the identity's prefix. The logger gains resume patience (retry flushes on
|
||||
the fat cadence, never abandon) and the ring-wrap gap marker. QEMU cases:
|
||||
|
||||
- `volume-replug`: yank the boot stick mid-run, replug, assert remount at the
|
||||
same prefixes and the logger appending to the SAME boot-stamp tree with the
|
||||
gap marked. Discrimination: against pre-V4, fat wedges (`mounted` forever)
|
||||
and no remount happens.
|
||||
- `volume-two-partitions`: a two-partition FAT image → two volumes, two
|
||||
filesystem processes, two mounts from one stick; yank once, both die; return
|
||||
once, both remount. Proves multi-volume and the reap breadth. (New image
|
||||
fixture beside make-fat-image.py.)
|
||||
- `volume-clone-policy`: two sticks with identical FAT serials — first keeps
|
||||
the mapped name, second mounts suffixed, loudly logged.
|
||||
- The **lifecycle conformance drill**, parameterized by filesystem: mount,
|
||||
serve, yank mid-write, verify honest loss (dirty flag set, gap said),
|
||||
replug, remount. FAT is implementation #1; the drill is the definition of
|
||||
"danos supports filesystem X".
|
||||
|
||||
## V5 — close-out
|
||||
|
||||
Full suite green; the architecture doc's *(planned)* markers flip to built;
|
||||
`fs_unmount` ownership, ranges, presence, identity, and the volume manager
|
||||
lose their future-tense; memory updated; Ryzen bench note: pull the stick,
|
||||
watch the log gap get marked, plug it back anywhere.
|
||||
|
||||
**Not in this track** (recorded so absence is deliberate): GPT parsing beyond
|
||||
the identity read (row exists in the prober's ladder; full GPT when a GPT
|
||||
medium matters), AHCI/NVMe drivers (NVMe gated on the shm-ring data plane),
|
||||
formatting/entropy, per-process namespaces.
|
||||
@@ -151,8 +151,9 @@ Its footprint was tiny: **five** call sites, all `unistd` file operations —
|
||||
(from the boot-log work) `init.zig` and `log-flush.zig`. `stdio.zig` was dead — nothing
|
||||
imported it. The plan — build `runtime.fs`, migrate those five to it, delete
|
||||
`library/posix/`, and drop the `posix` module from `build.zig`'s `addUserBinary` — has
|
||||
since been carried out: `library/` today holds only `mmio`, `runtime`, and
|
||||
`xkeyboard-config`.
|
||||
since been carried out: `library/` today holds `client`, `csv`, `device`,
|
||||
`kernel`, `protocol`, and `xkeyboard-config` (the `runtime` namespace was later
|
||||
reorganized into `library/kernel`, and `mmio` moved under `library/device`).
|
||||
|
||||
## Where danos stands: coverage vs. the gaps
|
||||
|
||||
|
||||
@@ -7,12 +7,25 @@
|
||||
//! `runtime.dma.alloc`), so whole sectors move without crossing the IPC size
|
||||
//! limit — the same handoff usb-storage uses toward the controller.
|
||||
|
||||
const std = @import("std");
|
||||
const envelope = @import("envelope");
|
||||
const ipc = @import("ipc");
|
||||
const block_protocol = @import("block-protocol");
|
||||
|
||||
const Protocol = block_protocol.Protocol;
|
||||
|
||||
/// The medium_changed event payload, re-exported so a consumer decodes it without
|
||||
/// reaching into the wire-format module.
|
||||
pub const MediumChanged = block_protocol.MediumChanged;
|
||||
|
||||
/// Decode a medium_changed event from a buffered-message payload a subscriber
|
||||
/// received (a `Received.isMessage` wake). Null if the bytes are too short to be
|
||||
/// one — a caller ignores anything that is not a well-formed event.
|
||||
pub fn decodeMediumChanged(payload: []const u8) ?MediumChanged {
|
||||
if (payload.len < envelope.prefix_size + @sizeOf(MediumChanged)) return null;
|
||||
return std.mem.bytesToValue(MediumChanged, payload[envelope.prefix_size..][0..@sizeOf(MediumChanged)]);
|
||||
}
|
||||
|
||||
pub const Geometry = struct { block_size: u32, block_count: u64 };
|
||||
|
||||
pub const Device = struct {
|
||||
@@ -35,6 +48,14 @@ pub const Device = struct {
|
||||
return self.call(.attach, {}, handle, &reply) != null;
|
||||
}
|
||||
|
||||
/// The reverse of `attach`: the buffer leaves the device's reach. The same
|
||||
/// region capability rides again (the kernel matches the region). Do not name
|
||||
/// the buffer's physical address in `read`/`write` after this.
|
||||
pub fn detach(self: Device, handle: ipc.Handle) bool {
|
||||
var reply: [block_protocol.message_maximum]u8 = undefined;
|
||||
return self.call(.detach, {}, handle, &reply) != null;
|
||||
}
|
||||
|
||||
/// Read `count` blocks starting at `lba` into the DMA buffer at `physical`.
|
||||
pub fn read(self: Device, lba: u64, count: u32, physical: u64) bool {
|
||||
var reply: [block_protocol.message_maximum]u8 = undefined;
|
||||
@@ -55,6 +76,15 @@ pub const Device = struct {
|
||||
return self.call(.flush, {}, null, &reply) != null;
|
||||
}
|
||||
|
||||
/// Confine the process `badge` to blocks `[base_lba, base_lba + block_count)`
|
||||
/// on this device — the volume manager's per-volume grant to a filesystem.
|
||||
/// The caller must itself be unconfined (whole-device); a confined caller is
|
||||
/// refused, so a filesystem cannot widen its own range. See `DefineRange`.
|
||||
pub fn defineRange(self: Device, badge: u32, base_lba: u64, block_count: u64) bool {
|
||||
var reply: [block_protocol.message_maximum]u8 = undefined;
|
||||
return self.call(.define_range, .{ .badge = badge, .base_lba = base_lba, .block_count = block_count }, null, &reply) != null;
|
||||
}
|
||||
|
||||
/// One request at the driver. `target` is always 0: one endpoint per device, so
|
||||
/// there is no object within the peer to address.
|
||||
fn call(
|
||||
@@ -71,6 +101,30 @@ pub const Device = struct {
|
||||
if (status.status != 0) return null;
|
||||
return reply[0..answer.len];
|
||||
}
|
||||
|
||||
/// Subscribe `subscriber` (an endpoint) to this device's medium_changed
|
||||
/// events: the reserved `subscribe` verb carries the subscriber's endpoint as
|
||||
/// the capability, and the driver then ipc.sends each medium transition to it.
|
||||
pub fn subscribeMedium(self: Device, subscriber: ipc.Handle) bool {
|
||||
var packet: [block_protocol.message_maximum]u8 = undefined;
|
||||
const framed = envelope.encodeSubscribe(0, &packet) orelse return false; // interest 0: every event (block has one)
|
||||
var reply: [block_protocol.message_maximum]u8 = undefined;
|
||||
const answer = ipc.callCap(self.endpoint, framed, &reply, subscriber) catch return false;
|
||||
const status = envelope.statusOf(reply[0..answer.len]) orelse return false;
|
||||
return status.status == 0;
|
||||
}
|
||||
|
||||
/// Unsubscribe from this device's medium_changed events. Call before closing
|
||||
/// the channel so the driver's bounded subscriber table frees the slot rather
|
||||
/// than holding a dead endpoint until an exit sweep notices.
|
||||
pub fn unsubscribeMedium(self: Device) bool {
|
||||
var packet: [block_protocol.message_maximum]u8 = undefined;
|
||||
const framed = envelope.encodeUnsubscribe(&packet) orelse return false;
|
||||
var reply: [block_protocol.message_maximum]u8 = undefined;
|
||||
const answer = ipc.callCap(self.endpoint, framed, &reply, null) catch return false;
|
||||
const status = envelope.statusOf(reply[0..answer.len]) orelse return false;
|
||||
return status.status == 0;
|
||||
}
|
||||
};
|
||||
|
||||
// There is deliberately no open-by-name here: `block` is not a registry name.
|
||||
|
||||
@@ -136,6 +136,15 @@ pub const Device = struct {
|
||||
return self.call(.dma_attach, {}, &.{}, handle, &reply) != null;
|
||||
}
|
||||
|
||||
/// The reverse of `attachDma`: unbind the buffer from the controller's IOMMU
|
||||
/// domain. The same region capability rides again — the kernel matches the
|
||||
/// region, so neither side kept state between the two calls. Do not name the
|
||||
/// buffer's physical address in any transfer after this.
|
||||
pub fn detachDma(self: *Device, handle: ipc.Handle) bool {
|
||||
var reply: [usb_transfer_protocol.message_maximum]u8 = undefined;
|
||||
return self.call(.dma_detach, {}, &.{}, handle, &reply) != null;
|
||||
}
|
||||
|
||||
/// One bulk transfer (IN or OUT per `endpoint_address`'s direction bit) to or
|
||||
/// from the caller's own DMA buffer at `physical`. Returns the bytes moved.
|
||||
pub fn bulk(self: *Device, endpoint_address: u8, physical: u64, length: u32) ?u32 {
|
||||
|
||||
@@ -90,7 +90,7 @@ pub fn build(b: *std.Build) void {
|
||||
// a provider that beat init to the mount needs). It also owns the subscriber
|
||||
// table and the fan-out, which are expressed in the envelope's vocabulary
|
||||
// (the reserved subscribe verb, the push floor) — hence envelope.
|
||||
_ = b.addModule("service", .{
|
||||
const service = b.addModule("service", .{
|
||||
.root_source_file = b.path("service.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "channel", .module = channel },
|
||||
@@ -99,6 +99,23 @@ pub fn build(b: *std.Build) void {
|
||||
.{ .name = "process", .module = process },
|
||||
},
|
||||
});
|
||||
// The filesystem serving harness (docs/file-system-development/storage-architecture.md):
|
||||
// the engine-agnostic half of a filesystem service. It specializes `service`
|
||||
// for the vfs protocol and is block-free (durability rides a caller closure),
|
||||
// so it needs nothing from the device domain — it sits beside its sibling.
|
||||
_ = b.addModule("file-system-harness", .{
|
||||
.root_source_file = b.path("file-system-harness.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "ipc", .module = ipc },
|
||||
.{ .name = "process", .module = process },
|
||||
.{ .name = "service", .module = service },
|
||||
.{ .name = "time", .module = time },
|
||||
.{ .name = "file-system", .module = file_system },
|
||||
.{ .name = "envelope", .module = protocol.module("envelope") },
|
||||
.{ .name = "vfs-protocol", .module = protocol.module("vfs-protocol") },
|
||||
.{ .name = "logging", .module = logging },
|
||||
},
|
||||
});
|
||||
_ = b.addModule("start", .{
|
||||
.root_source_file = b.path("start.zig"),
|
||||
.imports = &.{ .{ .name = "process", .module = process }, .{ .name = "logging", .module = logging } },
|
||||
|
||||
@@ -0,0 +1,347 @@
|
||||
//! The filesystem serving harness: the block-client-and-engine-agnostic half of
|
||||
//! a filesystem service (docs/file-system-development/storage-architecture.md).
|
||||
//! Everything a filesystem process does that is NOT its on-disk format lives
|
||||
//! here — establishment, the badge-scoped open-node table, the nine vfs-protocol
|
||||
//! handlers, mount registration, the not-mounted-yet politeness, the exit sweep,
|
||||
//! and the durable-on-close flush. A filesystem is then an ENGINE (the pure,
|
||||
//! host-testable format code behind a small method set) plus a `main` that wires
|
||||
//! it in, so a second filesystem reuses this wholesale — the reason it is a
|
||||
//! shared library rather than per-filesystem code.
|
||||
//!
|
||||
//! Placement: `library/kernel`, beside its sibling `service` (the generic
|
||||
//! serving harness this specializes for the vfs protocol). It is block-free —
|
||||
//! the caller's `Volume.flush` closure owns durability — so it needs nothing
|
||||
//! from the device domain and introduces no backwards dependency.
|
||||
//!
|
||||
//! `Server(Engine)` is generic over the engine TYPE, checked at compile time by
|
||||
//! the calls below. An engine must expose:
|
||||
//! - `pub const Node` with fields `is_directory: bool`, `size`, `mtime`;
|
||||
//! - `pub const Listing` with `name_buffer`, `name_len`, `is_directory`, `size`;
|
||||
//! - `current_time_epoch` a settable field (the harness stamps it per turn);
|
||||
//! - resolve, createFile, createDirectory, removeFile, rename, truncate,
|
||||
//! readFile, writeFile, listEntry — the signatures fat's engine.zig already has.
|
||||
|
||||
const std = @import("std");
|
||||
const ipc = @import("ipc");
|
||||
const process = @import("process");
|
||||
const service = @import("service");
|
||||
const time = @import("time");
|
||||
const file_system = @import("file-system");
|
||||
const envelope = @import("envelope");
|
||||
const vfs_protocol = @import("vfs-protocol");
|
||||
const logging = @import("logging");
|
||||
|
||||
/// One prefix this filesystem mounts into the kernel mount table. `rewrite` is
|
||||
/// the backend-relative prefix a path is rewritten to before it reaches the
|
||||
/// engine (empty = mount the volume root at `prefix`, the common case).
|
||||
pub const MountSpec = struct { prefix: []const u8, rewrite: []const u8 = "" };
|
||||
|
||||
/// The engine's node type, inferred from `resolve`'s return (`?Node`) so the
|
||||
/// engine need not re-export it as a member — engine.zig keeps `Node` at module
|
||||
/// scope, and this harness stays purely additive on the engine side.
|
||||
fn NodeType(comptime Engine: type) type {
|
||||
return @typeInfo(@typeInfo(@TypeOf(Engine.resolve)).@"fn".return_type.?).optional.child;
|
||||
}
|
||||
|
||||
pub fn Server(comptime Engine: type) type {
|
||||
return struct {
|
||||
const Node = NodeType(Engine);
|
||||
|
||||
/// What a bring-up produces: the mounted engine (a stable pointer the
|
||||
/// caller owns), the prefixes to install, and a durability closure the
|
||||
/// harness calls on every close (the caller checks its own dirty state).
|
||||
pub const Volume = struct {
|
||||
engine: *Engine,
|
||||
mounts: []const MountSpec,
|
||||
flush: *const fn () void,
|
||||
};
|
||||
|
||||
pub const Callbacks = struct {
|
||||
/// Acquire and mount the volume, or null to retry on the timer. The
|
||||
/// caller does the filesystem-specific bring-up (find the block
|
||||
/// device, set up DMA, mount the engine) and returns a `Volume`.
|
||||
bringUp: *const fn (endpoint: ipc.Handle) ?Volume,
|
||||
/// A contract name to bind under /protocol, or null to bind none. In
|
||||
/// the volume-manager era every filesystem is a per-volume process and
|
||||
/// clients reach it through the kernel mount table — fs_resolve routes
|
||||
/// a path to its backing endpoint by prefix — so no filesystem binds a
|
||||
/// shared name. Two volumes would collide on one: the second's bind is
|
||||
/// refused and service.run would exit, so its volume never mounts. The
|
||||
/// endpoint still serves as the mount backend without a name.
|
||||
service_name: ?[]const u8 = null,
|
||||
};
|
||||
|
||||
// --- the harness's own state, one set per instantiation ---------------
|
||||
// A filesystem binary instantiates Server once, so these globals are the
|
||||
// one server's state, exactly where fat's file-scoped globals were.
|
||||
|
||||
const Serve = vfs_protocol.Protocol.Provider(void);
|
||||
const Invocation = envelope.Invocation;
|
||||
const Answer = envelope.Answer;
|
||||
|
||||
/// What a handler returns when the thing asked for is not there — a bad
|
||||
/// node id, someone else's node, an unresolved path, a refused mutation.
|
||||
/// One errno for all: a filesystem's failures are all "no such thing" to
|
||||
/// the file API, and *someone else's* must be indistinguishable from
|
||||
/// *nobody's*, or the refusal would leak which ids are live.
|
||||
const refused: isize = -envelope.ENOENT;
|
||||
|
||||
/// How often to retry bring-up while unmounted. Storage arriving is
|
||||
/// event-shaped (the usb chain registering, maybe after a restart), but
|
||||
/// there is no subscription; a slow poll keeps the service responsive
|
||||
/// (ping, terminate) while it waits and alive to catch late storage.
|
||||
const mount_retry_ms = 500;
|
||||
|
||||
const OpenNode = struct { used: bool = false, node: Node = undefined, owner: u32 = 0 };
|
||||
var open_nodes = [_]OpenNode{.{}} ** 32;
|
||||
|
||||
var callbacks: Callbacks = undefined;
|
||||
var service_endpoint: ipc.Handle = 0;
|
||||
var engine_ptr: ?*Engine = null;
|
||||
var volume_flush: *const fn () void = undefined;
|
||||
var mounted: bool = false;
|
||||
|
||||
fn allocOpen() ?usize {
|
||||
for (&open_nodes, 0..) |*o, i| {
|
||||
if (!o.used) return i;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// The open node `id` names **for `owner`** — null unless in range, in
|
||||
/// use, and this client's own. Ids are small integers from a table of 32,
|
||||
/// trivially guessable, so this badge check is the scope
|
||||
/// (docs/os-development/protocol-namespace.md). The owner is a TASK, not a
|
||||
/// process, because the badge is: a threaded client reads a node from the
|
||||
/// thread that opened it, and the exit sweep releases a worker's handles.
|
||||
fn openFor(id: u64, owner: u32) ?*OpenNode {
|
||||
if (id >= open_nodes.len) return null;
|
||||
const o = &open_nodes[@intCast(id)];
|
||||
if (!o.used or o.owner != owner) return null;
|
||||
return o;
|
||||
}
|
||||
|
||||
const ParentLeaf = struct { parent: []const u8, leaf: []const u8 };
|
||||
|
||||
// Split a path: "/a/b" -> ("/a", "b"); "/b" -> ("/", "b"); "b" -> ("/", "b").
|
||||
fn splitParent(path: []const u8) ParentLeaf {
|
||||
const slash = std.mem.lastIndexOfScalar(u8, path, '/');
|
||||
return .{
|
||||
.parent = if (slash) |s| (if (s == 0) "/" else path[0..s]) else "/",
|
||||
.leaf = if (slash) |s| path[s + 1 ..] else path,
|
||||
};
|
||||
}
|
||||
|
||||
fn onOpen(_: void, invocation: Invocation(vfs_protocol.Open), answer: Answer(vfs_protocol.Opened)) isize {
|
||||
const fs = engine_ptr orelse return refused;
|
||||
const path = invocation.tail;
|
||||
const flags = invocation.request.flags;
|
||||
var node = fs.resolve(path);
|
||||
if (node == null and flags & vfs_protocol.create != 0) {
|
||||
const split = splitParent(path);
|
||||
const parent = fs.resolve(split.parent) orelse return refused;
|
||||
node = fs.createFile(parent, split.leaf);
|
||||
}
|
||||
var resolved = node orelse return refused;
|
||||
// O_TRUNC: replace contents rather than overwrite in place (frees the
|
||||
// old chain, so a shorter rewrite leaves no stale tail).
|
||||
if (flags & vfs_protocol.truncate != 0 and !resolved.is_directory) {
|
||||
fs.truncate(&resolved);
|
||||
}
|
||||
const index = allocOpen() orelse return refused;
|
||||
open_nodes[index] = .{ .used = true, .node = resolved, .owner = invocation.sender };
|
||||
answer.set(.{ .node = index });
|
||||
return 0;
|
||||
}
|
||||
|
||||
fn onRead(_: void, invocation: Invocation(vfs_protocol.Read), answer: Answer(void)) isize {
|
||||
const fs = engine_ptr orelse return refused;
|
||||
const o = openFor(invocation.target, invocation.sender) orelse return refused;
|
||||
const into = answer.tail();
|
||||
const want = @min(@as(usize, invocation.request.len), into.len);
|
||||
return @intCast(fs.readFile(o.node, @intCast(invocation.request.offset), into[0..want]));
|
||||
}
|
||||
|
||||
fn onWrite(_: void, invocation: Invocation(vfs_protocol.Write), answer: Answer(vfs_protocol.Written)) isize {
|
||||
const fs = engine_ptr orelse return refused;
|
||||
const o = openFor(invocation.target, invocation.sender) orelse return refused;
|
||||
const data = invocation.tail[0..@min(invocation.tail.len, invocation.request.len)];
|
||||
const n = fs.writeFile(&o.node, @intCast(invocation.request.offset), data);
|
||||
answer.set(.{ .count = @intCast(n) });
|
||||
return 0;
|
||||
}
|
||||
|
||||
fn onStatus(_: void, invocation: Invocation(void), answer: Answer(vfs_protocol.FileStatus)) isize {
|
||||
const o = openFor(invocation.target, invocation.sender) orelse return refused;
|
||||
const kind: vfs_protocol.NodeKind = if (o.node.is_directory) .directory else .regular;
|
||||
answer.set(.{ .size = o.node.size, .kind = @intFromEnum(kind), .mtime = o.node.mtime });
|
||||
return 0;
|
||||
}
|
||||
|
||||
/// One entry per call. End of directory — not a directory, or a cursor
|
||||
/// past the last child — is an entry with no name.
|
||||
fn onReaddir(_: void, invocation: Invocation(vfs_protocol.Readdir), answer: Answer(vfs_protocol.DirectoryEntry)) isize {
|
||||
const fs = engine_ptr orelse return refused;
|
||||
const o = openFor(invocation.target, invocation.sender) orelse return refused;
|
||||
if (!o.node.is_directory) {
|
||||
answer.set(.{});
|
||||
return 0;
|
||||
}
|
||||
const listing = fs.listEntry(o.node, @intCast(invocation.request.cursor)) orelse {
|
||||
answer.set(.{});
|
||||
return 0;
|
||||
};
|
||||
const kind: vfs_protocol.NodeKind = if (listing.is_directory) .directory else .regular;
|
||||
const into = answer.tail();
|
||||
const name_len = @min(listing.name_len, into.len);
|
||||
@memcpy(into[0..name_len], listing.name_buffer[0..name_len]);
|
||||
answer.set(.{ .kind = @intFromEnum(kind), .name_len = @intCast(name_len), .size = listing.size });
|
||||
return @intCast(name_len);
|
||||
}
|
||||
|
||||
/// Closing is scoped like any other node operation: a client releases its
|
||||
/// own handles and nobody else's, and a foreign/free/out-of-range id is
|
||||
/// refused identically so a close cannot probe which ids are live.
|
||||
fn onClose(_: void, invocation: Invocation(void), _: Answer(void)) isize {
|
||||
const o = openFor(invocation.target, invocation.sender) orelse return refused;
|
||||
o.used = false;
|
||||
// Durable-on-close: the caller's flush commits any device write cache
|
||||
// to stable media now. This is what makes init's shutdown log flush
|
||||
// survive a real power-off, and the right default for removable media.
|
||||
volume_flush();
|
||||
return 0;
|
||||
}
|
||||
|
||||
fn onMakeDirectory(_: void, invocation: Invocation(void), _: Answer(void)) isize {
|
||||
const fs = engine_ptr orelse return refused;
|
||||
const path = invocation.tail;
|
||||
if (fs.resolve(path) != null) return refused; // already exists
|
||||
const split = splitParent(path);
|
||||
const parent = fs.resolve(split.parent) orelse return refused;
|
||||
if (fs.createDirectory(parent, split.leaf) == null) return refused;
|
||||
return 0;
|
||||
}
|
||||
|
||||
fn onUnlink(_: void, invocation: Invocation(void), _: Answer(void)) isize {
|
||||
const fs = engine_ptr orelse return refused;
|
||||
const split = splitParent(invocation.tail);
|
||||
const parent = fs.resolve(split.parent) orelse return refused;
|
||||
if (!fs.removeFile(parent, split.leaf)) return refused;
|
||||
return 0;
|
||||
}
|
||||
|
||||
fn onRename(_: void, invocation: Invocation(void), _: Answer(void)) isize {
|
||||
const fs = engine_ptr orelse return refused;
|
||||
const both = invocation.tail;
|
||||
const separator = std.mem.indexOfScalar(u8, both, 0) orelse return refused;
|
||||
const old_split = splitParent(both[0..separator]);
|
||||
const new_split = splitParent(both[separator + 1 ..]);
|
||||
if (!std.mem.eql(u8, old_split.parent, new_split.parent)) return refused; // same-directory only
|
||||
const parent = fs.resolve(old_split.parent) orelse return refused;
|
||||
if (!fs.rename(parent, old_split.leaf, new_split.leaf)) return refused;
|
||||
return 0;
|
||||
}
|
||||
|
||||
/// The verbs this backend implements. `mount`/`unmount`/`bind` are absent
|
||||
/// on purpose — path routing is the kernel's, and only init serves `bind`.
|
||||
const handlers = Serve.Handlers{
|
||||
.open = onOpen,
|
||||
.close = onClose,
|
||||
.read = onRead,
|
||||
.write = onWrite,
|
||||
.status = onStatus,
|
||||
.readdir = onReaddir,
|
||||
.mkdir = onMakeDirectory,
|
||||
.unlink = onUnlink,
|
||||
.rename = onRename,
|
||||
};
|
||||
|
||||
/// The vfs protocol carries no capability, so `arrived` is never claimed
|
||||
/// — the harness's ownership rule then closes whatever a caller attached,
|
||||
/// so a request carrying one cannot spend a slot of this server's table.
|
||||
fn onMessage(message: []const u8, out: []u8, sender: u32, arrived: *ipc.Arrival) usize {
|
||||
_ = arrived;
|
||||
const fs = engine_ptr;
|
||||
// Storage not up yet: fail politely, whatever was asked — clients retry.
|
||||
if (!mounted or fs == null) {
|
||||
const status = envelope.Status{ .status = refused, .len = 0 };
|
||||
@memcpy(out[0..envelope.prefix_size], std.mem.asBytes(&status));
|
||||
return envelope.prefix_size;
|
||||
}
|
||||
// Stamp create/write with the current wall-clock time (mtime): cheap,
|
||||
// and it keeps the engine pure (it takes the time as data, not a call).
|
||||
fs.?.current_time_epoch = time.wallClock();
|
||||
return Serve.dispatch({}, handlers, message, sender, null, out);
|
||||
}
|
||||
|
||||
/// A process-exit event releases every open handle the dead client held,
|
||||
/// so a crashed reader cannot pin table slots.
|
||||
fn onNotification(badge: u64) void {
|
||||
const got = ipc.Received{ .len = 0, .badge = badge, .cap = null };
|
||||
if (got.isTimer()) {
|
||||
tryBringUp();
|
||||
if (!mounted) _ = time.timerOnce(service_endpoint, mount_retry_ms);
|
||||
return;
|
||||
}
|
||||
if (!got.isChildExit()) return;
|
||||
const dead = got.childProcessId();
|
||||
var released: u32 = 0;
|
||||
for (&open_nodes) |*o| {
|
||||
if (o.used and o.owner == dead) {
|
||||
o.* = .{};
|
||||
released += 1;
|
||||
}
|
||||
}
|
||||
if (released != 0) std.log.info("released {d} handle(s) for dead client {d}", .{ released, dead });
|
||||
}
|
||||
|
||||
/// One bring-up attempt: ask the caller for a mounted volume, and on
|
||||
/// success install its mounts and go live. A failure leaves everything
|
||||
/// untouched for the next tick.
|
||||
fn tryBringUp() void {
|
||||
if (mounted) return;
|
||||
const volume = callbacks.bringUp(service_endpoint) orelse return;
|
||||
engine_ptr = volume.engine;
|
||||
volume_flush = volume.flush;
|
||||
for (volume.mounts) |m| {
|
||||
const ok = if (m.rewrite.len == 0)
|
||||
file_system.mount(m.prefix, service_endpoint)
|
||||
else
|
||||
file_system.mountRewritten(m.prefix, service_endpoint, m.rewrite);
|
||||
if (ok) {
|
||||
std.log.info("mounted {s}", .{m.prefix});
|
||||
} else {
|
||||
// The failure diagnostic goes to the kernel ring directly, not
|
||||
// through std.log — the mount that failed may be /system/logs
|
||||
// itself, and a routed record would then have nowhere to land.
|
||||
// Three appends rather than a formatted line so there is no
|
||||
// scratch buffer (and so no fixed length to justify).
|
||||
_ = logging.write("file-system: could not mount ");
|
||||
_ = logging.write(m.prefix);
|
||||
_ = logging.write("\n");
|
||||
}
|
||||
}
|
||||
mounted = true;
|
||||
}
|
||||
|
||||
fn initialise(endpoint: ipc.Handle) bool {
|
||||
service_endpoint = endpoint;
|
||||
// Sweep a dead client's open handles via the published exit events —
|
||||
// clients hold OUR node ids directly, so a crash must not pin slots.
|
||||
_ = process.subscribeExits(endpoint);
|
||||
tryBringUp();
|
||||
if (!mounted) _ = time.timerOnce(endpoint, mount_retry_ms);
|
||||
return true; // serve regardless: requests fail politely until storage mounts
|
||||
}
|
||||
|
||||
pub fn run(cb: Callbacks) void {
|
||||
callbacks = cb;
|
||||
service.run(vfs_protocol.message_maximum, .{
|
||||
.service = cb.service_name,
|
||||
.init = initialise,
|
||||
.on_message = onMessage,
|
||||
.on_notification = onNotification,
|
||||
});
|
||||
}
|
||||
};
|
||||
}
|
||||
@@ -67,6 +67,15 @@ pub const Callbacks = struct {
|
||||
/// A notification that is not a signal — a subscribed exit event, a bound
|
||||
/// IRQ, a timer landing. The raw badge; decode with the ipc helpers.
|
||||
on_notification: ?*const fn (badge: u64) void = null,
|
||||
/// A buffered async message (`Received.isMessage`): a pushed event from a
|
||||
/// provider this service subscribed to, its payload in the receive buffer.
|
||||
/// Unlike `on_message`, it never goes through the protocol dispatch — so an
|
||||
/// event whose reserved op number collides with one of this service's own
|
||||
/// verbs (a `block` `medium_changed` reaching the volume manager, whose own
|
||||
/// protocol numbers `hello` the same) is decoded by hand here, not
|
||||
/// mis-dispatched. Default null: the badge alone still reaches
|
||||
/// `on_notification`, exactly as before this callback existed.
|
||||
on_buffered_message: ?*const fn (message: []const u8) void = null,
|
||||
/// The reload signal. Default: ignored.
|
||||
on_reload: ?*const fn () void = null,
|
||||
/// The terminate signal, called before the loop returns. The clean exit is
|
||||
@@ -372,6 +381,13 @@ pub fn run(comptime maximum_message: usize, callbacks: Callbacks) void {
|
||||
if (got.isChildExit()) {
|
||||
if (callbacks.subscribers) |subscribers| subscribers.forget(got.childProcessId());
|
||||
}
|
||||
// A buffered async message (a pushed event) carries a payload; hand it
|
||||
// to the service that asked for it. The badge still reaches
|
||||
// on_notification below, so a coalesced timer/exit riding the same wake
|
||||
// is not lost — and a service without this callback is unchanged.
|
||||
if (got.isMessage()) {
|
||||
if (callbacks.on_buffered_message) |onBuffered| onBuffered(receive[0..got.len]);
|
||||
}
|
||||
if (callbacks.on_notification) |onNotification| onNotification(got.badge);
|
||||
continue;
|
||||
}
|
||||
|
||||
@@ -37,9 +37,47 @@ pub const Transfer = extern struct {
|
||||
/// How many blocks a transfer actually moved.
|
||||
pub const Transferred = extern struct { count: u32 };
|
||||
|
||||
/// `define_range(badge, base_lba, block_count)`: confine the sender identified by
|
||||
/// `badge` to blocks `[base_lba, base_lba + block_count)`. The volume manager
|
||||
/// calls this for each filesystem process it hands a channel to — the badge is
|
||||
/// the filesystem's kernel-stamped task id, and the range is the partition it
|
||||
/// mounts. A confined sender's read/write LBAs are then volume-relative (the
|
||||
/// driver adds `base_lba`) and a transfer past `block_count` is refused. A
|
||||
/// sender with no range is unconfined (the whole device), the default until the
|
||||
/// volume manager defines one. The clamp lives at the provider because a channel
|
||||
/// must carry exactly the authority it grants (storage-architecture.md): handing
|
||||
/// a filesystem the whole disk plus a base offset would let it reach the
|
||||
/// neighbouring partition. **A confined caller may not call this** — a filesystem
|
||||
/// cannot redefine its own range and escape; only an unconfined party (the
|
||||
/// volume manager) confines others.
|
||||
pub const DefineRange = extern struct {
|
||||
badge: u32,
|
||||
_padding: u32 = 0,
|
||||
base_lba: u64,
|
||||
block_count: u64,
|
||||
};
|
||||
|
||||
/// The `medium_changed` event payload: whether a medium is now present, and a
|
||||
/// monotonic counter so a subscriber that missed an edge still sees that
|
||||
/// SOMETHING changed. Pushed by a driver whose transport can tell medium from
|
||||
/// device (a card reader, an ATAPI tray): the device stays, the medium comes and
|
||||
/// goes. The volume manager consumes it into the same unmount/remount path it
|
||||
/// runs on device death — one lifecycle, two triggers
|
||||
/// (docs/file-system-development/storage-architecture.md). Presence only, never
|
||||
/// content: the driver reports that the medium changed, not what is on it.
|
||||
pub const MediumChanged = extern struct {
|
||||
present: u8, // 1 present, 0 absent
|
||||
_padding: u8 = 0,
|
||||
_padding2: u16 = 0,
|
||||
change_count: u32,
|
||||
};
|
||||
|
||||
pub const Protocol = envelope.Define(.{
|
||||
.name = "block",
|
||||
.version = 1,
|
||||
.events = &.{
|
||||
.{ .name = "medium_changed", .payload = MediumChanged },
|
||||
},
|
||||
.operations = &.{
|
||||
.{ .name = "geometry", .reply = Geometry },
|
||||
.{ .name = "read", .request = Transfer, .reply = Transferred },
|
||||
@@ -54,8 +92,19 @@ pub const Protocol = envelope.Define(.{
|
||||
// physical addresses (named in later read/write) are reachable by the
|
||||
// device under an enforcing IOMMU. Call once per buffer before using it.
|
||||
.{ .name = "attach" },
|
||||
// detach(): the reverse — the same region capability rides the cap slot
|
||||
// (the caller still holds its handle; the kernel matches the region) and
|
||||
// the buffer leaves the device's domain. Every grant a live process
|
||||
// makes is revocable by the granter while alive; death remains the
|
||||
// mechanical backstop (storage-architecture.md, the lifecycle rule).
|
||||
.{ .name = "detach" },
|
||||
// define_range(): confine a sender to a block sub-range — the partition
|
||||
// it mounts. See `DefineRange`. Appended, so every verb above keeps its
|
||||
// number.
|
||||
.{ .name = "define_range", .request = DefineRange },
|
||||
},
|
||||
});
|
||||
|
||||
pub const Operation = Protocol.Operation;
|
||||
pub const Event = Protocol.Event;
|
||||
pub const message_maximum: usize = Protocol.message_maximum;
|
||||
|
||||
@@ -31,6 +31,7 @@ pub fn build(b: *std.Build) void {
|
||||
.{ .name = "block-protocol", .root = "block/block-protocol.zig" },
|
||||
.{ .name = "usb-transfer-protocol", .root = "usb-transfer/usb-transfer-protocol.zig" },
|
||||
.{ .name = "device-manager-protocol", .root = "device-manager/device-manager-protocol.zig" },
|
||||
.{ .name = "volume-manager-protocol", .root = "volume-manager/volume-manager-protocol.zig" },
|
||||
.{ .name = "display-protocol", .root = "display/display-protocol.zig" },
|
||||
.{ .name = "scanout-protocol", .root = "scanout/scanout-protocol.zig" },
|
||||
.{ .name = "power-protocol", .root = "power/power-protocol.zig" },
|
||||
|
||||
@@ -212,8 +212,9 @@ test "a tree report fits the push floor with the header folded in" {
|
||||
try std.testing.expectEqual(@as(usize, 48), @sizeOf(ChildAdded));
|
||||
try std.testing.expectEqual(envelope.post_maximum, Protocol.event_maximum);
|
||||
try std.testing.expect(Protocol.event_maximum <= envelope.post_maximum);
|
||||
// Ten records per enumerate reply — what the old count-header layout carried.
|
||||
try std.testing.expectEqual(@as(usize, 10), entries_per_reply);
|
||||
// Seven records per enumerate reply: (packet_maximum 256 - prefix 16) / 32.
|
||||
// (The old count-header layout carried ten; this asserts the current shape.)
|
||||
try std.testing.expectEqual(@as(usize, 7), entries_per_reply);
|
||||
}
|
||||
|
||||
test "the verb and event numbering, and the device id in the header" {
|
||||
|
||||
@@ -156,6 +156,11 @@ pub const Protocol = envelope.Define(.{
|
||||
// target says which caller's device is attaching, so there is nothing
|
||||
// left for a body to carry.
|
||||
.{ .name = "dma_attach" },
|
||||
// dma_detach: the reverse, same shape — the region capability rides the
|
||||
// cap slot again (the kernel matches the region; the provider retains
|
||||
// nothing between the two calls) and the buffer leaves the controller's
|
||||
// domain. Appended, so every existing verb keeps its number.
|
||||
.{ .name = "dma_detach" },
|
||||
},
|
||||
.events = &.{
|
||||
.{ .name = "interrupt_report", .payload = InterruptReport },
|
||||
@@ -188,6 +193,7 @@ test "the verb numbering, and the device token in the header" {
|
||||
try std.testing.expectEqual(@as(u32, 18), @intFromEnum(Operation.interrupt_subscribe));
|
||||
try std.testing.expectEqual(@as(u32, 19), @intFromEnum(Operation.bulk));
|
||||
try std.testing.expectEqual(@as(u32, 20), @intFromEnum(Operation.dma_attach));
|
||||
try std.testing.expectEqual(@as(u32, 21), @intFromEnum(Operation.dma_detach));
|
||||
try std.testing.expectEqual(@as(u32, 16), @intFromEnum(Event.interrupt_report));
|
||||
|
||||
var buffer: [message_maximum]u8 = undefined;
|
||||
|
||||
@@ -0,0 +1,115 @@
|
||||
//! The volume-manager protocol (docs/file-system-development/storage-architecture.md):
|
||||
//! what a filesystem service says to the volume manager over
|
||||
//! `/protocol/volume-manager`. Defined through the envelope, so every packet
|
||||
//! begins with the folded `Header`.
|
||||
//!
|
||||
//! One verb. A filesystem the volume manager spawned announces itself with the
|
||||
//! volume id it was given as argv[1] (folded into `Header.target`); the reply
|
||||
//! carries that volume's block channel — already range-confined to the
|
||||
//! filesystem's badge — as the call's returned capability. The filesystem never
|
||||
//! finds its storage by name and never sees the whole device; establishment is
|
||||
//! by lineage, exactly as a driver reaches its controller (communication.md
|
||||
//! "Establishment: two planes"). No channel in the reply means the volume is not
|
||||
//! ready yet — retryable, never a verdict.
|
||||
|
||||
const std = @import("std");
|
||||
const envelope = @import("envelope");
|
||||
|
||||
pub const version: u16 = 1;
|
||||
|
||||
/// The filesystem's handshake. Carries only its protocol version; the volume it
|
||||
/// serves is `Header.target`, and the block channel it needs comes back as the
|
||||
/// reply's capability.
|
||||
pub const Hello = extern struct {
|
||||
version: u16 = version,
|
||||
_padding: u16 = 0,
|
||||
};
|
||||
|
||||
/// A `volumes` query — no request fields; the reply's tail carries the volume's
|
||||
/// descriptor (`VolumeInfo`). The mechanism by which a shell or file manager
|
||||
/// reads a volume's display label: the mount path is its id (software's stable
|
||||
/// handle), the label is separate display metadata, the database id/name split.
|
||||
pub const Volumes = extern struct {
|
||||
_reserved: u32 = 0,
|
||||
};
|
||||
|
||||
/// The `volumes` reply: three length-prefixed strings packed into the reply tail
|
||||
/// — the volume's id (its mount path is /volumes/<id> unless overridden), its
|
||||
/// actual mount path, and its display label. `id` is what software keys on;
|
||||
/// `label` is what a UI shows.
|
||||
pub const VolumeInfo = struct {
|
||||
id: []const u8,
|
||||
mount_path: []const u8,
|
||||
label: []const u8,
|
||||
|
||||
const header_bytes = 6; // three u16 lengths, little-endian
|
||||
|
||||
/// Pack into `buf`, returning the used slice, or null if it does not fit.
|
||||
pub fn encode(self: VolumeInfo, buf: []u8) ?[]u8 {
|
||||
const total = header_bytes + self.id.len + self.mount_path.len + self.label.len;
|
||||
if (total > buf.len) return null;
|
||||
std.mem.writeInt(u16, buf[0..2], @intCast(self.id.len), .little);
|
||||
std.mem.writeInt(u16, buf[2..4], @intCast(self.mount_path.len), .little);
|
||||
std.mem.writeInt(u16, buf[4..6], @intCast(self.label.len), .little);
|
||||
var off: usize = header_bytes;
|
||||
@memcpy(buf[off..][0..self.id.len], self.id);
|
||||
off += self.id.len;
|
||||
@memcpy(buf[off..][0..self.mount_path.len], self.mount_path);
|
||||
off += self.mount_path.len;
|
||||
@memcpy(buf[off..][0..self.label.len], self.label);
|
||||
return buf[0..total];
|
||||
}
|
||||
|
||||
/// Decode a reply tail, or null if it is malformed (short or inconsistent).
|
||||
/// The returned slices point into `bytes`.
|
||||
pub fn decode(bytes: []const u8) ?VolumeInfo {
|
||||
if (bytes.len < header_bytes) return null;
|
||||
const id_len = std.mem.readInt(u16, bytes[0..2], .little);
|
||||
const path_len = std.mem.readInt(u16, bytes[2..4], .little);
|
||||
const label_len = std.mem.readInt(u16, bytes[4..6], .little);
|
||||
const total = header_bytes + @as(usize, id_len) + path_len + label_len;
|
||||
if (total > bytes.len) return null;
|
||||
var off: usize = header_bytes;
|
||||
const id = bytes[off..][0..id_len];
|
||||
off += id_len;
|
||||
const mount_path = bytes[off..][0..path_len];
|
||||
off += path_len;
|
||||
const label = bytes[off..][0..label_len];
|
||||
return .{ .id = id, .mount_path = mount_path, .label = label };
|
||||
}
|
||||
};
|
||||
|
||||
pub const Protocol = envelope.Define(.{
|
||||
.name = "volume-manager",
|
||||
.version = 1,
|
||||
.operations = &.{
|
||||
.{ .name = "hello", .request = Hello },
|
||||
.{ .name = "volumes", .request = Volumes },
|
||||
},
|
||||
});
|
||||
|
||||
pub const Operation = Protocol.Operation;
|
||||
pub const message_maximum: usize = Protocol.message_maximum;
|
||||
|
||||
// Named fixture sizes so the bounds gate (which flags literal array lengths)
|
||||
// stays quiet: test inputs, not runtime ceilings.
|
||||
const test_reply_bytes = 128;
|
||||
const test_tiny_bytes = 4;
|
||||
|
||||
test "VolumeInfo round-trips id, mount_path, and label" {
|
||||
var buf: [test_reply_bytes]u8 = undefined;
|
||||
const info = VolumeInfo{ .id = "fat-12345678", .mount_path = "/volumes/fat-12345678", .label = "DANOS" };
|
||||
const encoded = info.encode(&buf).?;
|
||||
const back = VolumeInfo.decode(encoded).?;
|
||||
try std.testing.expectEqualStrings("fat-12345678", back.id);
|
||||
try std.testing.expectEqualStrings("/volumes/fat-12345678", back.mount_path);
|
||||
try std.testing.expectEqualStrings("DANOS", back.label);
|
||||
}
|
||||
|
||||
test "VolumeInfo encode refuses a buffer that is too small; decode rejects a short tail" {
|
||||
var tiny: [test_tiny_bytes]u8 = undefined;
|
||||
const info = VolumeInfo{ .id = "fat-1", .mount_path = "/volumes/fat-1", .label = "" };
|
||||
try std.testing.expect(info.encode(&tiny) == null);
|
||||
try std.testing.expect(VolumeInfo.decode(&[_]u8{ 0, 0, 0 }) == null); // shorter than the header
|
||||
try std.testing.expect(VolumeInfo.decode(&[_]u8{ 0xFF, 0xFF, 0, 0, 0, 0 }) == null); // claims 65535 id bytes
|
||||
}
|
||||
@@ -0,0 +1,8 @@
|
||||
# The filesystem map: a probed volume's content signature -> the service binary
|
||||
# that serves it (docs/file-system-development/storage-architecture.md). The
|
||||
# volume manager reads this (the policy); a signature no row matches goes
|
||||
# unserved, never guessed. Adding a filesystem adds a row.
|
||||
#
|
||||
# signature, binary
|
||||
fat, /system/services/fat
|
||||
exfat, /system/services/exfat
|
||||
|
@@ -14,7 +14,9 @@
|
||||
# service args...
|
||||
/system/services/input
|
||||
/system/services/device-manager
|
||||
/system/services/fat
|
||||
# fat is not here: the volume manager spawns one filesystem per volume it finds,
|
||||
# confined to that volume's partition (docs/file-system-development/storage-architecture.md).
|
||||
/system/services/volume-manager
|
||||
/system/services/display
|
||||
/system/services/display-demo
|
||||
/system/services/logger
|
||||
|
||||
|
@@ -55,7 +55,10 @@
|
||||
# --- the services init spawns from init.csv ---------------------------------
|
||||
/system/services/input, /system/services/init, bind, input
|
||||
/system/services/device-manager, /system/services/init, bind, device-manager
|
||||
/system/services/fat, /system/services/init, bind, vfs
|
||||
/system/services/volume-manager, /system/services/init, bind, volume-manager
|
||||
# fat is spawned and supervised by the volume manager now, not init — the volume
|
||||
# manager confines it to its partition and hands it the block channel.
|
||||
/system/services/fat, /system/services/volume-manager, bind, vfs
|
||||
/system/services/display, /system/services/init, bind, display
|
||||
|
||||
# The discovery service ships under one neutral name per firmware (docs/discovery.md);
|
||||
@@ -76,6 +79,7 @@
|
||||
# 'kernel' as the supervisor. Nothing else changes: the binary must still match.
|
||||
/system/services/input, kernel, bind, input
|
||||
/system/services/device-manager, kernel, bind, device-manager
|
||||
/system/services/volume-manager, kernel, bind, volume-manager
|
||||
/system/services/fat, kernel, bind, vfs
|
||||
/system/services/display, kernel, bind, display
|
||||
/system/services/discovery, kernel, bind, power
|
||||
@@ -94,12 +98,19 @@
|
||||
# ============================================================================
|
||||
|
||||
# --- init's own services ----------------------------------------------------
|
||||
# fat reaches the device manager to be routed to its volume's block provider
|
||||
# (block is not a name — see the bind section); the compositor reaches the
|
||||
# scanout its driver announced, its own endpoint (the mouse-listener thread
|
||||
# opens /protocol/display like any other client — threads share no handles),
|
||||
# and the input stream that moves the cursor.
|
||||
/system/services/fat, /system/services/init, open, device-manager
|
||||
# fat reaches the volume manager to be handed its volume's block channel
|
||||
# (range-confined); the compositor reaches the scanout its driver announced, its
|
||||
# own endpoint (the mouse-listener thread opens /protocol/display like any other
|
||||
# client — threads share no handles), and the input stream that moves the cursor.
|
||||
/system/services/fat, /system/services/volume-manager, open, volume-manager
|
||||
# exfat reaches the volume manager the same way — the second engine, same lineage.
|
||||
/system/services/exfat, /system/services/volume-manager, open, volume-manager
|
||||
# The volume manager reaches the device manager to be routed to each storage
|
||||
# provider's block channel, then confines a filesystem to each volume.
|
||||
/system/services/volume-manager, /system/services/init, open, device-manager
|
||||
# ...and again under the kernel supervisor for the manual-tree drills (S5's
|
||||
# volume-driver-restart spawns the volume manager directly, not via init).
|
||||
/system/services/volume-manager, kernel, open, device-manager
|
||||
/system/services/display, /system/services/init, open, scanout
|
||||
/system/services/display, /system/services/init, open, display
|
||||
/system/services/display, /system/services/init, open, input
|
||||
|
||||
|
Can't render this file because it contains an unexpected character in line 12 and column 15.
|
@@ -0,0 +1,8 @@
|
||||
# The mount map (danos's fstab): a volume's content id -> a chosen mount prefix.
|
||||
# This is an OPTIONAL override, read by the volume manager. A volume with no row
|
||||
# mounts at its default /volumes/<id>, where <id> is the manager's rendered
|
||||
# content identity (e.g. fat-12345678, gpt-<guid>, mbr-<sig>-<index>) — stable,
|
||||
# unique, and never a port or a label. The label is display metadata, not here:
|
||||
# query it via the volume manager's `volumes` verb.
|
||||
#
|
||||
# id, mount_prefix
|
||||
|
@@ -25,9 +25,11 @@ const bot = @import("bulk-only-transport.zig");
|
||||
const envelope = @import("envelope");
|
||||
const block_protocol = @import("block-protocol");
|
||||
|
||||
/// The generated block dispatch. One device per process, so the handler context
|
||||
/// is empty and the geometry stays in this file's globals.
|
||||
const Serve = block_protocol.Protocol.Provider(void);
|
||||
/// The generated block dispatch plus the subscriber machinery the harness owns
|
||||
/// (subscribe/unsubscribe, the exit sweep, the fan-out) — usb-storage publishes
|
||||
/// `medium_changed`, so it is a Subscribers provider, not a bare Provider. One
|
||||
/// device per process, so the handler context is empty.
|
||||
const Serve = service.Subscribers(block_protocol.Protocol, void);
|
||||
|
||||
const Invocation = envelope.Invocation;
|
||||
const Answer = envelope.Answer;
|
||||
@@ -47,6 +49,21 @@ var next_tag: u32 = 1;
|
||||
var block_size: u32 = 512;
|
||||
var block_count: u64 = 0;
|
||||
|
||||
// --- medium presence --------------------------------------------------------
|
||||
//
|
||||
// A slow TEST UNIT READY poll tracks whether the medium is present; on a
|
||||
// transition the driver publishes `medium_changed` to its subscribers (the
|
||||
// volume manager). This is the second removal trigger — the DEVICE stays while
|
||||
// the MEDIUM leaves (a card reader, an ATAPI tray) — which channel death cannot
|
||||
// see (docs/file-system-development/storage-architecture.md). Presence only,
|
||||
// never content. A device that is genuinely unplugged is reaped by the device
|
||||
// manager instead; a poll failure just before that death publishes absent
|
||||
// harmlessly.
|
||||
var service_endpoint: ipc.Handle = 0;
|
||||
var medium_present: bool = true; // a successful bring-up means the medium is here
|
||||
var medium_change_count: u32 = 0;
|
||||
const presence_poll_ms = 1000;
|
||||
|
||||
/// One Bulk-Only-Transport command: send the CBW, run the data stage (to/from
|
||||
/// `data_physical`), read and validate the CSW. Returns true on a passed status.
|
||||
fn transact(cdb: []const u8, direction_in: bool, data_physical: u64, data_length: u32) bool {
|
||||
@@ -82,6 +99,7 @@ fn transact(cdb: []const u8, direction_in: bool, data_physical: u64, data_length
|
||||
var bring_up_failed = false;
|
||||
|
||||
fn initialise(endpoint: ipc.Handle) bool {
|
||||
service_endpoint = endpoint;
|
||||
// One hello, both directions: the block-serving endpoint goes UP (the
|
||||
// manager routes fat's consumer hello here — this driver serves one
|
||||
// volume, one process per stick, so `block` is never a registry name),
|
||||
@@ -154,6 +172,8 @@ fn initialise(endpoint: ipc.Handle) bool {
|
||||
const sector: [*]const u8 = @ptrFromInt(command_data.virtual);
|
||||
std.log.info("block 0 signature 0x{x:0>2}{x:0>2}", .{ sector[510], sector[511] });
|
||||
}
|
||||
// Bring-up succeeded, so the medium is present; start the presence poll.
|
||||
_ = time.timerOnce(endpoint, presence_poll_ms);
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -167,14 +187,73 @@ fn initialise(endpoint: ipc.Handle) bool {
|
||||
/// command did not complete — so there is one errno for all of them.
|
||||
const refused: isize = -envelope.ENOENT;
|
||||
|
||||
fn onGeometry(_: void, _: Invocation(void), answer: Answer(block_protocol.Geometry)) isize {
|
||||
// --- per-sender range confinement -------------------------------------------
|
||||
//
|
||||
// The volume manager confines each filesystem to the partition it mounts
|
||||
// (define_range); a confined sender addresses volume-relative LBAs from 0 and
|
||||
// the driver translates and bounds-checks against its range. A sender with no
|
||||
// range is unconfined — the whole device — which is the default until a range
|
||||
// is defined (behaviour-neutral for a single-volume boot), and is what the
|
||||
// volume manager itself uses to probe partitions before it confines anyone.
|
||||
|
||||
/// bound: filesystem processes confined to sub-ranges of this device at once
|
||||
/// decided-by: ours
|
||||
/// protects: the per-badge range table below
|
||||
/// at-limit: refuse - define_range past it returns -ENOSPC; a runaway detector for
|
||||
/// a compromised volume manager, not a real-partition limit (real disks carry a
|
||||
/// handful of volumes, far under this)
|
||||
/// observed-by: the -ENOSPC a define_range caller gets when the table is full
|
||||
const maximum_ranges = 64;
|
||||
|
||||
const Range = struct { used: bool = false, badge: u32 = 0, base: u64 = 0, count: u64 = 0 };
|
||||
var ranges = [_]Range{.{}} ** maximum_ranges;
|
||||
|
||||
/// The one party allowed to confine others — the first unconfined caller to
|
||||
/// define a range, which is the volume manager (it probes and confines every
|
||||
/// filesystem before handing it a channel). Without this, any unconfined
|
||||
/// opener could install a range for another live client's badge and silently
|
||||
/// redirect its I/O. Released on the controller's death so a restarted volume
|
||||
/// manager re-takes it.
|
||||
var range_controller: ?u32 = null;
|
||||
|
||||
fn rangeFor(badge: u32) ?*Range {
|
||||
for (&ranges) |*r| {
|
||||
if (r.used and r.badge == badge) return r;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Resolve a caller's transfer to an absolute LBA, or null if it falls outside
|
||||
/// the caller's confinement. Unconfined callers (no range) pass through against
|
||||
/// the whole device.
|
||||
///
|
||||
/// The bound is written to survive a hostile confined caller: `lba + count`
|
||||
/// would WRAP for an `lba` near u64 max, sail under a naive `> r.count` check,
|
||||
/// and translate to a wild absolute block — so the check is phrased as two
|
||||
/// subtractions that cannot overflow (`lba` within the range, and `count`
|
||||
/// within what remains). `r.base + lba` cannot overflow once `lba <= r.count`,
|
||||
/// because the volume manager sets `base + count` inside the device.
|
||||
fn resolveTransfer(sender: u32, lba: u64, count: u32) ?u64 {
|
||||
const r = rangeFor(sender) orelse return lba; // unconfined: whole device
|
||||
if (lba > r.count or r.count - lba < count) return null; // past the volume's end
|
||||
return r.base + lba;
|
||||
}
|
||||
|
||||
fn onGeometry(_: void, invocation: Invocation(void), answer: Answer(block_protocol.Geometry)) isize {
|
||||
// A confined caller sees ITS volume's size, not the device's — so a
|
||||
// filesystem mounts against the geometry it is actually allowed to touch.
|
||||
if (rangeFor(invocation.sender)) |r| {
|
||||
answer.set(.{ .block_size = block_size, .block_count = r.count });
|
||||
return 0;
|
||||
}
|
||||
answer.set(.{ .block_size = block_size, .block_count = block_count });
|
||||
return 0;
|
||||
}
|
||||
|
||||
fn onRead(_: void, invocation: Invocation(block_protocol.Transfer), answer: Answer(block_protocol.Transferred)) isize {
|
||||
const request = invocation.request;
|
||||
const cdb = scsi.read10(@intCast(request.lba), @intCast(request.count));
|
||||
const abs = resolveTransfer(invocation.sender, request.lba, request.count) orelse return refused;
|
||||
const cdb = scsi.read10(@intCast(abs), @intCast(request.count));
|
||||
if (!transact(&cdb, true, request.physical, request.count * block_size)) return refused;
|
||||
answer.set(.{ .count = request.count });
|
||||
return 0;
|
||||
@@ -182,12 +261,39 @@ fn onRead(_: void, invocation: Invocation(block_protocol.Transfer), answer: Answ
|
||||
|
||||
fn onWrite(_: void, invocation: Invocation(block_protocol.Transfer), answer: Answer(block_protocol.Transferred)) isize {
|
||||
const request = invocation.request;
|
||||
const cdb = scsi.write10(@intCast(request.lba), @intCast(request.count));
|
||||
const abs = resolveTransfer(invocation.sender, request.lba, request.count) orelse return refused;
|
||||
const cdb = scsi.write10(@intCast(abs), @intCast(request.count));
|
||||
if (!transact(&cdb, false, request.physical, request.count * block_size)) return refused;
|
||||
answer.set(.{ .count = request.count });
|
||||
return 0;
|
||||
}
|
||||
|
||||
/// Confine a sender to a block sub-range (the volume manager's per-volume grant).
|
||||
/// Refused if the CALLER is itself confined — a filesystem cannot widen its own
|
||||
/// range or confine anyone; only an unconfined party (the volume manager) may.
|
||||
fn onDefineRange(_: void, invocation: Invocation(block_protocol.DefineRange), _: Answer(void)) isize {
|
||||
const sender = invocation.sender;
|
||||
// A confined caller may never confine — no self-widening, no escape.
|
||||
if (rangeFor(sender) != null) return -envelope.EPERM;
|
||||
// Confinement has a single controller (the volume manager). Whoever defines
|
||||
// the first range takes it; only they may thereafter, so a second unconfined
|
||||
// opener cannot install a range for a badge it does not own.
|
||||
if (range_controller) |c| {
|
||||
if (sender != c) return -envelope.EPERM;
|
||||
} else {
|
||||
range_controller = sender;
|
||||
}
|
||||
const request = invocation.request;
|
||||
const slot = rangeFor(request.badge) orelse free: {
|
||||
for (&ranges) |*r| {
|
||||
if (!r.used) break :free r;
|
||||
}
|
||||
break :free null;
|
||||
} orelse return -envelope.ENOSPC;
|
||||
slot.* = .{ .used = true, .badge = request.badge, .base = request.base_lba, .count = request.block_count };
|
||||
return 0;
|
||||
}
|
||||
|
||||
/// SYNCHRONIZE CACHE: commit the device's write cache to flash. No data stage.
|
||||
/// Makes prior writes durable before a caller (init at shutdown) cuts power. A
|
||||
/// device without a volatile cache reports success anyway.
|
||||
@@ -204,18 +310,61 @@ fn onAttach(_: void, invocation: Invocation(void), _: Answer(void)) isize {
|
||||
return if (device.attachDma(handle)) 0 else refused;
|
||||
}
|
||||
|
||||
/// The reverse: forward the same region capability so the controller unbinds
|
||||
/// the buffer. As with attach, our copy stays the turn's to close.
|
||||
fn onDetach(_: void, invocation: Invocation(void), _: Answer(void)) isize {
|
||||
const handle = invocation.capability orelse return -envelope.EPROTO;
|
||||
return if (device.detachDma(handle)) 0 else refused;
|
||||
}
|
||||
|
||||
const handlers = Serve.Handlers{
|
||||
.geometry = onGeometry,
|
||||
.read = onRead,
|
||||
.write = onWrite,
|
||||
.flush = onFlush,
|
||||
.attach = onAttach,
|
||||
.detach = onDetach,
|
||||
.define_range = onDefineRange,
|
||||
};
|
||||
|
||||
fn onMessage(message: []const u8, reply: []u8, sender: u32, arrived: *ipc.Arrival) usize {
|
||||
// Peeked, never taken: `attach` forwards the capability and the controller's
|
||||
// binding takes its own reference, so this copy stays the turn's to close.
|
||||
return Serve.dispatch({}, handlers, message, sender, arrived.peek(), reply);
|
||||
// The harness peeks the arrival and takes it only if a handler (subscribe)
|
||||
// claimed it; attach/detach forward the capability without claiming, so the
|
||||
// turn still closes their copy after the controller took its own reference.
|
||||
return Serve.dispatch({}, handlers, message, sender, arrived, reply);
|
||||
}
|
||||
|
||||
/// A slow TEST UNIT READY poll: success means the medium is present, failure
|
||||
/// means it is not. On a transition, bump the counter and publish. (Sense-key
|
||||
/// inspection to tell "medium absent" from other transport errors is a
|
||||
/// refinement; a clean eject — what QEMU and a card reader produce — makes
|
||||
/// TEST UNIT READY report not-ready, which this reads correctly.)
|
||||
fn pollPresence() void {
|
||||
const ready = scsi.testUnitReady();
|
||||
const now = transact(&ready, false, 0, 0);
|
||||
if (now == medium_present) return;
|
||||
medium_present = now;
|
||||
medium_change_count +%= 1;
|
||||
std.log.info("medium {s}", .{if (now) "present" else "absent"});
|
||||
Serve.publish(.medium_changed, 0, .{ .present = @intFromBool(now), .change_count = medium_change_count });
|
||||
}
|
||||
|
||||
fn onNotification(badge: u64) void {
|
||||
const got = ipc.Received{ .len = 0, .badge = badge, .cap = null };
|
||||
if (got.isTimer()) {
|
||||
pollPresence();
|
||||
_ = time.timerOnce(service_endpoint, presence_poll_ms);
|
||||
return;
|
||||
}
|
||||
// A client died. The harness sweep (Serve.hooks) covers only the SUBSCRIBER
|
||||
// table; the per-badge range table is ours to reclaim, or confine/die cycles
|
||||
// (the medium-removal lifecycle) would exhaust it. A dead controller also
|
||||
// releases confinement authority to its successor.
|
||||
if (got.isChildExit()) {
|
||||
const dead = got.childProcessId();
|
||||
if (rangeFor(dead)) |r| r.* = .{};
|
||||
if (range_controller == dead) range_controller = null;
|
||||
}
|
||||
}
|
||||
|
||||
pub fn main(init: process.Init) void {
|
||||
@@ -233,6 +382,8 @@ pub fn main(init: process.Init) void {
|
||||
service.run(block_protocol.message_maximum, .{
|
||||
.init = initialise,
|
||||
.on_message = onMessage,
|
||||
.on_notification = onNotification,
|
||||
.subscribers = Serve.hooks,
|
||||
});
|
||||
// A failure exit (nonzero -> .aborted) tells the device manager to restart
|
||||
// us with backoff; a clean return means there was nothing to serve.
|
||||
|
||||
@@ -490,6 +490,14 @@ fn deviceIsHub(usb_device: *const library.Device) bool {
|
||||
fn tearDownPort(manager: ipc.Handle, engine: *library.Controller, port: u32) void {
|
||||
const usb_device = engine.deviceOnPort(port) orelse return;
|
||||
std.log.info("port {d} disconnected", .{port});
|
||||
// A hub yanked from a root port takes its whole subtree with it — children
|
||||
// first, recursively, exactly as tearDownHubDevice does for a hub leaving
|
||||
// one level down. Without this, the downstream slots stayed live against
|
||||
// vanished hardware, their class drivers were never reaped, and the
|
||||
// replugged hub found its port still occupied — nothing re-enumerated.
|
||||
if (usb_device.is_hub) {
|
||||
while (engine.nextChildOf(usb_device.slot_id, 0)) |child| tearDownHubDevice(manager, engine, child);
|
||||
}
|
||||
for (usb_device.interfaces[0..usb_device.interface_count]) |*interface| {
|
||||
if (interface.registered_device_id == 0) continue;
|
||||
reportRemoved(manager, (@as(u64, port) << 8) | interface.number, .{ .port = port, .interface = interface.number });
|
||||
@@ -585,6 +593,7 @@ const handlers = Serve.Handlers{
|
||||
.interrupt_subscribe = onInterruptSubscribe,
|
||||
.bulk = onBulk,
|
||||
.dma_attach = onDmaAttach,
|
||||
.dma_detach = onDmaDetach,
|
||||
};
|
||||
|
||||
/// open: the target is the class driver's assigned device id. Resolve it to an
|
||||
@@ -686,6 +695,14 @@ fn onDmaAttach(_: void, invocation: Invocation(void), _: Answer(void)) isize {
|
||||
return if (device.dmaBind(controller_id, handle)) 0 else refused;
|
||||
}
|
||||
|
||||
/// The reverse: the same region capability arrives again and the buffer leaves
|
||||
/// the controller's domain (`dma_unbind` matches the region — nothing was
|
||||
/// retained here between the two calls). The turn closes the arriving copy.
|
||||
fn onDmaDetach(_: void, invocation: Invocation(void), _: Answer(void)) isize {
|
||||
const handle = invocation.capability orelse return -envelope.EPROTO;
|
||||
return if (device.dmaUnbind(controller_id, handle)) 0 else refused;
|
||||
}
|
||||
|
||||
/// A timer tick or an MSI landed: drain the event ring, reconcile ports, and fan out.
|
||||
/// The timer arm re-arms itself (8 ms drain when polling, 250 ms reconcile under MSI);
|
||||
/// the MSI arm clears the interrupter's pending bit FIRST, then drains — so an event
|
||||
|
||||
@@ -2146,9 +2146,18 @@ fn systemFsMount(state: *architecture.CpuState) void {
|
||||
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
// The ownership gate. A LIVE owner's mount is its own: nobody else may
|
||||
// replace it — displacement-by-remount would be worse than unmounting.
|
||||
// A DEAD owner's mount is replaceable by anyone with a backend: that is
|
||||
// the restart story (a respawned filesystem is a new task retaking its
|
||||
// prefix). Kernel-installed mounts (owner 0) are never displaceable.
|
||||
if (vfs.mountOwner(prefix)) |owner| {
|
||||
const displaceable = owner != 0 and (owner == t.id or scheduler.taskByIdLocked(owner) == null);
|
||||
if (!displaceable) return failErr(state, ipc.EPERM);
|
||||
}
|
||||
const endpoint = ipc.resolveHandle(t, backend_handle) orelse return failErr(state, ipc.EBADF);
|
||||
endpoint.refcount += 1; // the mount table's reference
|
||||
if (!vfs.mountBackend(prefix, endpoint, rewrite)) {
|
||||
if (!vfs.mountBackend(prefix, endpoint, rewrite, t.id)) {
|
||||
ipc.dropRef(endpoint);
|
||||
return fail(state);
|
||||
}
|
||||
@@ -2166,6 +2175,12 @@ fn systemFsUnmount(state: *architecture.CpuState) void {
|
||||
if (!user_memory.copyFromUser(t.address_space, prefix_ptr, prefix)) return failErr(state, ipc.EFAULT);
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
// Only the mounting task unmounts. No dead-owner exception here: a dead
|
||||
// owner's mount is already being swept lazily by resolution, and a
|
||||
// stranger gains nothing legitimate by racing that — the restart story
|
||||
// goes through remount-replace, never through unmount.
|
||||
const owner = vfs.mountOwner(prefix) orelse return fail(state);
|
||||
if (owner != t.id) return failErr(state, ipc.EPERM);
|
||||
if (!vfs.unmount(prefix)) return fail(state);
|
||||
architecture.setSystemCallResult(state, 0);
|
||||
}
|
||||
|
||||
@@ -227,6 +227,10 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void {
|
||||
usbStorageTest(boot_information);
|
||||
} else if (eql(case, "fat-mount")) {
|
||||
fatMountTest(boot_information);
|
||||
} else if (eql(case, "exfat-volume")) {
|
||||
exfatVolumeTest(boot_information);
|
||||
} else if (eql(case, "volume-driver-restart")) {
|
||||
volumeDriverRestartTest(boot_information);
|
||||
} else if (eql(case, "device-list")) {
|
||||
deviceListTest(boot_information);
|
||||
} else if (eql(case, "pci-scan")) {
|
||||
@@ -265,6 +269,8 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void {
|
||||
deviceTransferTest(boot_information);
|
||||
} else if (eql(case, "device-authority")) {
|
||||
deviceAuthorityTest(boot_information);
|
||||
} else if (eql(case, "block-range")) {
|
||||
blockRangeTest(boot_information);
|
||||
} else if (eql(case, "device-manager")) {
|
||||
deviceManagerTest(boot_information);
|
||||
} else if (eql(case, "protocol-registry")) {
|
||||
@@ -3034,6 +3040,59 @@ fn fatMountTest(boot_information: *const BootInformation) void {
|
||||
result();
|
||||
}
|
||||
|
||||
/// The exFAT mount chain (S4): boot the full tree, which brings up the USB storage
|
||||
/// chain. The harness attaches a SECOND device — a data-only exFAT volume — beside
|
||||
/// the FAT boot volume, so the volume manager spawns the exfat service for it
|
||||
/// (content-routed, its id-path /volumes/exfat-<serial>). Then spawn exfat-test,
|
||||
/// which reads the seeded file and mutates through the mount. The reuse of the
|
||||
/// shared harness by a second engine is proven end to end here.
|
||||
fn exfatVolumeTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: exfat-volume\n", .{});
|
||||
if (boot_information.initial_ramdisk_len == 0) {
|
||||
check("bootloader handed over the initial_ramdisk", false);
|
||||
result();
|
||||
return;
|
||||
}
|
||||
const ramdisk = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||
const rd = initial_ramdisk.Reader.init(ramdisk) orelse {
|
||||
check("initial_ramdisk image is valid", false);
|
||||
result();
|
||||
return;
|
||||
};
|
||||
process.setInitialRamdisk(ramdisk);
|
||||
const init_ok = if (process.spawnBundled("/system/services/init")) true else |_| false;
|
||||
check("init spawned (boots the tree, incl. the volume manager)", init_ok);
|
||||
check("exfat-test client spawned", spawnNamed(rd, "exfat-test"));
|
||||
result();
|
||||
}
|
||||
|
||||
/// Per-sender range confinement (V2a, docs/volume-manager-plan.md): the fixture
|
||||
/// acquires the block channel, confines ITSELF to a sub-range, and asserts it
|
||||
/// cannot read past that range or widen it. Boots init in REGISTRY-ONLY mode
|
||||
/// plus the device manager (which brings up the USB storage chain) — deliberately
|
||||
/// NOT the full tree, because the volume manager would take the confinement
|
||||
/// controller first and refuse the fixture's define_range. Without it the fixture
|
||||
/// is the sole definer, exactly as the volume manager is in a real boot.
|
||||
fn blockRangeTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: block-range\n", .{});
|
||||
if (boot_information.initial_ramdisk_len == 0) {
|
||||
check("bootloader handed over the initial_ramdisk", false);
|
||||
result();
|
||||
return;
|
||||
}
|
||||
const ramdisk = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||
const rd = initial_ramdisk.Reader.init(ramdisk) orelse {
|
||||
check("initial_ramdisk image is valid", false);
|
||||
result();
|
||||
return;
|
||||
};
|
||||
process.setInitialRamdisk(ramdisk);
|
||||
check("registry (init) spawned", spawnRegistry(rd));
|
||||
check("device-manager spawned (boots the USB storage chain)", spawnNamed(rd, "device-manager"));
|
||||
check("block-range-test spawned", spawnNamedWithArg(rd, "block-range-test", "run"));
|
||||
result();
|
||||
}
|
||||
|
||||
fn bootServiceTreeTest(boot_information: *const BootInformation, comptime label: []const u8) void {
|
||||
log("DANOS-TEST-BEGIN: " ++ label ++ "\n", .{});
|
||||
if (boot_information.initial_ramdisk_len == 0) {
|
||||
@@ -3628,6 +3687,44 @@ fn displayReattachTest(boot_information: *const BootInformation) void {
|
||||
while (true) scheduler.yield();
|
||||
}
|
||||
|
||||
/// Storage-driver-crash rebuild (S5): the device manager runs in
|
||||
/// "test-storage-restart" mode and kills the usb-storage driver once, a moment
|
||||
/// after its volume has mounted. The driver's device stays in the tree, so the
|
||||
/// volume manager's presence poll alone would miss the death and leave fat wedged
|
||||
/// on a dead channel; its channel-liveness probe must notice, reap the volume, and
|
||||
/// rebuild on the restarted driver's fresh channel — a SECOND mount of the same
|
||||
/// id-path is the proof. (A pre-S5 manager, checking only device presence, never
|
||||
/// reaps, so the second mount never appears.)
|
||||
fn volumeDriverRestartTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: volume-driver-restart\n", .{});
|
||||
if (boot_information.initial_ramdisk_len == 0) {
|
||||
check("bootloader handed over an initial_ramdisk", false);
|
||||
result();
|
||||
return;
|
||||
}
|
||||
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||
check("initial_ramdisk image is valid", false);
|
||||
result();
|
||||
return;
|
||||
};
|
||||
process.setInitialRamdisk(image);
|
||||
_ = spawnRegistry(rd);
|
||||
var manager: u32 = 0;
|
||||
var i: u32 = 0;
|
||||
while (i < rd.count) : (i += 1) {
|
||||
const item = rd.entry(i) orelse continue;
|
||||
if (!eql(initial_ramdisk.basename(item.name), "device-manager")) continue;
|
||||
manager = process.spawnProcessSupervised(item.blob, 4, &.{ item.name, "test-storage-restart" }, scheduler.currentId(), null) catch 0;
|
||||
break;
|
||||
}
|
||||
check("device-manager spawned (test-storage-restart mode)", manager != 0);
|
||||
check("volume-manager spawned", spawnNamed(rd, "volume-manager"));
|
||||
check("fat-test client spawned", spawnNamed(rd, "fat-test"));
|
||||
scheduler.setPriority(1); // below the tree, so it runs
|
||||
while (true) scheduler.yield();
|
||||
}
|
||||
|
||||
/// Process arguments, end to end: spawn args-echo bare (its argv[0] is the
|
||||
/// initial-ramdisk name). Instance 1 sees argc == 1 and respawns itself through
|
||||
/// `system_spawn` with the extra arguments "alpha beta-42" — the syscall argument
|
||||
|
||||
+53
-12
@@ -55,7 +55,17 @@ fn tokenIndex(t: u64) u64 {
|
||||
|
||||
// --- the mount table ---------------------------------------------------------
|
||||
|
||||
pub const maximum_mounts = 8;
|
||||
/// bound: prefixes mounted in the kernel VFS table at once
|
||||
/// decided-by: ours
|
||||
/// protects: the `mounts` table below
|
||||
/// at-limit: refuse - installMount returns false and mountBackend propagates it;
|
||||
/// the mounting filesystem's harness logs "could not mount <prefix>" and the
|
||||
/// mount simply does not exist (no silent success). Budget: the initrd's
|
||||
/// top-level dirs (/system, /test) plus one id-path mount per volume and the
|
||||
/// system volume's two FHS rewrites — a few over the volume manager's
|
||||
/// maximum_volumes (16); 32 leaves headroom.
|
||||
/// observed-by: the harness "file-system: could not mount <prefix>" ring line
|
||||
pub const maximum_mounts = 32;
|
||||
const maximum_prefix = 64;
|
||||
const maximum_rewrite = 32;
|
||||
|
||||
@@ -69,6 +79,10 @@ const Mount = struct {
|
||||
backend: ?*ipc.Endpoint = null, // referenced while mounted
|
||||
rewrite: [maximum_rewrite]u8 = undefined,
|
||||
rewrite_len: usize = 0,
|
||||
// The task that mounted this prefix — the ownership `fs_unmount` and
|
||||
// remount-replace are gated on (storage-architecture.md, the lifecycle
|
||||
// rule). Zero for kernel-installed mounts, which no task may displace.
|
||||
owner: u32 = 0,
|
||||
|
||||
fn prefixSlice(self: *const Mount) []const u8 {
|
||||
return self.prefix[0..self.prefix_len];
|
||||
@@ -152,7 +166,10 @@ pub fn setInitialRamdisk(image: []const u8) void {
|
||||
for (directories[0..directory_count], 0..) |*d, index| {
|
||||
const parent = parentOf(d.slice());
|
||||
d.parent = directoryIndex(parent) orelse index;
|
||||
if (parent.len == 1) installMount(d.slice(), .kernel_initrd, null, "");
|
||||
// Boot-time install of one mount per top-level initrd dir (/system, /test):
|
||||
// provably few, far under maximum_mounts, so a full table here is
|
||||
// impossible — but discard the result explicitly rather than assume it.
|
||||
if (parent.len == 1) _ = installMount(d.slice(), .kernel_initrd, null, "");
|
||||
}
|
||||
}
|
||||
|
||||
@@ -163,7 +180,12 @@ fn directoryIndex(path: []const u8) ?usize {
|
||||
return null;
|
||||
}
|
||||
|
||||
fn installMount(prefix: []const u8, kind: MountKind, backend: ?*ipc.Endpoint, rewrite: []const u8) void {
|
||||
/// Install (or remount-replace) a prefix. Returns false when the table is full
|
||||
/// and no slot could be claimed — the caller must surface that, never report a
|
||||
/// dropped mount as success. A remount of an already-mounted prefix reuses its
|
||||
/// slot and always succeeds; a /protocol remount is refused-as-noop (returns
|
||||
/// true: the first mount stands, nothing is dropped).
|
||||
fn installMount(prefix: []const u8, kind: MountKind, backend: ?*ipc.Endpoint, rewrite: []const u8) bool {
|
||||
// Remount replaces: a restarted backend re-mounts its prefix.
|
||||
var slot: ?*Mount = null;
|
||||
for (&mounts) |*m| {
|
||||
@@ -172,19 +194,20 @@ fn installMount(prefix: []const u8, kind: MountKind, backend: ?*ipc.Endpoint, re
|
||||
// restarted FAT retakes /volumes/usb; letting it retake /protocol
|
||||
// would hand the whole naming layer to whoever asked second.
|
||||
// First mount wins, and init (PID 1) is always first.
|
||||
if (std.mem.eql(u8, prefix, protocol_root)) return;
|
||||
if (std.mem.eql(u8, prefix, protocol_root)) return true;
|
||||
if (m.backend) |old| ipc.dropRef(old);
|
||||
slot = m;
|
||||
break;
|
||||
}
|
||||
if (slot == null and !m.used) slot = m;
|
||||
}
|
||||
const m = slot orelse return;
|
||||
const m = slot orelse return false;
|
||||
m.* = .{ .used = true, .kind = kind, .backend = backend };
|
||||
@memcpy(m.prefix[0..prefix.len], prefix);
|
||||
m.prefix_len = prefix.len;
|
||||
@memcpy(m.rewrite[0..rewrite.len], rewrite);
|
||||
m.rewrite_len = rewrite.len;
|
||||
return true;
|
||||
}
|
||||
|
||||
// --- resolve -----------------------------------------------------------------
|
||||
@@ -375,11 +398,25 @@ fn refusesProtocolMount(prefix: []const u8) bool {
|
||||
return protocolBound(); // /protocol itself: first mount wins
|
||||
}
|
||||
|
||||
/// Mount `backend` at `prefix` with an optional backend-side `rewrite` prefix.
|
||||
/// The endpoint reference is taken by the caller (process.zig bumps it); refuses
|
||||
/// shadowing or replacing the initrd trees (/system, /test) — except the two
|
||||
/// carve-outs in `initrd_carve_outs`, the writable configuration/log subtrees.
|
||||
pub fn mountBackend(prefix: []const u8, backend: *ipc.Endpoint, rewrite: []const u8) bool {
|
||||
/// The task a backend mount at exactly `prefix` is recorded against, or null
|
||||
/// when nothing backend-shaped is mounted there. The syscall layer consults
|
||||
/// this before allowing a replace or an unmount — the ownership gate lives
|
||||
/// there, where the task table is; this table only remembers the fact.
|
||||
pub fn mountOwner(prefix: []const u8) ?u32 {
|
||||
for (&mounts) |*m| {
|
||||
if (m.used and m.kind == .backend and std.mem.eql(u8, m.prefixSlice(), prefix)) return m.owner;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Mount `backend` at `prefix` with an optional backend-side `rewrite` prefix,
|
||||
/// recorded against `owner`. The endpoint reference is taken by the caller
|
||||
/// (process.zig bumps it); refuses shadowing or replacing the initrd trees
|
||||
/// (/system, /test) — except the two carve-outs in `initrd_carve_outs`, the
|
||||
/// writable configuration/log subtrees. The replace-vs-refuse decision for an
|
||||
/// already-mounted prefix is the CALLER's (it can see task liveness); by the
|
||||
/// time this runs, replacing is decided.
|
||||
pub fn mountBackend(prefix: []const u8, backend: *ipc.Endpoint, rewrite: []const u8, owner: u32) bool {
|
||||
if (!isAbsolute(prefix) or prefix.len < 2 or prefix.len > maximum_prefix) return false;
|
||||
if (rewrite.len > maximum_rewrite) return false;
|
||||
if (refusesProtocolMount(prefix)) return false; // the registry's prefix is claimed once
|
||||
@@ -388,13 +425,17 @@ pub fn mountBackend(prefix: []const u8, backend: *ipc.Endpoint, rewrite: []const
|
||||
if (!isInitrdCarveOut(prefix)) return false;
|
||||
}
|
||||
}
|
||||
installMount(prefix, .backend, backend, rewrite);
|
||||
if (!installMount(prefix, .backend, backend, rewrite)) return false; // table full
|
||||
for (&mounts) |*m| {
|
||||
if (m.used and std.mem.eql(u8, m.prefixSlice(), prefix)) m.owner = owner;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
pub fn unmount(prefix: []const u8) bool {
|
||||
// Unmounting /protocol would delete the naming layer for everyone; nobody
|
||||
// may, init included. The mount lasts the boot.
|
||||
// may, init included. The mount lasts the boot. Ownership is checked by
|
||||
// the syscall layer (mountOwner) before this runs.
|
||||
if (std.mem.eql(u8, prefix, protocol_root)) return false;
|
||||
for (&mounts) |*m| {
|
||||
if (m.used and m.kind == .backend and std.mem.eql(u8, m.prefixSlice(), prefix)) {
|
||||
|
||||
@@ -170,6 +170,8 @@ var test_usb_killed = false;
|
||||
var test_pci_restart_mode = false;
|
||||
var test_scanout_restart_mode = false;
|
||||
var test_scanout_killed = false;
|
||||
var test_storage_restart_mode = false;
|
||||
var test_storage_killed = false;
|
||||
var test_kill_pid: u32 = 0;
|
||||
var test_kill_due_ns: u64 = 0;
|
||||
|
||||
@@ -600,6 +602,16 @@ fn onHello(_: void, invocation: Invocation(device_manager_protocol.Hello), _: An
|
||||
test_kill_due_ns = time.clock() + 1_500_000_000;
|
||||
_ = time.timerOnce(manager_endpoint, 1600);
|
||||
}
|
||||
// Storage-driver-crash drill (S5): once, a moment after usb-storage hellos —
|
||||
// long enough that its volume has mounted — kill it. The manager re-delegates
|
||||
// the still-present device to a restarted driver on a fresh channel; the volume
|
||||
// manager's channel-liveness probe must notice the dead channel and rebuild.
|
||||
if (test_storage_restart_mode and !test_storage_killed and std.mem.eql(u8, driver.name(), "/system/drivers/usb-storage")) {
|
||||
test_storage_killed = true;
|
||||
test_kill_pid = invocation.sender;
|
||||
test_kill_due_ns = time.clock() + 2_000_000_000; // after the ~0.6s mount
|
||||
_ = time.timerOnce(manager_endpoint, 2100);
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
@@ -706,6 +718,12 @@ fn onChildRemoved(_: void, invocation: Invocation(device_manager_protocol.ChildR
|
||||
for (&children) |*child| {
|
||||
if (child.used and child.parent == report.parent and child.bus_address == report.bus_address and child.reporter == invocation.sender) {
|
||||
std.log.info("child removed (device {d} port {d})", .{ child.parent, child.bus_address });
|
||||
// Unplug reaps exactly like reporter death (pruneChildrenOf): the
|
||||
// bound driver's device is gone and it cannot observe that — it
|
||||
// blocks on reports that will never come — and its stale entry
|
||||
// would make the dedupe refuse the respawn when the device is
|
||||
// PLUGGED BACK IN. Reap now, and a replug's re-report rebinds.
|
||||
reapDriverBoundTo(child.device_id);
|
||||
child.used = false;
|
||||
status = 0;
|
||||
}
|
||||
@@ -757,6 +775,7 @@ pub fn main(init: process.Init) void {
|
||||
test_usb_restart_mode = std.mem.eql(u8, mode, "test-usb-restart");
|
||||
test_pci_restart_mode = std.mem.eql(u8, mode, "test-pci-restart");
|
||||
test_scanout_restart_mode = std.mem.eql(u8, mode, "test-scanout-restart");
|
||||
test_storage_restart_mode = std.mem.eql(u8, mode, "test-storage-restart");
|
||||
}
|
||||
service.run(device_manager_protocol.message_maximum, .{
|
||||
.service = "device-manager",
|
||||
|
||||
@@ -0,0 +1,34 @@
|
||||
//! The exfat service as a binary package (docs/build-packages-plan.md):
|
||||
//! this file names the binary and EXACTLY the modules its source imports —
|
||||
//! build-support resolves each name from the domains this zon declares.
|
||||
|
||||
const std = @import("std");
|
||||
const build_support = @import("build-support");
|
||||
|
||||
pub fn build(b: *std.Build) void {
|
||||
const exe = build_support.userBinary(b, .{
|
||||
.name = "exfat",
|
||||
.root_source_file = b.path("exfat.zig"),
|
||||
.imports = &.{
|
||||
"block", "channel", "envelope", "file-system-harness",
|
||||
"ipc", "logging", "memory", "process",
|
||||
"time", "volume-manager-protocol",
|
||||
},
|
||||
});
|
||||
b.installArtifact(exe);
|
||||
|
||||
// Standalone `zig build test`; the root aggregate depends on this step.
|
||||
const test_step = b.step("test", "Run the exfat unit tests");
|
||||
for ([_][]const u8{
|
||||
"on-disk.zig", // exFAT on-disk struct sizes + geometry + checksums
|
||||
"engine.zig", // exFAT read/write over a RAM-backed image
|
||||
}) |test_root| {
|
||||
const unit_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path(test_root),
|
||||
.target = b.resolveTargetQuery(.{}),
|
||||
}),
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(unit_tests).step);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,16 @@
|
||||
.{
|
||||
.name = .exfat,
|
||||
.version = "0.0.0",
|
||||
.fingerprint = 0x5eafdf02d20f93dd, // Changing this has security and trust implications.
|
||||
.minimum_zig_version = "0.16.0",
|
||||
.dependencies = .{
|
||||
// build-support supplies the shared recipe; kernel is implicit in
|
||||
// every binary (the root shim + link script live there). The rest
|
||||
// are exactly the homes of this binary's declared imports.
|
||||
.@"build-support" = .{ .path = "../../../build-support" },
|
||||
.kernel = .{ .path = "../../../library/kernel" },
|
||||
.device = .{ .path = "../../../library/device" },
|
||||
.protocol = .{ .path = "../../../library/protocol" },
|
||||
},
|
||||
.paths = .{""},
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,200 @@
|
||||
//! system/services/exfat — the exFAT filesystem service. Like fat.zig, this is
|
||||
//! only the format-specific half: it finds its block device, sets up the DMA
|
||||
//! bounce buffer, mounts the exFAT engine on it, and hands the mounted volume to
|
||||
//! the shared filesystem harness (library/kernel/file-system-harness), which owns
|
||||
//! everything else — vfs serving, the open-node table, mount registration, the
|
||||
//! exit sweep, durable-on-close. The engine (engine.zig) is the pure,
|
||||
//! host-testable format code; on-disk.zig its byte layout.
|
||||
//!
|
||||
//! This service is a near-clone of fat.zig: the second engine reuses the harness
|
||||
//! wholesale, which is the reuse the storage architecture promised
|
||||
//! (docs/file-system-development/storage-architecture.md). The block data path
|
||||
//! never crosses IPC: a DMA bounce buffer is handed to the block driver by
|
||||
//! physical address, and the engine copies sectors in and out.
|
||||
|
||||
const std = @import("std");
|
||||
const channel = @import("channel");
|
||||
const volume_manager_protocol = @import("volume-manager-protocol");
|
||||
const ipc = @import("ipc");
|
||||
const process = @import("process");
|
||||
const block = @import("block");
|
||||
const memory = @import("memory");
|
||||
const logging = @import("logging");
|
||||
const time = @import("time");
|
||||
const engine = @import("engine.zig");
|
||||
const envelope = @import("envelope");
|
||||
const harness = @import("file-system-harness");
|
||||
|
||||
/// The serving harness, specialized for the exFAT engine. One volume per process.
|
||||
const Harness = harness.Server(engine.FileSystem);
|
||||
|
||||
// The engine's BlockDevice, backed by the `.block` driver plus a DMA bounce
|
||||
// buffer the driver reads/writes by physical address.
|
||||
const IpcBlock = struct {
|
||||
device: block.Device,
|
||||
bounce: memory.DmaRegion, // engine.max_transfer_sectors * 512 bytes
|
||||
|
||||
fn readBlocks(context: *anyopaque, lba: u64, count: u32, buffer: []u8) bool {
|
||||
const self: *IpcBlock = @ptrCast(@alignCast(context));
|
||||
if (count == 0 or count > engine.max_transfer_sectors) return false;
|
||||
const len = count * 512;
|
||||
if (!self.device.read(lba, count, self.bounce.physical)) return false;
|
||||
const source: [*]const u8 = @ptrFromInt(self.bounce.virtual);
|
||||
@memcpy(buffer[0..len], source[0..len]);
|
||||
return true;
|
||||
}
|
||||
fn writeBlocks(context: *anyopaque, lba: u64, count: u32, buffer: []const u8) bool {
|
||||
const self: *IpcBlock = @ptrCast(@alignCast(context));
|
||||
if (count == 0 or count > engine.max_transfer_sectors) return false;
|
||||
const len = count * 512;
|
||||
const destination: [*]u8 = @ptrFromInt(self.bounce.virtual);
|
||||
@memcpy(destination[0..len], buffer[0..len]);
|
||||
if (!self.device.write(lba, count, self.bounce.physical)) return false;
|
||||
device_dirty = true; // a block reached the device; a close will flush it
|
||||
return true;
|
||||
}
|
||||
};
|
||||
|
||||
var ipc_block: IpcBlock = undefined;
|
||||
// Set whenever a block is written, cleared when the device cache is flushed on a
|
||||
// file close — so writes are committed to stable media before a power-off.
|
||||
var device_dirty: bool = false;
|
||||
var filesystem: engine.FileSystem = undefined;
|
||||
/// The volume this exFAT process serves, its id given as argv[1] by the volume
|
||||
/// manager that spawned it. The startup hello names it so the manager returns the
|
||||
/// right volume's channel.
|
||||
var my_volume_id: u64 = 0;
|
||||
|
||||
/// The volume's own mount path, handed in as argv[2] by the volume manager: the
|
||||
/// volume's content id-path (e.g. /volumes/exfat-12345678). Defaults to
|
||||
/// /volumes/exfat only for a bare launch with no argument; the manager always
|
||||
/// passes it. The slice points into the entry block, valid for the process life.
|
||||
var volume_mount_prefix: []const u8 = "/volumes/exfat";
|
||||
|
||||
/// The mounts this volume installs: its own root, plus — only if it is the boot
|
||||
/// volume (it resolves /system/configuration) — the two FHS rewrites, so the
|
||||
/// logger's /system/logs stays decoupled from which volume backs it. Boot-volume
|
||||
/// detection is by content, so it works no matter which volume carries /system.
|
||||
/// bound: mounts one volume installs (its root + the two boot rewrites)
|
||||
/// decided-by: ours
|
||||
/// protects: the mount_specs array
|
||||
/// at-limit: truncate - unreachable today (fixed at 3); more configured mounts
|
||||
/// would need this raised, a deliberate change
|
||||
/// observed-by: a mount silently missing from the harness's mount log
|
||||
const maximum_mounts_per_volume = 4;
|
||||
var mount_specs: [maximum_mounts_per_volume]harness.MountSpec = undefined;
|
||||
|
||||
/// Get this volume's block channel from the volume manager (establishment by
|
||||
/// lineage — `block` is not a registry name). The manager spawned this process,
|
||||
/// confined it to its partition, and answers the hello with the channel; the
|
||||
/// channel is range-confined to this process's badge. Null until the manager has
|
||||
/// the volume ready — this retries.
|
||||
fn acquireVolume() ?block.Device {
|
||||
var attempts: u32 = 0;
|
||||
const vm = while (attempts < 500) : (attempts += 1) {
|
||||
if (channel.openEndpoint("volume-manager")) |handle| break handle;
|
||||
time.sleepMillis(20);
|
||||
} else return null;
|
||||
|
||||
attempts = 0;
|
||||
while (attempts < 500) : (attempts += 1) {
|
||||
var packet: [volume_manager_protocol.message_maximum]u8 = undefined;
|
||||
const framed = volume_manager_protocol.Protocol.encodeRequest(.hello, my_volume_id, .{}, &.{}, &packet) orelse return null;
|
||||
var reply: [volume_manager_protocol.message_maximum]u8 = undefined;
|
||||
const answered = ipc.callCap(vm, framed, &reply, null) catch return null;
|
||||
const status = envelope.statusOf(reply[0..answered.len]) orelse return null;
|
||||
if (status.status != 0) {
|
||||
if (answered.cap) |stray| _ = ipc.close(stray);
|
||||
_ = logging.write("/system/services/exfat: volume manager refused the hello\n");
|
||||
return null;
|
||||
}
|
||||
if (answered.cap) |bus| return .{ .endpoint = bus };
|
||||
// Acked with no channel: the volume is not ready yet — retry.
|
||||
time.sleepMillis(20);
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Durable-on-close: commit the device write cache if any block reached it since
|
||||
/// the last flush. The harness calls this on every close; the dirty check keeps
|
||||
/// it cheap.
|
||||
fn flushIfDirty() void {
|
||||
if (device_dirty) {
|
||||
_ = ipc_block.device.flush();
|
||||
device_dirty = false;
|
||||
}
|
||||
}
|
||||
|
||||
/// exFAT bring-up: find the block device, set up DMA, mount the engine, and hand
|
||||
/// the volume to the harness — or null to retry on the harness's timer.
|
||||
fn exfatBringUp(endpoint: ipc.Handle) ?Harness.Volume {
|
||||
_ = endpoint;
|
||||
const device = acquireVolume() orelse return null;
|
||||
const geometry = device.geometry() orelse {
|
||||
_ = logging.write("/system/services/exfat: block geometry unavailable\n");
|
||||
return null;
|
||||
};
|
||||
// Shareable so the buffer's capability can be attached down the chain, making
|
||||
// its physical addresses reachable under an enforcing IOMMU. No-op otherwise.
|
||||
const bounce = memory.dmaAlloc(engine.max_transfer_sectors * 512, memory.dma_coherent | memory.dma_shareable) orelse return null;
|
||||
if (bounce.handle) |handle| {
|
||||
// Attach, detach, and attach again: the round trip exercises BOTH verbs of
|
||||
// the DMA-window lifecycle through the whole chain on every boot.
|
||||
if (!device.attach(handle)) {
|
||||
_ = logging.write("/system/services/exfat: could not attach the DMA bounce buffer\n");
|
||||
return null;
|
||||
}
|
||||
if (!device.detach(handle)) {
|
||||
_ = logging.write("/system/services/exfat: could not detach the DMA bounce buffer\n");
|
||||
return null;
|
||||
}
|
||||
if (!device.attach(handle)) {
|
||||
_ = logging.write("/system/services/exfat: could not re-attach the DMA bounce buffer\n");
|
||||
return null;
|
||||
}
|
||||
_ = ipc.close(handle); // the binding holds its own reference now
|
||||
}
|
||||
ipc_block = .{ .device = device, .bounce = bounce };
|
||||
|
||||
const block_device = engine.BlockDevice{
|
||||
.context = &ipc_block,
|
||||
.block_size = geometry.block_size,
|
||||
.block_count = geometry.block_count,
|
||||
.readBlocksFn = IpcBlock.readBlocks,
|
||||
.writeBlocksFn = IpcBlock.writeBlocks,
|
||||
};
|
||||
filesystem = engine.FileSystem.mount(block_device) orelse {
|
||||
_ = logging.write("/system/services/exfat: not an exFAT filesystem\n");
|
||||
return null;
|
||||
};
|
||||
std.log.info("mounted exFAT ({d} clusters, {d} sectors/cluster, serial 0x{x})", .{ filesystem.geometry.cluster_count, filesystem.geometry.sectors_per_cluster, filesystem.geometry.volume_serial_number });
|
||||
|
||||
// The volume mounts at its id-path (argv[2]). The boot/system volume — the one
|
||||
// carrying the /system tree — additionally installs the two FHS rewrites, by
|
||||
// CONTENT: it resolves /system/configuration on its own media. A data volume
|
||||
// mounts only at its id-path and never shadows the running system.
|
||||
mount_specs[0] = .{ .prefix = volume_mount_prefix };
|
||||
var mount_count: usize = 1;
|
||||
if (filesystem.resolve("/system/configuration") != null) {
|
||||
std.log.info("volume {d} carries the system tree; backing /system/configuration and /system/logs", .{my_volume_id});
|
||||
mount_specs[1] = .{ .prefix = "/system/configuration", .rewrite = "/system/configuration" };
|
||||
mount_specs[2] = .{ .prefix = "/system/logs", .rewrite = "/system/logs" };
|
||||
mount_count = 3;
|
||||
} else {
|
||||
std.log.info("volume {d} is a data volume; mounted at {s}", .{ my_volume_id, volume_mount_prefix });
|
||||
}
|
||||
return .{ .engine = &filesystem, .mounts = mount_specs[0..mount_count], .flush = flushIfDirty };
|
||||
}
|
||||
|
||||
pub fn main(init: process.Init) void {
|
||||
// The volume manager spawns this process with its volume id as argv[1] and the
|
||||
// volume's mount path (its id-path) as argv[2].
|
||||
if (init.arguments.get(1)) |id| {
|
||||
my_volume_id = std.fmt.parseInt(u64, id, 10) catch 0;
|
||||
}
|
||||
if (init.arguments.get(2)) |prefix| {
|
||||
volume_mount_prefix = prefix;
|
||||
}
|
||||
_ = logging.write("/system/services/exfat: starting, waiting for a block device\n");
|
||||
Harness.run(.{ .bringUp = exfatBringUp });
|
||||
}
|
||||
@@ -0,0 +1,475 @@
|
||||
//! The on-disk layout of an exFAT filesystem — the Main Boot Sector (VBR) and the
|
||||
//! six 32-byte directory-entry types — as `align(1)` extern structs that bit-cast
|
||||
//! straight out of a sector (multi-byte fields little-endian). Pure data, plus the
|
||||
//! three exFAT checksums (boot region, up-case table, directory-entry set), the
|
||||
//! name hash, and the packed timestamp <-> Unix-epoch conversion. Host-testable.
|
||||
//!
|
||||
//! exFAT departs from FAT in three ways this file encodes: geometry lives in a
|
||||
//! MustBeZero-guarded VBR (byte 11 is zero, which is exactly why the FAT prober
|
||||
//! rejects an exFAT volume — it reads a zero bytes-per-sector); a file is a SET of
|
||||
//! entries (a File entry, a Stream Extension, and one or more File Name entries)
|
||||
//! validated by a rotate-right checksum; and names are compared case-folded through
|
||||
//! the volume's own on-disk up-case table (the folding itself lives in the engine,
|
||||
//! which holds the loaded table; the hash it feeds is here).
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
// exFAT's fixed on-disk widths — byte and UTF-16-unit counts the FORMAT defines,
|
||||
// not ceilings danos chooses. Named so the wire-format structs carry no bare
|
||||
// literal lengths; the values are facts of the spec.
|
||||
pub const entry_bytes: usize = 32; // every directory entry
|
||||
const jump_boot_bytes = 3;
|
||||
const filesystem_name_bytes = 8; // "EXFAT "
|
||||
const must_be_zero_bytes = 53; // the FAT-BPB overlap the format holds zero
|
||||
const boot_code_bytes = 390;
|
||||
const volume_label_units = 11;
|
||||
|
||||
// --- the Main Boot Sector (VBR, sector 0) ------------------------------------
|
||||
|
||||
/// The exFAT Main Boot Sector. `must_be_zero` (offset 11..64) overlaps where a
|
||||
/// FAT BPB keeps bytes-per-sector/sectors-per-cluster/etc.; exFAT holds it zero,
|
||||
/// so a FAT prober reading a zero bytes-per-sector rejects the volume — the
|
||||
/// mutual-exclusion the two engines rely on.
|
||||
pub const MainBootSector = extern struct {
|
||||
jump_boot: [jump_boot_bytes]u8, // 0
|
||||
filesystem_name: [filesystem_name_bytes]u8, // 3 "EXFAT "
|
||||
must_be_zero: [must_be_zero_bytes]u8, // 11
|
||||
partition_offset: u64 align(1), // 64 sectors, informational
|
||||
volume_length: u64 align(1), // 72 sectors
|
||||
fat_offset: u32 align(1), // 80 sectors from volume start
|
||||
fat_length: u32 align(1), // 84 sectors, per FAT
|
||||
cluster_heap_offset: u32 align(1), // 88 sectors from volume start
|
||||
cluster_count: u32 align(1), // 92
|
||||
first_cluster_of_root: u32 align(1), // 96
|
||||
volume_serial_number: u32 align(1), // 100
|
||||
filesystem_revision: u16 align(1), // 104
|
||||
volume_flags: u16 align(1), // 106 (skipped by the boot checksum)
|
||||
bytes_per_sector_shift: u8, // 108 9..12
|
||||
sectors_per_cluster_shift: u8, // 109
|
||||
number_of_fats: u8, // 110 1 (2 for TexFAT)
|
||||
drive_select: u8, // 111
|
||||
percent_in_use: u8, // 112 (skipped by the boot checksum)
|
||||
reserved: [7]u8, // 113
|
||||
boot_code: [boot_code_bytes]u8, // 120
|
||||
boot_signature: u16 align(1), // 510 0xAA55
|
||||
};
|
||||
|
||||
// --- directory entries (32 bytes each) ---------------------------------------
|
||||
|
||||
/// Entry-type bytes. The high bit (0x80) is InUse: a type with it clear is not in
|
||||
/// use, and 0x00 ends the directory. Deleting an entry clears bit 7 (0x85 -> 0x05).
|
||||
pub const entry_type_allocation_bitmap: u8 = 0x81;
|
||||
pub const entry_type_upcase_table: u8 = 0x82;
|
||||
pub const entry_type_volume_label: u8 = 0x83;
|
||||
pub const entry_type_file: u8 = 0x85;
|
||||
pub const entry_type_stream_extension: u8 = 0xC0;
|
||||
pub const entry_type_file_name: u8 = 0xC1;
|
||||
pub const entry_type_in_use_bit: u8 = 0x80;
|
||||
pub const entry_type_end_of_directory: u8 = 0x00;
|
||||
|
||||
/// A raw 32-byte entry, for type dispatch before it is reinterpreted as a
|
||||
/// specific entry.
|
||||
pub const RawEntry = extern struct {
|
||||
entry_type: u8,
|
||||
data: [entry_bytes - 1]u8,
|
||||
|
||||
pub fn inUse(self: RawEntry) bool {
|
||||
return self.entry_type & entry_type_in_use_bit != 0;
|
||||
}
|
||||
pub fn isEnd(self: RawEntry) bool {
|
||||
return self.entry_type == entry_type_end_of_directory;
|
||||
}
|
||||
};
|
||||
|
||||
/// 0x81 — the Allocation Bitmap: one bit per cluster (cluster 2 = bit 0), the
|
||||
/// authority for which clusters are free. The deepest departure from FAT, where
|
||||
/// the chain itself was the authority.
|
||||
pub const AllocationBitmapEntry = extern struct {
|
||||
entry_type: u8, // 0 0x81
|
||||
bitmap_flags: u8, // 1
|
||||
reserved: [18]u8, // 2
|
||||
first_cluster: u32 align(1), // 20
|
||||
data_length: u64 align(1), // 24
|
||||
};
|
||||
|
||||
/// 0x82 — the Up-case Table: the on-disk case-fold map (code unit -> uppercase),
|
||||
/// referenced by cluster and validated by `table_checksum`.
|
||||
pub const UpcaseTableEntry = extern struct {
|
||||
entry_type: u8, // 0 0x82
|
||||
reserved1: [3]u8, // 1
|
||||
table_checksum: u32 align(1), // 4
|
||||
reserved2: [12]u8, // 8
|
||||
first_cluster: u32 align(1), // 20
|
||||
data_length: u64 align(1), // 24
|
||||
};
|
||||
|
||||
/// 0x83 — the Volume Label (up to 11 UTF-16 units).
|
||||
pub const VolumeLabelEntry = extern struct {
|
||||
entry_type: u8, // 0 0x83
|
||||
character_count: u8, // 1
|
||||
volume_label: [volume_label_units]u16 align(1), // 2
|
||||
reserved: [8]u8, // 24
|
||||
};
|
||||
|
||||
/// 0x85 — the File entry: the head of a set, carrying the attributes,
|
||||
/// timestamps, the secondary-entry count, and the set checksum.
|
||||
pub const FileEntry = extern struct {
|
||||
entry_type: u8, // 0 0x85
|
||||
secondary_count: u8, // 1 stream (1) + name entries
|
||||
set_checksum: u16 align(1), // 2 over the whole set, skipping these two bytes
|
||||
file_attributes: u16 align(1), // 4
|
||||
reserved1: u16 align(1), // 6
|
||||
create_timestamp: u32 align(1), // 8
|
||||
last_modified_timestamp: u32 align(1), // 12
|
||||
last_accessed_timestamp: u32 align(1), // 16
|
||||
create_10ms: u8, // 20
|
||||
last_modified_10ms: u8, // 21
|
||||
create_utc_offset: u8, // 22
|
||||
last_modified_utc_offset: u8, // 23
|
||||
last_accessed_utc_offset: u8, // 24
|
||||
reserved2: [7]u8, // 25
|
||||
};
|
||||
|
||||
/// 0xC0 — the Stream Extension: the second entry of every file set, carrying the
|
||||
/// name length + hash and the data location (first cluster, sizes, the
|
||||
/// no-FAT-chain flag).
|
||||
pub const StreamExtensionEntry = extern struct {
|
||||
entry_type: u8, // 0 0xC0
|
||||
general_secondary_flags: u8, // 1
|
||||
reserved1: u8, // 2
|
||||
name_length: u8, // 3 UTF-16 units
|
||||
name_hash: u16 align(1), // 4
|
||||
reserved2: u16 align(1), // 6
|
||||
valid_data_length: u64 align(1), // 8
|
||||
reserved3: u32 align(1), // 16
|
||||
first_cluster: u32 align(1), // 20
|
||||
data_length: u64 align(1), // 24
|
||||
};
|
||||
|
||||
/// 0xC1 — a File Name entry: 15 UTF-16 units of the name; a set carries
|
||||
/// ceil(name_length / 15) of them.
|
||||
pub const FileNameEntry = extern struct {
|
||||
entry_type: u8, // 0 0xC1
|
||||
general_secondary_flags: u8, // 1
|
||||
file_name: [name_units_per_entry]u16 align(1), // 2
|
||||
};
|
||||
|
||||
pub const name_units_per_entry: usize = 15;
|
||||
|
||||
// General secondary flags (Stream Extension + File Name entries).
|
||||
pub const secondary_flag_allocation_possible: u8 = 0x01;
|
||||
pub const secondary_flag_no_fat_chain: u8 = 0x02;
|
||||
|
||||
// File attributes (same bit assignments as FAT).
|
||||
pub const attribute_read_only: u16 = 0x0001;
|
||||
pub const attribute_hidden: u16 = 0x0002;
|
||||
pub const attribute_system: u16 = 0x0004;
|
||||
pub const attribute_directory: u16 = 0x0010;
|
||||
pub const attribute_archive: u16 = 0x0020;
|
||||
|
||||
// FAT special cluster values (exFAT's FAT is 32-bit; used only for a fragmented
|
||||
// chain, i.e. when no_fat_chain is clear).
|
||||
pub const first_data_cluster: u32 = 2;
|
||||
pub const end_of_chain: u32 = 0xFFFFFFFF;
|
||||
pub const bad_cluster: u32 = 0xFFFFFFF7;
|
||||
|
||||
pub const boot_signature_offset: usize = 510; // 0x55 0xAA
|
||||
|
||||
// --- geometry ----------------------------------------------------------------
|
||||
|
||||
pub const Geometry = struct {
|
||||
bytes_per_sector: u32,
|
||||
sectors_per_cluster: u32,
|
||||
cluster_count: u32,
|
||||
fat_offset_sectors: u32, // from volume start
|
||||
fat_length_sectors: u32,
|
||||
cluster_heap_offset_sectors: u32, // from volume start
|
||||
first_cluster_of_root: u32,
|
||||
volume_serial_number: u32,
|
||||
volume_length_sectors: u64,
|
||||
number_of_fats: u32,
|
||||
};
|
||||
|
||||
/// Derive the geometry from a Main Boot Sector. Returns null unless it is a
|
||||
/// plausible exFAT VBR: the "EXFAT " name, an all-zero MustBeZero region, the
|
||||
/// 0xAA55 signature, and sane shifts. Accepting ONLY these is what keeps exFAT and
|
||||
/// FAT from ever claiming each other's volumes.
|
||||
pub fn geometryOf(sector: []const u8) ?Geometry {
|
||||
if (sector.len < 512) return null;
|
||||
if (sector[boot_signature_offset] != 0x55 or sector[boot_signature_offset + 1] != 0xAA) return null;
|
||||
const vbr = std.mem.bytesToValue(MainBootSector, sector[0..@sizeOf(MainBootSector)]);
|
||||
if (!std.mem.eql(u8, &vbr.filesystem_name, "EXFAT ")) return null;
|
||||
for (vbr.must_be_zero) |byte| if (byte != 0) return null;
|
||||
if (vbr.bytes_per_sector_shift < 9 or vbr.bytes_per_sector_shift > 12) return null;
|
||||
// The exFAT spec caps a cluster at 2^25 bytes (32 MiB): bytes-per-sector-shift
|
||||
// plus sectors-per-cluster-shift must not exceed 25. Enforcing it here is also
|
||||
// what keeps the engine's u32 cluster-byte arithmetic (sectors_per_cluster *
|
||||
// 512) from overflowing on a crafted VBR off untrusted removable media.
|
||||
if (@as(u16, vbr.bytes_per_sector_shift) + vbr.sectors_per_cluster_shift > 25) return null;
|
||||
// cluster_count is capped at 0xFFFFFFF5 (the spec's ClusterCount maximum), so
|
||||
// cluster_count + first_data_cluster cannot overflow u32 in the bounds checks.
|
||||
if (vbr.number_of_fats == 0 or vbr.cluster_count == 0 or vbr.cluster_count > 0xFFFFFFF5) return null;
|
||||
if (vbr.first_cluster_of_root < first_data_cluster) return null;
|
||||
return .{
|
||||
.bytes_per_sector = @as(u32, 1) << @intCast(vbr.bytes_per_sector_shift),
|
||||
.sectors_per_cluster = @as(u32, 1) << @intCast(vbr.sectors_per_cluster_shift),
|
||||
.cluster_count = vbr.cluster_count,
|
||||
.fat_offset_sectors = vbr.fat_offset,
|
||||
.fat_length_sectors = vbr.fat_length,
|
||||
.cluster_heap_offset_sectors = vbr.cluster_heap_offset,
|
||||
.first_cluster_of_root = vbr.first_cluster_of_root,
|
||||
.volume_serial_number = vbr.volume_serial_number,
|
||||
.volume_length_sectors = vbr.volume_length,
|
||||
.number_of_fats = vbr.number_of_fats,
|
||||
};
|
||||
}
|
||||
|
||||
// --- checksums and the name hash ---------------------------------------------
|
||||
|
||||
/// The directory-entry-SET checksum (a File entry's `set_checksum`), a 16-bit
|
||||
/// rotate-right sum over every byte of the set, skipping the two checksum bytes
|
||||
/// themselves (offset 2..3 of the first entry). `entries` is the whole set:
|
||||
/// (secondary_count + 1) * 32 bytes.
|
||||
pub fn setChecksum(entries: []const u8) u16 {
|
||||
var checksum: u16 = 0;
|
||||
for (entries, 0..) |byte, i| {
|
||||
if (i == 2 or i == 3) continue;
|
||||
checksum = std.math.rotr(u16, checksum, 1) +% byte;
|
||||
}
|
||||
return checksum;
|
||||
}
|
||||
|
||||
/// The up-case-table checksum (an Up-case entry's `table_checksum`), a 32-bit
|
||||
/// rotate-right sum over the table's on-disk bytes.
|
||||
pub fn upcaseChecksum(table_bytes: []const u8) u32 {
|
||||
var checksum: u32 = 0;
|
||||
for (table_bytes) |byte| checksum = std.math.rotr(u32, checksum, 1) +% byte;
|
||||
return checksum;
|
||||
}
|
||||
|
||||
/// The boot-region checksum — the u32 the checksum sector repeats — a 32-bit
|
||||
/// rotate-right sum over the first eleven sectors, skipping VolumeFlags (offset
|
||||
/// 106..107) and PercentInUse (offset 112) of the first sector.
|
||||
pub fn bootChecksum(region: []const u8) u32 {
|
||||
var checksum: u32 = 0;
|
||||
for (region, 0..) |byte, i| {
|
||||
if (i == 106 or i == 107 or i == 112) continue;
|
||||
checksum = std.math.rotr(u32, checksum, 1) +% byte;
|
||||
}
|
||||
return checksum;
|
||||
}
|
||||
|
||||
/// The name hash a Stream entry carries: a 16-bit rotate-right sum over the
|
||||
/// UP-CASED name's bytes (low byte then high byte of each UTF-16 unit). The caller
|
||||
/// up-cases through the volume's table first; a mismatch lets a lookup reject a
|
||||
/// name without reading its File Name entries.
|
||||
pub fn nameHash(upcased: []const u16) u16 {
|
||||
var hash: u16 = 0;
|
||||
for (upcased) |unit| {
|
||||
hash = std.math.rotr(u16, hash, 1) +% @as(u8, @truncate(unit));
|
||||
hash = std.math.rotr(u16, hash, 1) +% @as(u8, @truncate(unit >> 8));
|
||||
}
|
||||
return hash;
|
||||
}
|
||||
|
||||
// --- timestamps --------------------------------------------------------------
|
||||
//
|
||||
// exFAT packs a timestamp into one u32: the high 16 bits are a DOS date
|
||||
// (year-1980 | month | day), the low 16 a DOS time (hour | minute | second/2).
|
||||
// Same field layout as FAT, so the epoch math matches; there is no timezone in
|
||||
// the packed value (a separate UTC-offset byte carries that, which danos leaves
|
||||
// zero = UTC).
|
||||
|
||||
fn isLeapYear(year: u32) bool {
|
||||
return (year % 4 == 0 and year % 100 != 0) or (year % 400 == 0);
|
||||
}
|
||||
|
||||
const days_in_month = [_]u8{ 31, 28, 31, 30, 31, 30, 31, 31, 30, 31, 30, 31 };
|
||||
|
||||
/// Convert a packed exFAT timestamp to Unix epoch seconds (UTC). 0 for unset.
|
||||
pub fn timestampToEpoch(timestamp: u32) u64 {
|
||||
if (timestamp == 0) return 0;
|
||||
const date: u32 = timestamp >> 16;
|
||||
const time: u32 = timestamp & 0xFFFF;
|
||||
const day: u32 = date & 0x1F;
|
||||
const month: u32 = (date >> 5) & 0x0F;
|
||||
const year: u32 = 1980 + (date >> 9);
|
||||
if (month < 1 or month > 12 or day < 1) return 0;
|
||||
const second: u32 = (time & 0x1F) * 2;
|
||||
const minute: u32 = (time >> 5) & 0x3F;
|
||||
const hour: u32 = (time >> 11) & 0x1F;
|
||||
|
||||
var days: u64 = 0;
|
||||
var y: u32 = 1970;
|
||||
while (y < year) : (y += 1) days += if (isLeapYear(y)) 366 else 365;
|
||||
var m: u32 = 1;
|
||||
while (m < month) : (m += 1) {
|
||||
days += days_in_month[m - 1];
|
||||
if (m == 2 and isLeapYear(year)) days += 1;
|
||||
}
|
||||
days += day - 1;
|
||||
return ((days * 24 + hour) * 60 + minute) * 60 + second;
|
||||
}
|
||||
|
||||
/// Convert Unix epoch seconds (UTC) to a packed exFAT timestamp. 0 for epoch 0 or
|
||||
/// any time before 1980 (unrepresentable).
|
||||
pub fn epochToTimestamp(epoch: u64) u32 {
|
||||
if (epoch == 0) return 0;
|
||||
var remaining = epoch;
|
||||
const second: u32 = @intCast(remaining % 60);
|
||||
remaining /= 60;
|
||||
const minute: u32 = @intCast(remaining % 60);
|
||||
remaining /= 60;
|
||||
const hour: u32 = @intCast(remaining % 24);
|
||||
remaining /= 24;
|
||||
var days: u32 = @intCast(remaining);
|
||||
|
||||
var year: u32 = 1970;
|
||||
while (true) {
|
||||
const year_days: u32 = if (isLeapYear(year)) 366 else 365;
|
||||
if (days < year_days) break;
|
||||
days -= year_days;
|
||||
year += 1;
|
||||
}
|
||||
if (year < 1980) return 0;
|
||||
var month: u32 = 1;
|
||||
while (true) {
|
||||
var month_days: u32 = days_in_month[month - 1];
|
||||
if (month == 2 and isLeapYear(year)) month_days += 1;
|
||||
if (days < month_days) break;
|
||||
days -= month_days;
|
||||
month += 1;
|
||||
}
|
||||
const day = days + 1;
|
||||
const date: u32 = ((year - 1980) << 9) | (month << 5) | day;
|
||||
const time: u32 = (hour << 11) | (minute << 5) | (second / 2);
|
||||
return (date << 16) | time;
|
||||
}
|
||||
|
||||
// --- tests -------------------------------------------------------------------
|
||||
|
||||
test "on-disk struct sizes match the specification" {
|
||||
try std.testing.expectEqual(@as(usize, 512), @sizeOf(MainBootSector));
|
||||
try std.testing.expectEqual(@as(usize, 32), @sizeOf(RawEntry));
|
||||
try std.testing.expectEqual(@as(usize, 32), @sizeOf(AllocationBitmapEntry));
|
||||
try std.testing.expectEqual(@as(usize, 32), @sizeOf(UpcaseTableEntry));
|
||||
try std.testing.expectEqual(@as(usize, 32), @sizeOf(VolumeLabelEntry));
|
||||
try std.testing.expectEqual(@as(usize, 32), @sizeOf(FileEntry));
|
||||
try std.testing.expectEqual(@as(usize, 32), @sizeOf(StreamExtensionEntry));
|
||||
try std.testing.expectEqual(@as(usize, 32), @sizeOf(FileNameEntry));
|
||||
}
|
||||
|
||||
test "MainBootSector field offsets" {
|
||||
try std.testing.expectEqual(@as(usize, 3), @offsetOf(MainBootSector, "filesystem_name"));
|
||||
try std.testing.expectEqual(@as(usize, 11), @offsetOf(MainBootSector, "must_be_zero"));
|
||||
try std.testing.expectEqual(@as(usize, 80), @offsetOf(MainBootSector, "fat_offset"));
|
||||
try std.testing.expectEqual(@as(usize, 88), @offsetOf(MainBootSector, "cluster_heap_offset"));
|
||||
try std.testing.expectEqual(@as(usize, 96), @offsetOf(MainBootSector, "first_cluster_of_root"));
|
||||
try std.testing.expectEqual(@as(usize, 106), @offsetOf(MainBootSector, "volume_flags"));
|
||||
try std.testing.expectEqual(@as(usize, 112), @offsetOf(MainBootSector, "percent_in_use"));
|
||||
try std.testing.expectEqual(@as(usize, 510), @offsetOf(MainBootSector, "boot_signature"));
|
||||
// The Stream Extension's data location must sit where the spec places it.
|
||||
try std.testing.expectEqual(@as(usize, 20), @offsetOf(StreamExtensionEntry, "first_cluster"));
|
||||
try std.testing.expectEqual(@as(usize, 24), @offsetOf(StreamExtensionEntry, "data_length"));
|
||||
}
|
||||
|
||||
test "geometryOf accepts exFAT and the MustBeZero guard rejects a FAT-shaped sector" {
|
||||
var sector = [_]u8{0} ** 512;
|
||||
@memcpy(sector[3..11], "EXFAT ");
|
||||
sector[510] = 0x55;
|
||||
sector[511] = 0xAA;
|
||||
// fat_offset=128, fat_length=64, cluster_heap_offset=256, cluster_count=1000,
|
||||
// root cluster=5, bytes/sector=512 (shift 9), sectors/cluster=8 (shift 3), 1 FAT.
|
||||
std.mem.writeInt(u32, sector[80..84], 128, .little);
|
||||
std.mem.writeInt(u32, sector[84..88], 64, .little);
|
||||
std.mem.writeInt(u32, sector[88..92], 256, .little);
|
||||
std.mem.writeInt(u32, sector[92..96], 1000, .little);
|
||||
std.mem.writeInt(u32, sector[96..100], 5, .little);
|
||||
sector[108] = 9; // bytes_per_sector_shift
|
||||
sector[109] = 3; // sectors_per_cluster_shift
|
||||
sector[110] = 1; // number_of_fats
|
||||
const geo = geometryOf(§or) orelse return error.ShouldParse;
|
||||
try std.testing.expectEqual(@as(u32, 512), geo.bytes_per_sector);
|
||||
try std.testing.expectEqual(@as(u32, 8), geo.sectors_per_cluster);
|
||||
try std.testing.expectEqual(@as(u32, 1000), geo.cluster_count);
|
||||
try std.testing.expectEqual(@as(u32, 5), geo.first_cluster_of_root);
|
||||
|
||||
// A non-zero byte in MustBeZero (where a FAT BPB keeps bytes-per-sector) is
|
||||
// rejected — the mutual exclusion between the engines.
|
||||
sector[11] = 0x02;
|
||||
try std.testing.expect(geometryOf(§or) == null);
|
||||
sector[11] = 0;
|
||||
// Wrong name is rejected too.
|
||||
sector[3] = 'F';
|
||||
try std.testing.expect(geometryOf(§or) == null);
|
||||
sector[3] = 'E';
|
||||
}
|
||||
|
||||
test "geometryOf rejects crafted VBRs that would overflow u32 cluster arithmetic" {
|
||||
var sector = [_]u8{0} ** 512;
|
||||
@memcpy(sector[3..11], "EXFAT ");
|
||||
sector[510] = 0x55;
|
||||
sector[511] = 0xAA;
|
||||
std.mem.writeInt(u32, sector[92..96], 1000, .little); // cluster_count
|
||||
std.mem.writeInt(u32, sector[96..100], 5, .little); // root cluster
|
||||
sector[108] = 9; // bytes_per_sector_shift
|
||||
sector[110] = 1; // number_of_fats
|
||||
// A cluster shift past the exFAT ceiling (9 + 17 = 26 > 25) would make
|
||||
// sectors_per_cluster * 512 overflow u32 — rejected.
|
||||
sector[109] = 17;
|
||||
try std.testing.expect(geometryOf(§or) == null);
|
||||
sector[109] = 3; // sane again
|
||||
try std.testing.expect(geometryOf(§or) != null);
|
||||
// cluster_count above the spec maximum (0xFFFFFFF5) would overflow
|
||||
// cluster_count + first_data_cluster in the bounds checks — rejected.
|
||||
std.mem.writeInt(u32, sector[92..96], 0xFFFFFFFF, .little);
|
||||
try std.testing.expect(geometryOf(§or) == null);
|
||||
}
|
||||
|
||||
test "set checksum skips its own two bytes and depends on the rest" {
|
||||
var set = [_]u8{0} ** 64; // a File entry + one secondary
|
||||
set[0] = entry_type_file;
|
||||
set[1] = 1;
|
||||
set[4] = 0x20; // an attribute byte
|
||||
set[40] = 0xAB; // a byte in the secondary entry
|
||||
const base = setChecksum(&set);
|
||||
// Changing the checksum field itself must NOT change the computed checksum.
|
||||
set[2] = 0xFF;
|
||||
set[3] = 0xEE;
|
||||
try std.testing.expectEqual(base, setChecksum(&set));
|
||||
// Changing any other byte MUST change it.
|
||||
set[4] = 0x21;
|
||||
try std.testing.expect(setChecksum(&set) != base);
|
||||
}
|
||||
|
||||
test "name hash is deterministic and order-sensitive" {
|
||||
const readme = [_]u16{ 'R', 'E', 'A', 'D', 'M', 'E' };
|
||||
const different = [_]u16{ 'E', 'R', 'A', 'D', 'M', 'E' };
|
||||
try std.testing.expectEqual(nameHash(&readme), nameHash(&readme));
|
||||
try std.testing.expect(nameHash(&readme) != nameHash(&different));
|
||||
}
|
||||
|
||||
test "boot checksum skips VolumeFlags and PercentInUse" {
|
||||
var region = [_]u8{0} ** 1536; // three 512-byte sectors is enough to exercise the skips
|
||||
region[64] = 0x11;
|
||||
const base = bootChecksum(®ion);
|
||||
for ([_]usize{ 106, 107, 112 }) |skipped| {
|
||||
var copy = region;
|
||||
copy[skipped] = 0xFF;
|
||||
try std.testing.expectEqual(base, bootChecksum(©));
|
||||
}
|
||||
var copy = region;
|
||||
copy[108] = 0xFF; // a non-skipped byte
|
||||
try std.testing.expect(bootChecksum(©) != base);
|
||||
}
|
||||
|
||||
test "exFAT timestamp <-> Unix epoch round trip" {
|
||||
for ([_]u64{ 1_577_836_800, 1_700_000_000, 1_262_304_000, 1_783_971_244 }) |epoch| {
|
||||
try std.testing.expectEqual(epoch, timestampToEpoch(epochToTimestamp(epoch)));
|
||||
}
|
||||
// 1577836800 is 2020-01-01 00:00:00 UTC.
|
||||
const stamp = epochToTimestamp(1_577_836_800);
|
||||
try std.testing.expectEqual(@as(u32, 2020), 1980 + (stamp >> 16 >> 9));
|
||||
try std.testing.expectEqual(@as(u64, 0), timestampToEpoch(0));
|
||||
try std.testing.expectEqual(@as(u32, 0), epochToTimestamp(0));
|
||||
}
|
||||
@@ -10,10 +10,9 @@ pub fn build(b: *std.Build) void {
|
||||
.name = "fat",
|
||||
.root_source_file = b.path("fat.zig"),
|
||||
.imports = &.{
|
||||
"block", "channel", "device-manager-protocol", "driver",
|
||||
"envelope", "file-system", "ipc", "logging",
|
||||
"memory", "process", "service", "time",
|
||||
"vfs-protocol",
|
||||
"block", "channel", "envelope", "file-system-harness",
|
||||
"ipc", "logging", "memory", "process",
|
||||
"time", "volume-manager-protocol",
|
||||
},
|
||||
});
|
||||
b.installArtifact(exe);
|
||||
|
||||
@@ -1657,3 +1657,21 @@ test "short-name checksum matches the reference vector" {
|
||||
const c = FileSystem.shortChecksum("REDAME TXT".*);
|
||||
try std.testing.expect(a != c);
|
||||
}
|
||||
|
||||
test "the FAT engine rejects an exFAT volume (mutual exclusion at mount)" {
|
||||
const allocator = std.testing.allocator;
|
||||
const bytes = try allocator.alloc(u8, 5000 * sector_size);
|
||||
defer allocator.free(bytes);
|
||||
@memset(bytes, 0);
|
||||
// An exFAT boot sector: the "EXFAT " name and 0x55AA, but MustBeZero (offset
|
||||
// 11, where a FAT BPB keeps bytes-per-sector) stays zero — so this engine's
|
||||
// geometryOf reads a zero bytes-per-sector and rejects it.
|
||||
@memcpy(bytes[3..11], "EXFAT ");
|
||||
bytes[on_disk.boot_signature_offset] = 0x55;
|
||||
bytes[on_disk.boot_signature_offset + 1] = 0xAA;
|
||||
var disk = RamDisk{ .bytes = bytes };
|
||||
try std.testing.expect(FileSystem.mount(disk.device()) == null);
|
||||
// Control: a real FAT16 mounts.
|
||||
formatFat16(bytes);
|
||||
try std.testing.expect(FileSystem.mount(disk.device()) != null);
|
||||
}
|
||||
|
||||
+125
-351
@@ -1,40 +1,31 @@
|
||||
//! system/services/fat — the FAT filesystem server. Spawned as a boot service, it
|
||||
//! opens the block device (a USB stick via usb-storage) under `.block`, mounts the
|
||||
//! FAT filesystem on it (the pure engine in engine.zig), and mounts itself into
|
||||
//! the VFS at /volumes/usb. From then on the VFS forwards every open/read/write/
|
||||
//! status/readdir/close under /volumes/usb to this server, which serves the same
|
||||
//! vfs-protocol as a backend — turning block reads into file reads.
|
||||
//! system/services/fat — the FAT filesystem service. This is FAT's FAT-specific
|
||||
//! half: it finds its block device, sets up the DMA bounce buffer, mounts the
|
||||
//! FAT engine on it, and hands the mounted volume to the shared filesystem
|
||||
//! harness (library/kernel/file-system-harness), which owns everything else —
|
||||
//! the vfs-protocol serving, the open-node table, mount registration, the exit
|
||||
//! sweep, durable-on-close. The engine (engine.zig) is the pure, host-testable
|
||||
//! format code; on-disk.zig its byte layout. A second filesystem reuses the
|
||||
//! harness and supplies its own engine
|
||||
//! (docs/file-system-development/storage-architecture.md).
|
||||
//!
|
||||
//! The block data path never crosses IPC: a DMA bounce buffer is handed to the
|
||||
//! block driver by physical address, and the engine copies sectors in and out of
|
||||
//! it.
|
||||
//! block driver by physical address, and the engine copies sectors in and out.
|
||||
|
||||
const std = @import("std");
|
||||
const channel = @import("channel");
|
||||
const device_manager_protocol = @import("device-manager-protocol");
|
||||
const driver = @import("driver");
|
||||
const volume_manager_protocol = @import("volume-manager-protocol");
|
||||
const ipc = @import("ipc");
|
||||
const process = @import("process");
|
||||
const service = @import("service");
|
||||
const time = @import("time");
|
||||
const block = @import("block");
|
||||
const file_system = @import("file-system");
|
||||
const memory = @import("memory");
|
||||
const logging = @import("logging");
|
||||
const time = @import("time");
|
||||
const engine = @import("engine.zig");
|
||||
const on_disk = @import("on-disk.zig");
|
||||
const envelope = @import("envelope");
|
||||
const vfs_protocol = @import("vfs-protocol");
|
||||
const harness = @import("file-system-harness");
|
||||
|
||||
/// The generated vfs dispatch, bound to this server. There is one FAT volume per
|
||||
/// process, so the handler context is empty and the state stays where it was: in
|
||||
/// this file's globals.
|
||||
const Serve = vfs_protocol.Protocol.Provider(void);
|
||||
|
||||
const Invocation = envelope.Invocation;
|
||||
const Answer = envelope.Answer;
|
||||
|
||||
const mount_point = "/volumes/usb";
|
||||
/// The serving harness, specialized for the FAT engine. One volume per process.
|
||||
const Harness = harness.Server(engine.FileSystem);
|
||||
|
||||
// The engine's BlockDevice, backed by the `.block` driver plus a DMA bounce
|
||||
// buffer the driver reads/writes by physical address.
|
||||
@@ -68,137 +59,102 @@ var ipc_block: IpcBlock = undefined;
|
||||
// file close — so writes are committed to stable media before a power-off.
|
||||
var device_dirty: bool = false;
|
||||
var filesystem: engine.FileSystem = undefined;
|
||||
/// The volume this FAT process serves, its id given as argv[1] by the volume
|
||||
/// manager that spawned it. The startup hello names it so the manager returns
|
||||
/// the right volume's channel.
|
||||
var my_volume_id: u64 = 0;
|
||||
|
||||
// Open handles clients hold against this backend: each maps a node id to a
|
||||
// resolved engine node, and to the client that opened it. `owner` is the
|
||||
// kernel-stamped badge of the opening task — the only source identity there is.
|
||||
const OpenNode = struct { used: bool = false, node: engine.Node = undefined, owner: u32 = 0 };
|
||||
var open_nodes = [_]OpenNode{.{}} ** 32;
|
||||
/// The volume's own mount path, handed in as argv[2] by the volume manager: the
|
||||
/// volume's content id-path (e.g. /volumes/fat-12345678). Defaults to
|
||||
/// /volumes/usb only for a bare launch with no argument; the manager always
|
||||
/// passes it. The slice points into the entry block, valid for the process life.
|
||||
var volume_mount_prefix: []const u8 = "/volumes/usb";
|
||||
|
||||
fn allocOpen() ?usize {
|
||||
for (&open_nodes, 0..) |*o, i| {
|
||||
if (!o.used) return i;
|
||||
/// The mounts this volume installs: its own root, plus — only if it is the boot
|
||||
/// volume (it resolves /system/configuration) — the two FHS rewrites, so the
|
||||
/// logger's /system/logs stays decoupled from which volume backs it. Boot-volume
|
||||
/// detection is by content, so it works no matter which volume carries /system.
|
||||
/// bound: mounts one volume installs (its root + the two boot rewrites)
|
||||
/// decided-by: ours
|
||||
/// protects: the mount_specs array
|
||||
/// at-limit: truncate - unreachable today (fixed at 3); more configured mounts
|
||||
/// would need this raised, a deliberate change
|
||||
/// observed-by: a mount silently missing from the harness's mount log
|
||||
const maximum_mounts_per_volume = 4;
|
||||
var mount_specs: [maximum_mounts_per_volume]harness.MountSpec = undefined;
|
||||
|
||||
/// Get this volume's block channel from the volume manager (establishment by
|
||||
/// lineage, communication.md "Establishment: two planes" — `block` is not a
|
||||
/// registry name). The manager spawned this process, confined it to its
|
||||
/// partition, and answers the hello with the channel; the channel is
|
||||
/// range-confined to this process's badge, so reads and writes are
|
||||
/// volume-relative and cannot reach the neighbouring partition. Null until the
|
||||
/// manager has the volume ready — this retries.
|
||||
fn acquireVolume() ?block.Device {
|
||||
var attempts: u32 = 0;
|
||||
const vm = while (attempts < 500) : (attempts += 1) {
|
||||
if (channel.openEndpoint("volume-manager")) |handle| break handle;
|
||||
time.sleepMillis(20);
|
||||
} else return null;
|
||||
|
||||
attempts = 0;
|
||||
while (attempts < 500) : (attempts += 1) {
|
||||
var packet: [volume_manager_protocol.message_maximum]u8 = undefined;
|
||||
const framed = volume_manager_protocol.Protocol.encodeRequest(.hello, my_volume_id, .{}, &.{}, &packet) orelse return null;
|
||||
var reply: [volume_manager_protocol.message_maximum]u8 = undefined;
|
||||
const answered = ipc.callCap(vm, framed, &reply, null) catch return null;
|
||||
const status = envelope.statusOf(reply[0..answered.len]) orelse return null;
|
||||
if (status.status != 0) {
|
||||
if (answered.cap) |stray| _ = ipc.close(stray);
|
||||
_ = logging.write("/system/services/fat: volume manager refused the hello\n");
|
||||
return null;
|
||||
}
|
||||
if (answered.cap) |bus| return .{ .endpoint = bus };
|
||||
// Acked with no channel: the volume is not ready yet — retry.
|
||||
time.sleepMillis(20);
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// The open node `id` names **for `owner`** — null unless the id is in range, in
|
||||
/// use, and this client's own. Node ids are small integers drawn from a table of
|
||||
/// thirty-two, so they are trivially guessable; before this check every client
|
||||
/// honoured every other client's ids, which is the hole
|
||||
/// docs/os-development/protocol-namespace.md names ("handles must be scoped per
|
||||
/// client — validated against the badge"). Nothing else about them changed: they
|
||||
/// are still per-session, still swept when their owner dies.
|
||||
///
|
||||
/// The owner is a *task*, not a process, because the badge is: a threaded client
|
||||
/// reads and writes a node from the thread that opened it, exactly as the exit
|
||||
/// sweep already released a worker thread's handles when that thread died.
|
||||
fn openFor(id: u64, owner: u32) ?*OpenNode {
|
||||
if (id >= open_nodes.len) return null;
|
||||
const o = &open_nodes[@intCast(id)];
|
||||
if (!o.used or o.owner != owner) return null;
|
||||
return o;
|
||||
}
|
||||
|
||||
/// What a handler returns when the thing asked for is not there — a bad node id,
|
||||
/// a node that is someone else's, a path that does not resolve, a mutation the
|
||||
/// volume refused. One errno for all of them, because a filesystem's failures are
|
||||
/// all "no such thing" as far as the file API can act on them — and because
|
||||
/// *someone else's* must be indistinguishable from *nobody's*, or the refusal
|
||||
/// would itself tell a prober which ids are live (the same discipline the
|
||||
/// protocol namespace's refused open follows).
|
||||
const refused: isize = -envelope.ENOENT;
|
||||
|
||||
/// How often to look for a block device while none is mounted. Storage arriving
|
||||
/// is EVENT-shaped (the usb chain registering, possibly after a driver restart),
|
||||
/// but the registry has no subscription — a slow poll from our own harness loop
|
||||
/// keeps the service responsive (ping, terminate) while it waits, and keeps it
|
||||
/// alive to catch storage that appears LATE (a restarted usb-storage after a
|
||||
/// transient failure — the resilience half of docs/logging.md's storage story).
|
||||
const mount_retry_ms = 500;
|
||||
|
||||
var mounted = false;
|
||||
var service_endpoint: ipc.Handle = 0;
|
||||
/// The one channel to the device manager, opened on first need and kept — the
|
||||
/// poll retries on it, never spending a handle-table slot per attempt.
|
||||
var manager_handle: ?ipc.Handle = null;
|
||||
|
||||
/// Find the volume's provider through the device manager (establishment by
|
||||
/// lineage, communication.md "Establishment: two planes" — `block` is not a
|
||||
/// registry name; one storage process serves each stick): enumerate the
|
||||
/// manager's tree, take the FIRST usb mass-storage child by enumeration order
|
||||
/// (deterministic within a boot; single-volume by construction, and choosing
|
||||
/// the BOOT volume by content when two sticks are present is the M21 remount
|
||||
/// track), and consumer-hello for the channel of the driver bound to it.
|
||||
/// Null until the chain is up — the caller's poll retries.
|
||||
fn acquireVolume() ?block.Device {
|
||||
const manager = manager_handle orelse opened: {
|
||||
const handle = channel.openEndpoint("device-manager") orelse return null;
|
||||
manager_handle = handle;
|
||||
break :opened handle;
|
||||
};
|
||||
|
||||
// The envelope's reserved `enumerate` verb, PAGED: one reply carries only
|
||||
// a handful of entries and a real tree (a dozen ACPI nodes before the
|
||||
// first USB child) is bigger, so `Header.target` is the start cursor and
|
||||
// a short page is the end. Identity is the bus's native triple, for USB
|
||||
// (base << 16) | (class << 8) | protocol — mass storage is base 0x08,
|
||||
// subclass 0x06 (SCSI transparent), the same key devices.csv matches on.
|
||||
const Entry = device_manager_protocol.ChildEntry;
|
||||
var start: u64 = 0;
|
||||
while (true) {
|
||||
const enumerate = envelope.Header{ .operation = envelope.operation_enumerate, .target = start };
|
||||
var reply: [device_manager_protocol.message_maximum]u8 = undefined;
|
||||
const length = ipc.call(manager, std.mem.asBytes(&enumerate), &reply) catch return null;
|
||||
const status = envelope.statusOf(reply[0..length]) orelse return null;
|
||||
if (status.status != 0) return null;
|
||||
|
||||
const carried = @min(@as(usize, status.len), length -| envelope.prefix_size);
|
||||
const tail = reply[envelope.prefix_size..][0..carried];
|
||||
const count = tail.len / @sizeOf(Entry);
|
||||
if (count == 0) return null; // the tree is exhausted; no volume yet
|
||||
var index: usize = 0;
|
||||
while (index < count) : (index += 1) {
|
||||
const entry = std.mem.bytesToValue(Entry, tail[index * @sizeOf(Entry) ..][0..@sizeOf(Entry)]);
|
||||
if (entry.device_id == device_manager_protocol.no_device) continue;
|
||||
if ((entry.identity >> 16) & 0xff != 0x08 or (entry.identity >> 8) & 0xff != 0x06) continue;
|
||||
const exchanged = driver.helloOn(manager, .consumer, entry.device_id, null, true) orelse return null;
|
||||
const provider = exchanged.channel orelse continue; // its driver not up yet — next tick
|
||||
return .{ .endpoint = provider };
|
||||
}
|
||||
start += count;
|
||||
/// Durable-on-close: commit the device write cache if any block reached it since
|
||||
/// the last flush. The harness calls this on every close; the dirty check keeps
|
||||
/// it cheap. `device_dirty` lives here because `IpcBlock.writeBlocks` sets it.
|
||||
fn flushIfDirty() void {
|
||||
if (device_dirty) {
|
||||
_ = ipc_block.device.flush();
|
||||
device_dirty = false;
|
||||
}
|
||||
}
|
||||
|
||||
fn initialise(endpoint: ipc.Handle) bool {
|
||||
service_endpoint = endpoint;
|
||||
_ = logging.write("/system/services/fat: starting, waiting for a block device\n");
|
||||
// With the router in the kernel, clients hold OUR node ids directly; sweep
|
||||
// a dead client's open handles via the published exit events (the pattern
|
||||
// the old userspace router used for its own table).
|
||||
_ = process.subscribeExits(endpoint);
|
||||
tryBringUp();
|
||||
if (!mounted) _ = time.timerOnce(endpoint, mount_retry_ms);
|
||||
return true; // serve regardless: requests fail politely until storage mounts
|
||||
}
|
||||
|
||||
/// One storage bring-up attempt: block device -> FAT mount -> VFS mounts. Sets
|
||||
/// `mounted` on success; a failure leaves everything untouched for the next tick.
|
||||
fn tryBringUp() void {
|
||||
if (mounted) return;
|
||||
const device = acquireVolume() orelse return;
|
||||
/// FAT bring-up: find the block device, set up DMA, mount the engine, and hand
|
||||
/// the volume to the harness — or null to retry on the harness's timer.
|
||||
fn fatBringUp(endpoint: ipc.Handle) ?Harness.Volume {
|
||||
_ = endpoint;
|
||||
const device = acquireVolume() orelse return null;
|
||||
const geometry = device.geometry() orelse {
|
||||
_ = logging.write("/system/services/fat: block geometry unavailable\n");
|
||||
return;
|
||||
return null;
|
||||
};
|
||||
// Shareable so the buffer's capability can be attached down the chain (block server
|
||||
// -> controller), making its physical addresses reachable by the device under an
|
||||
// enforcing IOMMU. No-op binding otherwise.
|
||||
const bounce = memory.dmaAlloc(engine.max_transfer_sectors * 512, memory.dma_coherent | memory.dma_shareable) orelse return;
|
||||
// Shareable so the buffer's capability can be attached down the chain (block
|
||||
// server -> controller), making its physical addresses reachable by the
|
||||
// device under an enforcing IOMMU. No-op binding otherwise.
|
||||
const bounce = memory.dmaAlloc(engine.max_transfer_sectors * 512, memory.dma_coherent | memory.dma_shareable) orelse return null;
|
||||
if (bounce.handle) |handle| {
|
||||
// Attach, detach, and attach again: the round trip exercises BOTH verbs
|
||||
// of the DMA-window lifecycle through the whole chain (fat -> storage ->
|
||||
// bus -> kernel) on every boot, so a broken detach fails every fat case
|
||||
// rather than lying dormant until the first buffer replacement.
|
||||
if (!device.attach(handle)) {
|
||||
_ = logging.write("/system/services/fat: could not attach the DMA bounce buffer\n");
|
||||
return;
|
||||
return null;
|
||||
}
|
||||
if (!device.detach(handle)) {
|
||||
_ = logging.write("/system/services/fat: could not detach the DMA bounce buffer\n");
|
||||
return null;
|
||||
}
|
||||
if (!device.attach(handle)) {
|
||||
_ = logging.write("/system/services/fat: could not re-attach the DMA bounce buffer\n");
|
||||
return null;
|
||||
}
|
||||
_ = ipc.close(handle); // the binding holds its own reference now
|
||||
}
|
||||
@@ -213,222 +169,40 @@ fn tryBringUp() void {
|
||||
};
|
||||
filesystem = engine.FileSystem.mount(block_device) orelse {
|
||||
_ = logging.write("/system/services/fat: not a FAT filesystem\n");
|
||||
return;
|
||||
return null;
|
||||
};
|
||||
std.log.info("mounted FAT ({s}, {d} clusters, partition lba {d})", .{ @tagName(filesystem.geometry.fat_type), filesystem.geometry.cluster_count, filesystem.base_lba });
|
||||
|
||||
// Mount ourselves into the kernel VFS at /volumes/usb — and serve
|
||||
// /system/configuration and /system/logs from the volume's identically-named
|
||||
// subtrees (the boot volume is hierarchy-shaped, so rewrite == prefix), so
|
||||
// hierarchy paths (the logger's /system/logs) stay decoupled from which
|
||||
// volume carries them.
|
||||
if (file_system.mount(mount_point, endpointForMount())) {
|
||||
std.log.info("mounted {s}", .{mount_point});
|
||||
// Every volume mounts at its own id-path (argv[2]). The boot/system volume —
|
||||
// the one carrying the /system tree — ADDITIONALLY installs the two FHS
|
||||
// rewrites, so hierarchy paths (config reads, the logger's persistent
|
||||
// /system/logs) stay decoupled from which volume backs them. Detection is by
|
||||
// CONTENT, not spawn order: a volume is the system volume iff /system/
|
||||
// configuration resolves on its own media. A data volume has no /system, so it
|
||||
// mounts only at its id-path and never shadows the running system's config or
|
||||
// logs with a dead mount.
|
||||
mount_specs[0] = .{ .prefix = volume_mount_prefix };
|
||||
var mount_count: usize = 1;
|
||||
if (filesystem.resolve("/system/configuration") != null) {
|
||||
std.log.info("volume {d} carries the system tree; backing /system/configuration and /system/logs", .{my_volume_id});
|
||||
mount_specs[1] = .{ .prefix = "/system/configuration", .rewrite = "/system/configuration" };
|
||||
mount_specs[2] = .{ .prefix = "/system/logs", .rewrite = "/system/logs" };
|
||||
mount_count = 3;
|
||||
} else {
|
||||
_ = logging.write("/system/services/fat: could not mount /volumes/usb\n");
|
||||
std.log.info("volume {d} is a data volume; mounted at {s}", .{ my_volume_id, volume_mount_prefix });
|
||||
}
|
||||
if (file_system.mountRewritten("/system/configuration", endpointForMount(), "/system/configuration")) {
|
||||
std.log.info("mounted /system/configuration", .{});
|
||||
} else {
|
||||
_ = logging.write("/system/services/fat: could not mount /system/configuration\n");
|
||||
return .{ .engine = &filesystem, .mounts = mount_specs[0..mount_count], .flush = flushIfDirty };
|
||||
}
|
||||
|
||||
pub fn main(init: process.Init) void {
|
||||
// The volume manager spawns this process with its volume id as argv[1] and
|
||||
// the volume's mount path (its id-path) as argv[2].
|
||||
if (init.arguments.get(1)) |id| {
|
||||
my_volume_id = std.fmt.parseInt(u64, id, 10) catch 0;
|
||||
}
|
||||
if (file_system.mountRewritten("/system/logs", endpointForMount(), "/system/logs")) {
|
||||
std.log.info("mounted /system/logs", .{});
|
||||
} else {
|
||||
_ = logging.write("/system/services/fat: could not mount /system/logs\n");
|
||||
if (init.arguments.get(2)) |prefix| {
|
||||
volume_mount_prefix = prefix;
|
||||
}
|
||||
mounted = true;
|
||||
}
|
||||
|
||||
fn endpointForMount() ipc.Handle {
|
||||
return service_endpoint;
|
||||
}
|
||||
|
||||
/// A subscribed process-exit event: release every open handle the dead client
|
||||
/// held, so a crashed reader can't pin table slots (or, later, locks).
|
||||
fn onNotification(badge: u64) void {
|
||||
const got = ipc.Received{ .len = 0, .badge = badge, .cap = null };
|
||||
if (got.isTimer()) {
|
||||
tryBringUp();
|
||||
if (!mounted) _ = time.timerOnce(service_endpoint, mount_retry_ms);
|
||||
return;
|
||||
}
|
||||
if (!got.isChildExit()) return;
|
||||
const dead = got.childProcessId();
|
||||
var released: u32 = 0;
|
||||
for (&open_nodes) |*o| {
|
||||
if (o.used and o.owner == dead) {
|
||||
o.* = .{};
|
||||
released += 1;
|
||||
}
|
||||
}
|
||||
if (released != 0) std.log.info("released {d} handle(s) for dead client {d}", .{ released, dead });
|
||||
}
|
||||
|
||||
const ParentLeaf = struct { parent: []const u8, leaf: []const u8 };
|
||||
|
||||
// Split a path into its parent directory and final component: "/a/b" -> ("/a",
|
||||
// "b"); "/b" -> ("/", "b"); "b" -> ("/", "b").
|
||||
fn splitParent(path: []const u8) ParentLeaf {
|
||||
const slash = std.mem.lastIndexOfScalar(u8, path, '/');
|
||||
return .{
|
||||
.parent = if (slash) |s| (if (s == 0) "/" else path[0..s]) else "/",
|
||||
.leaf = if (slash) |s| path[s + 1 ..] else path,
|
||||
};
|
||||
}
|
||||
|
||||
fn onOpen(_: void, invocation: Invocation(vfs_protocol.Open), answer: Answer(vfs_protocol.Opened)) isize {
|
||||
const path = invocation.tail;
|
||||
const flags = invocation.request.flags;
|
||||
var node = filesystem.resolve(path);
|
||||
if (node == null and flags & vfs_protocol.create != 0) {
|
||||
const split = splitParent(path);
|
||||
const parent = filesystem.resolve(split.parent) orelse return refused;
|
||||
node = filesystem.createFile(parent, split.leaf);
|
||||
}
|
||||
var resolved = node orelse return refused;
|
||||
// O_TRUNC: replace an existing file's contents rather than overwriting in place
|
||||
// (frees the old chain, so a shorter rewrite leaves no stale tail).
|
||||
if (flags & vfs_protocol.truncate != 0 and !resolved.is_directory) {
|
||||
filesystem.truncate(&resolved);
|
||||
}
|
||||
const index = allocOpen() orelse return refused;
|
||||
open_nodes[index] = .{ .used = true, .node = resolved, .owner = invocation.sender };
|
||||
answer.set(.{ .node = index });
|
||||
return 0;
|
||||
}
|
||||
|
||||
fn onRead(_: void, invocation: Invocation(vfs_protocol.Read), answer: Answer(void)) isize {
|
||||
const o = openFor(invocation.target, invocation.sender) orelse return refused;
|
||||
const into = answer.tail();
|
||||
const want = @min(@as(usize, invocation.request.len), into.len);
|
||||
return @intCast(filesystem.readFile(o.node, @intCast(invocation.request.offset), into[0..want]));
|
||||
}
|
||||
|
||||
fn onWrite(_: void, invocation: Invocation(vfs_protocol.Write), answer: Answer(vfs_protocol.Written)) isize {
|
||||
const o = openFor(invocation.target, invocation.sender) orelse return refused;
|
||||
const data = invocation.tail[0..@min(invocation.tail.len, invocation.request.len)];
|
||||
const n = filesystem.writeFile(&o.node, @intCast(invocation.request.offset), data);
|
||||
answer.set(.{ .count = @intCast(n) });
|
||||
return 0;
|
||||
}
|
||||
|
||||
fn onStatus(_: void, invocation: Invocation(void), answer: Answer(vfs_protocol.FileStatus)) isize {
|
||||
const o = openFor(invocation.target, invocation.sender) orelse return refused;
|
||||
const kind: vfs_protocol.NodeKind = if (o.node.is_directory) .directory else .regular;
|
||||
answer.set(.{ .size = o.node.size, .kind = @intFromEnum(kind), .mtime = o.node.mtime });
|
||||
return 0;
|
||||
}
|
||||
|
||||
/// One entry per call. End of directory — a node that is not a directory, or a
|
||||
/// cursor past the last child — is an entry with no name, which is how the
|
||||
/// protocol spells it now that the reply's length always counts the fixed part.
|
||||
fn onReaddir(_: void, invocation: Invocation(vfs_protocol.Readdir), answer: Answer(vfs_protocol.DirectoryEntry)) isize {
|
||||
const o = openFor(invocation.target, invocation.sender) orelse return refused;
|
||||
if (!o.node.is_directory) {
|
||||
answer.set(.{});
|
||||
return 0;
|
||||
}
|
||||
const listing = filesystem.listEntry(o.node, @intCast(invocation.request.cursor)) orelse {
|
||||
answer.set(.{});
|
||||
return 0;
|
||||
};
|
||||
const kind: vfs_protocol.NodeKind = if (listing.is_directory) .directory else .regular;
|
||||
const into = answer.tail();
|
||||
const name_len = @min(listing.name_len, into.len);
|
||||
@memcpy(into[0..name_len], listing.name_buffer[0..name_len]);
|
||||
answer.set(.{ .kind = @intFromEnum(kind), .name_len = @intCast(name_len), .size = listing.size });
|
||||
return @intCast(name_len);
|
||||
}
|
||||
|
||||
/// Closing is an operation on a node like any other, so it is scoped like any
|
||||
/// other: a client may release its own handles and nobody else's. An id that is
|
||||
/// not the caller's — free, out of range, or another client's — is refused
|
||||
/// identically, so a close cannot be used to ask which ids are live either.
|
||||
fn onClose(_: void, invocation: Invocation(void), _: Answer(void)) isize {
|
||||
const o = openFor(invocation.target, invocation.sender) orelse return refused;
|
||||
o.used = false;
|
||||
// Durable-on-close: if any block reached the device since the last flush,
|
||||
// commit its cache to stable media now (best-effort). This is what makes
|
||||
// init's shutdown log flush survive a real power-off, and is the right
|
||||
// default for removable media the user may unplug.
|
||||
if (device_dirty) {
|
||||
_ = ipc_block.device.flush();
|
||||
device_dirty = false;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
fn onMakeDirectory(_: void, invocation: Invocation(void), _: Answer(void)) isize {
|
||||
const path = invocation.tail;
|
||||
if (filesystem.resolve(path) != null) return refused; // already exists — no duplicate entries
|
||||
const split = splitParent(path);
|
||||
const parent = filesystem.resolve(split.parent) orelse return refused;
|
||||
if (filesystem.createDirectory(parent, split.leaf) == null) return refused;
|
||||
return 0;
|
||||
}
|
||||
|
||||
fn onUnlink(_: void, invocation: Invocation(void), _: Answer(void)) isize {
|
||||
const split = splitParent(invocation.tail);
|
||||
const parent = filesystem.resolve(split.parent) orelse return refused;
|
||||
if (!filesystem.removeFile(parent, split.leaf)) return refused;
|
||||
return 0;
|
||||
}
|
||||
|
||||
fn onRename(_: void, invocation: Invocation(void), _: Answer(void)) isize {
|
||||
const both = invocation.tail;
|
||||
const separator = std.mem.indexOfScalar(u8, both, 0) orelse return refused;
|
||||
const old_split = splitParent(both[0..separator]);
|
||||
const new_split = splitParent(both[separator + 1 ..]);
|
||||
// Same-directory rename only.
|
||||
if (!std.mem.eql(u8, old_split.parent, new_split.parent)) return refused;
|
||||
const parent = filesystem.resolve(old_split.parent) orelse return refused;
|
||||
if (!filesystem.rename(parent, old_split.leaf, new_split.leaf)) return refused;
|
||||
return 0;
|
||||
}
|
||||
|
||||
/// The verbs this backend implements. The three it leaves out — `mount`,
|
||||
/// `unmount`, `bind` — answer `-ENOSYS` from the generated dispatch, which is
|
||||
/// exactly right: path routing is the kernel's now, and only init implements
|
||||
/// `bind` (docs/os-development/protocol-namespace.md). `describe` is the
|
||||
/// envelope's own.
|
||||
const handlers = Serve.Handlers{
|
||||
.open = onOpen,
|
||||
.close = onClose,
|
||||
.read = onRead,
|
||||
.write = onWrite,
|
||||
.status = onStatus,
|
||||
.readdir = onReaddir,
|
||||
.mkdir = onMakeDirectory,
|
||||
.unlink = onUnlink,
|
||||
.rename = onRename,
|
||||
};
|
||||
|
||||
/// The vfs protocol has no operation that takes a capability, so `arrived` is
|
||||
/// never claimed here — which, under the harness's ownership rule, means the
|
||||
/// loop closes whatever a caller attached. That is the point of the rule: this
|
||||
/// callback used to discard a `?ipc.Handle` and every request carrying one — a
|
||||
/// legal thing for any client to do — spent a slot of the VFS server's
|
||||
/// thirty-two until it could accept no capability at all.
|
||||
fn onMessage(message: []const u8, out: []u8, sender: u32, arrived: *ipc.Arrival) usize {
|
||||
_ = arrived;
|
||||
// Storage not up (yet): fail politely, whatever was asked — clients retry.
|
||||
if (!mounted) {
|
||||
const status = envelope.Status{ .status = refused, .len = 0 };
|
||||
@memcpy(out[0..envelope.prefix_size], std.mem.asBytes(&status));
|
||||
return envelope.prefix_size;
|
||||
}
|
||||
// Stamp create/write with the current wall-clock time (mtime). Cheap, and it
|
||||
// keeps the engine pure (it takes the time as data, not a syscall).
|
||||
filesystem.current_time_epoch = time.wallClock();
|
||||
return Serve.dispatch({}, handlers, message, sender, null, out);
|
||||
}
|
||||
|
||||
pub fn main() void {
|
||||
service.run(vfs_protocol.message_maximum, .{
|
||||
.service = "vfs",
|
||||
.init = initialise,
|
||||
.on_message = onMessage,
|
||||
.on_notification = onNotification,
|
||||
});
|
||||
_ = logging.write("/system/services/fat: starting, waiting for a block device\n");
|
||||
Harness.run(.{ .bringUp = fatBringUp });
|
||||
}
|
||||
|
||||
@@ -0,0 +1,55 @@
|
||||
//! The volume-manager service as a binary package (docs/build-packages-plan.md):
|
||||
//! this file names the binary and EXACTLY the modules its source imports —
|
||||
//! build-support resolves each name from the domains this zon declares.
|
||||
|
||||
const std = @import("std");
|
||||
const build_support = @import("build-support");
|
||||
|
||||
pub fn build(b: *std.Build) void {
|
||||
const exe = build_support.userBinary(b, .{
|
||||
.name = "volume-manager",
|
||||
.root_source_file = b.path("volume-manager.zig"),
|
||||
.imports = &.{
|
||||
"block", "channel", "csv", "device-manager-protocol",
|
||||
"driver", "envelope", "file-system", "ipc",
|
||||
"logging", "memory", "process", "service",
|
||||
"time", "volume-manager-protocol",
|
||||
},
|
||||
});
|
||||
b.installArtifact(exe);
|
||||
|
||||
// Standalone `zig build test` for the parser + mount-map modules; the root
|
||||
// build keeps its aggregate test step.
|
||||
const csv = b.dependency("csv", .{});
|
||||
const test_step = b.step("test", "Run the partition parser + mount-map unit tests");
|
||||
|
||||
const partition_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("partition.zig"),
|
||||
.target = b.resolveTargetQuery(.{}),
|
||||
}),
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(partition_tests).step);
|
||||
|
||||
// filesystem-map imports csv (and, by path, partition.zig), so its test
|
||||
// module needs csv wired.
|
||||
const filesystem_map_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("filesystem-map.zig"),
|
||||
.target = b.resolveTargetQuery(.{}),
|
||||
.imports = &.{.{ .name = "csv", .module = csv.module("csv") }},
|
||||
}),
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(filesystem_map_tests).step);
|
||||
|
||||
// volume-map imports csv (and, by path, partition.zig) for the id-path
|
||||
// deriver and the volumes.csv override parser.
|
||||
const volume_map_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("volume-map.zig"),
|
||||
.target = b.resolveTargetQuery(.{}),
|
||||
.imports = &.{.{ .name = "csv", .module = csv.module("csv") }},
|
||||
}),
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(volume_map_tests).step);
|
||||
}
|
||||
@@ -0,0 +1,18 @@
|
||||
.{
|
||||
.name = .volume_manager,
|
||||
.version = "0.0.0",
|
||||
.fingerprint = 0x911b01e1eaeabcf5, // Changing this has security and trust implications.
|
||||
.minimum_zig_version = "0.16.0",
|
||||
.dependencies = .{
|
||||
// build-support supplies the shared recipe; kernel is implicit in every
|
||||
// binary. device (block, driver) and protocol (device-manager-protocol,
|
||||
// envelope) are the homes of this service's remaining imports.
|
||||
.@"build-support" = .{ .path = "../../../build-support" },
|
||||
.kernel = .{ .path = "../../../library/kernel" },
|
||||
.device = .{ .path = "../../../library/device" },
|
||||
.protocol = .{ .path = "../../../library/protocol" },
|
||||
// csv parses filesystems.csv / volumes.csv, the mount-map configuration.
|
||||
.csv = .{ .path = "../../../library/csv" },
|
||||
},
|
||||
.paths = .{""},
|
||||
}
|
||||
@@ -0,0 +1,121 @@
|
||||
//! filesystem-map — parse `/system/configuration/filesystems.csv` into
|
||||
//! content-signature → service-binary rules, and pick the binary for a probed
|
||||
//! volume's signature. The data-driven replacement for the volume manager's
|
||||
//! hardcoded `filesystem_binary` const: a signature no row matches goes unserved
|
||||
//! (logged), never guessed — the same discipline the device registry uses.
|
||||
//!
|
||||
//! Pure logic: no hardware, no syscalls, no allocator. The `binary` slice points
|
||||
//! into the CSV source, which the manager holds in a static buffer for the life
|
||||
//! of the process (zero-copy), so the source must outlive the rules.
|
||||
//!
|
||||
//! Format: one rule per line, two comma-separated fields, `#` comments (whole-
|
||||
//! line or trailing), blank lines ignored:
|
||||
//!
|
||||
//! signature, binary
|
||||
//!
|
||||
//! `signature` is a filesystem token (`fat`; `exfat` lands with S4); `binary` is
|
||||
//! a full ramdisk path.
|
||||
|
||||
const std = @import("std");
|
||||
const csv = @import("csv");
|
||||
const partition = @import("partition.zig");
|
||||
|
||||
/// One parsed row: a content signature and the service binary that serves it.
|
||||
pub const Rule = struct {
|
||||
kind: partition.FilesystemKind,
|
||||
binary: []const u8,
|
||||
};
|
||||
|
||||
/// How many rules landed, how many non-blank lines were malformed (for the
|
||||
/// manager to log), and whether there were more rules than the buffer could hold.
|
||||
pub const ParseResult = struct { count: usize, malformed: usize, truncated: bool };
|
||||
|
||||
const Line = union(enum) { rule: Rule, ignorable, malformed };
|
||||
|
||||
fn parseLine(line: []const u8) Line {
|
||||
const body = csv.stripComment(line);
|
||||
if (body.len == 0) return .ignorable;
|
||||
var it = csv.fields(body);
|
||||
const sig = it.next() orelse return .malformed;
|
||||
const binary = it.next() orelse return .malformed;
|
||||
if (it.next() != null) return .malformed; // too many columns
|
||||
if (binary.len == 0) return .malformed;
|
||||
const kind = partition.FilesystemKind.fromToken(sig);
|
||||
if (kind == .unknown) return .malformed; // an unrecognised signature token
|
||||
return .{ .rule = .{ .kind = kind, .binary = binary } };
|
||||
}
|
||||
|
||||
/// Parse a whole `filesystems.csv` into `out_rules`. The `binary` slices point
|
||||
/// into `source`, which must outlive them.
|
||||
pub fn parse(source: []const u8, out_rules: []Rule) ParseResult {
|
||||
var result: ParseResult = .{ .count = 0, .malformed = 0, .truncated = false };
|
||||
var lines = std.mem.splitScalar(u8, source, '\n');
|
||||
while (lines.next()) |line| {
|
||||
switch (parseLine(line)) {
|
||||
.ignorable => {},
|
||||
.malformed => result.malformed += 1,
|
||||
.rule => |rule| {
|
||||
if (result.count >= out_rules.len) {
|
||||
result.truncated = true;
|
||||
continue;
|
||||
}
|
||||
out_rules[result.count] = rule;
|
||||
result.count += 1;
|
||||
},
|
||||
}
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
/// The service binary for a probed volume's signature — the first matching row,
|
||||
/// or null (the volume goes unserved, like a device no registry row matches).
|
||||
pub fn match(rules: []const Rule, kind: partition.FilesystemKind) ?[]const u8 {
|
||||
for (rules) |rule| {
|
||||
if (rule.kind == kind) return rule.binary;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
// --- tests -------------------------------------------------------------------
|
||||
|
||||
const testing = std.testing;
|
||||
|
||||
// A fixture-sized rule buffer for the tests, named so the bounds gate (which
|
||||
// flags literal array lengths) stays quiet: this is a test input, not a runtime
|
||||
// ceiling — the real one is maximum_filesystem_rules in the volume manager.
|
||||
const test_rule_slots = 4;
|
||||
|
||||
test "a fat signature maps to its binary; an unmatched signature is null" {
|
||||
const text =
|
||||
\\# signature, binary
|
||||
\\fat, /system/services/fat
|
||||
;
|
||||
var rules: [test_rule_slots]Rule = undefined;
|
||||
const parsed = parse(text, &rules);
|
||||
try testing.expectEqual(@as(usize, 1), parsed.count);
|
||||
try testing.expectEqual(@as(usize, 0), parsed.malformed);
|
||||
try testing.expectEqualStrings("/system/services/fat", match(rules[0..parsed.count], .fat).?);
|
||||
try testing.expect(match(rules[0..parsed.count], .unknown) == null);
|
||||
}
|
||||
|
||||
test "the binary is chosen by content, not hardcoded" {
|
||||
// Point the fat row at a different binary and confirm that binary is chosen —
|
||||
// a constant could not satisfy this, which is the whole point of the map.
|
||||
const text = "fat, /system/services/other-fat\n";
|
||||
var rules: [test_rule_slots]Rule = undefined;
|
||||
const parsed = parse(text, &rules);
|
||||
try testing.expectEqualStrings("/system/services/other-fat", match(rules[0..parsed.count], .fat).?);
|
||||
}
|
||||
|
||||
test "malformed rows are counted, not bound" {
|
||||
const text =
|
||||
\\fat, /system/services/fat
|
||||
\\bogusfs, /system/services/x
|
||||
\\fat,
|
||||
\\fat, /a, /b
|
||||
;
|
||||
var rules: [test_rule_slots]Rule = undefined;
|
||||
const parsed = parse(text, &rules);
|
||||
try testing.expectEqual(@as(usize, 1), parsed.count); // only the first fat row
|
||||
try testing.expectEqual(@as(usize, 3), parsed.malformed); // bad token, empty binary, too many columns
|
||||
}
|
||||
@@ -0,0 +1,620 @@
|
||||
//! Partition-table parsing, the policy the storage architecture places above the
|
||||
//! block driver and below the filesystem (docs/file-system-development/
|
||||
//! storage-architecture.md): read the medium, decide what block sub-ranges are
|
||||
//! volumes, and read each volume's content identity. The block DRIVER never does
|
||||
//! this — it clamps ranges it is told about; this is what tells it the numbers.
|
||||
//!
|
||||
//! Reads happen through a `SectorReader` (not one preloaded block-0 slice) so the
|
||||
//! parser can reach GPT metadata at LBA 1, the entry array beyond it, and each
|
||||
//! partition's VBR on demand. The identity it returns is a tagged `Identity`: the
|
||||
//! `key` is the id (the mount path is derived from it — a stable, unique,
|
||||
//! content-derived handle), and `label` is display metadata (the FAT volume label
|
||||
//! or the GPT partition name), never part of the id. Today's rung is MBR/bare-FAT;
|
||||
//! GPT (rung 1) and the FAT serial (rung 3) slot in without changing the shape.
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
/// A single 512-byte sector's worth of bytes. The parser assumes 512-byte
|
||||
/// logical sectors (4Kn media is a separate concern, noted in the plan).
|
||||
pub const sector_bytes = 512;
|
||||
|
||||
/// bound: bytes of a volume's display label the parser records (a GPT partition
|
||||
/// name is 36 UTF-16 units; a FAT volume label is 11 bytes; 36 covers both)
|
||||
/// decided-by: hardware
|
||||
/// protects: the Identity.label buffer
|
||||
/// at-limit: degrade - a longer name is truncated to this many ASCII bytes
|
||||
/// observed-by: a volume whose displayed label is clipped
|
||||
pub const label_maximum = 36;
|
||||
|
||||
/// Which rung of the identity ladder produced this identity. The rung tags the
|
||||
/// `key` namespace so a FAT serial and an MBR signature that happen to share bits
|
||||
/// stay distinct, and it drives how the mount path is rendered from the id.
|
||||
pub const Rung = enum(u8) {
|
||||
gpt_guid = 1,
|
||||
filesystem_uuid = 2, // reserved: no engine reads a superblock UUID yet
|
||||
fat_serial = 3,
|
||||
mbr_index = 4,
|
||||
anonymous = 5,
|
||||
exfat_serial = 6, // exFAT's VolumeSerialNumber — content-strong like fat_serial
|
||||
};
|
||||
|
||||
/// A volume's content identity. `key` is the ID — the stable, unique handle the
|
||||
/// mount path is derived from and the mount map keys on. `label` is DISPLAY
|
||||
/// metadata (FAT volume label / GPT partition name), exposed to a UI but never
|
||||
/// part of the path; two volumes with the same label but different keys are
|
||||
/// different volumes. Derived from the medium, never from a port.
|
||||
pub const Identity = struct {
|
||||
rung: Rung,
|
||||
key: u128 = 0,
|
||||
label: [label_maximum]u8 = [_]u8{0} ** label_maximum,
|
||||
label_len: u8 = 0,
|
||||
|
||||
pub fn labelSlice(self: *const Identity) []const u8 {
|
||||
return self.label[0..self.label_len];
|
||||
}
|
||||
|
||||
/// Identity equality is the ID (rung + key) only — the label is display
|
||||
/// metadata and does not enter it. Same rung + same key means the same
|
||||
/// volume (the dd-cloned-media case the duplicate policy is for).
|
||||
pub fn eql(a: Identity, b: Identity) bool {
|
||||
return a.rung == b.rung and a.key == b.key;
|
||||
}
|
||||
};
|
||||
|
||||
/// Which filesystem a volume's content is — the key `filesystems.csv` maps to a
|
||||
/// service binary. FAT and exFAT are recognized by their VBRs; content that is
|
||||
/// neither falls back to `.fat`, the volume manager's historical hand-off.
|
||||
pub const FilesystemKind = enum {
|
||||
fat,
|
||||
exfat,
|
||||
unknown,
|
||||
|
||||
pub fn fromToken(token: []const u8) FilesystemKind {
|
||||
if (std.mem.eql(u8, token, "fat")) return .fat;
|
||||
if (std.mem.eql(u8, token, "exfat")) return .exfat;
|
||||
return .unknown;
|
||||
}
|
||||
};
|
||||
|
||||
/// One volume the parser found on the device: the block sub-range it occupies,
|
||||
/// its content identity, and which filesystem its content is (the signature the
|
||||
/// `filesystems.csv` map keys on to pick the service binary).
|
||||
pub const Volume = struct {
|
||||
base_lba: u64,
|
||||
block_count: u64,
|
||||
identity: Identity,
|
||||
signature: FilesystemKind = .fat,
|
||||
};
|
||||
|
||||
/// Read sectors on demand. `context` + `readFn` mirror the FAT engine's
|
||||
/// `BlockDevice` vtable; `readFn` returns false past the end of the device or on
|
||||
/// an I/O error, which the parser treats as "no volume".
|
||||
pub const SectorReader = struct {
|
||||
context: *anyopaque,
|
||||
readFn: *const fn (context: *anyopaque, lba: u64, buffer: *[sector_bytes]u8) bool,
|
||||
|
||||
pub fn read(self: SectorReader, lba: u64, buffer: *[sector_bytes]u8) bool {
|
||||
return self.readFn(self.context, lba, buffer);
|
||||
}
|
||||
};
|
||||
|
||||
/// The MBR disk signature (offset 440, 4 bytes LE) — a 32-bit id written at
|
||||
/// partition time. Weak (dd-cloned disks share it) but on the medium, and the
|
||||
/// last rung of the identity ladder; the fuller rungs (GPT GUID, FAT serial)
|
||||
/// take precedence when present.
|
||||
fn diskSignature(block0: []const u8) u32 {
|
||||
if (block0.len < 444) return 0;
|
||||
return std.mem.readInt(u32, block0[440..444], .little);
|
||||
}
|
||||
|
||||
/// The rung-4 identity of the volume at partition `index`: the disk signature
|
||||
/// paired with the index, so two partitions of one disk stay distinct. For a
|
||||
/// bare FAT (no table) the index is 0. Carries no label.
|
||||
fn mbrIdentity(block0: []const u8, index: u8) Identity {
|
||||
return .{ .rung = .mbr_index, .key = (@as(u128, diskSignature(block0)) << 8) | index };
|
||||
}
|
||||
|
||||
/// Whether a block looks like a boot sector / partition table (the 0x55AA boot
|
||||
/// signature). A bare FAT also carries it, so the caller distinguishes by whether
|
||||
/// any partition entry is non-empty.
|
||||
fn hasBootSignature(block0: []const u8) bool {
|
||||
return block0.len >= 512 and block0[510] == 0x55 and block0[511] == 0xAA;
|
||||
}
|
||||
|
||||
/// GPT header signature at LBA 1.
|
||||
const gpt_signature = "EFI PART";
|
||||
|
||||
/// bound: GPT partition entries scanned before the prober gives up
|
||||
/// decided-by: ours
|
||||
/// protects: the entry-array scan loop from an untrusted num_partition_entries
|
||||
/// at-limit: degrade - stop scanning; a device whose usable entry sits past the
|
||||
/// cap is treated as having no GPT volume (real tables carry <=128 entries)
|
||||
/// observed-by: the gpt-entry-past-device host test
|
||||
const gpt_entry_scan_maximum = 128;
|
||||
|
||||
/// Reflected CRC-32 (polynomial 0xEDB88320) — the ISO-HDLC variant GPT uses for
|
||||
/// its header checksum. Inlined so the parser and the host fixtures compute it
|
||||
/// the same way and never drift onto a magic constant.
|
||||
fn crc32(bytes: []const u8) u32 {
|
||||
var c: u32 = 0xFFFFFFFF;
|
||||
for (bytes) |b| {
|
||||
c ^= b;
|
||||
var k: u8 = 0;
|
||||
while (k < 8) : (k += 1) {
|
||||
c = if (c & 1 != 0) (c >> 1) ^ 0xEDB88320 else c >> 1;
|
||||
}
|
||||
}
|
||||
return c ^ 0xFFFFFFFF;
|
||||
}
|
||||
|
||||
/// A GPT disk carries a protective MBR: a boot-signed block 0 with a partition
|
||||
/// entry of type 0xEE. Its presence routes probing to the GPT (authoritative).
|
||||
fn isProtectiveMbr(block0: []const u8) bool {
|
||||
if (!hasBootSignature(block0)) return false;
|
||||
var index: usize = 0;
|
||||
while (index < 4) : (index += 1) {
|
||||
if (block0[446 + index * 16 + 4] == 0xEE) return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/// Copy the GPT partition name (36 UTF-16LE units, the 72 bytes at entry+56)
|
||||
/// into the identity's display label as ASCII, dropping non-ASCII units.
|
||||
fn setLabelFromUtf16(id: *Identity, name_bytes: []const u8) void {
|
||||
var out: usize = 0;
|
||||
var i: usize = 0;
|
||||
while (i + 1 < name_bytes.len and out < label_maximum) : (i += 2) {
|
||||
const unit = std.mem.readInt(u16, name_bytes[i..][0..2], .little);
|
||||
if (unit == 0) break;
|
||||
if (unit < 0x80) {
|
||||
id.label[out] = @intCast(unit);
|
||||
out += 1;
|
||||
}
|
||||
}
|
||||
id.label_len = @intCast(out);
|
||||
}
|
||||
|
||||
/// Append every valid GPT volume to `out` (up to `out.len`), returning the count
|
||||
/// (0 if LBA 1 is not a valid GPT header). The header CRC-32 and the per-entry
|
||||
/// overflow-safe range check are the confinement-safety guards the driver's clamp
|
||||
/// rests on — the invariant documented for MBR, extended to untrusted GPT
|
||||
/// metadata. The entry-array CRC is deferred (correctness-only; the range check
|
||||
/// carries safety).
|
||||
fn gptAllVolumes(reader: SectorReader, device_blocks: u64, out: []Volume) usize {
|
||||
var header: [sector_bytes]u8 = undefined;
|
||||
if (!reader.read(1, &header)) return 0;
|
||||
if (!std.mem.eql(u8, header[0..8], gpt_signature)) return 0;
|
||||
const header_size = std.mem.readInt(u32, header[12..16], .little);
|
||||
if (header_size < 92 or header_size > sector_bytes) return 0;
|
||||
const stored_crc = std.mem.readInt(u32, header[16..20], .little);
|
||||
var check: [sector_bytes]u8 = undefined;
|
||||
@memcpy(check[0..header_size], header[0..header_size]);
|
||||
@memset(check[16..20], 0);
|
||||
if (crc32(check[0..header_size]) != stored_crc) return 0;
|
||||
|
||||
const entry_lba = std.mem.readInt(u64, header[72..80], .little);
|
||||
const num_entries = std.mem.readInt(u32, header[80..84], .little);
|
||||
const entry_size = std.mem.readInt(u32, header[84..88], .little);
|
||||
if (entry_size != 128 and entry_size != 256 and entry_size != 512) return 0;
|
||||
if (entry_lba == 0 or entry_lba >= device_blocks) return 0;
|
||||
|
||||
const scan = @min(num_entries, gpt_entry_scan_maximum);
|
||||
var sector_buf: [sector_bytes]u8 = undefined;
|
||||
var loaded: u64 = std.math.maxInt(u64);
|
||||
var count: usize = 0;
|
||||
var i: u32 = 0;
|
||||
while (i < scan and count < out.len) : (i += 1) {
|
||||
const abs = @as(u64, i) * entry_size;
|
||||
const lba = entry_lba + abs / sector_bytes;
|
||||
const off = @as(usize, @intCast(abs % sector_bytes));
|
||||
if (lba != loaded) {
|
||||
if (!reader.read(lba, §or_buf)) break; // return what we have
|
||||
loaded = lba;
|
||||
}
|
||||
const entry = sector_buf[off..][0..128]; // the fields we read live in the first 128 bytes
|
||||
var type_nonzero = false;
|
||||
for (entry[0..16]) |b| {
|
||||
if (b != 0) {
|
||||
type_nonzero = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (!type_nonzero) continue;
|
||||
const start = std.mem.readInt(u64, entry[32..40], .little);
|
||||
const end = std.mem.readInt(u64, entry[40..48], .little); // inclusive last LBA
|
||||
// Untrusted range from removable media: overflow-safe validation. Reject a
|
||||
// partition that starts at 0, is reversed, or ends outside the device; only
|
||||
// then is start + count <= device_blocks guaranteed for the driver's clamp.
|
||||
if (start == 0 or end < start or end >= device_blocks) continue;
|
||||
var id = Identity{ .rung = .gpt_guid, .key = std.mem.readInt(u128, entry[16..32], .little) };
|
||||
setLabelFromUtf16(&id, entry[56..128]);
|
||||
out[count] = .{ .base_lba = start, .block_count = end - start + 1, .identity = id, .signature = signatureAt(reader, start) };
|
||||
count += 1;
|
||||
}
|
||||
return count;
|
||||
}
|
||||
|
||||
/// Trim trailing spaces (FAT labels are space-padded) and copy into the display
|
||||
/// label, clamped to label_maximum.
|
||||
fn setFatLabel(id: *Identity, label: []const u8) void {
|
||||
var end: usize = label.len;
|
||||
while (end > 0 and label[end - 1] == ' ') : (end -= 1) {}
|
||||
const n = @min(end, label_maximum);
|
||||
@memcpy(id.label[0..n], label[0..n]);
|
||||
id.label_len = @intCast(n);
|
||||
}
|
||||
|
||||
/// The FAT volume serial (BS_VolID) + label (BS_VolLab) read from the VBR at
|
||||
/// `start_lba` — rung 3, stronger than the MBR disk signature. Null if the
|
||||
/// sector is not an extended FAT boot record (no 0x55AA, or no 0x28/0x29
|
||||
/// extended boot signature). FAT32 is distinguished by fat_size_16 == 0; the
|
||||
/// serial and label live at different EBR offsets for FAT12/16 vs FAT32 (the
|
||||
/// offsets are cross-checked against system/services/fat/on-disk.zig).
|
||||
fn fatIdentity(reader: SectorReader, start_lba: u64) ?Identity {
|
||||
var vbr: [sector_bytes]u8 = undefined;
|
||||
if (!reader.read(start_lba, &vbr)) return null;
|
||||
if (vbr[510] != 0x55 or vbr[511] != 0xAA) return null;
|
||||
const is_fat32 = std.mem.readInt(u16, vbr[22..24], .little) == 0;
|
||||
const sig_off: usize = if (is_fat32) 66 else 38;
|
||||
if (vbr[sig_off] != 0x28 and vbr[sig_off] != 0x29) return null;
|
||||
const id_off: usize = if (is_fat32) 67 else 39;
|
||||
const label_off: usize = if (is_fat32) 71 else 43;
|
||||
var id = Identity{ .rung = .fat_serial, .key = std.mem.readInt(u32, vbr[id_off..][0..4], .little) };
|
||||
setFatLabel(&id, vbr[label_off..][0..11]);
|
||||
return id;
|
||||
}
|
||||
|
||||
/// The exFAT VolumeSerialNumber (offset 100) read from the Main Boot Sector at
|
||||
/// `start_lba` — its content identity, rung `exfat_serial`. Null unless the sector
|
||||
/// is an exFAT VBR (the "EXFAT " name at offset 3 + the 0x55AA signature; the
|
||||
/// name is where a FAT BPB keeps its OEM string, so the two never collide). The
|
||||
/// label lives in a root-directory entry, not the VBR, so it is left empty here.
|
||||
fn exfatIdentity(reader: SectorReader, start_lba: u64) ?Identity {
|
||||
var vbr: [sector_bytes]u8 = undefined;
|
||||
if (!reader.read(start_lba, &vbr)) return null;
|
||||
if (vbr[510] != 0x55 or vbr[511] != 0xAA) return null;
|
||||
if (!std.mem.eql(u8, vbr[3..11], "EXFAT ")) return null;
|
||||
return .{ .rung = .exfat_serial, .key = std.mem.readInt(u32, vbr[100..104], .little) };
|
||||
}
|
||||
|
||||
const Recognized = struct { identity: Identity, signature: FilesystemKind };
|
||||
|
||||
/// Recognize the filesystem at `start_lba` by its VBR: exFAT first (its serial and
|
||||
/// the `.exfat` signature), else FAT (its serial), else unknown content that keeps
|
||||
/// the MBR disk-signature identity and the historical `.fat` hand-off.
|
||||
fn recognize(reader: SectorReader, start_lba: u64, block0: *const [sector_bytes]u8, index: u8) Recognized {
|
||||
if (exfatIdentity(reader, start_lba)) |id| return .{ .identity = id, .signature = .exfat };
|
||||
if (fatIdentity(reader, start_lba)) |id| return .{ .identity = id, .signature = .fat };
|
||||
return .{ .identity = mbrIdentity(block0, index), .signature = .fat };
|
||||
}
|
||||
|
||||
/// The filesystem signature at `start_lba` when the identity is decided elsewhere
|
||||
/// (a GPT partition keeps its GUID identity but still needs its content's kind).
|
||||
fn signatureAt(reader: SectorReader, start_lba: u64) FilesystemKind {
|
||||
return if (exfatIdentity(reader, start_lba) != null) .exfat else .fat;
|
||||
}
|
||||
|
||||
/// Append every volume on the device `reader` addresses, whose whole-device size
|
||||
/// is `device_blocks`, to `out` (up to `out.len`), returning the count. A GPT
|
||||
/// disk (protective MBR) is enumerated by GPT, authoritatively — a zero count is
|
||||
/// final. Otherwise every fitting MBR entry is a volume; a boot signature with no
|
||||
/// partition entries is a bare FAT spanning the whole device. Each volume's
|
||||
/// [start, count) is validated overflow-safe (the confinement invariant the
|
||||
/// driver's clamp rests on), and each prefers its FAT serial identity over the
|
||||
/// disk signature.
|
||||
pub fn allVolumes(reader: SectorReader, device_blocks: u64, out: []Volume) usize {
|
||||
var block0: [sector_bytes]u8 = undefined;
|
||||
if (!reader.read(0, &block0)) return 0;
|
||||
if (!hasBootSignature(&block0)) return 0;
|
||||
if (isProtectiveMbr(&block0)) return gptAllVolumes(reader, device_blocks, out);
|
||||
var count: usize = 0;
|
||||
var index: u8 = 0;
|
||||
while (index < 4 and count < out.len) : (index += 1) {
|
||||
const entry = block0[446 + @as(usize, index) * 16 ..][0..16];
|
||||
const kind = entry[4];
|
||||
const start = std.mem.readInt(u32, entry[8..12], .little);
|
||||
const size = std.mem.readInt(u32, entry[12..16], .little);
|
||||
if (kind == 0 or start == 0 or size == 0) continue;
|
||||
// These bytes come off an untrusted removable medium. A partition that
|
||||
// does not fit inside the device is not a partition — skip it. This is
|
||||
// where the driver's confinement-safety invariant is established: the
|
||||
// clamp's overflow-safety rests on base + count staying inside the
|
||||
// device (usb-storage.zig resolveTransfer), which only holds because the
|
||||
// range handed down is validated here. The subtraction cannot overflow.
|
||||
if (start > device_blocks or device_blocks - start < size) continue;
|
||||
const found = recognize(reader, start, &block0, index);
|
||||
out[count] = .{ .base_lba = start, .block_count = size, .identity = found.identity, .signature = found.signature };
|
||||
count += 1;
|
||||
}
|
||||
if (count == 0 and out.len > 0) {
|
||||
// No partition entries: a bare FAT or exFAT spanning the device.
|
||||
const found = recognize(reader, 0, &block0, 0);
|
||||
out[0] = .{ .base_lba = 0, .block_count = device_blocks, .identity = found.identity, .signature = found.signature };
|
||||
return 1;
|
||||
}
|
||||
return count;
|
||||
}
|
||||
|
||||
/// firstVolume is allVolumes into a one-element buffer.
|
||||
const one_volume_slot = 1;
|
||||
|
||||
/// The first volume on the device, or null — the single-volume case of
|
||||
/// `allVolumes`, kept for callers that want just one.
|
||||
pub fn firstVolume(reader: SectorReader, device_blocks: u64) ?Volume {
|
||||
var one: [one_volume_slot]Volume = undefined;
|
||||
return if (allVolumes(reader, device_blocks, &one) > 0) one[0] else null;
|
||||
}
|
||||
|
||||
/// A read-only RAM disk over a byte slice of sectors, for the host tests.
|
||||
const RamDisk = struct {
|
||||
sectors: []const u8,
|
||||
|
||||
fn readFn(context: *anyopaque, lba: u64, buffer: *[sector_bytes]u8) bool {
|
||||
const self: *const RamDisk = @ptrCast(@alignCast(context));
|
||||
const off = lba * sector_bytes;
|
||||
if (off + sector_bytes > self.sectors.len) return false;
|
||||
@memcpy(buffer, self.sectors[off..][0..sector_bytes]);
|
||||
return true;
|
||||
}
|
||||
|
||||
fn reader(self: *const RamDisk) SectorReader {
|
||||
return .{ .context = @constCast(self), .readFn = readFn };
|
||||
}
|
||||
};
|
||||
|
||||
test "an MBR with one partition yields its range and a distinct identity" {
|
||||
var block0 = [_]u8{0} ** 512;
|
||||
block0[510] = 0x55;
|
||||
block0[511] = 0xAA;
|
||||
std.mem.writeInt(u32, block0[440..444], 0xDEADBEEF, .little);
|
||||
// partition 0: type 0x0c (FAT32 LBA), start 2048, size 100000
|
||||
block0[446 + 4] = 0x0c;
|
||||
std.mem.writeInt(u32, block0[446 + 8 ..][0..4], 2048, .little);
|
||||
std.mem.writeInt(u32, block0[446 + 12 ..][0..4], 100000, .little);
|
||||
const disk = RamDisk{ .sectors = &block0 };
|
||||
const v = firstVolume(disk.reader(), 200000).?;
|
||||
try std.testing.expectEqual(@as(u64, 2048), v.base_lba);
|
||||
try std.testing.expectEqual(@as(u64, 100000), v.block_count);
|
||||
try std.testing.expectEqual(Rung.mbr_index, v.identity.rung);
|
||||
try std.testing.expectEqual((@as(u128, 0xDEADBEEF) << 8) | 0, v.identity.key);
|
||||
}
|
||||
|
||||
test "a boot signature with no partitions is a bare FAT over the whole device" {
|
||||
var block0 = [_]u8{0} ** 512;
|
||||
block0[510] = 0x55;
|
||||
block0[511] = 0xAA;
|
||||
const disk = RamDisk{ .sectors = &block0 };
|
||||
const v = firstVolume(disk.reader(), 65536).?;
|
||||
try std.testing.expectEqual(@as(u64, 0), v.base_lba);
|
||||
try std.testing.expectEqual(@as(u64, 65536), v.block_count);
|
||||
}
|
||||
|
||||
test "no boot signature is no volume" {
|
||||
const block0 = [_]u8{0} ** 512;
|
||||
const disk = RamDisk{ .sectors = &block0 };
|
||||
try std.testing.expect(firstVolume(disk.reader(), 65536) == null);
|
||||
}
|
||||
test "a bare exFAT volume is recognized by its VBR, with its serial as the id" {
|
||||
var block0 = [_]u8{0} ** 512;
|
||||
@memcpy(block0[3..11], "EXFAT ");
|
||||
block0[510] = 0x55;
|
||||
block0[511] = 0xAA;
|
||||
std.mem.writeInt(u32, block0[100..104], 0xDA7A0001, .little); // VolumeSerialNumber
|
||||
const disk = RamDisk{ .sectors = &block0 };
|
||||
const v = firstVolume(disk.reader(), 65536).?;
|
||||
try std.testing.expectEqual(FilesystemKind.exfat, v.signature);
|
||||
try std.testing.expectEqual(Rung.exfat_serial, v.identity.rung);
|
||||
try std.testing.expectEqual(@as(u128, 0xDA7A0001), v.identity.key);
|
||||
}
|
||||
test "a FAT VBR is recognized as fat, not exfat — the signatures never collide" {
|
||||
var block0 = [_]u8{0} ** 512;
|
||||
@memcpy(block0[3..11], "MSWIN4.1"); // a FAT OEM name, not "EXFAT "
|
||||
block0[510] = 0x55;
|
||||
block0[511] = 0xAA;
|
||||
std.mem.writeInt(u16, block0[22..24], 16, .little); // fat_size_16 != 0 -> FAT16 shape
|
||||
block0[38] = 0x29; // extended boot signature
|
||||
std.mem.writeInt(u32, block0[39..43], 0x12345678, .little); // volume id
|
||||
const disk = RamDisk{ .sectors = &block0 };
|
||||
const v = firstVolume(disk.reader(), 65536).?;
|
||||
try std.testing.expectEqual(FilesystemKind.fat, v.signature);
|
||||
try std.testing.expectEqual(Rung.fat_serial, v.identity.rung);
|
||||
}
|
||||
|
||||
test "a partition that runs past the device is skipped, not trusted" {
|
||||
var block0 = [_]u8{0} ** 512;
|
||||
block0[510] = 0x55;
|
||||
block0[511] = 0xAA;
|
||||
// partition 0: start 0xFFFFFF00, size 0x400 — far past a 200000-block device.
|
||||
block0[446 + 4] = 0x0c;
|
||||
std.mem.writeInt(u32, block0[446 + 8 ..][0..4], 0xFFFFFF00, .little);
|
||||
std.mem.writeInt(u32, block0[446 + 12 ..][0..4], 0x400, .little);
|
||||
// partition 1: start 2048, size 1000 — fits.
|
||||
block0[462 + 4] = 0x0c;
|
||||
std.mem.writeInt(u32, block0[462 + 8 ..][0..4], 2048, .little);
|
||||
std.mem.writeInt(u32, block0[462 + 12 ..][0..4], 1000, .little);
|
||||
const disk = RamDisk{ .sectors = &block0 };
|
||||
const v = firstVolume(disk.reader(), 200000).?;
|
||||
try std.testing.expectEqual(@as(u64, 2048), v.base_lba); // the fitting one, not the overflowing one
|
||||
try std.testing.expectEqual(@as(u64, 1000), v.block_count);
|
||||
}
|
||||
|
||||
/// A single 128-byte GPT partition entry for the tests.
|
||||
fn gptEntry(type_nonzero: bool, unique_guid: u128, start: u64, end: u64) [128]u8 {
|
||||
var e = [_]u8{0} ** 128;
|
||||
if (type_nonzero) e[0] = 0x01; // any non-zero byte makes the type GUID non-zero
|
||||
std.mem.writeInt(u128, e[16..32], unique_guid, .little);
|
||||
std.mem.writeInt(u64, e[32..40], start, .little);
|
||||
std.mem.writeInt(u64, e[40..48], end, .little);
|
||||
return e;
|
||||
}
|
||||
|
||||
/// Lay out a disk with `entry_size`-spaced GPT entries: protective MBR (LBA 0),
|
||||
/// GPT header with a correct CRC (LBA 1), the entry array (LBA 2+).
|
||||
fn buildGptDiskSized(disk: []u8, entries: []const [128]u8, entry_size: u32) void {
|
||||
@memset(disk, 0);
|
||||
disk[510] = 0x55;
|
||||
disk[511] = 0xAA;
|
||||
disk[446 + 4] = 0xEE; // protective entry type
|
||||
std.mem.writeInt(u32, disk[446 + 8 ..][0..4], 1, .little);
|
||||
std.mem.writeInt(u32, disk[446 + 12 ..][0..4], 0xFFFFFFFF, .little);
|
||||
const h = disk[sector_bytes..][0..sector_bytes];
|
||||
@memcpy(h[0..8], gpt_signature);
|
||||
std.mem.writeInt(u32, h[12..16], 92, .little); // header_size
|
||||
std.mem.writeInt(u64, h[72..80], 2, .little); // partition_entry_lba
|
||||
std.mem.writeInt(u32, h[80..84], @intCast(entries.len), .little);
|
||||
std.mem.writeInt(u32, h[84..88], entry_size, .little); // size_of_partition_entry
|
||||
@memset(h[16..20], 0);
|
||||
std.mem.writeInt(u32, h[16..20], crc32(h[0..92]), .little);
|
||||
const step: usize = @intCast(entry_size);
|
||||
var i: usize = 0;
|
||||
while (i < entries.len) : (i += 1) {
|
||||
const abs = 2 * sector_bytes + i * step;
|
||||
@memcpy(disk[abs..][0..128], &entries[i]);
|
||||
}
|
||||
}
|
||||
|
||||
/// The common 128-byte-entry case.
|
||||
fn buildGptDisk(disk: []u8, entries: []const [128]u8) void {
|
||||
buildGptDiskSized(disk, entries, 128);
|
||||
}
|
||||
|
||||
test "a GPT disk yields the partition GUID as the identity id" {
|
||||
var disk = [_]u8{0} ** (4 * sector_bytes);
|
||||
const guid: u128 = 0x112233445566778899AABBCCDDEEFF00;
|
||||
const entries = [_][128]u8{gptEntry(true, guid, 2048, 4095)};
|
||||
buildGptDisk(&disk, &entries);
|
||||
const rd = RamDisk{ .sectors = &disk };
|
||||
const v = firstVolume(rd.reader(), 200000).?;
|
||||
try std.testing.expectEqual(@as(u64, 2048), v.base_lba);
|
||||
try std.testing.expectEqual(@as(u64, 2048), v.block_count); // 4095 - 2048 + 1
|
||||
try std.testing.expectEqual(Rung.gpt_guid, v.identity.rung);
|
||||
try std.testing.expectEqual(guid, v.identity.key);
|
||||
}
|
||||
|
||||
test "a GPT entry past the device is skipped; an all-out-of-range table is no volume" {
|
||||
var disk = [_]u8{0} ** (4 * sector_bytes);
|
||||
const entries = [_][128]u8{
|
||||
gptEntry(true, 0xAAA, 2048, 999999), // ends past a 200000-block device
|
||||
gptEntry(true, 0xBBB, 4096, 8191), // fits
|
||||
};
|
||||
buildGptDisk(&disk, &entries);
|
||||
const rd = RamDisk{ .sectors = &disk };
|
||||
const v = firstVolume(rd.reader(), 200000).?;
|
||||
try std.testing.expectEqual(@as(u64, 4096), v.base_lba); // the fitting one, not the overflowing one
|
||||
try std.testing.expectEqual(@as(u128, 0xBBB), v.identity.key);
|
||||
|
||||
var solo_disk = [_]u8{0} ** (4 * sector_bytes);
|
||||
const solo = [_][128]u8{gptEntry(true, 0xAAA, 2048, 999999)};
|
||||
buildGptDisk(&solo_disk, &solo);
|
||||
const rd2 = RamDisk{ .sectors = &solo_disk };
|
||||
try std.testing.expect(firstVolume(rd2.reader(), 200000) == null);
|
||||
}
|
||||
|
||||
test "a protective MBR with a broken GPT header is not a volume" {
|
||||
var disk = [_]u8{0} ** (4 * sector_bytes);
|
||||
const entries = [_][128]u8{gptEntry(true, 0xCCC, 2048, 4095)};
|
||||
buildGptDisk(&disk, &entries);
|
||||
disk[sector_bytes] = 'X'; // wreck the 'EFI PART' signature
|
||||
const rd = RamDisk{ .sectors = &disk };
|
||||
try std.testing.expect(firstVolume(rd.reader(), 200000) == null);
|
||||
|
||||
var bad_crc = [_]u8{0} ** (4 * sector_bytes);
|
||||
buildGptDisk(&bad_crc, &entries);
|
||||
bad_crc[sector_bytes + 16] ^= 0xFF; // corrupt a header-CRC byte
|
||||
const rd2 = RamDisk{ .sectors = &bad_crc };
|
||||
try std.testing.expect(firstVolume(rd2.reader(), 200000) == null);
|
||||
}
|
||||
|
||||
test "a bare FAT32 reports its volume serial and label as the identity" {
|
||||
var block0 = [_]u8{0} ** 512;
|
||||
block0[510] = 0x55;
|
||||
block0[511] = 0xAA;
|
||||
std.mem.writeInt(u16, block0[22..24], 0, .little); // fat_size_16 == 0 → FAT32
|
||||
block0[66] = 0x29; // FAT32 extended boot signature
|
||||
std.mem.writeInt(u32, block0[67..71], 0x12345678, .little); // BS_VolID
|
||||
@memcpy(block0[71..82], "DANOS "); // BS_VolLab, space-padded to 11
|
||||
const disk = RamDisk{ .sectors = &block0 };
|
||||
const v = firstVolume(disk.reader(), 65536).?;
|
||||
try std.testing.expectEqual(@as(u64, 0), v.base_lba);
|
||||
try std.testing.expectEqual(Rung.fat_serial, v.identity.rung);
|
||||
try std.testing.expectEqual(@as(u128, 0x12345678), v.identity.key);
|
||||
try std.testing.expectEqualStrings("DANOS", v.identity.labelSlice());
|
||||
}
|
||||
|
||||
test "an MBR FAT partition prefers the volume serial; a non-FAT partition keeps rung 4" {
|
||||
var disk = [_]u8{0} ** (3 * 512);
|
||||
disk[510] = 0x55;
|
||||
disk[511] = 0xAA;
|
||||
std.mem.writeInt(u32, disk[440..444], 0xDEADBEEF, .little);
|
||||
disk[446 + 4] = 0x0c; // FAT32-LBA partition
|
||||
std.mem.writeInt(u32, disk[446 + 8 ..][0..4], 1, .little); // start LBA 1
|
||||
std.mem.writeInt(u32, disk[446 + 12 ..][0..4], 2, .little); // size 2
|
||||
const vbr = disk[512..][0..512]; // a FAT16 VBR at the partition start
|
||||
vbr[510] = 0x55;
|
||||
vbr[511] = 0xAA;
|
||||
std.mem.writeInt(u16, vbr[22..24], 0x0080, .little); // fat_size_16 != 0 → FAT16
|
||||
vbr[38] = 0x29; // FAT12/16 extended boot signature
|
||||
std.mem.writeInt(u32, vbr[39..43], 0xCAFEBABE, .little);
|
||||
@memcpy(vbr[43..54], "MYVOL ");
|
||||
const rd = RamDisk{ .sectors = &disk };
|
||||
const v = firstVolume(rd.reader(), 200000).?;
|
||||
try std.testing.expectEqual(@as(u64, 1), v.base_lba);
|
||||
try std.testing.expectEqual(Rung.fat_serial, v.identity.rung);
|
||||
try std.testing.expectEqual(@as(u128, 0xCAFEBABE), v.identity.key);
|
||||
try std.testing.expectEqualStrings("MYVOL", v.identity.labelSlice());
|
||||
|
||||
// A partition whose VBR is not an extended FAT falls back to the rung-4 id.
|
||||
var plain = [_]u8{0} ** (3 * 512);
|
||||
@memcpy(plain[0..512], disk[0..512]); // same MBR; LBA 1 left blank
|
||||
const rd2 = RamDisk{ .sectors = &plain };
|
||||
const v2 = firstVolume(rd2.reader(), 200000).?;
|
||||
try std.testing.expectEqual(Rung.mbr_index, v2.identity.rung);
|
||||
try std.testing.expectEqual((@as(u128, 0xDEADBEEF) << 8) | 0, v2.identity.key);
|
||||
}
|
||||
|
||||
test "GPT with 256-byte entries reads the non-128 offset arithmetic correctly" {
|
||||
// With entry_size 256, entry 1 lands at offset 256 of the same sector (LBA 2).
|
||||
// Put the only valid entry at index 1 so the off = (i*entry_size) % 512 path
|
||||
// (256, not 0) is exercised — the sharp edge the 128-byte tests never hit.
|
||||
var disk = [_]u8{0} ** (5 * sector_bytes);
|
||||
const entries = [_][128]u8{
|
||||
gptEntry(false, 0, 0, 0), // index 0: unused (type GUID zero)
|
||||
gptEntry(true, 0xF00D, 4096, 8191), // index 1: at offset 256
|
||||
};
|
||||
buildGptDiskSized(&disk, &entries, 256);
|
||||
const rd = RamDisk{ .sectors = &disk };
|
||||
const v = firstVolume(rd.reader(), 200000).?;
|
||||
try std.testing.expectEqual(@as(u64, 4096), v.base_lba);
|
||||
try std.testing.expectEqual(Rung.gpt_guid, v.identity.rung);
|
||||
try std.testing.expectEqual(@as(u128, 0xF00D), v.identity.key);
|
||||
}
|
||||
|
||||
// A fixture-sized volume buffer for the multi-volume tests, named so the bounds
|
||||
// gate (which flags literal array lengths) stays quiet: a test input.
|
||||
const test_volume_slots = 4;
|
||||
|
||||
test "allVolumes returns every fitting MBR partition with distinct identities" {
|
||||
var block0 = [_]u8{0} ** 512;
|
||||
block0[510] = 0x55;
|
||||
block0[511] = 0xAA;
|
||||
std.mem.writeInt(u32, block0[440..444], 0xDEADBEEF, .little);
|
||||
// partition 0: start 2048, size 1000
|
||||
block0[446 + 4] = 0x0c;
|
||||
std.mem.writeInt(u32, block0[446 + 8 ..][0..4], 2048, .little);
|
||||
std.mem.writeInt(u32, block0[446 + 12 ..][0..4], 1000, .little);
|
||||
// partition 1: start 4096, size 2000
|
||||
block0[462 + 4] = 0x0c;
|
||||
std.mem.writeInt(u32, block0[462 + 8 ..][0..4], 4096, .little);
|
||||
std.mem.writeInt(u32, block0[462 + 12 ..][0..4], 2000, .little);
|
||||
const disk = RamDisk{ .sectors = &block0 };
|
||||
var vols: [test_volume_slots]Volume = undefined;
|
||||
const n = allVolumes(disk.reader(), 200000, &vols);
|
||||
try std.testing.expectEqual(@as(usize, 2), n); // both partitions, not just the first
|
||||
try std.testing.expectEqual(@as(u64, 2048), vols[0].base_lba);
|
||||
try std.testing.expectEqual(@as(u64, 4096), vols[1].base_lba);
|
||||
// distinct rung-4 identities (no FAT VBR at those LBAs): index 0 vs 1.
|
||||
try std.testing.expectEqual((@as(u128, 0xDEADBEEF) << 8) | 0, vols[0].identity.key);
|
||||
try std.testing.expectEqual((@as(u128, 0xDEADBEEF) << 8) | 1, vols[1].identity.key);
|
||||
// firstVolume (the 1-buffer case) still returns just the first.
|
||||
try std.testing.expectEqual(@as(u64, 2048), firstVolume(disk.reader(), 200000).?.base_lba);
|
||||
}
|
||||
@@ -0,0 +1,604 @@
|
||||
//! system/services/volume-manager — the storage layer's policy home
|
||||
//! (docs/file-system-development/storage-architecture.md). Beside the device
|
||||
//! manager: that owns the DEVICE tree, this owns the VOLUME layer. It probes a
|
||||
//! storage provider's partition table, confines each filesystem to its
|
||||
//! partition, spawns one filesystem per volume, and answers that filesystem's
|
||||
//! startup hello with the range-confined block channel — so the filesystem
|
||||
//! never finds its storage by name and never sees the whole device. It
|
||||
//! supervises the filesystems it spawns, exactly as the device manager
|
||||
//! supervises drivers.
|
||||
//!
|
||||
//! The manager holds a table of adopted storage DEVICES and a table of the
|
||||
//! VOLUMES on them: it adopts every storage device the device-manager tree
|
||||
//! carries, probes each one's whole partition table, and spawns one filesystem
|
||||
//! process per volume — each confined to its partition's badge-scoped block
|
||||
//! range, each supervised with its own budget. A device leaving the tree takes
|
||||
//! its volumes with it.
|
||||
|
||||
const std = @import("std");
|
||||
const channel = @import("channel");
|
||||
const device_manager_protocol = @import("device-manager-protocol");
|
||||
const volume_manager_protocol = @import("volume-manager-protocol");
|
||||
const driver = @import("driver");
|
||||
const ipc = @import("ipc");
|
||||
const block = @import("block");
|
||||
const memory = @import("memory");
|
||||
const logging = @import("logging");
|
||||
const process = @import("process");
|
||||
const service = @import("service");
|
||||
const time = @import("time");
|
||||
const envelope = @import("envelope");
|
||||
const fs = @import("file-system");
|
||||
const partition = @import("partition.zig");
|
||||
const filesystem_map = @import("filesystem-map.zig");
|
||||
const volume_map = @import("volume-map.zig");
|
||||
|
||||
const Serve = volume_manager_protocol.Protocol.Provider(void);
|
||||
const Invocation = envelope.Invocation;
|
||||
const Answer = envelope.Answer;
|
||||
|
||||
/// One adopted storage device: the block channel to its provider (opened once and
|
||||
/// shared — refcounted per confined filesystem via the hello reply) and the
|
||||
/// device-manager id it serves. A device leaving the tree takes its volumes.
|
||||
const StorageDevice = struct {
|
||||
used: bool = false,
|
||||
device_id: u64 = 0,
|
||||
channel: block.Device = undefined,
|
||||
};
|
||||
|
||||
/// One volume: which device serves it, its block sub-range, its content
|
||||
/// identity, the id it is addressed by, the service binary + mount path it was
|
||||
/// spawned with, the filesystem process serving it, and its own supervision
|
||||
/// budget (so one volume's crash loop never touches another's).
|
||||
const Volume = struct {
|
||||
used: bool = false,
|
||||
device_id: u64 = 0,
|
||||
base_lba: u64 = 0,
|
||||
block_count: u64 = 0,
|
||||
identity: partition.Identity = .{ .rung = .anonymous },
|
||||
id: u64 = 0,
|
||||
binary: []const u8 = "",
|
||||
mount_prefix: []const u8 = "",
|
||||
filesystem_pid: u32 = 0,
|
||||
// Per-volume supervision, mirroring the device manager's: a clean exit is not
|
||||
// restarted, a fault restarts with backoff, a fast crash loop gives up.
|
||||
restarts: u32 = 0,
|
||||
spawn_ns: u64 = 0,
|
||||
failed: bool = false,
|
||||
restart_pending: bool = false,
|
||||
restart_due_ns: u64 = 0,
|
||||
};
|
||||
|
||||
// The mount map, read from configuration at boot (the policy home, storage-
|
||||
// architecture.md): filesystems.csv (content signature -> service binary) and
|
||||
// volumes.csv (an optional id -> mount-prefix override). The sources are held
|
||||
// for the process life so the parsed rules' slices into them stay valid.
|
||||
/// bound: bytes of filesystems.csv / volumes.csv the manager reads
|
||||
/// decided-by: ours
|
||||
/// protects: the config source buffers below
|
||||
/// at-limit: truncate - a longer file is cut; a row split by the cut is malformed
|
||||
/// observed-by: the per-file "malformed/truncated" log line
|
||||
const config_source_bytes = 2048;
|
||||
var filesystems_source: [config_source_bytes]u8 = undefined;
|
||||
var volumes_source: [config_source_bytes]u8 = undefined;
|
||||
/// bound: filesystem-map rules held (one per content signature)
|
||||
/// decided-by: ours
|
||||
/// protects: the filesystem_rules table
|
||||
/// at-limit: truncate - extra rows are dropped and the "truncated" note logged
|
||||
/// observed-by: the "truncated" log line
|
||||
const maximum_filesystem_rules = 8;
|
||||
/// bound: volumes.csv override rows held (one per pinned volume id)
|
||||
/// decided-by: ours
|
||||
/// protects: the volume_rules table
|
||||
/// at-limit: truncate - extra rows are dropped and the "truncated" note logged
|
||||
/// observed-by: the "truncated" log line
|
||||
const maximum_volume_rules = 64;
|
||||
var filesystem_rules: [maximum_filesystem_rules]filesystem_map.Rule = undefined;
|
||||
var filesystem_rule_count: usize = 0;
|
||||
var volume_rules: [maximum_volume_rules]volume_map.Override = undefined;
|
||||
var volume_rule_count: usize = 0;
|
||||
/// bound: bytes of a composed /volumes/<id> mount path
|
||||
/// decided-by: ours
|
||||
/// protects: the per-volume mount_prefix buffers below
|
||||
/// at-limit: truncate - bufPrint fails; the volume mounts at a fallback path (logged)
|
||||
/// observed-by: the fallback path in the log
|
||||
const mount_path_maximum = 64;
|
||||
|
||||
/// bound: volumes the manager serves at once
|
||||
/// decided-by: ours
|
||||
/// protects: the volumes table and its per-volume mount-path buffers
|
||||
/// at-limit: truncate - a further partition is left unserved and logged (real
|
||||
/// machines carry a handful of volumes, far under this)
|
||||
/// observed-by: the "volume table full" log line
|
||||
const maximum_volumes = 16;
|
||||
/// bound: storage devices the manager adopts at once
|
||||
/// decided-by: ours
|
||||
/// protects: the devices table
|
||||
/// at-limit: truncate - a further device is left unadopted and logged
|
||||
/// observed-by: the "device table full" log line
|
||||
const maximum_devices = 8;
|
||||
var devices = [_]StorageDevice{.{}} ** maximum_devices;
|
||||
var volumes = [_]Volume{.{}} ** maximum_volumes;
|
||||
/// Each volume's composed default mount path lives in its slot's buffer; a
|
||||
/// volumes.csv override is used in place (a slice into volumes_source, no buffer).
|
||||
var mount_prefix_bufs: [maximum_volumes][mount_path_maximum]u8 = undefined;
|
||||
var next_volume_id: u64 = 1; // monotonic — never reused, so a stale id can't address the wrong child
|
||||
|
||||
var service_endpoint: ipc.Handle = 0;
|
||||
var manager_handle: ?ipc.Handle = null;
|
||||
var bounce: memory.DmaRegion = undefined;
|
||||
var bounce_ready = false;
|
||||
/// How often the poll checks device presence and fires due restarts. Fast enough
|
||||
/// that an unplug unmounts promptly; the poll is a bare device-manager enumerate,
|
||||
/// no channel work, so it is cheap to run continuously.
|
||||
const poll_interval_ms = 500;
|
||||
|
||||
// Filesystem supervision, mirroring the device manager's (device-manager.zig).
|
||||
const fast_death_ns: u64 = 2_000_000_000;
|
||||
const crash_loop_cap: u32 = 3;
|
||||
const backoff_base_ms: u64 = 300;
|
||||
/// bytes to format a u64 volume id as decimal (20 digits fit)
|
||||
const id_decimal_bytes = 24;
|
||||
|
||||
// --- table lookups -----------------------------------------------------------
|
||||
|
||||
fn deviceById(id: u64) ?*StorageDevice {
|
||||
for (&devices) |*d| if (d.used and d.device_id == id) return d;
|
||||
return null;
|
||||
}
|
||||
fn claimDevice() ?*StorageDevice {
|
||||
for (&devices) |*d| if (!d.used) return d;
|
||||
return null;
|
||||
}
|
||||
fn volumeById(id: u64) ?*Volume {
|
||||
for (&volumes) |*v| if (v.used and v.id == id) return v;
|
||||
return null;
|
||||
}
|
||||
fn volumeByPid(pid: u32) ?*Volume {
|
||||
for (&volumes) |*v| if (v.used and v.filesystem_pid == pid) return v;
|
||||
return null;
|
||||
}
|
||||
fn firstUsedVolume() ?*Volume {
|
||||
for (&volumes) |*v| if (v.used) return v;
|
||||
return null;
|
||||
}
|
||||
fn claimVolumeIndex() ?usize {
|
||||
for (&volumes, 0..) |*v, i| if (!v.used) return i;
|
||||
return null;
|
||||
}
|
||||
|
||||
// --- device-manager plumbing -------------------------------------------------
|
||||
|
||||
fn deviceManager() ?ipc.Handle {
|
||||
if (manager_handle) |h| return h;
|
||||
const handle = channel.openEndpoint("device-manager") orelse return null;
|
||||
manager_handle = handle;
|
||||
return handle;
|
||||
}
|
||||
|
||||
const OpenedStorage = struct { device_id: u64, device: block.Device };
|
||||
|
||||
/// The first mass-storage provider whose block channel opens and is NOT already
|
||||
/// adopted, with its device id. A device-manager tree can carry more than one
|
||||
/// entry of the mass-storage identity — a phantom that no driver is bound to
|
||||
/// answers a consumer hello with NO channel — so this tries each and takes the
|
||||
/// first that yields a channel. Skips already-adopted devices so a re-poll does
|
||||
/// not re-open a device it already serves.
|
||||
fn openAnyStorage() ?OpenedStorage {
|
||||
const manager = deviceManager() orelse return null;
|
||||
const Entry = device_manager_protocol.ChildEntry;
|
||||
var start: u64 = 0;
|
||||
while (true) {
|
||||
const enumerate = envelope.Header{ .operation = envelope.operation_enumerate, .target = start };
|
||||
var reply: [device_manager_protocol.message_maximum]u8 = undefined;
|
||||
const length = ipc.call(manager, std.mem.asBytes(&enumerate), &reply) catch return null;
|
||||
const status = envelope.statusOf(reply[0..length]) orelse return null;
|
||||
if (status.status != 0) return null;
|
||||
const carried = @min(@as(usize, status.len), length -| envelope.prefix_size);
|
||||
const tail = reply[envelope.prefix_size..][0..carried];
|
||||
const count = tail.len / @sizeOf(Entry);
|
||||
if (count == 0) return null;
|
||||
var index: usize = 0;
|
||||
while (index < count) : (index += 1) {
|
||||
const entry = std.mem.bytesToValue(Entry, tail[index * @sizeOf(Entry) ..][0..@sizeOf(Entry)]);
|
||||
if (entry.device_id == device_manager_protocol.no_device) continue;
|
||||
if ((entry.identity >> 16) & 0xff != 0x08 or (entry.identity >> 8) & 0xff != 0x06) continue;
|
||||
if (deviceById(entry.device_id) != null) continue; // already adopted
|
||||
const exchanged = driver.helloOn(manager, .consumer, entry.device_id, null, true) orelse continue;
|
||||
const provider = exchanged.channel orelse continue; // a phantom / not-yet-bound entry
|
||||
return .{ .device_id = entry.device_id, .device = .{ .endpoint = provider } };
|
||||
}
|
||||
start += count;
|
||||
}
|
||||
}
|
||||
|
||||
/// Whether `device_id` is still in the device-manager tree — a bare enumerate,
|
||||
/// no consumer-hello, so it is cheap to call every poll. This is how removal is
|
||||
/// detected: the specific device a mounted volume sits on disappears.
|
||||
fn isDevicePresent(device_id: u64) bool {
|
||||
const manager = deviceManager() orelse return false;
|
||||
const Entry = device_manager_protocol.ChildEntry;
|
||||
var start: u64 = 0;
|
||||
while (true) {
|
||||
const enumerate = envelope.Header{ .operation = envelope.operation_enumerate, .target = start };
|
||||
var reply: [device_manager_protocol.message_maximum]u8 = undefined;
|
||||
const length = ipc.call(manager, std.mem.asBytes(&enumerate), &reply) catch return false;
|
||||
const status = envelope.statusOf(reply[0..length]) orelse return false;
|
||||
if (status.status != 0) return false;
|
||||
const carried = @min(@as(usize, status.len), length -| envelope.prefix_size);
|
||||
const tail = reply[envelope.prefix_size..][0..carried];
|
||||
const count = tail.len / @sizeOf(Entry);
|
||||
if (count == 0) return false;
|
||||
var index: usize = 0;
|
||||
while (index < count) : (index += 1) {
|
||||
const entry = std.mem.bytesToValue(Entry, tail[index * @sizeOf(Entry) ..][0..@sizeOf(Entry)]);
|
||||
if (entry.device_id == device_id) return true;
|
||||
}
|
||||
start += count;
|
||||
}
|
||||
}
|
||||
|
||||
// --- lifecycle ---------------------------------------------------------------
|
||||
|
||||
/// Spawn the filesystem for `v`, confine it to the volume's range on its device's
|
||||
/// channel, and record its pid. The confinement is defined for the fresh pid
|
||||
/// BEFORE the filesystem runs, so its first read is already bounded; the volume
|
||||
/// manager is the confinement controller (it defines the first range on the
|
||||
/// device).
|
||||
fn spawnFilesystem(v: *Volume) void {
|
||||
if (v.failed) return;
|
||||
const dev = deviceById(v.device_id) orelse return; // its device left — poll will clean up
|
||||
var id_str_buf: [id_decimal_bytes]u8 = undefined;
|
||||
const id_str = std.fmt.bufPrint(&id_str_buf, "{d}", .{v.id}) catch "1";
|
||||
const pid = process.spawnSupervised(v.binary, &.{ id_str, v.mount_prefix }, service_endpoint) orelse {
|
||||
_ = logging.write("volume-manager: could not spawn the filesystem; retrying\n");
|
||||
armRestart(v);
|
||||
return;
|
||||
};
|
||||
if (!dev.channel.defineRange(pid, v.base_lba, v.block_count)) {
|
||||
_ = logging.write("volume-manager: could not confine the filesystem to its volume; retrying\n");
|
||||
_ = process.kill(pid);
|
||||
armRestart(v);
|
||||
return;
|
||||
}
|
||||
v.filesystem_pid = pid;
|
||||
v.spawn_ns = time.clock();
|
||||
std.log.info("volume 0x{x} -> {s} (pid {d}), lba {d}, {d} blocks", .{ v.identity.key, v.binary, pid, v.base_lba, v.block_count });
|
||||
}
|
||||
|
||||
/// Schedule a restart for `v` after backoff; the poll loop performs it once due.
|
||||
fn armRestart(v: *Volume) void {
|
||||
const delay = if (v.restarts == 0) backoff_base_ms else backoff_base_ms << @intCast(@min(v.restarts - 1, 5));
|
||||
v.restart_due_ns = time.clock() + delay * 1_000_000;
|
||||
v.restart_pending = true;
|
||||
}
|
||||
|
||||
/// Compose a volume's mount path (its id-path `/volumes/<id>`, or a volumes.csv
|
||||
/// override) into its slot's buffer, and return the slice.
|
||||
fn composeMountPrefix(slot: usize, identity: partition.Identity) []const u8 {
|
||||
var id_buf: [volume_map.id_maximum]u8 = undefined;
|
||||
const id = volume_map.idString(identity, &id_buf);
|
||||
return volume_map.overrideFor(volume_rules[0..volume_rule_count], id) orelse
|
||||
(std.fmt.bufPrint(&mount_prefix_bufs[slot], "/volumes/{s}", .{id}) catch "/volumes/unknown");
|
||||
}
|
||||
|
||||
/// Adopt the next present, not-yet-adopted storage device: take its channel,
|
||||
/// probe its whole partition table, and spawn a filesystem per volume it carries.
|
||||
/// Returns true when it consumed a device (so the caller can loop to adopt every
|
||||
/// present device in one tick), false when none remain or the device table is full.
|
||||
///
|
||||
/// A device is adopted exactly once and kept until it leaves the tree — even when
|
||||
/// it carries no volume we can serve, or its geometry cannot be read. Keeping the
|
||||
/// empty/unreadable device adopted (rather than dropping and re-probing) is what
|
||||
/// lets openAnyStorage advance PAST it to the devices behind it; dropping it would
|
||||
/// make openAnyStorage hand back the same unservable device every tick and starve
|
||||
/// the rest. A genuine removal frees the slot (removeDevice); a re-insert gets a
|
||||
/// fresh device id and is probed anew.
|
||||
fn bringUpVolume() bool {
|
||||
if (!bounce_ready) {
|
||||
bounce = memory.dmaAlloc(512, memory.dma_coherent | memory.dma_shareable) orelse return false;
|
||||
bounce_ready = true;
|
||||
}
|
||||
const opened = openAnyStorage() orelse return false;
|
||||
const dev = claimDevice() orelse {
|
||||
_ = logging.write("volume-manager: device table full; a storage device is left unadopted\n");
|
||||
_ = ipc.close(opened.device.endpoint);
|
||||
return false;
|
||||
};
|
||||
dev.* = .{ .used = true, .device_id = opened.device_id, .channel = opened.device };
|
||||
// Consume this device's medium_changed events (the second of the removal
|
||||
// lifecycle's two triggers: the device stays in the tree while its medium
|
||||
// leaves — a card reader, an eject). Best effort: a provider that never
|
||||
// publishes the event simply never wakes us, and device-pull is still caught
|
||||
// by the presence poll.
|
||||
_ = opened.device.subscribeMedium(service_endpoint);
|
||||
const device = opened.device;
|
||||
// Attach the read buffer to THIS device (a no-op without an enforcing IOMMU).
|
||||
// The handle is kept, not closed, so it can be re-attached after a replug. A
|
||||
// failed attach or geometry read leaves the device adopted but empty — we just
|
||||
// cannot read it, and the slot still watches it for removal.
|
||||
if (bounce.handle) |handle| {
|
||||
if (!device.attach(handle)) {
|
||||
_ = logging.write("volume-manager: could not attach the read buffer to a storage device; no volume served\n");
|
||||
return true;
|
||||
}
|
||||
}
|
||||
const geometry = device.geometry() orelse {
|
||||
_ = logging.write("volume-manager: could not read a storage device's geometry; no volume served\n");
|
||||
return true;
|
||||
};
|
||||
const ProbeReader = struct {
|
||||
device: block.Device,
|
||||
fn readSector(context: *anyopaque, lba: u64, buffer: *[partition.sector_bytes]u8) bool {
|
||||
const self: *@This() = @ptrCast(@alignCast(context));
|
||||
if (!self.device.read(lba, 1, bounce.physical)) return false;
|
||||
const src: [*]const u8 = @ptrFromInt(bounce.virtual);
|
||||
@memcpy(buffer, src[0..partition.sector_bytes]);
|
||||
return true;
|
||||
}
|
||||
};
|
||||
var probe = ProbeReader{ .device = device };
|
||||
const reader = partition.SectorReader{ .context = &probe, .readFn = ProbeReader.readSector };
|
||||
var found: [maximum_volumes]partition.Volume = undefined;
|
||||
const n = partition.allVolumes(reader, geometry.block_count, found[0..]);
|
||||
if (n == 0) {
|
||||
std.log.info("device {d} present but carries no recognizable volume", .{dev.device_id});
|
||||
return true;
|
||||
}
|
||||
for (found[0..n]) |fv| {
|
||||
// Pick the service binary from the volume's content signature. A signature
|
||||
// no filesystems.csv row serves goes unserved (logged), like an unbound
|
||||
// device — the manager does not guess.
|
||||
const binary = filesystem_map.match(filesystem_rules[0..filesystem_rule_count], fv.signature) orelse {
|
||||
_ = logging.write("volume-manager: no filesystem serves this volume's content; unserved\n");
|
||||
continue;
|
||||
};
|
||||
const slot = claimVolumeIndex() orelse {
|
||||
_ = logging.write("volume-manager: volume table full; a volume is left unserved\n");
|
||||
break;
|
||||
};
|
||||
volumes[slot] = .{
|
||||
.used = true,
|
||||
.device_id = dev.device_id,
|
||||
.base_lba = fv.base_lba,
|
||||
.block_count = fv.block_count,
|
||||
.identity = fv.identity,
|
||||
.id = next_volume_id,
|
||||
.binary = binary,
|
||||
.mount_prefix = composeMountPrefix(slot, fv.identity),
|
||||
};
|
||||
next_volume_id += 1;
|
||||
spawnFilesystem(&volumes[slot]);
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Close a device's channel and free its slot. No volumes are touched (the caller
|
||||
/// ensures none remain, or there never were any).
|
||||
fn dropDevice(dev: *StorageDevice) void {
|
||||
// Free the driver's subscriber slot before the channel closes. On a still-live
|
||||
// channel (a medium eject) this frees the slot; on a dead one (a device pull)
|
||||
// the call fails fast and the exit sweep frees it anyway.
|
||||
_ = dev.channel.unsubscribeMedium();
|
||||
_ = ipc.close(dev.channel.endpoint);
|
||||
dev.* = .{};
|
||||
}
|
||||
|
||||
/// Retire one volume: kill its filesystem so its mounts are retired. Retirement
|
||||
/// is lazy, not an eager death-time sweep — killing the process marks the
|
||||
/// filesystem's backend endpoint dead, and the VFS router drops each mount that
|
||||
/// endpoint backed on the next path resolution under it (that resolve frees the
|
||||
/// slot and returns not_found). Then free the volume slot.
|
||||
fn removeVolumeState(v: *Volume) void {
|
||||
std.log.info("storage for volume {d} removed; unmounting", .{v.id});
|
||||
if (v.filesystem_pid != 0) _ = process.kill(v.filesystem_pid);
|
||||
v.* = .{};
|
||||
}
|
||||
|
||||
/// A storage device left the tree (a pulled stick): retire every volume it served
|
||||
/// and drop its channel. One removal path, whether the device is pulled cleanly
|
||||
/// or vanishes.
|
||||
fn removeDevice(dev: *StorageDevice) void {
|
||||
for (&volumes) |*v| {
|
||||
if (v.used and v.device_id == dev.device_id) removeVolumeState(v);
|
||||
}
|
||||
dropDevice(dev);
|
||||
}
|
||||
|
||||
/// Whether a device's block channel still answers — a geometry() probe. A storage
|
||||
/// driver that DIED while its device stays in the tree (it crashed; the device
|
||||
/// manager will re-delegate the device to a restarted driver on a FRESH channel)
|
||||
/// leaves a dead channel here, even though isDevicePresent still reports the device
|
||||
/// present. geometry() on the dead endpoint fails fast, so this catches the crash
|
||||
/// that presence-polling alone cannot — the V4 review's open edge.
|
||||
fn channelAlive(dev: *StorageDevice) bool {
|
||||
return dev.channel.geometry() != null;
|
||||
}
|
||||
|
||||
/// One poll tick. Device removal is reconciled FIRST and supersedes a pending
|
||||
/// restart: a volume whose device left (a pull) OR whose driver died on a channel
|
||||
/// that no longer answers is retired before its restart could fire, so nothing
|
||||
/// respawns against a dead channel. Dropping the device frees its slot, so the
|
||||
/// adopt loop below re-adopts the still-present device on the restarted driver's
|
||||
/// fresh channel — the rebuild. Then due restarts fire for present volumes.
|
||||
fn pollTick() void {
|
||||
for (&devices) |*dev| {
|
||||
if (dev.used and (!isDevicePresent(dev.device_id) or !channelAlive(dev))) removeDevice(dev);
|
||||
}
|
||||
for (&volumes) |*v| {
|
||||
if (v.used and v.restart_pending and time.clock() >= v.restart_due_ns) {
|
||||
v.restart_pending = false;
|
||||
spawnFilesystem(v);
|
||||
}
|
||||
}
|
||||
// Adopt every present, not-yet-adopted storage device. Each call consumes at
|
||||
// most one device (openAnyStorage skips the adopted), so the loop terminates
|
||||
// once none remain; the maximum_devices guard is insurance against a logic
|
||||
// slip, never the normal exit.
|
||||
var adopted: usize = 0;
|
||||
while (adopted < maximum_devices and bringUpVolume()) : (adopted += 1) {}
|
||||
}
|
||||
|
||||
/// A filesystem announces itself for the volume it was spawned to serve. Reply
|
||||
/// with that volume's block channel (already range-confined to this filesystem's
|
||||
/// badge) as the call's returned capability. No channel means the volume is not
|
||||
/// ready — the filesystem retries.
|
||||
fn onHello(_: void, invocation: Invocation(volume_manager_protocol.Hello), _: Answer(void)) isize {
|
||||
const v = volumeById(invocation.target) orelse return 0; // not probed yet — retryable, no cap
|
||||
if (invocation.sender != v.filesystem_pid) {
|
||||
// Not the filesystem we spawned for this volume. Refuse: only the confined
|
||||
// filesystem gets the channel.
|
||||
std.log.info("refused hello for volume {d} from process {d}", .{ invocation.target, invocation.sender });
|
||||
return -envelope.EPERM;
|
||||
}
|
||||
const dev = deviceById(v.device_id) orelse return 0; // its device left — retryable
|
||||
service.replyWithCapability(dev.channel.endpoint);
|
||||
std.log.info("handed volume {d} to pid {d}", .{ v.id, invocation.sender });
|
||||
return 0;
|
||||
}
|
||||
|
||||
/// Answer a `volumes` query with a mounted volume's descriptor — its id (its
|
||||
/// mount path is /volumes/<id> unless overridden), its actual mount path, and its
|
||||
/// display label. Software keys on the id; a UI shows the label. Returns the first
|
||||
/// mounted volume for now; a full enumerate is a later refinement. Empty reply
|
||||
/// means no volume is mounted.
|
||||
fn onVolumes(_: void, _: Invocation(volume_manager_protocol.Volumes), answer: Answer(void)) isize {
|
||||
const v = firstUsedVolume() orelse return 0;
|
||||
var id_buf: [volume_map.id_maximum]u8 = undefined;
|
||||
const info = volume_manager_protocol.VolumeInfo{
|
||||
.id = volume_map.idString(v.identity, &id_buf),
|
||||
.mount_path = v.mount_prefix,
|
||||
.label = v.identity.labelSlice(),
|
||||
};
|
||||
const encoded = info.encode(answer.tail()) orelse return 0;
|
||||
return @intCast(encoded.len);
|
||||
}
|
||||
|
||||
const handlers = Serve.Handlers{ .hello = onHello, .volumes = onVolumes };
|
||||
|
||||
fn onMessage(message: []const u8, out: []u8, sender: u32, arrived: *ipc.Arrival) usize {
|
||||
// No verb takes a capability up, so the turn closes whatever arrives.
|
||||
return Serve.dispatch({}, handlers, message, sender, arrived.peek(), out);
|
||||
}
|
||||
|
||||
/// Read a config file into `buf`, returning the byte count (0 if missing).
|
||||
fn readConfig(path: []const u8, buf: []u8) usize {
|
||||
var file = fs.open(path, .{}) orelse {
|
||||
std.log.info("volume-manager: {s} missing", .{path});
|
||||
return 0;
|
||||
};
|
||||
defer file.close();
|
||||
var used: usize = 0;
|
||||
while (used < buf.len) {
|
||||
const nn = file.read(buf[used..]) orelse break;
|
||||
if (nn == 0) break;
|
||||
used += nn;
|
||||
}
|
||||
return used;
|
||||
}
|
||||
|
||||
/// Load the mount map from configuration once at boot (mirrors the device
|
||||
/// manager's registry load). A missing or empty filesystems.csv means no volume
|
||||
/// is served; volumes.csv is optional — no rows means every volume takes its
|
||||
/// default /volumes/<id> path.
|
||||
fn loadTables() void {
|
||||
const fs_used = readConfig("/system/configuration/filesystems.csv", &filesystems_source);
|
||||
const fr = filesystem_map.parse(filesystems_source[0..fs_used], &filesystem_rules);
|
||||
filesystem_rule_count = fr.count;
|
||||
if (fr.malformed != 0 or fr.truncated) std.log.info("filesystems.csv: {d} malformed, truncated={}", .{ fr.malformed, fr.truncated });
|
||||
|
||||
const vol_used = readConfig("/system/configuration/volumes.csv", &volumes_source);
|
||||
const vr = volume_map.parse(volumes_source[0..vol_used], &volume_rules);
|
||||
volume_rule_count = vr.count;
|
||||
if (vr.malformed != 0 or vr.truncated) std.log.info("volumes.csv: {d} malformed, truncated={}", .{ vr.malformed, vr.truncated });
|
||||
}
|
||||
|
||||
fn initialise(endpoint: ipc.Handle) bool {
|
||||
service_endpoint = endpoint;
|
||||
_ = logging.write("volume-manager: starting, waiting for a storage device\n");
|
||||
loadTables();
|
||||
_ = process.subscribeExits(endpoint);
|
||||
pollTick();
|
||||
_ = time.timerOnce(endpoint, poll_interval_ms); // the poll runs for the life of the boot
|
||||
return true;
|
||||
}
|
||||
|
||||
fn onNotification(badge: u64) void {
|
||||
const got = ipc.Received{ .len = 0, .badge = badge, .cap = null };
|
||||
if (got.isTimer()) {
|
||||
pollTick();
|
||||
_ = time.timerOnce(service_endpoint, poll_interval_ms); // always re-arm: presence is watched continuously
|
||||
return;
|
||||
}
|
||||
// A filesystem died. The exit reason drives the decision, exactly as the
|
||||
// device manager supervises drivers: a clean exit meant to stop; a fault
|
||||
// restarts with backoff until a fast crash loop gives up. The old range is
|
||||
// reclaimed by the driver on the same death; the respawn confines afresh.
|
||||
if (got.isChildExit()) {
|
||||
const dead = got.childProcessId();
|
||||
const v = volumeByPid(dead) orelse return;
|
||||
v.filesystem_pid = 0;
|
||||
const reason = process.exitReason(dead) orelse .fault;
|
||||
if (reason == .exited) {
|
||||
std.log.info("filesystem for volume {d} exited cleanly; not restarting", .{v.id});
|
||||
return;
|
||||
}
|
||||
const alive = time.clock() -| v.spawn_ns;
|
||||
v.restarts = if (alive < fast_death_ns) v.restarts + 1 else 1;
|
||||
if (v.restarts >= crash_loop_cap) {
|
||||
v.failed = true;
|
||||
std.log.info("filesystem for volume {d} is failing repeatedly; giving up", .{v.id});
|
||||
return;
|
||||
}
|
||||
std.log.info("filesystem for volume {d} died ({s}); restarting", .{ v.id, @tagName(reason) });
|
||||
armRestart(v);
|
||||
}
|
||||
}
|
||||
|
||||
fn anyVolumeOn(device_id: u64) bool {
|
||||
for (&volumes) |*v| if (v.used and v.device_id == device_id) return true;
|
||||
return false;
|
||||
}
|
||||
|
||||
/// A storage device published `medium_changed` — the second removal trigger: the
|
||||
/// device stays in the tree while its medium leaves or returns (a card reader, an
|
||||
/// eject). This arrives as a buffered async message, NOT a protocol request, so it
|
||||
/// never reaches `Serve.dispatch` (its event op number collides with the manager's
|
||||
/// own `hello`); it is decoded here by hand. Single-volume scope: the event names
|
||||
/// no device, so `absent` retires every adopted device (its volumes unmount and
|
||||
/// the poll re-adopts the still-present device with its now-empty medium), and
|
||||
/// `present` frees any empty adopted device so the poll re-probes and remounts it.
|
||||
///
|
||||
/// We act on every edge and do NOT dedup on `change_count`. The driver publishes
|
||||
/// exactly once per transition, each with a unique monotonic count, so a count is
|
||||
/// never legitimately re-sent within one subscription — an equality dedup could
|
||||
/// only ever fire spuriously, and it did: `change_count` restarts at 0 in each
|
||||
/// driver instance (usb-storage.zig), so a global "last count" carried across a
|
||||
/// driver restart (S5's own crash-rebuild) mistook the fresh instance's first
|
||||
/// edge for a re-delivery and dropped a real eject, wedging a mount over absent
|
||||
/// media. Both branches are idempotent (a freed device stops matching `dev.used`)
|
||||
/// and the poll reconciles, so reacting to each genuine edge is safe.
|
||||
fn onMediumEvent(payload: []const u8) void {
|
||||
const event = block.decodeMediumChanged(payload) orelse return;
|
||||
if (event.present == 0) {
|
||||
std.log.info("medium left a storage device; unmounting its volume(s)", .{});
|
||||
for (&devices) |*dev| {
|
||||
if (dev.used) removeDevice(dev);
|
||||
}
|
||||
} else {
|
||||
for (&devices) |*dev| {
|
||||
if (dev.used and !anyVolumeOn(dev.device_id)) removeDevice(dev);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub fn main(init: process.Init) void {
|
||||
_ = init;
|
||||
service.run(volume_manager_protocol.message_maximum, .{
|
||||
.service = "volume-manager",
|
||||
.init = initialise,
|
||||
.on_message = onMessage,
|
||||
.on_notification = onNotification,
|
||||
.on_buffered_message = onMediumEvent,
|
||||
});
|
||||
}
|
||||
@@ -0,0 +1,139 @@
|
||||
//! volume-map — render a volume's identity into its stable mount id-string, and
|
||||
//! parse `/system/configuration/volumes.csv` (danos's fstab) into optional
|
||||
//! id → mount-prefix overrides. This is where the id/label split becomes the
|
||||
//! path: a volume's mount point is derived from its content identity (the id),
|
||||
//! never from a port or a label. Two distinct volumes that share a label get
|
||||
//! distinct id-strings automatically; only identical ids (dd-cloned media) can
|
||||
//! collide, which is the narrow case the manager's duplicate policy is for.
|
||||
//!
|
||||
//! `volumes.csv` is an OPTIONAL override: a row `id, mount_prefix` pins a volume
|
||||
//! (by its id-string) to a chosen path. A volume with no row takes its default
|
||||
//! `/volumes/<id>`. The label is display metadata, exposed by the manager's
|
||||
//! `volumes` query, and never appears here.
|
||||
//!
|
||||
//! Pure logic: no syscalls, no allocator. Override slices point into the CSV
|
||||
//! source, which the manager holds in a static buffer for the process life.
|
||||
|
||||
const std = @import("std");
|
||||
const csv = @import("csv");
|
||||
const partition = @import("partition.zig");
|
||||
|
||||
/// bound: bytes of the longest volume id-string the deriver renders
|
||||
/// decided-by: ours
|
||||
/// protects: the caller's id-string buffer
|
||||
/// at-limit: truncate - bufPrint fails and idString returns ""; the volume goes
|
||||
/// unnamed and the manager logs it rather than mounting at an empty path
|
||||
/// observed-by: a volume with an empty id in the `volumes` query / the log
|
||||
pub const id_maximum = 40; // "gpt-" (4) or "uuid-" (5) + 32 hex fits in 40
|
||||
|
||||
/// One parsed override row: a volume id-string and the mount prefix it pins to.
|
||||
pub const Override = struct { id: []const u8, prefix: []const u8 };
|
||||
|
||||
/// How many overrides landed, how many non-blank lines were malformed, and
|
||||
/// whether there were more rows than the buffer could hold.
|
||||
pub const ParseResult = struct { count: usize, malformed: usize, truncated: bool };
|
||||
|
||||
/// Render a volume's identity into its id-string — the content-derived, unique,
|
||||
/// order-independent token whose default mount path is `/volumes/<id>`. The rung
|
||||
/// tags the scheme so ids never collide across rungs; the key is the content id,
|
||||
/// so a moved drive keeps its id (and thus its path).
|
||||
pub fn idString(identity: partition.Identity, buf: []u8) []const u8 {
|
||||
return switch (identity.rung) {
|
||||
.gpt_guid => std.fmt.bufPrint(buf, "gpt-{x:0>32}", .{identity.key}) catch "",
|
||||
.filesystem_uuid => std.fmt.bufPrint(buf, "uuid-{x:0>32}", .{identity.key}) catch "",
|
||||
.fat_serial => std.fmt.bufPrint(buf, "fat-{x:0>8}", .{@as(u32, @truncate(identity.key))}) catch "",
|
||||
.exfat_serial => std.fmt.bufPrint(buf, "exfat-{x:0>8}", .{@as(u32, @truncate(identity.key))}) catch "",
|
||||
.mbr_index => std.fmt.bufPrint(buf, "mbr-{x}-{d}", .{
|
||||
@as(u32, @truncate(identity.key >> 8)),
|
||||
@as(u8, @truncate(identity.key & 0xff)),
|
||||
}) catch "",
|
||||
.anonymous => std.fmt.bufPrint(buf, "anon-{x}", .{identity.key}) catch "",
|
||||
};
|
||||
}
|
||||
|
||||
const Line = union(enum) { override: Override, ignorable, malformed };
|
||||
|
||||
fn parseLine(line: []const u8) Line {
|
||||
const body = csv.stripComment(line);
|
||||
if (body.len == 0) return .ignorable;
|
||||
var it = csv.fields(body);
|
||||
const id = it.next() orelse return .malformed;
|
||||
const prefix = it.next() orelse return .malformed;
|
||||
if (it.next() != null) return .malformed; // too many columns
|
||||
if (id.len == 0 or prefix.len == 0) return .malformed;
|
||||
return .{ .override = .{ .id = id, .prefix = prefix } };
|
||||
}
|
||||
|
||||
/// Parse a whole `volumes.csv` into `out_rules`. The slices point into `source`,
|
||||
/// which must outlive them.
|
||||
pub fn parse(source: []const u8, out_rules: []Override) ParseResult {
|
||||
var result: ParseResult = .{ .count = 0, .malformed = 0, .truncated = false };
|
||||
var lines = std.mem.splitScalar(u8, source, '\n');
|
||||
while (lines.next()) |line| {
|
||||
switch (parseLine(line)) {
|
||||
.ignorable => {},
|
||||
.malformed => result.malformed += 1,
|
||||
.override => |ov| {
|
||||
if (result.count >= out_rules.len) {
|
||||
result.truncated = true;
|
||||
continue;
|
||||
}
|
||||
out_rules[result.count] = ov;
|
||||
result.count += 1;
|
||||
},
|
||||
}
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
/// The override mount prefix for a volume whose id-string is `id`, or null (the
|
||||
/// volume takes its default `/volumes/<id>` path). First matching row wins.
|
||||
pub fn overrideFor(rules: []const Override, id: []const u8) ?[]const u8 {
|
||||
for (rules) |rule| {
|
||||
if (std.mem.eql(u8, rule.id, id)) return rule.prefix;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
// --- tests -------------------------------------------------------------------
|
||||
|
||||
const testing = std.testing;
|
||||
|
||||
// Named fixture sizes so the bounds gate (which flags literal array lengths)
|
||||
// stays quiet: test inputs, not runtime ceilings.
|
||||
const test_override_slots = 4;
|
||||
|
||||
test "idString renders each rung's id token" {
|
||||
var buf: [id_maximum]u8 = undefined;
|
||||
try testing.expectEqualStrings("fat-12345678", idString(.{ .rung = .fat_serial, .key = 0x12345678 }, &buf));
|
||||
try testing.expectEqualStrings("exfat-da7a0001", idString(.{ .rung = .exfat_serial, .key = 0xDA7A0001 }, &buf));
|
||||
try testing.expectEqualStrings("mbr-deadbeef-1", idString(.{ .rung = .mbr_index, .key = (@as(u128, 0xDEADBEEF) << 8) | 1 }, &buf));
|
||||
const guid: u128 = 0x00112233445566778899AABBCCDDEEFF;
|
||||
try testing.expectEqualStrings("gpt-00112233445566778899aabbccddeeff", idString(.{ .rung = .gpt_guid, .key = guid }, &buf));
|
||||
}
|
||||
|
||||
test "overrideFor returns the mapped prefix, else null" {
|
||||
const text =
|
||||
\\# id, mount_prefix
|
||||
\\fat-12345678, /mnt/boot
|
||||
;
|
||||
var rules: [test_override_slots]Override = undefined;
|
||||
const parsed = parse(text, &rules);
|
||||
try testing.expectEqual(@as(usize, 1), parsed.count);
|
||||
try testing.expectEqual(@as(usize, 0), parsed.malformed);
|
||||
try testing.expectEqualStrings("/mnt/boot", overrideFor(rules[0..parsed.count], "fat-12345678").?);
|
||||
try testing.expect(overrideFor(rules[0..parsed.count], "fat-99999999") == null);
|
||||
}
|
||||
|
||||
test "malformed volume rows are counted, not bound" {
|
||||
const text =
|
||||
\\fat-1, /mnt/a
|
||||
\\onlyonecolumn
|
||||
\\fat-2,
|
||||
\\fat-3, /a, /b
|
||||
;
|
||||
var rules: [test_override_slots]Override = undefined;
|
||||
const parsed = parse(text, &rules);
|
||||
try testing.expectEqual(@as(usize, 1), parsed.count); // only the first valid row
|
||||
try testing.expectEqual(@as(usize, 3), parsed.malformed); // one column, empty prefix, too many columns
|
||||
}
|
||||
+354
-11
@@ -79,7 +79,9 @@ ARCHES = {
|
||||
"-device", "usb-kbd,bus=xhci.0",
|
||||
"-device", "usb-mouse,bus=xhci.0",
|
||||
"-drive", f"if=none,id=bootusb,format=raw,file={boot_volume}",
|
||||
"-device", "usb-storage,bus=xhci.0,drive=bootusb,removable=on,bootindex=0",
|
||||
# id=bootstorage + an explicit port so the volume-replug drill can
|
||||
# device_del/device_add it back onto the same freed root port.
|
||||
"-device", "usb-storage,bus=xhci.0,port=3,drive=bootusb,removable=on,bootindex=0,id=bootstorage",
|
||||
"-net", "none",
|
||||
"-vga", "none", "-device", "VGA,edid=on,xres=1280,yres=720",
|
||||
"-display", "none",
|
||||
@@ -177,7 +179,7 @@ CASES = [
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"qemu_extra": ["-device", "intel-iommu,intremap=off"],
|
||||
"expect": r"(?s)(?=.*/system/kernel: iommu online)(?=.*fat: mounted /volumes/usb)(?=.*fat-test: ok)",
|
||||
"expect": r"(?s)(?=.*/system/kernel: iommu online)(?=.*fat: mounted /volumes/fat-12345678)(?=.*fat-test: ok)",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL|DANOS-IOMMU-FAULT"},
|
||||
# DMA + MSI under translation: interrupt-IN reports arrive through translated DMA and
|
||||
# the xHC's MSI/MSI-X still delivers (the 0xFEE00000 interrupt window bypasses second-
|
||||
@@ -213,7 +215,7 @@ CASES = [
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"qemu_extra": ["-device", "amd-iommu,dma-remap=on,intremap=off"],
|
||||
"expect": r"(?s)(?=.*iommu online \(AMD-Vi\))(?=.*fat: mounted /volumes/usb)(?=.*fat-test: ok)",
|
||||
"expect": r"(?s)(?=.*iommu online \(AMD-Vi\))(?=.*fat: mounted /volumes/fat-12345678)(?=.*fat-test: ok)",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL|DANOS-IOMMU-FAULT"},
|
||||
# Port I/O grants: a claimed device's io_port resource lets a driver read/write its
|
||||
# ports (PS/2 status 0x64), gated by the claim; out-of-range/unclaimed is refused.
|
||||
@@ -747,14 +749,167 @@ CASES = [
|
||||
"expect": r"usb-storage: ready[\s\S]*usb-storage: block 0 signature 0x55aa",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# FAT mount end to end: the fat server mounts the boot usb-storage device (the
|
||||
# FAT32 image) into the VFS at /volumes/usb. A fat-test client then lists and reads
|
||||
# FAT32 image) into the VFS at /volumes/fat-12345678. A fat-test client then lists and reads
|
||||
# through the mount — proof of the whole stack: block device -> FAT parse ->
|
||||
# VFS routing -> file read.
|
||||
{"name": "fat-mount",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"expect": r"fat: mounted /volumes/usb[\s\S]*fat-test: ok",
|
||||
"expect": r"fat: mounted /volumes/fat-12345678[\s\S]*fat-test: ok",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# The id-path naming (S2, storage-stack-plan.md). The boot volume mounts at
|
||||
# its CONTENT-derived id-path (/volumes/fat-12345678, from the FAT32 serial
|
||||
# 0x12345678) — never a port name — and keeps its FHS rewrites so /system/logs
|
||||
# persistence still rides the volume. Discrimination: before S2's flip fat
|
||||
# hardcoded /volumes/usb, so the id-path mount line never appears. (Making the
|
||||
# rewrites content-conditional on which volume carries the system is S3.)
|
||||
{"name": "volume-identity-name",
|
||||
"build_case": "fat-mount",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"expect": r"(?s)fat: mounted /volumes/fat-12345678"
|
||||
r"[\s\S]*fat: mounted /system/logs",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# The removal lifecycle (V4, docs/volume-manager-plan.md): pull the boot stick
|
||||
# mid-run. device_del the usb-storage device -> the bus reports the port empty
|
||||
# -> the device manager reaps usb-storage -> the mass-storage child leaves the
|
||||
# tree -> the volume manager's poll sees it gone and kills the FAT service, so
|
||||
# its mounts retire (an honest unmount). The tail (mounted -> removed) can only
|
||||
# be the removal, since the mount precedes the unplug. Discrimination: before
|
||||
# V4 the volume manager stopped polling after the first probe, so it never
|
||||
# noticed the removal — this line is absent.
|
||||
#
|
||||
# The RE-mount on replug is not asserted here: QEMU's device_add of usb-storage
|
||||
# to the boot xHCI controller is not re-presented to the guest (no port-connect
|
||||
# on any port), so it cannot drive the reappearance in this harness. On real
|
||||
# hardware the bus's per-tick port poll catches a reconnect's PORTSC change
|
||||
# (H1/usb-root-replug proves reconnect works on a second controller), and the
|
||||
# volume manager's bringUpVolume remounts when the device returns — bench-
|
||||
# verified, not QEMU-verified. So this case proves the unmount half.
|
||||
{"name": "volume-removal",
|
||||
"build_case": "fat-mount",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"qmp_sequence": [
|
||||
{"delay": 8, "command": "device_del", "arguments": {"id": "bootstorage"}},
|
||||
],
|
||||
"expect": r"(?s)fat: mounted /volumes/fat-12345678"
|
||||
r"[\s\S]*volume-manager: storage for volume \d+ removed; unmounting",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# S5 medium_changed: the SECOND removal trigger. QMP-eject the MEDIUM (the
|
||||
# block backend, not the device) — the usb-storage device stays in the tree,
|
||||
# but its TEST UNIT READY poll reports not-ready and publishes medium_changed
|
||||
# (absent). The volume manager, now a subscriber, runs the same unmount path as
|
||||
# a device pull. Discrimination: before S5 the manager never subscribed, so the
|
||||
# event reached no one and the mount persisted (device-presence polling cannot
|
||||
# see a medium leave while the device stays). One lifecycle, two triggers.
|
||||
{"name": "volume-medium-change",
|
||||
"build_case": "fat-mount",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"qmp_sequence": [
|
||||
{"delay": 8, "command": "eject", "arguments": {"device": "bootusb", "force": True}},
|
||||
],
|
||||
"expect": r"(?s)fat: mounted /volumes/fat-12345678"
|
||||
r"[\s\S]*usb-storage: medium absent"
|
||||
r"[\s\S]*volume-manager: medium left a storage device; unmounting"
|
||||
r"[\s\S]*volume-manager: storage for volume \d+ removed; unmounting",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# S5 storage-driver-crash rebuild. The device manager (test-storage-restart
|
||||
# mode) kills usb-storage once, ~2s in — after its volume mounted. The device
|
||||
# stays in the tree, so device-presence polling alone would leave fat wedged on
|
||||
# the dead channel; the volume manager's channel-liveness probe (a geometry()
|
||||
# that fails on the dead endpoint) must notice, reap the volume, and rebuild on
|
||||
# the restarted driver's fresh channel — a SECOND mount of the same id-path.
|
||||
# Discrimination: a pre-S5 manager checks only isDevicePresent (still true), so
|
||||
# it never reaps and the second mount never appears (it would restart fat on
|
||||
# the stale channel and crash-loop).
|
||||
{"name": "volume-driver-restart",
|
||||
"build_case": "volume-driver-restart",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"expect": r"(?s)fat: mounted /volumes/fat-12345678"
|
||||
r"[\s\S]*volume-manager: storage for volume \d+ removed; unmounting"
|
||||
r"[\s\S]*fat: mounted /volumes/fat-12345678",
|
||||
"fail": r"failing repeatedly; giving up|\[FAIL\]|DANOS-TEST-RESULT: FAIL"},
|
||||
# Volume-manager discovery + probe (V3a, docs/volume-manager-plan.md). Reuses
|
||||
# the fat-mount kernel build (the default boot now spawns the volume manager
|
||||
# from init.csv). It acquires the mass-storage block channel through the
|
||||
# device manager, reads block 0, and parses the first volume out of it — the
|
||||
# partition-table walk that used to live in the FAT engine, now above the
|
||||
# driver where it belongs. Before V3a the service did not exist, so this line
|
||||
# is absent.
|
||||
{"name": "volume-probe",
|
||||
"build_case": "fat-mount",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
# The volume manager probes the partition table, then confines a filesystem
|
||||
# to the volume and hands it over. Since S1 rung 3 the identity is the boot
|
||||
# image's real FAT32 serial (0x12345678, from make-fat-image.py) — before
|
||||
# rung 3 it was the rung-4 pseudo-signature read from VBR offset 440 (~0x0),
|
||||
# so this tightened value is an on-image witness that the enriched identity
|
||||
# reaches the running log, not just the host tests.
|
||||
"expect": r"volume-manager: volume 0x0*12345678 -> \S+ \(pid \d+\), lba \d+, \d+ blocks"
|
||||
r"[\s\S]*volume-manager: handed volume \d+ to pid \d+",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# S3 multi-volume: a SECOND usb-storage device (a generated data volume, serial
|
||||
# da7a0001, an empty FAT with no /system) plugged in beside the boot volume.
|
||||
# Proves the volume manager adopts BOTH devices and spawns a confined fat per
|
||||
# volume, each mounted at its own CONTENT id-path (/volumes/fat-<serial>); and
|
||||
# that boot-volume detection is by content — only the volume that carries
|
||||
# /system backs /system/configuration, while the data volume mounts at its
|
||||
# id-path alone. Against the pre-S3 one-device/one-volume manager the data
|
||||
# volume never mounts, so the da7a0001 lookaheads fail (toggle-demonstrated by
|
||||
# checking out the step-2 volume-manager.zig).
|
||||
{"name": "two-volumes",
|
||||
"build_case": "fat-mount",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"data_volume": {"serial": "DA7A0001", "label": "DATAVOL", "size_mib": 64},
|
||||
"expect": r"(?s)(?=.*volume-manager: volume 0x0*12345678 -> )"
|
||||
r"(?=.*volume-manager: volume 0x0*da7a0001 -> )"
|
||||
r"(?=.*fat: mounted /volumes/fat-12345678)"
|
||||
r"(?=.*fat: mounted /volumes/fat-da7a0001)"
|
||||
r"(?=.*carries the system tree)"
|
||||
r"(?=.*data volume; mounted at /volumes/fat-da7a0001)",
|
||||
"fail": r"data volume; mounted at /volumes/fat-12345678|DANOS-TEST-RESULT: FAIL"},
|
||||
# S3 shared-channel multi-volume: ONE usb-storage device carrying an MBR with
|
||||
# TWO FAT partitions (da7a0001 at lba 2048, da7a0002 at lba 83968). allVolumes
|
||||
# walks the table and the manager spawns a confined fat per partition on the
|
||||
# SAME block channel, each clamped to its own LBA range (usb-storage's
|
||||
# per-badge range table) — the path a pair of single-volume sticks (the
|
||||
# two-volumes case) does NOT exercise. The two mount lines sit at two DISTINCT
|
||||
# non-zero base_lbas on one device. Against the pre-uncap allVolumes (S3 step
|
||||
# 2, capped to one partition) only da7a0001 mounts, so the da7a0002 lookaheads
|
||||
# fail.
|
||||
{"name": "partitioned-volume",
|
||||
"build_case": "fat-mount",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"data_volume": {"partitions": [{"serial": "DA7A0001", "size_mib": 40},
|
||||
{"serial": "DA7A0002", "size_mib": 40}]},
|
||||
"expect": r"(?s)(?=.*volume 0x0*da7a0001 -> \S+ \(pid \d+\), lba 2048, )"
|
||||
r"(?=.*volume 0x0*da7a0002 -> \S+ \(pid \d+\), lba 83968, )"
|
||||
r"(?=.*fat: mounted /volumes/fat-da7a0001)"
|
||||
r"(?=.*fat: mounted /volumes/fat-da7a0002)",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# S4 second engine: a bare exFAT data device (serial e0fa0001) attached beside
|
||||
# the FAT boot volume. The volume manager content-routes it to the exFAT
|
||||
# service (not fat), which mounts it at its id-path /volumes/exfat-e0fa0001;
|
||||
# the exfat-test client then reads the seeded HELLO.TXT and mutates through the
|
||||
# mount (mkdir/write/rename/read/remove). Proves the second engine reuses the
|
||||
# shared harness end to end. Fails against pre-S4 (no exfat binary, csv row, or
|
||||
# VBR recognizer — the device would go to fat, which rejects the exFAT VBR).
|
||||
{"name": "exfat-volume",
|
||||
"build_case": "exfat-volume",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"data_volume": {"exfat": True, "serial": "E0FA0001", "size_mib": 48},
|
||||
"expect": r"(?s)(?=.*volume 0x0*e0fa0001 -> /system/services/exfat )"
|
||||
r"(?=.*exfat: mounted /volumes/exfat-e0fa0001)"
|
||||
r"(?=.*exfat-test: read HELLO.TXT ok)"
|
||||
r"(?=.*exfat-test: ok)",
|
||||
"fail": r"exfat-test: FAILED|DANOS-TEST-RESULT: FAIL"},
|
||||
# Phase 2b: mkdir/unlink through the mount. Reuses the fat-mount build — the
|
||||
# fat-test client, after listing, makes a directory, writes+reads a file inside
|
||||
# it, then removes the file, exercising the whole VFS -> fat mutation path.
|
||||
@@ -955,10 +1110,16 @@ CASES = [
|
||||
"expect": r"(?s)(?=.*hub slot \d+ port \d+ device:.*0x0409)"
|
||||
r"(?=.*usb-hid-keyboard: ok \(device 3)",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# Hub-downstream disconnect (B4c): device_del the keyboard behind the hub;
|
||||
# the hub's status-change endpoint reports it, the device is torn down
|
||||
# (ChildRemoved + Disable Slot). QEMU's hub DOES raise downstream changes
|
||||
# (unlike root-port hot-plug).
|
||||
# Hub-downstream disconnect AND replug (B4c + establishment): device_del the
|
||||
# keyboard behind the hub — the hub's status-change endpoint reports it, the
|
||||
# device is torn down (ChildRemoved + Disable Slot), and the manager REAPS
|
||||
# the bound class driver (it blocks on reports that will never come, and its
|
||||
# stale entry would make the dedupe refuse the respawn). Then device_add
|
||||
# plugs it back: the re-report re-registers the same identity and the
|
||||
# matcher spawns a fresh driver, which binds — the ordered tail (reaping →
|
||||
# child removed → delegated → ok) can only be satisfied by the SECOND
|
||||
# generation, since the boot keyboards' ok lines all precede the unplug.
|
||||
# QEMU's hub DOES raise downstream changes (unlike root-port hot-plug).
|
||||
{"name": "usb-hub-unplug",
|
||||
"build_case": "usb-hid",
|
||||
"smp": 4,
|
||||
@@ -966,9 +1127,146 @@ CASES = [
|
||||
"qemu_extra": ["-device", "qemu-xhci,id=xhci2",
|
||||
"-device", "usb-hub,bus=xhci2.0,port=1",
|
||||
"-device", "usb-kbd,bus=xhci2.0,port=1.1,id=dkbd"],
|
||||
"qmp_after": {"delay": 8, "command": "device_del", "arguments": {"id": "dkbd"}},
|
||||
"qmp_sequence": [
|
||||
{"delay": 8, "command": "device_del", "arguments": {"id": "dkbd"}},
|
||||
{"delay": 14, "command": "device_add",
|
||||
"arguments": {"driver": "usb-kbd", "bus": "xhci2.0", "port": "1.1", "id": "dkbd2"}},
|
||||
],
|
||||
"expect": r"(?s)(?=.*usb-hid-keyboard: ok \(device 3)"
|
||||
r"(?=.*slot \d+ disconnected)",
|
||||
r"(?=.*slot \d+ disconnected"
|
||||
r"[\s\S]*device-manager: child removed"
|
||||
r"[\s\S]*device-manager: reaping \S*usb-hid-keyboard"
|
||||
r"[\s\S]*device-manager: delegated device \d+ to /system/drivers/usb-hid-keyboard"
|
||||
r"[\s\S]*usb-hid-keyboard: ok)",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# Hot-plug matrix H1 (docs/hot-plug-matrix-plan.md): ROOT-port unplug and
|
||||
# replug on the same port. Different teardown path than the hub case
|
||||
# (tearDownPort, not tearDownHubDevice), and QEMU raises no port-change
|
||||
# events for root ports — the bus's 250 ms reconcile tick must notice the
|
||||
# PORTSC change on its own. The ordered tail (removed → reaping → delegated
|
||||
# → ok) can only be the second generation: every boot device's ok precedes
|
||||
# the unplug.
|
||||
{"name": "usb-root-replug",
|
||||
"build_case": "usb-hid",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"qemu_extra": ["-device", "qemu-xhci,id=xhci2",
|
||||
"-device", "usb-kbd,bus=xhci2.0,port=2,id=rkbd"],
|
||||
"qmp_sequence": [
|
||||
{"delay": 8, "command": "device_del", "arguments": {"id": "rkbd"}},
|
||||
{"delay": 14, "command": "device_add",
|
||||
"arguments": {"driver": "usb-kbd", "bus": "xhci2.0", "port": "2", "id": "rkbd2"}},
|
||||
],
|
||||
"expect": r"(?s)(?=.*usb-hid-keyboard: ok \(device (\d+)\b"
|
||||
r"[\s\S]*port \d+ disconnected"
|
||||
r"[\s\S]*device-manager: child removed"
|
||||
r"[\s\S]*device-manager: reaping \S*usb-hid-keyboard"
|
||||
r"[\s\S]*device-manager: delegated device \d+ to /system/drivers/usb-hid-keyboard"
|
||||
r"[\s\S]*usb-hid-keyboard: ok)",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# Hot-plug matrix H2 (docs/hot-plug-matrix-plan.md): yank a POPULATED hub.
|
||||
# One device_del removes the hub with a keyboard and a mouse behind it — the
|
||||
# bus's recursive teardown (tearDownHubDevice, children first) must report
|
||||
# every downstream interface removed, the manager must reap BOTH bound
|
||||
# drivers, and re-adding the hub and both devices must rebuild the subtree.
|
||||
# The reaping lines only exist post-yank, so `reaping X ... X: ok` chains
|
||||
# are unambiguously second-generation.
|
||||
{"name": "usb-hub-yank",
|
||||
"build_case": "usb-hid",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"qemu_extra": ["-device", "qemu-xhci,id=xhci2",
|
||||
"-device", "usb-hub,bus=xhci2.0,port=1,id=yhub",
|
||||
"-device", "usb-kbd,bus=xhci2.0,port=1.1,id=ykbd",
|
||||
"-device", "usb-mouse,bus=xhci2.0,port=1.2,id=ymouse"],
|
||||
"qmp_sequence": [
|
||||
{"delay": 8, "command": "device_del", "arguments": {"id": "yhub"}},
|
||||
{"delay": 14, "command": "device_add",
|
||||
"arguments": {"driver": "usb-hub", "bus": "xhci2.0", "port": "1", "id": "yhub2"}},
|
||||
{"delay": 17, "command": "device_add",
|
||||
"arguments": {"driver": "usb-kbd", "bus": "xhci2.0", "port": "1.1", "id": "ykbd2"}},
|
||||
{"delay": 18, "command": "device_add",
|
||||
"arguments": {"driver": "usb-mouse", "bus": "xhci2.0", "port": "1.2", "id": "ymouse2"}},
|
||||
],
|
||||
"expect": r"(?s)(?=.*hub device slot \d+ disconnected)"
|
||||
r"(?=.*device-manager: reaping \S*usb-hid-keyboard[\s\S]*usb-hid-keyboard: ok)"
|
||||
r"(?=.*device-manager: reaping \S*usb-hid-mouse[\s\S]*usb-hid-mouse: ok)",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# Hot-plug matrix H3 (docs/hot-plug-matrix-plan.md): yank a NESTED hub tree.
|
||||
# hub → hub → keyboard, one device_del of the outer hub — the recursion must
|
||||
# go two levels (inner hub torn down as a child, ITS keyboard first), the
|
||||
# keyboard's driver reaped, and re-adding all three rebuilds and rebinds.
|
||||
{"name": "usb-hub-nested-yank",
|
||||
"build_case": "usb-hid",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"qemu_extra": ["-device", "qemu-xhci,id=xhci2",
|
||||
"-device", "usb-hub,bus=xhci2.0,port=1,id=ohub",
|
||||
"-device", "usb-hub,bus=xhci2.0,port=1.1,id=ihub",
|
||||
"-device", "usb-kbd,bus=xhci2.0,port=1.1.1,id=nkbd"],
|
||||
"qmp_sequence": [
|
||||
{"delay": 8, "command": "device_del", "arguments": {"id": "ohub"}},
|
||||
{"delay": 14, "command": "device_add",
|
||||
"arguments": {"driver": "usb-hub", "bus": "xhci2.0", "port": "1", "id": "ohub2"}},
|
||||
{"delay": 17, "command": "device_add",
|
||||
"arguments": {"driver": "usb-hub", "bus": "xhci2.0", "port": "1.1", "id": "ihub2"}},
|
||||
{"delay": 20, "command": "device_add",
|
||||
"arguments": {"driver": "usb-kbd", "bus": "xhci2.0", "port": "1.1.1", "id": "nkbd2"}},
|
||||
],
|
||||
"expect": r"(?s)(?=.*hub device slot \d+ disconnected)"
|
||||
r"(?=.*device-manager: reaping \S*usb-hid-keyboard[\s\S]*usb-hid-keyboard: ok)",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# Hot-plug matrix H4 (docs/hot-plug-matrix-plan.md): replug on a DIFFERENT
|
||||
# port. Registration identity is per-port, so a moved device is simply a new
|
||||
# device: the old child is removed and its driver reaped; the new port's
|
||||
# child gets a fresh id and a fresh driver. Nothing may tie a driver to the
|
||||
# old port. The tail (reaping → hub slot N port 2 device → delegated → ok)
|
||||
# is unambiguous: port 2 of the hub had nothing before the move.
|
||||
{"name": "usb-replug-moved",
|
||||
"build_case": "usb-hid",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"qemu_extra": ["-device", "qemu-xhci,id=xhci2",
|
||||
"-device", "usb-hub,bus=xhci2.0,port=1,id=mhub",
|
||||
"-device", "usb-kbd,bus=xhci2.0,port=1.1,id=mkbd"],
|
||||
"qmp_sequence": [
|
||||
{"delay": 8, "command": "device_del", "arguments": {"id": "mkbd"}},
|
||||
{"delay": 14, "command": "device_add",
|
||||
"arguments": {"driver": "usb-kbd", "bus": "xhci2.0", "port": "1.2", "id": "mkbd2"}},
|
||||
],
|
||||
"expect": r"(?s)(?=.*device-manager: reaping \S*usb-hid-keyboard"
|
||||
r"[\s\S]*hub slot \d+ port 2 device:"
|
||||
r"[\s\S]*device-manager: delegated device \d+ to /system/drivers/usb-hid-keyboard"
|
||||
r"[\s\S]*usb-hid-keyboard: ok)",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# Hot-plug matrix H5 (docs/hot-plug-matrix-plan.md): three unplug/replug
|
||||
# cycles of the same keyboard on the same hub port. Each cycle must reap the
|
||||
# old generation and bind a new one — three reaping lines, and a bind after
|
||||
# the last — proving no state leaks across generations: xHCI slots, the
|
||||
# bus's open table, and the manager's driver entries must all be reusable.
|
||||
{"name": "usb-replug-cycles",
|
||||
"build_case": "usb-hid",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"qemu_extra": ["-device", "qemu-xhci,id=xhci2",
|
||||
"-device", "usb-hub,bus=xhci2.0,port=1,id=chub",
|
||||
"-device", "usb-kbd,bus=xhci2.0,port=1.1,id=ckbd0"],
|
||||
"qmp_sequence": [
|
||||
{"delay": 8, "command": "device_del", "arguments": {"id": "ckbd0"}},
|
||||
{"delay": 12, "command": "device_add",
|
||||
"arguments": {"driver": "usb-kbd", "bus": "xhci2.0", "port": "1.1", "id": "ckbd1"}},
|
||||
{"delay": 18, "command": "device_del", "arguments": {"id": "ckbd1"}},
|
||||
{"delay": 22, "command": "device_add",
|
||||
"arguments": {"driver": "usb-kbd", "bus": "xhci2.0", "port": "1.1", "id": "ckbd2"}},
|
||||
{"delay": 28, "command": "device_del", "arguments": {"id": "ckbd2"}},
|
||||
{"delay": 32, "command": "device_add",
|
||||
"arguments": {"driver": "usb-kbd", "bus": "xhci2.0", "port": "1.1", "id": "ckbd3"}},
|
||||
],
|
||||
"expect": r"(?s)(?=.*device-manager: reaping \S*usb-hid-keyboard"
|
||||
r"[\s\S]*device-manager: reaping \S*usb-hid-keyboard"
|
||||
r"[\s\S]*device-manager: reaping \S*usb-hid-keyboard"
|
||||
r"[\s\S]*device-manager: delegated device \d+ to /system/drivers/usb-hid-keyboard"
|
||||
r"[\s\S]*usb-hid-keyboard: ok)",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# The kernel VFS root (M-F): the mount table serves the initrd at /system —
|
||||
# path resolution, node status/read (an ELF magic), and directory listing,
|
||||
@@ -1091,6 +1389,23 @@ CASES = [
|
||||
{"name": "device-authority",
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# Per-sender range confinement (V2a, docs/volume-manager-plan.md): a process
|
||||
# confines ITSELF to a block sub-range (as the volume manager confines a
|
||||
# filesystem), then proves it cannot read past the range nor widen it. The
|
||||
# security assertions are named explicitly so the case cannot pass without
|
||||
# them; a confined read crossing the range must be REFUSED and a confined
|
||||
# define_range must be REFUSED. Against pre-clamp usb-storage the define_range
|
||||
# verb does not exist, so the fixture fails to arm confinement at all.
|
||||
{"name": "block-range",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"expect": r"(?s)(?=.*block-range: ok in-range-read)"
|
||||
r"(?=.*block-range: ok out-of-range-refused)"
|
||||
r"(?=.*block-range: ok wrap-refused)"
|
||||
r"(?=.*block-range: ok geometry-is-confined)"
|
||||
r"(?=.*block-range: ok confined-cannot-redefine)"
|
||||
r"(?=.*block-range: VERDICT done)",
|
||||
"fail": r"block-range: FAILED|DANOS-TEST-RESULT: FAIL"},
|
||||
# IRQ teardown: an exiting driver's line is masked and its slot cleared (so no
|
||||
# ISR notifies a freed endpoint), and a sibling owner sharing that endpoint
|
||||
# keeps its own binding. A long-running driver never reaches this teardown path.
|
||||
@@ -1201,6 +1516,34 @@ def run_case(arch, case):
|
||||
cmd[cmd.index("-m") + 1] = case["mem"]
|
||||
if case.get("qemu_extra"): # extra qemu args, e.g. -device intel-iommu for the IOMMU case
|
||||
cmd += case["qemu_extra"]
|
||||
# A multi-volume case attaches a second usb-storage device backed by a freshly
|
||||
# GENERATED data volume: a distinct-serial FAT32 with no /system tree, so the
|
||||
# volume manager mounts it at its own id-path and the fat process marks it a
|
||||
# data volume (never a system volume). Regenerated per run — no image is
|
||||
# committed to the tree (the user keeps the boot files copyable, not baked in).
|
||||
if case.get("data_volume"):
|
||||
dv = case["data_volume"]
|
||||
data_img = os.path.join(WORK, "data-volume.img")
|
||||
if dv.get("partitions"):
|
||||
# One device, an MBR with several FAT partitions: several volumes share
|
||||
# ONE block channel, each confined to its own LBA range.
|
||||
gen = [sys.executable, os.path.join(REPO, "tools", "make-partitioned-image.py"), data_img]
|
||||
for part in dv["partitions"]:
|
||||
gen += [part["serial"], str(part.get("size_mib", 40))]
|
||||
elif dv.get("exfat"):
|
||||
# One device, a bare exFAT volume — the second engine's medium.
|
||||
gen = [sys.executable, os.path.join(REPO, "tools", "make-exfat-image.py"),
|
||||
"--serial", dv["serial"], data_img, str(dv.get("size_mib", 48))]
|
||||
else:
|
||||
# One device, one bare FAT volume.
|
||||
gen = [sys.executable, os.path.join(REPO, "tools", "make-fat-image.py"),
|
||||
"--serial", dv["serial"], "--label", dv.get("label", "DATAVOL"),
|
||||
data_img, str(dv.get("size_mib", 64))]
|
||||
subprocess.run(gen, check=True, stdout=subprocess.DEVNULL)
|
||||
cmd += [
|
||||
"-drive", f"if=none,id=datausb,format=raw,file={data_img}",
|
||||
"-device", "usb-storage,bus=xhci.0,port=4,drive=datausb,removable=on,id=datastorage",
|
||||
]
|
||||
# A QMP control socket, always present (additive): how a case's `qmp_after`
|
||||
# hook injects host-side events into the guest mid-run. Kept under a short temp
|
||||
# dir, not WORK: a unix socket path is capped at ~104 bytes (sun_path), and a
|
||||
|
||||
@@ -38,7 +38,7 @@ const time = @import("time");
|
||||
/// A scratch file on the volume, so the node the intruder tries to write through
|
||||
/// is one nothing else reads. (A foreign write that *succeeded* would prove the
|
||||
/// bug — it must not also damage the boot volume proving it.)
|
||||
const held_path = "/volumes/usb/BADGE.TXT";
|
||||
const held_path = "/volumes/fat-12345678/BADGE.TXT";
|
||||
const held_contents = "held";
|
||||
|
||||
fn line(comptime format: []const u8, arguments: anytype) void {
|
||||
@@ -46,12 +46,12 @@ fn line(comptime format: []const u8, arguments: anytype) void {
|
||||
_ = logging.write(std.fmt.bufPrint(&buffer, format, arguments) catch return);
|
||||
}
|
||||
|
||||
/// The fat server mounts /volumes/usb only after the whole USB storage chain is
|
||||
/// The fat server mounts /volumes/fat-12345678 only after the whole USB storage chain is
|
||||
/// up, and both instances race it.
|
||||
fn waitForVolume() bool {
|
||||
var tries: u32 = 0;
|
||||
while (tries < 1400) : (tries += 1) {
|
||||
if (fs.openDirectory("/volumes/usb")) |opened| {
|
||||
if (fs.openDirectory("/volumes/fat-12345678")) |opened| {
|
||||
var directory = opened;
|
||||
directory.close();
|
||||
return true;
|
||||
@@ -79,7 +79,7 @@ pub fn main(init: process.Init) void {
|
||||
|
||||
fn own() void {
|
||||
if (!waitForVolume()) {
|
||||
_ = logging.write("badge-scope-test: FAILED (/volumes/usb never became available)\n");
|
||||
_ = logging.write("badge-scope-test: FAILED (/volumes/fat-12345678 never became available)\n");
|
||||
return;
|
||||
}
|
||||
var held = fs.open(held_path, .{ .create = true, .truncate = true }) orelse {
|
||||
@@ -142,7 +142,7 @@ fn own() void {
|
||||
|
||||
fn intrude(foreign_node: u64, foreign_layer: u32) void {
|
||||
if (!waitForVolume()) {
|
||||
_ = logging.write("badge-scope-test: FAILED (/volumes/usb never became available)\n");
|
||||
_ = logging.write("badge-scope-test: FAILED (/volumes/fat-12345678 never became available)\n");
|
||||
return;
|
||||
}
|
||||
const node_verdict = probeNode(foreign_node);
|
||||
|
||||
@@ -0,0 +1,154 @@
|
||||
//! block-range-test — the discrimination fixture for per-sender range
|
||||
//! confinement (V2a, docs/volume-manager-plan.md). It gets a block channel the
|
||||
//! way a filesystem does (consumer-hello the device manager for the mass-storage
|
||||
//! provider), then proves the two properties the clamp exists for:
|
||||
//!
|
||||
//! 1. an UNCONFINED caller may define a range on its own badge (the volume
|
||||
//! manager is unconfined — this stands in for it);
|
||||
//! 2. once confined, a transfer PAST the range is refused, and the volume
|
||||
//! relative LBA 0 maps inside the range (the clamp translates + bounds);
|
||||
//! 3. a CONFINED caller may NOT call define_range again (the gate — a
|
||||
//! filesystem cannot widen its own range or escape).
|
||||
//!
|
||||
//! Against pre-clamp usb-storage the verb does not exist, so (1) already fails —
|
||||
//! which is exactly the discrimination: the fixture cannot even arm confinement,
|
||||
//! let alone see a transfer refused for crossing it.
|
||||
//!
|
||||
//! It coexists with the FAT service in the same boot: ranges are per-badge, so
|
||||
//! confining THIS process touches nothing fat does on its own channel.
|
||||
|
||||
const std = @import("std");
|
||||
const channel = @import("channel");
|
||||
const device_manager_protocol = @import("device-manager-protocol");
|
||||
const driver = @import("driver");
|
||||
const ipc = @import("ipc");
|
||||
const block = @import("block");
|
||||
const memory = @import("memory");
|
||||
const logging = @import("logging");
|
||||
const process = @import("process");
|
||||
const time = @import("time");
|
||||
const envelope = @import("envelope");
|
||||
|
||||
fn verdict(ok: bool, name: []const u8) void {
|
||||
_ = logging.write("block-range: ");
|
||||
_ = logging.write(if (ok) "ok " else "FAILED ");
|
||||
_ = logging.write(name);
|
||||
_ = logging.write("\n");
|
||||
}
|
||||
|
||||
/// The mass-storage provider's block channel, via the device manager's tree —
|
||||
/// the same lineage acquisition the FAT service uses (block is not a name).
|
||||
fn acquireBlock() ?block.Device {
|
||||
var tries: u32 = 0;
|
||||
const manager = while (tries < 200) : (tries += 1) {
|
||||
if (channel.openEndpoint("device-manager")) |h| break h;
|
||||
time.sleepMillis(20);
|
||||
} else return null;
|
||||
|
||||
// The whole USB storage chain (enumeration, bring-up) takes a few seconds to
|
||||
// appear in the manager's tree, so retry the enumerate-and-hello with a pause
|
||||
// between rounds — 500 x 20 ms ~ 10 s, well within the case timeout.
|
||||
const Entry = device_manager_protocol.ChildEntry;
|
||||
var attempt: u32 = 0;
|
||||
while (attempt < 500) : (attempt += 1) {
|
||||
var start: u64 = 0;
|
||||
while (true) {
|
||||
const enumerate = envelope.Header{ .operation = envelope.operation_enumerate, .target = start };
|
||||
var reply: [device_manager_protocol.message_maximum]u8 = undefined;
|
||||
const length = ipc.call(manager, std.mem.asBytes(&enumerate), &reply) catch break;
|
||||
const status = envelope.statusOf(reply[0..length]) orelse break;
|
||||
if (status.status != 0) break;
|
||||
const carried = @min(@as(usize, status.len), length -| envelope.prefix_size);
|
||||
const tail = reply[envelope.prefix_size..][0..carried];
|
||||
const count = tail.len / @sizeOf(Entry);
|
||||
if (count == 0) break;
|
||||
var index: usize = 0;
|
||||
while (index < count) : (index += 1) {
|
||||
const entry = std.mem.bytesToValue(Entry, tail[index * @sizeOf(Entry) ..][0..@sizeOf(Entry)]);
|
||||
if (entry.device_id == device_manager_protocol.no_device) continue;
|
||||
if ((entry.identity >> 16) & 0xff != 0x08 or (entry.identity >> 8) & 0xff != 0x06) continue;
|
||||
const exchanged = driver.helloOn(manager, .consumer, entry.device_id, null, true) orelse break;
|
||||
const provider = exchanged.channel orelse continue;
|
||||
return .{ .endpoint = provider };
|
||||
}
|
||||
start += count;
|
||||
}
|
||||
time.sleepMillis(20);
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
pub fn main(init: process.Init) void {
|
||||
// Bundled fixtures are swept up and spawned bare on every boot; stay silent
|
||||
// unless the kernel test explicitly runs us, or we would contend for the
|
||||
// block channel and print markers into unrelated cases.
|
||||
const arg = init.arguments.get(1) orelse return;
|
||||
if (!std.mem.eql(u8, arg, "run")) return;
|
||||
|
||||
const device = acquireBlock() orelse {
|
||||
verdict(false, "acquire-block");
|
||||
return;
|
||||
};
|
||||
const geometry = device.geometry() orelse {
|
||||
verdict(false, "geometry");
|
||||
return;
|
||||
};
|
||||
// Need at least a few blocks to carve a range out of; every FAT image is far
|
||||
// larger, so this only guards a nonsense device.
|
||||
if (geometry.block_count < 4) {
|
||||
verdict(false, "device-too-small");
|
||||
return;
|
||||
}
|
||||
|
||||
// A one-block DMA buffer for the positive-control read. Shareable so it can be
|
||||
// attached under an enforcing IOMMU (a no-op success otherwise).
|
||||
const bounce = memory.dmaAlloc(512, memory.dma_coherent | memory.dma_shareable) orelse {
|
||||
verdict(false, "dma-alloc");
|
||||
return;
|
||||
};
|
||||
if (bounce.handle) |handle| {
|
||||
if (!device.attach(handle)) {
|
||||
verdict(false, "attach");
|
||||
return;
|
||||
}
|
||||
_ = ipc.close(handle);
|
||||
}
|
||||
|
||||
// Baseline: an unconfined read of block 0 succeeds — so a later refusal is
|
||||
// the clamp, not a broken read path.
|
||||
verdict(device.read(0, 1, bounce.physical), "unconfined-read");
|
||||
|
||||
const me = process.taskId();
|
||||
|
||||
// (1) An unconfined caller confines itself to blocks [1, 3). Against pre-clamp
|
||||
// usb-storage this verb does not exist and the call fails here.
|
||||
if (!device.defineRange(me, 1, 2)) {
|
||||
verdict(false, "define-range");
|
||||
return;
|
||||
}
|
||||
verdict(true, "define-range");
|
||||
|
||||
// (2) Confined now: volume-relative LBA 0 maps to device block 1 (inside the
|
||||
// range) and succeeds; LBA 2 would reach device block 3, past the 2-block
|
||||
// range, and must be refused.
|
||||
verdict(device.read(0, 1, bounce.physical), "in-range-read");
|
||||
verdict(!device.read(2, 1, bounce.physical), "out-of-range-refused");
|
||||
// The wrap attack, calibrated to be exploitable against a naive bound: this
|
||||
// process is confined with base 1, so a volume-relative LBA of maxInt(u64)
|
||||
// makes base + lba wrap to absolute block 0 — a real, readable block OUTSIDE
|
||||
// the range (the boot sector). A naive `lba + count > count` check also
|
||||
// wraps to 0 and waves it through; the overflow-safe bound refuses it.
|
||||
verdict(!device.read(std.math.maxInt(u64), 1, bounce.physical), "wrap-refused");
|
||||
|
||||
// Geometry now reports the CONFINED size, not the device's.
|
||||
const confined = device.geometry() orelse {
|
||||
verdict(false, "confined-geometry");
|
||||
return;
|
||||
};
|
||||
verdict(confined.block_count == 2, "geometry-is-confined");
|
||||
|
||||
// (3) The gate: a confined caller cannot define_range — no widening, no escape.
|
||||
verdict(!device.defineRange(me, 0, geometry.block_count), "confined-cannot-redefine");
|
||||
|
||||
_ = logging.write("block-range: VERDICT done\n");
|
||||
}
|
||||
@@ -0,0 +1,19 @@
|
||||
//! The block-range-test fixture as a binary package (docs/build-packages-plan.md):
|
||||
//! this file names the binary and EXACTLY the modules its source imports —
|
||||
//! build-support resolves each name from the domains this zon declares.
|
||||
|
||||
const std = @import("std");
|
||||
const build_support = @import("build-support");
|
||||
|
||||
pub fn build(b: *std.Build) void {
|
||||
const exe = build_support.userBinary(b, .{
|
||||
.name = "block-range-test",
|
||||
.root_source_file = b.path("block-range-test.zig"),
|
||||
.imports = &.{
|
||||
"block", "channel", "device-manager-protocol", "driver",
|
||||
"envelope", "ipc", "logging", "memory",
|
||||
"process", "time",
|
||||
},
|
||||
});
|
||||
b.installArtifact(exe);
|
||||
}
|
||||
@@ -0,0 +1,16 @@
|
||||
.{
|
||||
.name = .block_range_test,
|
||||
.version = "0.0.0",
|
||||
.fingerprint = 0xa7f2045ed72783c3, // Changing this has security and trust implications.
|
||||
.minimum_zig_version = "0.16.0",
|
||||
.dependencies = .{
|
||||
// build-support supplies the shared recipe; kernel is implicit in every
|
||||
// binary. device (block, driver) and protocol (device-manager-protocol,
|
||||
// envelope) are the homes of this fixture's remaining imports.
|
||||
.@"build-support" = .{ .path = "../../../../build-support" },
|
||||
.kernel = .{ .path = "../../../../library/kernel" },
|
||||
.device = .{ .path = "../../../../library/device" },
|
||||
.protocol = .{ .path = "../../../../library/protocol" },
|
||||
},
|
||||
.paths = .{""},
|
||||
}
|
||||
@@ -0,0 +1,15 @@
|
||||
//! The exfat-test test fixture as a binary package (docs/build-packages-plan.md):
|
||||
//! this file names the binary and EXACTLY the modules its source imports —
|
||||
//! build-support resolves each name from the domains this zon declares.
|
||||
|
||||
const std = @import("std");
|
||||
const build_support = @import("build-support");
|
||||
|
||||
pub fn build(b: *std.Build) void {
|
||||
const exe = build_support.userBinary(b, .{
|
||||
.name = "exfat-test",
|
||||
.root_source_file = b.path("exfat-test.zig"),
|
||||
.imports = &.{ "file-system", "logging", "process", "time" },
|
||||
});
|
||||
b.installArtifact(exe);
|
||||
}
|
||||
@@ -0,0 +1,14 @@
|
||||
.{
|
||||
.name = .exfat_test,
|
||||
.version = "0.0.0",
|
||||
.fingerprint = 0x77b19e3f7ee43ce3, // Changing this has security and trust implications.
|
||||
.minimum_zig_version = "0.16.0",
|
||||
.dependencies = .{
|
||||
// build-support supplies the shared recipe; kernel is implicit in
|
||||
// every binary (the root shim + link script live there). The rest
|
||||
// are exactly the homes of this binary's declared imports.
|
||||
.@"build-support" = .{ .path = "../../../../build-support" },
|
||||
.kernel = .{ .path = "../../../../library/kernel" },
|
||||
},
|
||||
.paths = .{""},
|
||||
}
|
||||
@@ -0,0 +1,88 @@
|
||||
//! test/system/services/exfat-test — a client that proves the exFAT mount end to
|
||||
//! end: it waits for the exfat server to mount the volume at /volumes/exfat-
|
||||
//! e0fa0001, reads the seeded HELLO.TXT off it, and exercises mkdir / write /
|
||||
//! read / rename / remove through the VFS (which routes the id-path to the exfat
|
||||
//! backend). Shipped in the initial_ramdisk; the `exfat-volume` QEMU case spawns
|
||||
//! it alongside init with an exFAT data device attached beside the FAT boot volume.
|
||||
|
||||
const std = @import("std");
|
||||
const fs = @import("file-system");
|
||||
const process = @import("process");
|
||||
const time = @import("time");
|
||||
const logging = @import("logging");
|
||||
|
||||
// The exFAT data device's content id-path — its VolumeSerialNumber is 0xE0FA0001
|
||||
// (the `exfat-volume` case passes --serial E0FA0001 to make-exfat-image.py).
|
||||
const mount = "/volumes/exfat-e0fa0001";
|
||||
|
||||
fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||
var line: [128]u8 = undefined;
|
||||
_ = logging.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
||||
}
|
||||
|
||||
pub fn main(init: process.Init) void {
|
||||
_ = init;
|
||||
|
||||
// Wait for the exfat server to bring up the USB storage chain and mount.
|
||||
var opened: ?fs.Directory = null;
|
||||
var tries: u32 = 0;
|
||||
while (opened == null and tries < 1400) : (tries += 1) {
|
||||
opened = fs.openDirectory(mount);
|
||||
if (opened == null) time.sleepMillis(50);
|
||||
}
|
||||
var dir = opened orelse {
|
||||
_ = logging.write("exfat-test: " ++ mount ++ " never became available\n");
|
||||
return;
|
||||
};
|
||||
var count: u32 = 0;
|
||||
var entry: fs.Entry = .{};
|
||||
while (dir.next(&entry)) {
|
||||
writeLine("exfat-test: entry '{s}' size={d}\n", .{ entry.name(), entry.size });
|
||||
count += 1;
|
||||
if (count > 32) break;
|
||||
}
|
||||
dir.close();
|
||||
|
||||
// Read the seeded HELLO.TXT (make-exfat-image.py writes "exfat hello danos\n").
|
||||
var read_ok = false;
|
||||
if (fs.open(mount ++ "/HELLO.TXT", .{})) |opened_file| {
|
||||
var file = opened_file;
|
||||
var buf: [32]u8 = undefined;
|
||||
const n = file.read(&buf) orelse 0;
|
||||
file.close();
|
||||
read_ok = std.mem.startsWith(u8, buf[0..n], "exfat hello danos");
|
||||
}
|
||||
if (read_ok) _ = logging.write("exfat-test: read HELLO.TXT ok\n");
|
||||
|
||||
// Mutation through the mount: mkdir, create + write, rename, read back, remove
|
||||
// — proof the write path reaches the engine over a real device.
|
||||
var mut_ok = false;
|
||||
if (fs.makeDirectory(mount ++ "/TESTDIR")) {
|
||||
var wrote = false;
|
||||
if (fs.open(mount ++ "/TESTDIR/W.TXT", .{ .create = true, .truncate = true })) |created| {
|
||||
var f = created;
|
||||
wrote = (f.writeAll("exfat-mutation-ok") orelse 0) == "exfat-mutation-ok".len;
|
||||
f.close();
|
||||
}
|
||||
const renamed = fs.rename(mount ++ "/TESTDIR/W.TXT", mount ++ "/TESTDIR/R.TXT");
|
||||
var readback = false;
|
||||
if (fs.open(mount ++ "/TESTDIR/R.TXT", .{})) |reopened| {
|
||||
var f = reopened;
|
||||
var buf: [32]u8 = undefined;
|
||||
const got = f.read(&buf) orelse 0;
|
||||
f.close();
|
||||
readback = std.mem.eql(u8, buf[0..got], "exfat-mutation-ok");
|
||||
}
|
||||
const removed = fs.remove(mount ++ "/TESTDIR/R.TXT");
|
||||
mut_ok = wrote and renamed and readback and removed;
|
||||
}
|
||||
if (mut_ok) _ = logging.write("exfat-test: mutations ok\n");
|
||||
|
||||
if (read_ok and mut_ok) {
|
||||
while (true) {
|
||||
_ = logging.write("exfat-test: ok\n");
|
||||
time.sleepMillis(1000);
|
||||
}
|
||||
}
|
||||
writeLine("exfat-test: FAILED (read={} mutations={})\n", .{ read_ok, mut_ok });
|
||||
}
|
||||
@@ -1,6 +1,6 @@
|
||||
//! test/system/services/fat-test — a client that proves the FAT mount end to end:
|
||||
//! it waits for the fat server to mount the USB volume at /volumes/usb, lists the
|
||||
//! root directory through the VFS (which routes /volumes/usb to the fat backend), and
|
||||
//! it waits for the fat server to mount the USB volume at /volumes/fat-12345678, lists the
|
||||
//! root directory through the VFS (which routes /volumes/fat-12345678 to the fat backend), and
|
||||
//! reads a known file off it. Shipped in the initial_ramdisk; the `fat-mount`
|
||||
//! kernel test spawns it alongside init.
|
||||
|
||||
@@ -18,16 +18,16 @@ fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||
pub fn main(init: process.Init) void {
|
||||
_ = init;
|
||||
|
||||
// Wait for /volumes/usb to be mounted — the fat server races us at boot (it must
|
||||
// Wait for /volumes/fat-12345678 to be mounted — the fat server races us at boot (it must
|
||||
// bring up the whole USB storage chain first).
|
||||
var opened: ?fs.Directory = null;
|
||||
var tries: u32 = 0;
|
||||
while (opened == null and tries < 1400) : (tries += 1) {
|
||||
opened = fs.openDirectory("/volumes/usb");
|
||||
opened = fs.openDirectory("/volumes/fat-12345678");
|
||||
if (opened == null) time.sleepMillis(50);
|
||||
}
|
||||
var dir = opened orelse {
|
||||
_ = logging.write("fat-test: /volumes/usb never became available\n");
|
||||
_ = logging.write("fat-test: /volumes/fat-12345678 never became available\n");
|
||||
return;
|
||||
};
|
||||
|
||||
@@ -43,56 +43,56 @@ pub fn main(init: process.Init) void {
|
||||
|
||||
// Read a known file off the boot volume through the mount (best effort): the
|
||||
// kernel image is an ELF, so its first bytes are the ELF magic.
|
||||
if (fs.open("/volumes/usb/system/kernel", .{})) |opened_file| {
|
||||
if (fs.open("/volumes/fat-12345678/system/kernel", .{})) |opened_file| {
|
||||
var file = opened_file;
|
||||
var magic: [4]u8 = undefined;
|
||||
const n = file.read(&magic) orelse 0;
|
||||
file.close();
|
||||
if (n == 4 and magic[0] == 0x7F and magic[1] == 'E' and magic[2] == 'L' and magic[3] == 'F') {
|
||||
_ = logging.write("fat-test: read /volumes/usb/system/kernel ELF magic ok\n");
|
||||
_ = logging.write("fat-test: read /volumes/fat-12345678/system/kernel ELF magic ok\n");
|
||||
} else {
|
||||
writeLine("fat-test: /volumes/usb/system/kernel read {d} bytes (not ELF magic)\n", .{n});
|
||||
writeLine("fat-test: /volumes/fat-12345678/system/kernel read {d} bytes (not ELF magic)\n", .{n});
|
||||
}
|
||||
}
|
||||
|
||||
// Exercise directory + file mutation through the mount: mkdir, create a file
|
||||
// inside it, read it back, then remove it — proof mkdir/unlink reach the engine.
|
||||
if (fs.makeDirectory("/volumes/usb/TESTDIR")) {
|
||||
if (fs.makeDirectory("/volumes/fat-12345678/TESTDIR")) {
|
||||
var wrote = false;
|
||||
if (fs.open("/volumes/usb/TESTDIR/HELLO.TXT", .{ .create = true, .truncate = true })) |created| {
|
||||
if (fs.open("/volumes/fat-12345678/TESTDIR/HELLO.TXT", .{ .create = true, .truncate = true })) |created| {
|
||||
var f = created;
|
||||
wrote = (f.writeAll("mutation-ok") orelse 0) == "mutation-ok".len;
|
||||
f.close();
|
||||
}
|
||||
// The created file carries a real modification time (stamped from the RTC).
|
||||
var mtime_ok = false;
|
||||
if (fs.attributes("/volumes/usb/TESTDIR/HELLO.TXT")) |attrs| {
|
||||
if (fs.attributes("/volumes/fat-12345678/TESTDIR/HELLO.TXT")) |attrs| {
|
||||
writeLine("fat-test: mtime {d}\n", .{attrs.mtime});
|
||||
mtime_ok = attrs.mtime > 1_577_836_800; // after 2020-01-01
|
||||
}
|
||||
if (mtime_ok) _ = logging.write("fat-test: mtime ok\n");
|
||||
|
||||
// Rename it, then read from the new name and confirm the old name is gone.
|
||||
const renamed = fs.rename("/volumes/usb/TESTDIR/HELLO.TXT", "/volumes/usb/TESTDIR/RENAMED.TXT");
|
||||
const old_gone = !fs.exists("/volumes/usb/TESTDIR/HELLO.TXT");
|
||||
const renamed = fs.rename("/volumes/fat-12345678/TESTDIR/HELLO.TXT", "/volumes/fat-12345678/TESTDIR/RENAMED.TXT");
|
||||
const old_gone = !fs.exists("/volumes/fat-12345678/TESTDIR/HELLO.TXT");
|
||||
if (renamed and old_gone) _ = logging.write("fat-test: rename ok\n");
|
||||
var readback = false;
|
||||
if (fs.open("/volumes/usb/TESTDIR/RENAMED.TXT", .{})) |reopened| {
|
||||
if (fs.open("/volumes/fat-12345678/TESTDIR/RENAMED.TXT", .{})) |reopened| {
|
||||
var f = reopened;
|
||||
var buf: [16]u8 = undefined;
|
||||
const got = f.read(&buf) orelse 0;
|
||||
f.close();
|
||||
readback = std.mem.eql(u8, buf[0..got], "mutation-ok");
|
||||
}
|
||||
const removed = fs.remove("/volumes/usb/TESTDIR/RENAMED.TXT");
|
||||
const gone = !fs.exists("/volumes/usb/TESTDIR/RENAMED.TXT");
|
||||
const removed = fs.remove("/volumes/fat-12345678/TESTDIR/RENAMED.TXT");
|
||||
const gone = !fs.exists("/volumes/fat-12345678/TESTDIR/RENAMED.TXT");
|
||||
if (wrote and mtime_ok and renamed and old_gone and readback and removed and gone) {
|
||||
_ = logging.write("fat-test: mutations ok\n");
|
||||
} else {
|
||||
writeLine("fat-test: mutations FAILED (wrote={} mtime={} renamed={} oldgone={} read={} removed={} gone={})\n", .{ wrote, mtime_ok, renamed, old_gone, readback, removed, gone });
|
||||
}
|
||||
} else {
|
||||
_ = logging.write("fat-test: mkdir /volumes/usb/TESTDIR failed\n");
|
||||
_ = logging.write("fat-test: mkdir /volumes/fat-12345678/TESTDIR failed\n");
|
||||
}
|
||||
|
||||
if (count > 0) {
|
||||
|
||||
@@ -77,13 +77,35 @@ fn park() void {
|
||||
var parked: ?fs.File = null;
|
||||
var tries: u32 = 0;
|
||||
while (parked == null and tries < 1000) : (tries += 1) {
|
||||
parked = fs.open("/volumes/usb/parked", .{ .create = true });
|
||||
parked = fs.open("/volumes/fat-12345678/parked", .{ .create = true });
|
||||
if (parked == null) time.sleepMillis(20);
|
||||
}
|
||||
if (parked == null) {
|
||||
_ = logging.write("vfstest: park open failed\n");
|
||||
return;
|
||||
}
|
||||
|
||||
// Mount ownership (V0, docs/volume-manager-plan.md): the volume is
|
||||
// provably mounted (the parked file just opened on it), it is FAT's mount,
|
||||
// and this process is not fat — unmounting it must be REFUSED and the
|
||||
// subtree must still resolve afterwards. Bailing here withholds the
|
||||
// "parked" marker, which fails the vfs-client-death case: before the
|
||||
// ownership gate existed, any process could unmount any prefix, and this
|
||||
// fixture would have deleted the volume out from under the whole boot.
|
||||
if (fs.fsUnmount("/volumes/fat-12345678")) {
|
||||
_ = logging.write("vfstest: foreign unmount was ALLOWED\n");
|
||||
return;
|
||||
}
|
||||
if (fs.open("/volumes/fat-12345678/parked", .{})) |resolved| {
|
||||
var verification = resolved;
|
||||
verification.close(); // the park below must be the client's ONLY open
|
||||
// handle — the kernel test string-matches "released 1 handle(s)".
|
||||
} else {
|
||||
_ = logging.write("vfstest: /volumes/fat-12345678 gone after refused unmount\n");
|
||||
return;
|
||||
}
|
||||
_ = logging.write("vfstest: foreign unmount refused\n");
|
||||
|
||||
while (true) {
|
||||
_ = logging.write("vfstest: parked\n");
|
||||
time.sleepMillis(500);
|
||||
|
||||
@@ -194,7 +194,6 @@ system/kernel/process.zig:write_buffer
|
||||
system/kernel/scheduler.zig:ipc_maximum_handles
|
||||
system/kernel/scheduler.zig:maximum_space_mappings
|
||||
system/kernel/vfs.zig:maximum_directories
|
||||
system/kernel/vfs.zig:maximum_mounts
|
||||
system/kernel/vfs.zig:maximum_prefix
|
||||
system/kernel/vfs.zig:maximum_rewrite
|
||||
system/services/acpi/acpi.zig:blocks
|
||||
|
||||
@@ -0,0 +1,307 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Format a real exFAT image from scratch — the danos exFAT test volume.
|
||||
|
||||
Pure Python 3 standard library (no mkfs.exfat / mtools). It writes a valid exFAT
|
||||
filesystem — a Main Boot Sector + its boot-region checksum + a backup region, the
|
||||
32-bit FAT, an allocation bitmap, an up-case table (with its checksum), and a root
|
||||
directory whose entry sets a real exFAT reader (and the danos exfat engine) mount
|
||||
and walk. Mirrors tools/make-fat-image.py in spirit.
|
||||
|
||||
make-exfat-image.py [--serial <hex>] [--label <name>] <out.img> <size-MiB>
|
||||
make-exfat-image.py --verify <out.img>
|
||||
|
||||
The image seeds one file, HELLO.TXT, so a mount can be proven by reading it.
|
||||
"""
|
||||
|
||||
import struct
|
||||
import sys
|
||||
|
||||
SECTOR = 512
|
||||
UPCASE_UNITS = 256 # a-z -> A-Z, the rest identity; covers ASCII names
|
||||
|
||||
|
||||
def align_up(value, to):
|
||||
return (value + to - 1) // to * to
|
||||
|
||||
|
||||
def rotr16(v):
|
||||
return ((v >> 1) | (v << 15)) & 0xFFFF
|
||||
|
||||
|
||||
def rotr32(v):
|
||||
return ((v >> 1) | (v << 31)) & 0xFFFFFFFF
|
||||
|
||||
|
||||
def boot_checksum(region):
|
||||
"""32-bit rotate-right sum over the boot region, skipping VolumeFlags
|
||||
(106,107) and PercentInUse (112) of the first sector."""
|
||||
checksum = 0
|
||||
for i, byte in enumerate(region):
|
||||
if i in (106, 107, 112):
|
||||
continue
|
||||
checksum = (rotr32(checksum) + byte) & 0xFFFFFFFF
|
||||
return checksum
|
||||
|
||||
|
||||
def upcase_checksum(table_bytes):
|
||||
checksum = 0
|
||||
for byte in table_bytes:
|
||||
checksum = (rotr32(checksum) + byte) & 0xFFFFFFFF
|
||||
return checksum
|
||||
|
||||
|
||||
def set_checksum(entries):
|
||||
"""16-bit rotate-right sum over a directory-entry set, skipping its own two
|
||||
checksum bytes (offset 2..3 of the first entry)."""
|
||||
checksum = 0
|
||||
for i, byte in enumerate(entries):
|
||||
if i in (2, 3):
|
||||
continue
|
||||
checksum = (rotr16(checksum) + byte) & 0xFFFF
|
||||
return checksum
|
||||
|
||||
|
||||
def name_hash(upcased_units):
|
||||
h = 0
|
||||
for unit in upcased_units:
|
||||
h = (rotr16(h) + (unit & 0xFF)) & 0xFFFF
|
||||
h = (rotr16(h) + (unit >> 8)) & 0xFFFF
|
||||
return h
|
||||
|
||||
|
||||
def ascii_upper(unit):
|
||||
return unit - ord("a") + ord("A") if ord("a") <= unit <= ord("z") else unit
|
||||
|
||||
|
||||
def solve_geometry(total_sectors, spc):
|
||||
"""Solve for cluster count / FAT length / heap offset that fit. The FAT sits
|
||||
after the main + backup boot regions (24 sectors)."""
|
||||
fat_offset = 24
|
||||
fat_length = 1
|
||||
while True:
|
||||
heap_offset = align_up(fat_offset + fat_length, spc)
|
||||
cluster_count = (total_sectors - heap_offset) // spc
|
||||
needed = ((cluster_count + 2) * 4 + SECTOR - 1) // SECTOR
|
||||
if needed <= fat_length:
|
||||
return cluster_count, fat_offset, fat_length, heap_offset
|
||||
fat_length = needed
|
||||
|
||||
|
||||
class ExfatImage:
|
||||
def __init__(self, size_mib, volume_id=0x1234ABCD, label="DANOS"):
|
||||
self.total_sectors = size_mib * 1024 * 1024 // SECTOR
|
||||
self.spc = 8 # 4 KiB clusters
|
||||
self.volume_id = volume_id & 0xFFFFFFFF
|
||||
self.label = label
|
||||
self.cluster_count, self.fat_offset, self.fat_length, self.heap_offset = solve_geometry(self.total_sectors, self.spc)
|
||||
if self.cluster_count < 16:
|
||||
sys.exit(f"error: image too small for exFAT ({self.cluster_count} clusters)")
|
||||
self.cluster_bytes = self.spc * SECTOR
|
||||
# Layout: the allocation bitmap (as many clusters as it needs — one per
|
||||
# 8*cluster_bytes clusters of the volume), then the up-case table, the root
|
||||
# directory, and the seeded file. A single-cluster bitmap (small volumes,
|
||||
# e.g. the 48 MiB fixture) puts root at cluster 4, as before.
|
||||
self.bitmap_bytes = (self.cluster_count + 7) // 8
|
||||
self.bitmap_clusters = (self.bitmap_bytes + self.cluster_bytes - 1) // self.cluster_bytes
|
||||
self.bitmap_cluster = 2
|
||||
self.upcase_cluster = self.bitmap_cluster + self.bitmap_clusters
|
||||
self.root_cluster = self.upcase_cluster + 1
|
||||
self.hello_cluster = self.root_cluster + 1
|
||||
self.image = bytearray(self.total_sectors * SECTOR)
|
||||
|
||||
def cluster_offset(self, cluster):
|
||||
return (self.heap_offset + (cluster - 2) * self.spc) * SECTOR
|
||||
|
||||
def set_fat(self, cluster, value):
|
||||
struct.pack_into("<I", self.image, self.fat_offset * SECTOR + cluster * 4, value)
|
||||
|
||||
def mark_allocated(self, cluster):
|
||||
bit = cluster - 2
|
||||
pos = self.cluster_offset(2) + bit // 8
|
||||
self.image[pos] |= 1 << (bit % 8)
|
||||
|
||||
def main_boot_sector(self):
|
||||
sector = bytearray(SECTOR)
|
||||
sector[0:3] = b"\xEB\x76\x90" # jump boot
|
||||
sector[3:11] = b"EXFAT " # filesystem name
|
||||
# 11..64 MustBeZero (already zero)
|
||||
struct.pack_into("<Q", sector, 72, self.total_sectors) # volume length
|
||||
struct.pack_into("<I", sector, 80, self.fat_offset) # fat offset
|
||||
struct.pack_into("<I", sector, 84, self.fat_length) # fat length
|
||||
struct.pack_into("<I", sector, 88, self.heap_offset) # cluster heap offset
|
||||
struct.pack_into("<I", sector, 92, self.cluster_count) # cluster count
|
||||
struct.pack_into("<I", sector, 96, self.root_cluster) # first cluster of root
|
||||
struct.pack_into("<I", sector, 100, self.volume_id) # volume serial number
|
||||
struct.pack_into("<H", sector, 104, 0x0100) # filesystem revision 1.0
|
||||
sector[108] = 9 # bytes per sector shift (512)
|
||||
sector[109] = self.spc.bit_length() - 1 # sectors per cluster shift
|
||||
sector[110] = 1 # number of FATs
|
||||
sector[111] = 0x80 # drive select
|
||||
sector[112] = 0xFF # percent in use (unknown)
|
||||
sector[510] = 0x55
|
||||
sector[511] = 0xAA
|
||||
return sector
|
||||
|
||||
def build(self):
|
||||
# Main boot region (sectors 0..11): VBR, eight extended boot sectors, OEM
|
||||
# parameters, reserved, then the checksum sector.
|
||||
vbr = self.main_boot_sector()
|
||||
self.image[0:SECTOR] = vbr
|
||||
for s in range(1, 9): # extended boot sectors carry the 0xAA550000 signature
|
||||
struct.pack_into("<I", self.image, s * SECTOR + 508, 0xAA550000)
|
||||
# sectors 9 (OEM) and 10 (reserved) stay zero
|
||||
region = bytes(self.image[0 : 11 * SECTOR])
|
||||
checksum = boot_checksum(region)
|
||||
for i in range(SECTOR // 4):
|
||||
struct.pack_into("<I", self.image, 11 * SECTOR + i * 4, checksum)
|
||||
# Backup boot region (sectors 12..23) is a copy of 0..11.
|
||||
self.image[12 * SECTOR : 24 * SECTOR] = self.image[0 : 12 * SECTOR]
|
||||
|
||||
# FAT: reserved entries, then a single-cluster chain per metadata object,
|
||||
# except the bitmap which spans self.bitmap_clusters (a real FAT chain).
|
||||
self.set_fat(0, 0xFFFFFFF8)
|
||||
self.set_fat(1, 0xFFFFFFFF)
|
||||
used = []
|
||||
for i in range(self.bitmap_clusters):
|
||||
cluster = self.bitmap_cluster + i
|
||||
self.set_fat(cluster, 0xFFFFFFFF if i == self.bitmap_clusters - 1 else cluster + 1)
|
||||
used.append(cluster)
|
||||
for cluster in (self.upcase_cluster, self.root_cluster, self.hello_cluster):
|
||||
self.set_fat(cluster, 0xFFFFFFFF)
|
||||
used.append(cluster)
|
||||
|
||||
# Allocation bitmap: every metadata/file cluster in use.
|
||||
for cluster in used:
|
||||
self.mark_allocated(cluster)
|
||||
|
||||
# Up-case table: 256 explicit units, a-z -> A-Z.
|
||||
upcase = bytearray(UPCASE_UNITS * 2)
|
||||
for i in range(UPCASE_UNITS):
|
||||
struct.pack_into("<H", upcase, i * 2, ascii_upper(i))
|
||||
off = self.cluster_offset(self.upcase_cluster)
|
||||
self.image[off : off + len(upcase)] = upcase
|
||||
table_checksum = upcase_checksum(upcase)
|
||||
|
||||
# Seed file HELLO.TXT (contiguous, one cluster).
|
||||
content = b"exfat hello danos\n"
|
||||
off = self.cluster_offset(self.hello_cluster)
|
||||
self.image[off : off + len(content)] = content
|
||||
|
||||
# Root directory: bitmap, up-case, volume label, HELLO set.
|
||||
root = self.cluster_offset(self.root_cluster)
|
||||
# 0x81 Allocation Bitmap
|
||||
struct.pack_into("<BBB", self.image, root, 0x81, 0, 0)
|
||||
struct.pack_into("<I", self.image, root + 20, self.bitmap_cluster)
|
||||
struct.pack_into("<Q", self.image, root + 24, self.bitmap_bytes)
|
||||
# 0x82 Up-case Table
|
||||
struct.pack_into("<B", self.image, root + 32, 0x82)
|
||||
struct.pack_into("<I", self.image, root + 32 + 4, table_checksum)
|
||||
struct.pack_into("<I", self.image, root + 32 + 20, self.upcase_cluster)
|
||||
struct.pack_into("<Q", self.image, root + 32 + 24, UPCASE_UNITS * 2)
|
||||
# 0x83 Volume Label
|
||||
label_units = [ord(c) for c in self.label[:11]]
|
||||
struct.pack_into("<BB", self.image, root + 64, 0x83, len(label_units))
|
||||
for i, u in enumerate(label_units):
|
||||
struct.pack_into("<H", self.image, root + 64 + 2 + i * 2, u)
|
||||
# HELLO.TXT set: File (0x85) + Stream (0xC0) + Name (0xC1)
|
||||
name = "HELLO.TXT"
|
||||
self.write_file_set(root + 96, name, first_cluster=self.hello_cluster, length=len(content))
|
||||
|
||||
def write_file_set(self, offset, name, first_cluster, length):
|
||||
entries = bytearray(32 * 3)
|
||||
# File entry
|
||||
entries[0] = 0x85
|
||||
entries[1] = 2 # stream + one name entry
|
||||
struct.pack_into("<H", entries, 4, 0x20) # attributes: archive
|
||||
# Stream entry
|
||||
entries[32 + 0] = 0xC0
|
||||
entries[32 + 1] = 0x01 | 0x02 # allocation possible + no FAT chain (contiguous)
|
||||
entries[32 + 3] = len(name)
|
||||
upname = [ascii_upper(ord(c)) for c in name]
|
||||
struct.pack_into("<H", entries, 32 + 4, name_hash(upname))
|
||||
struct.pack_into("<Q", entries, 32 + 8, length) # valid data length
|
||||
struct.pack_into("<I", entries, 32 + 20, first_cluster)
|
||||
struct.pack_into("<Q", entries, 32 + 24, length) # data length
|
||||
# File Name entry
|
||||
entries[64 + 0] = 0xC1
|
||||
for i, c in enumerate(name):
|
||||
struct.pack_into("<H", entries, 64 + 2 + i * 2, ord(c))
|
||||
struct.pack_into("<H", entries, 2, set_checksum(entries))
|
||||
self.image[offset : offset + len(entries)] = entries
|
||||
|
||||
def serialize(self):
|
||||
self.build()
|
||||
return bytes(self.image)
|
||||
|
||||
|
||||
def verify(path):
|
||||
with open(path, "rb") as handle:
|
||||
data = handle.read()
|
||||
if len(data) < 512 or data[510] != 0x55 or data[511] != 0xAA:
|
||||
sys.exit("verify: missing 0x55AA boot signature")
|
||||
if data[3:11] != b"EXFAT ":
|
||||
sys.exit("verify: not an exFAT boot sector")
|
||||
if any(data[11:64]):
|
||||
sys.exit("verify: MustBeZero region is not zero")
|
||||
fat_offset = struct.unpack_from("<I", data, 80)[0]
|
||||
heap_offset = struct.unpack_from("<I", data, 88)[0]
|
||||
cluster_count = struct.unpack_from("<I", data, 92)[0]
|
||||
root_cluster = struct.unpack_from("<I", data, 96)[0]
|
||||
spc = 1 << data[109]
|
||||
# Boot checksum sector 11 must match a fresh checksum over sectors 0..10.
|
||||
expected = boot_checksum(data[0 : 11 * SECTOR])
|
||||
got = struct.unpack_from("<I", data, 11 * SECTOR)[0]
|
||||
if expected != got:
|
||||
sys.exit(f"verify: boot checksum mismatch (0x{got:08X} != 0x{expected:08X})")
|
||||
# Walk the root directory for the HELLO.TXT set and check its checksum.
|
||||
root = (heap_offset + (root_cluster - 2) * spc) * SECTOR
|
||||
found = False
|
||||
for i in range(spc * SECTOR // 32):
|
||||
entry = root + i * 32
|
||||
if data[entry] == 0x00:
|
||||
break
|
||||
if data[entry] == 0x85:
|
||||
secondary = data[entry + 1]
|
||||
total = (secondary + 1) * 32
|
||||
stored = struct.unpack_from("<H", data, entry + 2)[0]
|
||||
if set_checksum(data[entry : entry + total]) != stored:
|
||||
sys.exit("verify: a file set checksum is wrong")
|
||||
found = True
|
||||
if not found:
|
||||
sys.exit("verify: no file set in the root directory")
|
||||
print(f"make-exfat-image: {path} OK "
|
||||
f"({cluster_count} clusters of {spc * SECTOR} bytes, fat@{fat_offset}, heap@{heap_offset})")
|
||||
|
||||
|
||||
def main(argv):
|
||||
if len(argv) == 3 and argv[1] == "--verify":
|
||||
verify(argv[2])
|
||||
return 0
|
||||
argv = list(argv)
|
||||
volume_id = 0x1234ABCD
|
||||
label = "DANOS"
|
||||
i = 1
|
||||
while i < len(argv):
|
||||
if argv[i] == "--serial" and i + 1 < len(argv):
|
||||
volume_id = int(argv[i + 1], 16)
|
||||
del argv[i : i + 2]
|
||||
elif argv[i] == "--label" and i + 1 < len(argv):
|
||||
label = argv[i + 1]
|
||||
del argv[i : i + 2]
|
||||
else:
|
||||
i += 1
|
||||
if len(argv) != 3:
|
||||
sys.exit("usage: make-exfat-image.py [--serial <hex>] [--label <name>] <out.img> <size-MiB>\n"
|
||||
" make-exfat-image.py --verify <out.img>")
|
||||
out_path = argv[1]
|
||||
size_mib = int(argv[2])
|
||||
image = ExfatImage(size_mib, volume_id, label)
|
||||
with open(out_path, "wb") as handle:
|
||||
handle.write(image.serialize())
|
||||
print(f"make-exfat-image: wrote {out_path} "
|
||||
f"({size_mib} MiB exFAT, {image.cluster_count} clusters, serial 0x{image.volume_id:08X})")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main(sys.argv))
|
||||
+31
-7
@@ -47,8 +47,14 @@ def fat32_geometry(total_sectors):
|
||||
|
||||
|
||||
class Fat32Image:
|
||||
def __init__(self, total_sectors):
|
||||
def __init__(self, total_sectors, volume_id=0x12345678, label="DANOS"):
|
||||
self.total_sectors = total_sectors
|
||||
# The FAT volume serial (its content identity — the /volumes/fat-<id>
|
||||
# mount path danos derives from it) and the display label. A second image
|
||||
# needs a distinct serial so its id-path does not collide with the boot
|
||||
# volume's.
|
||||
self.volume_id = volume_id & 0xFFFFFFFF
|
||||
self.label = label
|
||||
self.fat_size, self.cluster_count = fat32_geometry(total_sectors)
|
||||
if self.cluster_count < 65525:
|
||||
sys.exit(f"error: image too small for FAT32 ({self.cluster_count} clusters "
|
||||
@@ -146,8 +152,8 @@ class Fat32Image:
|
||||
0x80, # drive number
|
||||
0, # reserved
|
||||
0x29, # extended boot signature
|
||||
0x12345678, # volume id
|
||||
b"DANOS ", # volume label
|
||||
self.volume_id, # volume id
|
||||
self.label.encode("ascii", "replace")[:11].ljust(11, b" "), # volume label
|
||||
b"FAT32 ", # filesystem type
|
||||
)
|
||||
sector[510] = 0x55
|
||||
@@ -276,9 +282,9 @@ def build_tree(pairs):
|
||||
return root
|
||||
|
||||
|
||||
def build(out_path, size_mib, pairs):
|
||||
def build(out_path, size_mib, pairs, volume_id=0x12345678, label="DANOS"):
|
||||
total_sectors = size_mib * 1024 * 1024 // SECTOR
|
||||
image = Fat32Image(total_sectors)
|
||||
image = Fat32Image(total_sectors, volume_id, label)
|
||||
tree = build_tree(pairs)
|
||||
write_directory(image, 2, tree, 0, True)
|
||||
with open(out_path, "wb") as handle:
|
||||
@@ -350,14 +356,32 @@ def main(argv):
|
||||
if len(argv) == 3 and argv[1] == "--verify":
|
||||
verify(argv[2])
|
||||
return 0
|
||||
# Optional flags ahead of the positionals: --serial <hex> sets the FAT volume
|
||||
# id (the /volumes/fat-<id> content identity), --label <name> its display
|
||||
# label. A second FAT image passes a distinct --serial so its id-path cannot
|
||||
# collide with the boot volume's.
|
||||
argv = list(argv)
|
||||
volume_id = 0x12345678
|
||||
label = "DANOS"
|
||||
i = 1
|
||||
while i < len(argv):
|
||||
if argv[i] == "--serial" and i + 1 < len(argv):
|
||||
volume_id = int(argv[i + 1], 16)
|
||||
del argv[i:i + 2]
|
||||
elif argv[i] == "--label" and i + 1 < len(argv):
|
||||
label = argv[i + 1]
|
||||
del argv[i:i + 2]
|
||||
else:
|
||||
i += 1
|
||||
if len(argv) < 3 or (len(argv) - 3) % 2 != 0:
|
||||
sys.exit("usage: make-fat-image.py <out.img> <size-MiB> [<dest> <host>]...\n"
|
||||
sys.exit("usage: make-fat-image.py [--serial <hex>] [--label <name>] "
|
||||
"<out.img> <size-MiB> [<dest> <host>]...\n"
|
||||
" make-fat-image.py --verify <out.img>")
|
||||
out_path = argv[1]
|
||||
size_mib = int(argv[2])
|
||||
rest = argv[3:]
|
||||
pairs = [(rest[i], rest[i + 1]) for i in range(0, len(rest), 2)]
|
||||
build(out_path, size_mib, pairs)
|
||||
build(out_path, size_mib, pairs, volume_id, label)
|
||||
return 0
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,88 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Assemble an MBR-partitioned disk image from N FAT32 partitions — the danos
|
||||
multi-volume test disk.
|
||||
|
||||
Pure Python 3 stdlib (no mtools / parted). Each partition is a real FAT32
|
||||
filesystem produced by make-fat-image.py, laid out behind a classic MBR so the
|
||||
danos partition prober (partition.allVolumes) walks the table and the volume
|
||||
manager spawns one confined filesystem per partition — several volumes sharing
|
||||
ONE block channel, each clamped to its own LBA range. That shared-channel,
|
||||
per-partition path is what a single stick with two partitions exercises and a
|
||||
pair of single-volume sticks does not.
|
||||
|
||||
make-partitioned-image.py <out.img> [<serial-hex> <size-MiB>]...
|
||||
|
||||
Each partition is an empty FAT32 with the given volume serial (its /volumes/
|
||||
fat-<serial> content id). Partitions are 1-MiB aligned; the MBR marks each
|
||||
type 0x0C (FAT32 LBA). At most four (an MBR holds four primaries).
|
||||
"""
|
||||
|
||||
import os
|
||||
import struct
|
||||
import subprocess
|
||||
import sys
|
||||
import tempfile
|
||||
|
||||
SECTOR = 512
|
||||
ALIGN = 2048 # sectors (1 MiB) — standard partition alignment, and the gap the MBR sits in
|
||||
MBR_TYPE_FAT32_LBA = 0x0C
|
||||
MAX_PRIMARY_PARTITIONS = 4
|
||||
HERE = os.path.dirname(os.path.abspath(__file__))
|
||||
|
||||
|
||||
def align_up(sectors, to=ALIGN):
|
||||
return (sectors + to - 1) // to * to
|
||||
|
||||
|
||||
def main(argv):
|
||||
if len(argv) < 4 or (len(argv) - 2) % 2 != 0:
|
||||
sys.exit("usage: make-partitioned-image.py <out.img> [<serial-hex> <size-MiB>]...")
|
||||
out_path = argv[1]
|
||||
specs = [(argv[i], int(argv[i + 1])) for i in range(2, len(argv), 2)]
|
||||
if len(specs) > MAX_PRIMARY_PARTITIONS:
|
||||
sys.exit(f"error: an MBR holds at most {MAX_PRIMARY_PARTITIONS} primary partitions")
|
||||
|
||||
# Generate each partition's FAT32 image, then place it at its aligned start.
|
||||
partitions = [] # (start_sector, sector_count, bytes)
|
||||
cursor = ALIGN # leave the first 1 MiB for the MBR + alignment gap
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
for idx, (serial, size_mib) in enumerate(specs):
|
||||
part_path = os.path.join(tmp, f"p{idx}.img")
|
||||
subprocess.run(
|
||||
[sys.executable, os.path.join(HERE, "make-fat-image.py"),
|
||||
"--serial", serial, "--label", f"DATA{idx}",
|
||||
part_path, str(size_mib)],
|
||||
check=True, stdout=subprocess.DEVNULL)
|
||||
with open(part_path, "rb") as handle:
|
||||
data = handle.read()
|
||||
count = len(data) // SECTOR
|
||||
partitions.append((cursor, count, data))
|
||||
cursor = align_up(cursor + count)
|
||||
|
||||
total_sectors = cursor
|
||||
disk = bytearray(total_sectors * SECTOR)
|
||||
# The MBR: a disk signature, one partition entry per FAT partition, 0x55AA.
|
||||
# No boot code (this disk is data, never booted); danos's mount() sees the
|
||||
# signature but no BPB at LBA 0 and takes the MBR-walk path.
|
||||
struct.pack_into("<I", disk, 440, 0x0D05DA05) # arbitrary but fixed disk signature
|
||||
for idx, (start, count, data) in enumerate(partitions):
|
||||
entry = 446 + idx * 16
|
||||
disk[entry + 0] = 0x00 # not bootable
|
||||
disk[entry + 1:entry + 4] = b"\xFE\xFF\xFF" # CHS start (LBA-aware tools ignore)
|
||||
disk[entry + 4] = MBR_TYPE_FAT32_LBA
|
||||
disk[entry + 5:entry + 8] = b"\xFE\xFF\xFF" # CHS end
|
||||
struct.pack_into("<I", disk, entry + 8, start) # start LBA
|
||||
struct.pack_into("<I", disk, entry + 12, count) # sector count
|
||||
disk[start * SECTOR:start * SECTOR + len(data)] = data
|
||||
disk[510] = 0x55
|
||||
disk[511] = 0xAA
|
||||
|
||||
with open(out_path, "wb") as handle:
|
||||
handle.write(disk)
|
||||
print(f"make-partitioned-image: wrote {out_path} "
|
||||
f"({total_sectors * SECTOR // (1024 * 1024)} MiB, {len(partitions)} partitions)")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main(sys.argv))
|
||||
Reference in New Issue
Block a user