Compare commits
30
Commits
ea8ccf65d0
..
main
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
4037e746aa | ||
|
|
4dfb5012c0 | ||
|
|
061eb7c004 | ||
|
|
5dc966838a | ||
|
|
700452dc4e | ||
|
|
0faa0fd21b | ||
|
|
c47215821c | ||
|
|
6ddb08091d | ||
|
|
77b64229c2 | ||
|
|
90906bcefe | ||
|
|
2c2745e9e5 | ||
|
|
2a6d604577 | ||
|
|
e81a4e6f1d | ||
|
|
e240341bfb | ||
|
|
62eb2a748a | ||
|
|
56bd2e7678 | ||
|
|
0b25cd2c94 | ||
|
|
d59279422e | ||
|
|
bf9f8560c6 | ||
|
|
da7dcce64e | ||
|
|
9750db14da | ||
|
|
b2a5a0a3c6 | ||
|
|
7efe7b72d8 | ||
|
|
d4b544d66b | ||
|
|
6d4992ae02 | ||
|
|
df61693065 | ||
|
|
f1e79d0eeb | ||
|
|
167e9c7a9e | ||
|
|
5bfdb75e12 | ||
|
|
3fb8a9b96f |
@@ -122,6 +122,7 @@ fn driverArtifact(comptime package: []const u8, comptime artifact: []const u8) S
|
||||
/// (docs/build-packages-plan.md).
|
||||
const production_ship = [_]ShipRow{
|
||||
service("fat"),
|
||||
service("exfat"),
|
||||
service("display"),
|
||||
service("display-demo"),
|
||||
service("device-manager"),
|
||||
@@ -317,6 +318,11 @@ pub fn build(b: *std.Build) void {
|
||||
// out of the same read-only initrd, before it spawns anything — the registrar
|
||||
// has to know its policy before the first provider asks.
|
||||
bundled_list.append(b.allocator, .{ .path = "system/configuration/protocol.csv", .binary = b.path("system/configuration/protocol.csv") }) catch @panic("OOM");
|
||||
// The storage mount map (docs/file-system-development/storage-architecture.md):
|
||||
// filesystems.csv (content signature -> service binary) and volumes.csv (the
|
||||
// optional id -> mount-prefix override), both read by the volume manager.
|
||||
bundled_list.append(b.allocator, .{ .path = "system/configuration/filesystems.csv", .binary = b.path("system/configuration/filesystems.csv") }) catch @panic("OOM");
|
||||
bundled_list.append(b.allocator, .{ .path = "system/configuration/volumes.csv", .binary = b.path("system/configuration/volumes.csv") }) catch @panic("OOM");
|
||||
// A no-option build assumes neither -Dtest-case nor -Ddiagnose: it ships the
|
||||
// production set only. The userspace test fixtures under /test join in only
|
||||
// for a test build — which the QEMU harness signals by passing
|
||||
@@ -327,6 +333,7 @@ pub fn build(b: *std.Build) void {
|
||||
if (test_case != null) for ([_][]const u8{
|
||||
"vfs-test", // the user-space VFS round-trip client
|
||||
"fat-test",
|
||||
"exfat-test", // the exFAT mount round-trip client (S4)
|
||||
"badge-scope-test", // the guessable-id probe: a second process names the first's node and layer
|
||||
"shared-memory-server",
|
||||
"shared-memory-client",
|
||||
@@ -442,6 +449,7 @@ pub fn build(b: *std.Build) void {
|
||||
csv_library,
|
||||
xkeyboard_config_library,
|
||||
b.dependency("fat", .{}),
|
||||
b.dependency("exfat", .{}),
|
||||
b.dependency("volume-manager", .{}),
|
||||
b.dependency("display", .{}),
|
||||
b.dependency("ps2-bus", .{}),
|
||||
|
||||
@@ -46,6 +46,7 @@
|
||||
.@"pci-bus" = .{ .path = "system/drivers/pci-bus" },
|
||||
.init = .{ .path = "system/services/init" },
|
||||
.fat = .{ .path = "system/services/fat" },
|
||||
.exfat = .{ .path = "system/services/exfat" },
|
||||
.display = .{ .path = "system/services/display" },
|
||||
.@"display-demo" = .{ .path = "system/services/display-demo" },
|
||||
.@"device-manager" = .{ .path = "system/services/device-manager" },
|
||||
@@ -64,6 +65,7 @@
|
||||
.@"virtio-gpu" = .{ .path = "system/drivers/virtio-gpu" },
|
||||
.@"vfs-test" = .{ .path = "test/system/services/vfs-test", .lazy = true },
|
||||
.@"fat-test" = .{ .path = "test/system/services/fat-test", .lazy = true },
|
||||
.@"exfat-test" = .{ .path = "test/system/services/exfat-test", .lazy = true },
|
||||
.@"badge-scope-test" = .{ .path = "test/system/services/badge-scope-test", .lazy = true },
|
||||
.@"shared-memory-server" = .{ .path = "test/system/services/shared-memory-server", .lazy = true },
|
||||
.@"shared-memory-client" = .{ .path = "test/system/services/shared-memory-client", .lazy = true },
|
||||
|
||||
@@ -3,18 +3,37 @@
|
||||
> **Status:** the layered model below is the settled design
|
||||
> ([storage-design-rationale.md](storage-design-rationale.md) records how it was
|
||||
> reached, and [volume-manager-plan.md](../volume-manager-plan.md) how it was
|
||||
> built). **Built** (the volume-manager track, V0–V4): the data path, the driver
|
||||
> built). **Built** (V0–V4 + the storage-stack S1/S2): the data path, the driver
|
||||
> range confinement (per-sender clamp + the confinement gate), the `medium_changed`
|
||||
> presence event, the volume manager itself — it probes the partition table,
|
||||
> confines each filesystem to its partition, spawns one filesystem per volume, and
|
||||
> supervises it — and the removal half of the lifecycle (a pulled stick unmounts).
|
||||
> **Still pending**: the fuller identity ladder and the `volumes.csv` mount map,
|
||||
> multi-volume (one FAT volume today), the volume manager *consuming*
|
||||
> `medium_changed` (removal is detected by device-presence polling; the event is
|
||||
> published but only a card-reader medium change needs the subscription), and the
|
||||
> remount-on-replug end-to-end (the logic is in place; QEMU can't re-present the
|
||||
> boot-controller device, so it is bench-verified). A few markers below are left
|
||||
> where a duty is still pending.
|
||||
> supervises it — the removal half of the lifecycle (a pulled stick unmounts), the
|
||||
> identity ladder (GPT GUID + name, FAT serial + label, MBR), and the mount map:
|
||||
> `filesystems.csv` (signature → binary) + `volumes.csv` (identity → optional
|
||||
> override), a volume's mount path IS its content id (`/volumes/<id>`), with the
|
||||
> label as display metadata a `volumes` query returns. Multi-volume is **built**:
|
||||
> the manager adopts every storage device, probes each device's whole partition
|
||||
> table, and spawns one range-confined FAT per volume — several volumes across
|
||||
> several devices, or several partitions sharing one device's channel — each at
|
||||
> its own `/volumes/<id>` path with its own supervision. The boot volume is
|
||||
> identified by **content** (a volume backs `/system/configuration` + `/system/logs`
|
||||
> only when it resolves `/system/configuration` on its own media), so it works as
|
||||
> any partition of any device. exFAT is **built** as a second engine
|
||||
> (`system/services/exfat`): full read + write, directories, rename, and on-disk
|
||||
> up-case folding, reusing `library/kernel/file-system-harness` wholesale — the
|
||||
> reuse claim, proven — and a volume routes to fat or exfat by its VBR, at an
|
||||
> `exfat-<serial>` id-path. Removal is robust to all three triggers now: a
|
||||
> pulled device (presence polling), a medium that leaves while its device stays
|
||||
> (the volume manager CONSUMES `medium_changed`), and a storage driver that
|
||||
> crashes while its device stays present (a channel-liveness `geometry()` probe
|
||||
> reaps the volume and rebuilds it on the restarted driver's fresh channel). The
|
||||
> re-adopt-and-remount path is QEMU-proven by the driver-crash rebuild; a physical
|
||||
> unplug/replug exercises the same path but is bench-pending (QEMU cannot
|
||||
> re-present a usb-storage `device_add`). **Still pending**: the `filesystem UUID`
|
||||
> rung (ext-family superblocks, which need such an engine); and arbitration when
|
||||
> two volumes both resolve the boot markers (S3 mounts both and logs each claim;
|
||||
> picking one is deferred). A few
|
||||
> markers below are left where a duty is still pending.
|
||||
|
||||
## The model
|
||||
|
||||
@@ -85,16 +104,17 @@ manager's tree for a storage provider; when one appears it consumer-hellos for
|
||||
the block channel, reads the partition table and the first blocks itself
|
||||
(**it** is the prober), defines the volume's sub-range on the driver, spawns the
|
||||
matching filesystem service confined to that range, and supervises it (backoff,
|
||||
crash-loop cap). *(Pending)*: it decides mount placement from `volumes.csv` and
|
||||
picks the filesystem binary from `filesystems.csv` — today it hands every
|
||||
FAT-shaped volume to the FAT service and the FAT service carries hardcoded mount
|
||||
prefixes. Those tables are CSV configuration, read by it (the policy), enforced
|
||||
by nobody else:
|
||||
crash-loop cap). *(Built)*: it picks the filesystem binary from
|
||||
`filesystems.csv` by the volume's content signature, and mounts the volume at its
|
||||
content id (`/volumes/<id>`) — or a `volumes.csv` override. The label is display
|
||||
metadata the `volumes` query returns, never the path. Those tables are CSV
|
||||
configuration, read by it (the policy), enforced by nobody else:
|
||||
|
||||
- `filesystems.csv` *(pending)* — content signature → filesystem binary. Adding
|
||||
- `filesystems.csv` *(built)* — content signature → filesystem binary. Adding
|
||||
a filesystem adds a row.
|
||||
- `volumes.csv` *(pending)* — the mount map, danos's fstab: **volume identity → mount
|
||||
prefix**, keyed on content identity and never on port, path, or arrival
|
||||
- `volumes.csv` *(built)* — the mount map, danos's fstab: an OPTIONAL **volume
|
||||
identity → mount prefix** override (a volume with no row mounts at its default
|
||||
`/volumes/<id>`), keyed on content identity and never on port, path, or arrival
|
||||
order (the lesson of Linux's `/dev/sda1`-era fstab, which broke on every
|
||||
port move until `UUID=` replaced it). Identity is read off the medium by
|
||||
the prober, strongest first: GPT partition GUID → filesystem UUID → FAT
|
||||
@@ -105,8 +125,8 @@ by nobody else:
|
||||
identity (cloned sticks, together) is policy: first keeps the name, the
|
||||
second mounts suffixed and is logged loudly. The boot volume is the
|
||||
recorded identity of the volume carrying `/system/configuration` and
|
||||
`/system/logs`, findable on any port. Unknown volumes mount under
|
||||
`/volumes/<derived name>`.
|
||||
`/system/logs`, findable on any port. Every volume's default mount is
|
||||
`/volumes/<id>` — its rendered content identity.
|
||||
|
||||
**Filesystem service** (the FAT service today; one process per volume): the
|
||||
proven unit — block-client + engine + file-protocol provider in one binary. It
|
||||
@@ -114,12 +134,17 @@ receives its block channel at spawn; it never discovers devices. It registers
|
||||
its own mounts with the kernel; its write cache lives inside the process, so a
|
||||
write error is observed by the code that owns the volume and surfaces on the
|
||||
owning channel (the anti-fsyncgate rule — never a system-wide dirty pool).
|
||||
*(Today, interim:)* fat still hardcodes its mount prefixes (`/volumes/usb` plus
|
||||
the two boot-volume hierarchy subtrees it rewrites in place); a `volumes.csv`
|
||||
mount map will migrate that to the volume manager. It no longer self-acquires a
|
||||
volume — the V3b flip made it receive its volume id at spawn and its block
|
||||
channel from the volume manager's hello reply, consistent with "it never
|
||||
discovers devices" above.
|
||||
*(Built:)* fat receives its mount path as `argv[2]` from the volume manager (the
|
||||
volume's id-path, e.g. `/volumes/fat-12345678`) and mounts its root there. It
|
||||
installs the two `/system` hierarchy rewrites (`/system/configuration`,
|
||||
`/system/logs`) only when it is the boot volume — decided by **content**: it
|
||||
resolves `/system/configuration` on its own media at mount, so a data volume
|
||||
mounts at its id-path alone and never shadows the running system. It no longer
|
||||
self-acquires a volume — the V3b flip made it receive its volume id and block
|
||||
channel from the volume manager, consistent with "it never discovers devices"
|
||||
above. Because several volumes now serve at once, no filesystem binds a shared
|
||||
service name; clients reach each through the kernel mount table (`fs_resolve`
|
||||
routes by prefix to the backing endpoint).
|
||||
|
||||
**Kernel** (mechanism only): the mount table routes paths to backend
|
||||
endpoints — resolve and redirect, never data. Remount-replace is the restart
|
||||
@@ -165,25 +190,26 @@ surprise-removal path — kill the filesystem process, retire its mounts,
|
||||
respawn on return. No half-alive states, no `remount-ro`, no mounts that
|
||||
error forever (Plan 9's dead-server wart).
|
||||
|
||||
The path has **two triggers, one lifecycle**: the *device* leaving (the
|
||||
storage driver dies — channel death, the table below), and the *medium*
|
||||
leaving while the device stays (an SD card pulled from its reader, an ATAPI
|
||||
tray opened — including USB card readers today). The second trigger is the
|
||||
pushed `medium_changed` event on the block protocol — published today from a
|
||||
TEST UNIT READY poll; still *planned* is the volume manager *consuming* it
|
||||
(today removal is driven only by device-presence polling) and translating the
|
||||
transport's native signal (SCSI UNIT ATTENTION, AHCI PxSSTS, NVMe
|
||||
namespace-change AER) in place of the poll. On the event the volume manager
|
||||
runs the same kill-retire path, then re-probes on medium return exactly as on
|
||||
device return. Without it, a swapped card would be served with the previous
|
||||
card's filesystem state.
|
||||
The path folds **three triggers into one lifecycle**: the *device* leaving (a
|
||||
pulled stick — presence polling); the *medium* leaving while the device stays
|
||||
(an SD card pulled from its reader, an ATAPI tray opened, a USB card reader);
|
||||
and a storage *driver crashing* while its device stays in the tree. The second
|
||||
trigger is the pushed `medium_changed` event on the block protocol, published
|
||||
from a TEST UNIT READY poll — the volume manager now **consumes** it (subscribed
|
||||
per device), running the same kill-retire path and re-probing on medium return,
|
||||
so a swapped card is never served with the previous card's filesystem state. The
|
||||
third is caught by a channel-liveness `geometry()` probe: presence polling alone
|
||||
sees the device still present, but the channel is dead, so the manager reaps the
|
||||
volume and rebuilds it on the restarted driver's fresh channel. Still *planned*
|
||||
is translating the transport's native signal (SCSI UNIT ATTENTION, AHCI PxSSTS,
|
||||
NVMe namespace-change AER) in place of the presence poll.
|
||||
|
||||
| Layer | Observes | Must do | Guarantees |
|
||||
|---|---|---|---|
|
||||
| Bus driver | port/hub status change | tear down the device's slots (children first, recursively — built, hot-plug matrix), report `child_removed` per interface | the device tree is honest within one reconcile tick |
|
||||
| Device manager | `child_removed` / reporter death | prune the child; **reap the bound driver** (built) — the storage driver for that stick dies now, not never | no zombie storage processes; re-report rebinds |
|
||||
| Storage driver | its own death (it IS the removed device's driver) | nothing — dying is its removal handling; DMA/IOMMU/claims release mechanically at death | in-flight transfers fail visibly to callers, never hang |
|
||||
| Volume manager *(removal built; remount bench-pending)* | the storage device leaving the device-manager tree (poll) | kill the filesystem service of that device's volume; its kernel mounts retire | one removal path; mounts never dangle; log persistence stops *cleanly* |
|
||||
| Volume manager *(built)* | a device leaving the tree (poll), a `medium_changed` event, or a dead channel under a still-present device (a crashed driver — `geometry()` liveness probe) | kill that volume's filesystem service (its mounts retire), then re-adopt + remount on return or on the restarted driver's fresh channel | one removal path for all three triggers; mounts never dangle; the manager never serves from behind a dead channel |
|
||||
| Filesystem service | its block channel dies (`EPEER`) mid-operation, or it is killed by the volume manager | if it observes the death first: flush nothing (the medium is gone), answer in-flight requests with errors, exit; dirty write-back data is **lost and said to be lost** | the unflushed write-back window is dropped on a surprise yank — danos writes no on-disk dirty/clean-shutdown marker today; the process never serves from behind a dead channel |
|
||||
| Kernel | backend endpoint death | lazy mount-slot sweep on next resolve (built); ownership-gated `fs_unmount` (built, V0) | resolution under a dead mount is `not_found`, not a hang |
|
||||
| Application | `not_found` / error on paths under the vanished mount | its own error handling — the contract is honest absence, identical to the path never existing | no operation blocks forever on removed media |
|
||||
|
||||
@@ -104,12 +104,19 @@ and /system/logs), closing the two-sticks question honestly.
|
||||
**Filesystems (per volume, one process).** The proven unit everywhere from
|
||||
Plan 9's `dossrv` to Minix to Fuchsia: block-client + engine + file-protocol
|
||||
provider in one binary, one process per volume (9front practice; per-volume
|
||||
fault isolation is what our supervision makes cheap). fat's shell becomes a
|
||||
shared *filesystem harness* library before a second engine is written; a
|
||||
partition walk is added in the volume manager (`partition.zig`) — the engine's
|
||||
own MBR walk currently remains alongside it; write caching stays
|
||||
inside the process (the anti-fsyncgate rule). Each mounts its prefixes into the
|
||||
kernel mount table itself, exactly as today.
|
||||
fault isolation is what our supervision makes cheap). fat's shell became a
|
||||
shared *filesystem harness* library (`library/kernel/file-system-harness`), and
|
||||
the second engine — **exFAT**, `system/services/exfat` — now reuses it wholesale:
|
||||
the reuse this design promised, proven. exfat is nothing but the exFAT engine +
|
||||
a near-clone of fat's thin service, full read + write + directories + rename +
|
||||
on-disk up-case folding, differing only in the format it wraps. A partition walk
|
||||
lives in the volume manager (`partition.zig`), which recognizes fat vs exFAT by
|
||||
VBR and routes each to its engine; write caching stays inside the process (the
|
||||
anti-fsyncgate rule). Each mounts its prefixes into the kernel mount table
|
||||
itself. Two surface limits are shared across both engines and are the vfs
|
||||
layer's, not an exFAT shortcut: file offsets are u32 (a 4 GiB addressable cap),
|
||||
and file names are ASCII bytes (a non-ASCII unit becomes `?`) — teaching the vfs
|
||||
name layer UTF-8 is a separate cross-cutting change.
|
||||
|
||||
**Kernel: two small changes only.** `fs_unmount` gains ownership (only the
|
||||
mounting endpoint's holder may unmount — possession-is-capability, consistent
|
||||
@@ -145,9 +152,13 @@ matrix-proven shape; genuinely open.
|
||||
|
||||
**The pressure points, honestly:**
|
||||
|
||||
1. **Multi-volume providers are reserved, not implemented.** The volume
|
||||
manager flow assumes one provider, one volume; NVMe namespaces make
|
||||
endpoint-per-volume real work with hardware demanding it.
|
||||
1. **Multi-volume is built; multi-namespace-per-provider is untried.** The
|
||||
volume manager adopts every device and spawns one range-confined FAT per
|
||||
partition — several volumes across several devices, or several partitions
|
||||
sharing one device's channel, both proven on USB. What is untried is a single
|
||||
provider exposing several volumes as *namespaces* (NVMe): the endpoint and
|
||||
per-badge range machinery generalizes, but no such driver exists yet to
|
||||
exercise it.
|
||||
2. **The current transport will bottleneck NVMe.** Synchronous call/reply,
|
||||
one operation in flight, one bounce buffer — fine for a USB2 stick,
|
||||
forfeits an NVMe drive's queue depth and per-queue MSI-X. Correctness
|
||||
@@ -198,18 +209,23 @@ matrix-proven shape; genuinely open.
|
||||
names it as the 256-byte ceiling's unlock — Fuchsia's FIFO+VMO is the
|
||||
precedent); format-level crash honesty (a Power-Safe-style journaling or COW
|
||||
filesystem) once danos outgrows FAT; per-process namespaces.
|
||||
7. **The media-presence event** (settled in principle; lands with the volume
|
||||
manager): the block protocol gains a pushed event — `medium_changed`, with
|
||||
present/absent and a change counter — produced by the storage driver from
|
||||
its transport's native signal (SCSI UNIT ATTENTION / TEST UNIT READY for
|
||||
USB and ATAPI, PxSSTS for AHCI, namespace-change AER for NVMe) and
|
||||
consumed by the volume manager, which runs the SAME kill-retire-remount
|
||||
path it runs on channel death — one lifecycle, two triggers. The driver
|
||||
reports presence, never content; a pushed event carries no capability,
|
||||
which the kernel already guarantees. The device staying while its medium
|
||||
leaves is the one removable-media case the channel-death trigger cannot
|
||||
see; without this event a swapped SD card would be served with the old
|
||||
card's filesystem state.
|
||||
7. **The media-presence event** (the consuming half is BUILT; the
|
||||
transport-native signal stays future): the block protocol carries a pushed
|
||||
event — `medium_changed`, with present/absent and a change counter —
|
||||
produced today by the storage driver from a TEST UNIT READY poll (the
|
||||
transport's native signal — SCSI UNIT ATTENTION, PxSSTS for AHCI,
|
||||
namespace-change AER for NVMe — is the future refinement in place of the
|
||||
poll) and now **consumed** by the volume manager, which subscribes per
|
||||
device and runs the SAME kill-retire-remount path it runs on channel death.
|
||||
The driver reports presence, never content; a pushed event carries no
|
||||
capability, which the kernel already guarantees. The device staying while
|
||||
its medium leaves is the one removable-media case the channel-death trigger
|
||||
cannot see; without this event a swapped SD card would be served with the
|
||||
old card's filesystem state. A THIRD trigger closes the last gap — a
|
||||
storage driver that *crashes* while its device stays present: channel death
|
||||
there is invisible to presence polling, so the volume manager probes channel
|
||||
liveness (`geometry()`) each tick and reaps-then-rebuilds the volume on the
|
||||
restarted driver's fresh channel. One lifecycle, three triggers.
|
||||
8. **Volume identity, and the mount map as danos's fstab** (settled). The
|
||||
lesson is Linux's own history: fstab keyed on `/dev/sda1` for years and
|
||||
broke whenever a drive changed ports or enumeration order; `UUID=` entries
|
||||
@@ -218,9 +234,10 @@ matrix-proven shape; genuinely open.
|
||||
**content identity, never port or discovery order**. Build status: rungs 1,
|
||||
3, and 4 (GPT partition GUID, FAT serial + label, MBR signature + index) are
|
||||
implemented (S1); rung 2 waits on a non-FAT engine. The `volumes.csv` map and
|
||||
the id-derived mount path land with S2, so today a single volume still mounts
|
||||
at the fixed `/volumes/usb` and its recorded identity is not yet consulted to
|
||||
pick a path. The ladder the prober reads off the medium, strongest first:
|
||||
the id-derived mount path are built (S2): a volume's mount path IS its content
|
||||
id (`/volumes/<id>`, e.g. `/volumes/fat-12345678`), or a `volumes.csv`
|
||||
override; the label is display metadata a `volumes` query returns, never the
|
||||
path. The ladder the prober reads off the medium, strongest first:
|
||||
1. GPT partition GUID — 128-bit, unique, stable for the volume's life — **built (S1)**;
|
||||
2. filesystem UUID (ext-family and most modern formats, in the superblock) *(planned)*;
|
||||
3. FAT volume serial + label — 32 bits, weak (dd-cloned sticks share it)
|
||||
@@ -232,13 +249,16 @@ matrix-proven shape; genuinely open.
|
||||
Consequences, each mechanical once identity keys the map: **moving a drive
|
||||
to a different port changes nothing** — same identity, same mount point,
|
||||
whether USB port, hub depth, SATA port, or a stick that left as USB and
|
||||
returned in a SATA dock; **replug remounts at the same path** (a map lookup
|
||||
once the map exists; today's single volume re-probes and remounts at the
|
||||
fixed prefix, and remount-on-replug is bench-verified, not QEMU-tested); **the boot volume** is the
|
||||
recorded identity of the volume carrying `/system/configuration`, findable
|
||||
on any port; and **duplicate identity is a policy case, not a surprise** —
|
||||
two cloned sticks at once: first keeps the mapped name, second mounts
|
||||
suffixed and is logged loudly, never silently shadowed. Unknown identities
|
||||
returned in a SATA dock; **replug remounts at the same path** (the id-path is
|
||||
content-derived, so a volume returns to `/volumes/<id>` wherever it reappears;
|
||||
the re-adopt+remount code path is QEMU-proven by the driver-crash rebuild, but
|
||||
remount on a *physical* replug end-to-end is bench-pending, not QEMU-testable,
|
||||
because QEMU can't re-present the boot-controller device); **the boot volume** is the volume
|
||||
that resolves `/system/configuration` on its own media, findable on any port or
|
||||
partition; and **duplicate identity is a known S4 gap** — two cloned sticks
|
||||
share one content id, so today they collide on `/volumes/<id>` (the kernel
|
||||
remount-replaces; the last wins) and each boot-volume claim is logged loudly.
|
||||
Distinguishing them with a suffix is arbitration, deferred to S4. Unknown identities
|
||||
mount under a derived name (sanitized label, else generated) at
|
||||
`/volumes/<name>` — the hierarchy's documented home for attached media,
|
||||
which stands: `/system` is what danos IS; attached media is what it isn't.
|
||||
|
||||
@@ -7,12 +7,25 @@
|
||||
//! `runtime.dma.alloc`), so whole sectors move without crossing the IPC size
|
||||
//! limit — the same handoff usb-storage uses toward the controller.
|
||||
|
||||
const std = @import("std");
|
||||
const envelope = @import("envelope");
|
||||
const ipc = @import("ipc");
|
||||
const block_protocol = @import("block-protocol");
|
||||
|
||||
const Protocol = block_protocol.Protocol;
|
||||
|
||||
/// The medium_changed event payload, re-exported so a consumer decodes it without
|
||||
/// reaching into the wire-format module.
|
||||
pub const MediumChanged = block_protocol.MediumChanged;
|
||||
|
||||
/// Decode a medium_changed event from a buffered-message payload a subscriber
|
||||
/// received (a `Received.isMessage` wake). Null if the bytes are too short to be
|
||||
/// one — a caller ignores anything that is not a well-formed event.
|
||||
pub fn decodeMediumChanged(payload: []const u8) ?MediumChanged {
|
||||
if (payload.len < envelope.prefix_size + @sizeOf(MediumChanged)) return null;
|
||||
return std.mem.bytesToValue(MediumChanged, payload[envelope.prefix_size..][0..@sizeOf(MediumChanged)]);
|
||||
}
|
||||
|
||||
pub const Geometry = struct { block_size: u32, block_count: u64 };
|
||||
|
||||
pub const Device = struct {
|
||||
@@ -88,6 +101,30 @@ pub const Device = struct {
|
||||
if (status.status != 0) return null;
|
||||
return reply[0..answer.len];
|
||||
}
|
||||
|
||||
/// Subscribe `subscriber` (an endpoint) to this device's medium_changed
|
||||
/// events: the reserved `subscribe` verb carries the subscriber's endpoint as
|
||||
/// the capability, and the driver then ipc.sends each medium transition to it.
|
||||
pub fn subscribeMedium(self: Device, subscriber: ipc.Handle) bool {
|
||||
var packet: [block_protocol.message_maximum]u8 = undefined;
|
||||
const framed = envelope.encodeSubscribe(0, &packet) orelse return false; // interest 0: every event (block has one)
|
||||
var reply: [block_protocol.message_maximum]u8 = undefined;
|
||||
const answer = ipc.callCap(self.endpoint, framed, &reply, subscriber) catch return false;
|
||||
const status = envelope.statusOf(reply[0..answer.len]) orelse return false;
|
||||
return status.status == 0;
|
||||
}
|
||||
|
||||
/// Unsubscribe from this device's medium_changed events. Call before closing
|
||||
/// the channel so the driver's bounded subscriber table frees the slot rather
|
||||
/// than holding a dead endpoint until an exit sweep notices.
|
||||
pub fn unsubscribeMedium(self: Device) bool {
|
||||
var packet: [block_protocol.message_maximum]u8 = undefined;
|
||||
const framed = envelope.encodeUnsubscribe(&packet) orelse return false;
|
||||
var reply: [block_protocol.message_maximum]u8 = undefined;
|
||||
const answer = ipc.callCap(self.endpoint, framed, &reply, null) catch return false;
|
||||
const status = envelope.statusOf(reply[0..answer.len]) orelse return false;
|
||||
return status.status == 0;
|
||||
}
|
||||
};
|
||||
|
||||
// There is deliberately no open-by-name here: `block` is not a registry name.
|
||||
|
||||
@@ -61,10 +61,14 @@ pub fn Server(comptime Engine: type) type {
|
||||
/// caller does the filesystem-specific bring-up (find the block
|
||||
/// device, set up DMA, mount the engine) and returns a `Volume`.
|
||||
bringUp: *const fn (endpoint: ipc.Handle) ?Volume,
|
||||
/// The vfs contract name to bind. A filesystem serving one volume
|
||||
/// binds "vfs" today; the volume-manager era hands each per-volume
|
||||
/// process its own establishment and this fades.
|
||||
service_name: ?[]const u8 = "vfs",
|
||||
/// A contract name to bind under /protocol, or null to bind none. In
|
||||
/// the volume-manager era every filesystem is a per-volume process and
|
||||
/// clients reach it through the kernel mount table — fs_resolve routes
|
||||
/// a path to its backing endpoint by prefix — so no filesystem binds a
|
||||
/// shared name. Two volumes would collide on one: the second's bind is
|
||||
/// refused and service.run would exit, so its volume never mounts. The
|
||||
/// endpoint still serves as the mount backend without a name.
|
||||
service_name: ?[]const u8 = null,
|
||||
};
|
||||
|
||||
// --- the harness's own state, one set per instantiation ---------------
|
||||
|
||||
@@ -67,6 +67,15 @@ pub const Callbacks = struct {
|
||||
/// A notification that is not a signal — a subscribed exit event, a bound
|
||||
/// IRQ, a timer landing. The raw badge; decode with the ipc helpers.
|
||||
on_notification: ?*const fn (badge: u64) void = null,
|
||||
/// A buffered async message (`Received.isMessage`): a pushed event from a
|
||||
/// provider this service subscribed to, its payload in the receive buffer.
|
||||
/// Unlike `on_message`, it never goes through the protocol dispatch — so an
|
||||
/// event whose reserved op number collides with one of this service's own
|
||||
/// verbs (a `block` `medium_changed` reaching the volume manager, whose own
|
||||
/// protocol numbers `hello` the same) is decoded by hand here, not
|
||||
/// mis-dispatched. Default null: the badge alone still reaches
|
||||
/// `on_notification`, exactly as before this callback existed.
|
||||
on_buffered_message: ?*const fn (message: []const u8) void = null,
|
||||
/// The reload signal. Default: ignored.
|
||||
on_reload: ?*const fn () void = null,
|
||||
/// The terminate signal, called before the loop returns. The clean exit is
|
||||
@@ -372,6 +381,13 @@ pub fn run(comptime maximum_message: usize, callbacks: Callbacks) void {
|
||||
if (got.isChildExit()) {
|
||||
if (callbacks.subscribers) |subscribers| subscribers.forget(got.childProcessId());
|
||||
}
|
||||
// A buffered async message (a pushed event) carries a payload; hand it
|
||||
// to the service that asked for it. The badge still reaches
|
||||
// on_notification below, so a coalesced timer/exit riding the same wake
|
||||
// is not lost — and a service without this callback is unchanged.
|
||||
if (got.isMessage()) {
|
||||
if (callbacks.on_buffered_message) |onBuffered| onBuffered(receive[0..got.len]);
|
||||
}
|
||||
if (callbacks.on_notification) |onNotification| onNotification(got.badge);
|
||||
continue;
|
||||
}
|
||||
|
||||
@@ -12,6 +12,7 @@
|
||||
//! "Establishment: two planes"). No channel in the reply means the volume is not
|
||||
//! ready yet — retryable, never a verdict.
|
||||
|
||||
const std = @import("std");
|
||||
const envelope = @import("envelope");
|
||||
|
||||
pub const version: u16 = 1;
|
||||
@@ -24,13 +25,91 @@ pub const Hello = extern struct {
|
||||
_padding: u16 = 0,
|
||||
};
|
||||
|
||||
/// A `volumes` query — no request fields; the reply's tail carries the volume's
|
||||
/// descriptor (`VolumeInfo`). The mechanism by which a shell or file manager
|
||||
/// reads a volume's display label: the mount path is its id (software's stable
|
||||
/// handle), the label is separate display metadata, the database id/name split.
|
||||
pub const Volumes = extern struct {
|
||||
_reserved: u32 = 0,
|
||||
};
|
||||
|
||||
/// The `volumes` reply: three length-prefixed strings packed into the reply tail
|
||||
/// — the volume's id (its mount path is /volumes/<id> unless overridden), its
|
||||
/// actual mount path, and its display label. `id` is what software keys on;
|
||||
/// `label` is what a UI shows.
|
||||
pub const VolumeInfo = struct {
|
||||
id: []const u8,
|
||||
mount_path: []const u8,
|
||||
label: []const u8,
|
||||
|
||||
const header_bytes = 6; // three u16 lengths, little-endian
|
||||
|
||||
/// Pack into `buf`, returning the used slice, or null if it does not fit.
|
||||
pub fn encode(self: VolumeInfo, buf: []u8) ?[]u8 {
|
||||
const total = header_bytes + self.id.len + self.mount_path.len + self.label.len;
|
||||
if (total > buf.len) return null;
|
||||
std.mem.writeInt(u16, buf[0..2], @intCast(self.id.len), .little);
|
||||
std.mem.writeInt(u16, buf[2..4], @intCast(self.mount_path.len), .little);
|
||||
std.mem.writeInt(u16, buf[4..6], @intCast(self.label.len), .little);
|
||||
var off: usize = header_bytes;
|
||||
@memcpy(buf[off..][0..self.id.len], self.id);
|
||||
off += self.id.len;
|
||||
@memcpy(buf[off..][0..self.mount_path.len], self.mount_path);
|
||||
off += self.mount_path.len;
|
||||
@memcpy(buf[off..][0..self.label.len], self.label);
|
||||
return buf[0..total];
|
||||
}
|
||||
|
||||
/// Decode a reply tail, or null if it is malformed (short or inconsistent).
|
||||
/// The returned slices point into `bytes`.
|
||||
pub fn decode(bytes: []const u8) ?VolumeInfo {
|
||||
if (bytes.len < header_bytes) return null;
|
||||
const id_len = std.mem.readInt(u16, bytes[0..2], .little);
|
||||
const path_len = std.mem.readInt(u16, bytes[2..4], .little);
|
||||
const label_len = std.mem.readInt(u16, bytes[4..6], .little);
|
||||
const total = header_bytes + @as(usize, id_len) + path_len + label_len;
|
||||
if (total > bytes.len) return null;
|
||||
var off: usize = header_bytes;
|
||||
const id = bytes[off..][0..id_len];
|
||||
off += id_len;
|
||||
const mount_path = bytes[off..][0..path_len];
|
||||
off += path_len;
|
||||
const label = bytes[off..][0..label_len];
|
||||
return .{ .id = id, .mount_path = mount_path, .label = label };
|
||||
}
|
||||
};
|
||||
|
||||
pub const Protocol = envelope.Define(.{
|
||||
.name = "volume-manager",
|
||||
.version = 1,
|
||||
.operations = &.{
|
||||
.{ .name = "hello", .request = Hello },
|
||||
.{ .name = "volumes", .request = Volumes },
|
||||
},
|
||||
});
|
||||
|
||||
pub const Operation = Protocol.Operation;
|
||||
pub const message_maximum: usize = Protocol.message_maximum;
|
||||
|
||||
// Named fixture sizes so the bounds gate (which flags literal array lengths)
|
||||
// stays quiet: test inputs, not runtime ceilings.
|
||||
const test_reply_bytes = 128;
|
||||
const test_tiny_bytes = 4;
|
||||
|
||||
test "VolumeInfo round-trips id, mount_path, and label" {
|
||||
var buf: [test_reply_bytes]u8 = undefined;
|
||||
const info = VolumeInfo{ .id = "fat-12345678", .mount_path = "/volumes/fat-12345678", .label = "DANOS" };
|
||||
const encoded = info.encode(&buf).?;
|
||||
const back = VolumeInfo.decode(encoded).?;
|
||||
try std.testing.expectEqualStrings("fat-12345678", back.id);
|
||||
try std.testing.expectEqualStrings("/volumes/fat-12345678", back.mount_path);
|
||||
try std.testing.expectEqualStrings("DANOS", back.label);
|
||||
}
|
||||
|
||||
test "VolumeInfo encode refuses a buffer that is too small; decode rejects a short tail" {
|
||||
var tiny: [test_tiny_bytes]u8 = undefined;
|
||||
const info = VolumeInfo{ .id = "fat-1", .mount_path = "/volumes/fat-1", .label = "" };
|
||||
try std.testing.expect(info.encode(&tiny) == null);
|
||||
try std.testing.expect(VolumeInfo.decode(&[_]u8{ 0, 0, 0 }) == null); // shorter than the header
|
||||
try std.testing.expect(VolumeInfo.decode(&[_]u8{ 0xFF, 0xFF, 0, 0, 0, 0 }) == null); // claims 65535 id bytes
|
||||
}
|
||||
|
||||
@@ -0,0 +1,8 @@
|
||||
# The filesystem map: a probed volume's content signature -> the service binary
|
||||
# that serves it (docs/file-system-development/storage-architecture.md). The
|
||||
# volume manager reads this (the policy); a signature no row matches goes
|
||||
# unserved, never guessed. Adding a filesystem adds a row.
|
||||
#
|
||||
# signature, binary
|
||||
fat, /system/services/fat
|
||||
exfat, /system/services/exfat
|
||||
|
@@ -79,6 +79,7 @@
|
||||
# 'kernel' as the supervisor. Nothing else changes: the binary must still match.
|
||||
/system/services/input, kernel, bind, input
|
||||
/system/services/device-manager, kernel, bind, device-manager
|
||||
/system/services/volume-manager, kernel, bind, volume-manager
|
||||
/system/services/fat, kernel, bind, vfs
|
||||
/system/services/display, kernel, bind, display
|
||||
/system/services/discovery, kernel, bind, power
|
||||
@@ -102,9 +103,14 @@
|
||||
# own endpoint (the mouse-listener thread opens /protocol/display like any other
|
||||
# client — threads share no handles), and the input stream that moves the cursor.
|
||||
/system/services/fat, /system/services/volume-manager, open, volume-manager
|
||||
# exfat reaches the volume manager the same way — the second engine, same lineage.
|
||||
/system/services/exfat, /system/services/volume-manager, open, volume-manager
|
||||
# The volume manager reaches the device manager to be routed to each storage
|
||||
# provider's block channel, then confines a filesystem to each volume.
|
||||
/system/services/volume-manager, /system/services/init, open, device-manager
|
||||
# ...and again under the kernel supervisor for the manual-tree drills (S5's
|
||||
# volume-driver-restart spawns the volume manager directly, not via init).
|
||||
/system/services/volume-manager, kernel, open, device-manager
|
||||
/system/services/display, /system/services/init, open, scanout
|
||||
/system/services/display, /system/services/init, open, display
|
||||
/system/services/display, /system/services/init, open, input
|
||||
|
||||
|
Can't render this file because it contains an unexpected character in line 12 and column 15.
|
@@ -0,0 +1,8 @@
|
||||
# The mount map (danos's fstab): a volume's content id -> a chosen mount prefix.
|
||||
# This is an OPTIONAL override, read by the volume manager. A volume with no row
|
||||
# mounts at its default /volumes/<id>, where <id> is the manager's rendered
|
||||
# content identity (e.g. fat-12345678, gpt-<guid>, mbr-<sig>-<index>) — stable,
|
||||
# unique, and never a port or a label. The label is display metadata, not here:
|
||||
# query it via the volume manager's `volumes` verb.
|
||||
#
|
||||
# id, mount_prefix
|
||||
|
@@ -227,6 +227,10 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void {
|
||||
usbStorageTest(boot_information);
|
||||
} else if (eql(case, "fat-mount")) {
|
||||
fatMountTest(boot_information);
|
||||
} else if (eql(case, "exfat-volume")) {
|
||||
exfatVolumeTest(boot_information);
|
||||
} else if (eql(case, "volume-driver-restart")) {
|
||||
volumeDriverRestartTest(boot_information);
|
||||
} else if (eql(case, "device-list")) {
|
||||
deviceListTest(boot_information);
|
||||
} else if (eql(case, "pci-scan")) {
|
||||
@@ -3036,6 +3040,32 @@ fn fatMountTest(boot_information: *const BootInformation) void {
|
||||
result();
|
||||
}
|
||||
|
||||
/// The exFAT mount chain (S4): boot the full tree, which brings up the USB storage
|
||||
/// chain. The harness attaches a SECOND device — a data-only exFAT volume — beside
|
||||
/// the FAT boot volume, so the volume manager spawns the exfat service for it
|
||||
/// (content-routed, its id-path /volumes/exfat-<serial>). Then spawn exfat-test,
|
||||
/// which reads the seeded file and mutates through the mount. The reuse of the
|
||||
/// shared harness by a second engine is proven end to end here.
|
||||
fn exfatVolumeTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: exfat-volume\n", .{});
|
||||
if (boot_information.initial_ramdisk_len == 0) {
|
||||
check("bootloader handed over the initial_ramdisk", false);
|
||||
result();
|
||||
return;
|
||||
}
|
||||
const ramdisk = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||
const rd = initial_ramdisk.Reader.init(ramdisk) orelse {
|
||||
check("initial_ramdisk image is valid", false);
|
||||
result();
|
||||
return;
|
||||
};
|
||||
process.setInitialRamdisk(ramdisk);
|
||||
const init_ok = if (process.spawnBundled("/system/services/init")) true else |_| false;
|
||||
check("init spawned (boots the tree, incl. the volume manager)", init_ok);
|
||||
check("exfat-test client spawned", spawnNamed(rd, "exfat-test"));
|
||||
result();
|
||||
}
|
||||
|
||||
/// Per-sender range confinement (V2a, docs/volume-manager-plan.md): the fixture
|
||||
/// acquires the block channel, confines ITSELF to a sub-range, and asserts it
|
||||
/// cannot read past that range or widen it. Boots init in REGISTRY-ONLY mode
|
||||
@@ -3657,6 +3687,44 @@ fn displayReattachTest(boot_information: *const BootInformation) void {
|
||||
while (true) scheduler.yield();
|
||||
}
|
||||
|
||||
/// Storage-driver-crash rebuild (S5): the device manager runs in
|
||||
/// "test-storage-restart" mode and kills the usb-storage driver once, a moment
|
||||
/// after its volume has mounted. The driver's device stays in the tree, so the
|
||||
/// volume manager's presence poll alone would miss the death and leave fat wedged
|
||||
/// on a dead channel; its channel-liveness probe must notice, reap the volume, and
|
||||
/// rebuild on the restarted driver's fresh channel — a SECOND mount of the same
|
||||
/// id-path is the proof. (A pre-S5 manager, checking only device presence, never
|
||||
/// reaps, so the second mount never appears.)
|
||||
fn volumeDriverRestartTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: volume-driver-restart\n", .{});
|
||||
if (boot_information.initial_ramdisk_len == 0) {
|
||||
check("bootloader handed over an initial_ramdisk", false);
|
||||
result();
|
||||
return;
|
||||
}
|
||||
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||
check("initial_ramdisk image is valid", false);
|
||||
result();
|
||||
return;
|
||||
};
|
||||
process.setInitialRamdisk(image);
|
||||
_ = spawnRegistry(rd);
|
||||
var manager: u32 = 0;
|
||||
var i: u32 = 0;
|
||||
while (i < rd.count) : (i += 1) {
|
||||
const item = rd.entry(i) orelse continue;
|
||||
if (!eql(initial_ramdisk.basename(item.name), "device-manager")) continue;
|
||||
manager = process.spawnProcessSupervised(item.blob, 4, &.{ item.name, "test-storage-restart" }, scheduler.currentId(), null) catch 0;
|
||||
break;
|
||||
}
|
||||
check("device-manager spawned (test-storage-restart mode)", manager != 0);
|
||||
check("volume-manager spawned", spawnNamed(rd, "volume-manager"));
|
||||
check("fat-test client spawned", spawnNamed(rd, "fat-test"));
|
||||
scheduler.setPriority(1); // below the tree, so it runs
|
||||
while (true) scheduler.yield();
|
||||
}
|
||||
|
||||
/// Process arguments, end to end: spawn args-echo bare (its argv[0] is the
|
||||
/// initial-ramdisk name). Instance 1 sees argc == 1 and respawns itself through
|
||||
/// `system_spawn` with the extra arguments "alpha beta-42" — the syscall argument
|
||||
|
||||
+25
-6
@@ -55,7 +55,17 @@ fn tokenIndex(t: u64) u64 {
|
||||
|
||||
// --- the mount table ---------------------------------------------------------
|
||||
|
||||
pub const maximum_mounts = 8;
|
||||
/// bound: prefixes mounted in the kernel VFS table at once
|
||||
/// decided-by: ours
|
||||
/// protects: the `mounts` table below
|
||||
/// at-limit: refuse - installMount returns false and mountBackend propagates it;
|
||||
/// the mounting filesystem's harness logs "could not mount <prefix>" and the
|
||||
/// mount simply does not exist (no silent success). Budget: the initrd's
|
||||
/// top-level dirs (/system, /test) plus one id-path mount per volume and the
|
||||
/// system volume's two FHS rewrites — a few over the volume manager's
|
||||
/// maximum_volumes (16); 32 leaves headroom.
|
||||
/// observed-by: the harness "file-system: could not mount <prefix>" ring line
|
||||
pub const maximum_mounts = 32;
|
||||
const maximum_prefix = 64;
|
||||
const maximum_rewrite = 32;
|
||||
|
||||
@@ -156,7 +166,10 @@ pub fn setInitialRamdisk(image: []const u8) void {
|
||||
for (directories[0..directory_count], 0..) |*d, index| {
|
||||
const parent = parentOf(d.slice());
|
||||
d.parent = directoryIndex(parent) orelse index;
|
||||
if (parent.len == 1) installMount(d.slice(), .kernel_initrd, null, "");
|
||||
// Boot-time install of one mount per top-level initrd dir (/system, /test):
|
||||
// provably few, far under maximum_mounts, so a full table here is
|
||||
// impossible — but discard the result explicitly rather than assume it.
|
||||
if (parent.len == 1) _ = installMount(d.slice(), .kernel_initrd, null, "");
|
||||
}
|
||||
}
|
||||
|
||||
@@ -167,7 +180,12 @@ fn directoryIndex(path: []const u8) ?usize {
|
||||
return null;
|
||||
}
|
||||
|
||||
fn installMount(prefix: []const u8, kind: MountKind, backend: ?*ipc.Endpoint, rewrite: []const u8) void {
|
||||
/// Install (or remount-replace) a prefix. Returns false when the table is full
|
||||
/// and no slot could be claimed — the caller must surface that, never report a
|
||||
/// dropped mount as success. A remount of an already-mounted prefix reuses its
|
||||
/// slot and always succeeds; a /protocol remount is refused-as-noop (returns
|
||||
/// true: the first mount stands, nothing is dropped).
|
||||
fn installMount(prefix: []const u8, kind: MountKind, backend: ?*ipc.Endpoint, rewrite: []const u8) bool {
|
||||
// Remount replaces: a restarted backend re-mounts its prefix.
|
||||
var slot: ?*Mount = null;
|
||||
for (&mounts) |*m| {
|
||||
@@ -176,19 +194,20 @@ fn installMount(prefix: []const u8, kind: MountKind, backend: ?*ipc.Endpoint, re
|
||||
// restarted FAT retakes /volumes/usb; letting it retake /protocol
|
||||
// would hand the whole naming layer to whoever asked second.
|
||||
// First mount wins, and init (PID 1) is always first.
|
||||
if (std.mem.eql(u8, prefix, protocol_root)) return;
|
||||
if (std.mem.eql(u8, prefix, protocol_root)) return true;
|
||||
if (m.backend) |old| ipc.dropRef(old);
|
||||
slot = m;
|
||||
break;
|
||||
}
|
||||
if (slot == null and !m.used) slot = m;
|
||||
}
|
||||
const m = slot orelse return;
|
||||
const m = slot orelse return false;
|
||||
m.* = .{ .used = true, .kind = kind, .backend = backend };
|
||||
@memcpy(m.prefix[0..prefix.len], prefix);
|
||||
m.prefix_len = prefix.len;
|
||||
@memcpy(m.rewrite[0..rewrite.len], rewrite);
|
||||
m.rewrite_len = rewrite.len;
|
||||
return true;
|
||||
}
|
||||
|
||||
// --- resolve -----------------------------------------------------------------
|
||||
@@ -406,7 +425,7 @@ pub fn mountBackend(prefix: []const u8, backend: *ipc.Endpoint, rewrite: []const
|
||||
if (!isInitrdCarveOut(prefix)) return false;
|
||||
}
|
||||
}
|
||||
installMount(prefix, .backend, backend, rewrite);
|
||||
if (!installMount(prefix, .backend, backend, rewrite)) return false; // table full
|
||||
for (&mounts) |*m| {
|
||||
if (m.used and std.mem.eql(u8, m.prefixSlice(), prefix)) m.owner = owner;
|
||||
}
|
||||
|
||||
@@ -170,6 +170,8 @@ var test_usb_killed = false;
|
||||
var test_pci_restart_mode = false;
|
||||
var test_scanout_restart_mode = false;
|
||||
var test_scanout_killed = false;
|
||||
var test_storage_restart_mode = false;
|
||||
var test_storage_killed = false;
|
||||
var test_kill_pid: u32 = 0;
|
||||
var test_kill_due_ns: u64 = 0;
|
||||
|
||||
@@ -600,6 +602,16 @@ fn onHello(_: void, invocation: Invocation(device_manager_protocol.Hello), _: An
|
||||
test_kill_due_ns = time.clock() + 1_500_000_000;
|
||||
_ = time.timerOnce(manager_endpoint, 1600);
|
||||
}
|
||||
// Storage-driver-crash drill (S5): once, a moment after usb-storage hellos —
|
||||
// long enough that its volume has mounted — kill it. The manager re-delegates
|
||||
// the still-present device to a restarted driver on a fresh channel; the volume
|
||||
// manager's channel-liveness probe must notice the dead channel and rebuild.
|
||||
if (test_storage_restart_mode and !test_storage_killed and std.mem.eql(u8, driver.name(), "/system/drivers/usb-storage")) {
|
||||
test_storage_killed = true;
|
||||
test_kill_pid = invocation.sender;
|
||||
test_kill_due_ns = time.clock() + 2_000_000_000; // after the ~0.6s mount
|
||||
_ = time.timerOnce(manager_endpoint, 2100);
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
@@ -763,6 +775,7 @@ pub fn main(init: process.Init) void {
|
||||
test_usb_restart_mode = std.mem.eql(u8, mode, "test-usb-restart");
|
||||
test_pci_restart_mode = std.mem.eql(u8, mode, "test-pci-restart");
|
||||
test_scanout_restart_mode = std.mem.eql(u8, mode, "test-scanout-restart");
|
||||
test_storage_restart_mode = std.mem.eql(u8, mode, "test-storage-restart");
|
||||
}
|
||||
service.run(device_manager_protocol.message_maximum, .{
|
||||
.service = "device-manager",
|
||||
|
||||
@@ -0,0 +1,34 @@
|
||||
//! The exfat service as a binary package (docs/build-packages-plan.md):
|
||||
//! this file names the binary and EXACTLY the modules its source imports —
|
||||
//! build-support resolves each name from the domains this zon declares.
|
||||
|
||||
const std = @import("std");
|
||||
const build_support = @import("build-support");
|
||||
|
||||
pub fn build(b: *std.Build) void {
|
||||
const exe = build_support.userBinary(b, .{
|
||||
.name = "exfat",
|
||||
.root_source_file = b.path("exfat.zig"),
|
||||
.imports = &.{
|
||||
"block", "channel", "envelope", "file-system-harness",
|
||||
"ipc", "logging", "memory", "process",
|
||||
"time", "volume-manager-protocol",
|
||||
},
|
||||
});
|
||||
b.installArtifact(exe);
|
||||
|
||||
// Standalone `zig build test`; the root aggregate depends on this step.
|
||||
const test_step = b.step("test", "Run the exfat unit tests");
|
||||
for ([_][]const u8{
|
||||
"on-disk.zig", // exFAT on-disk struct sizes + geometry + checksums
|
||||
"engine.zig", // exFAT read/write over a RAM-backed image
|
||||
}) |test_root| {
|
||||
const unit_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path(test_root),
|
||||
.target = b.resolveTargetQuery(.{}),
|
||||
}),
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(unit_tests).step);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,16 @@
|
||||
.{
|
||||
.name = .exfat,
|
||||
.version = "0.0.0",
|
||||
.fingerprint = 0x5eafdf02d20f93dd, // Changing this has security and trust implications.
|
||||
.minimum_zig_version = "0.16.0",
|
||||
.dependencies = .{
|
||||
// build-support supplies the shared recipe; kernel is implicit in
|
||||
// every binary (the root shim + link script live there). The rest
|
||||
// are exactly the homes of this binary's declared imports.
|
||||
.@"build-support" = .{ .path = "../../../build-support" },
|
||||
.kernel = .{ .path = "../../../library/kernel" },
|
||||
.device = .{ .path = "../../../library/device" },
|
||||
.protocol = .{ .path = "../../../library/protocol" },
|
||||
},
|
||||
.paths = .{""},
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,200 @@
|
||||
//! system/services/exfat — the exFAT filesystem service. Like fat.zig, this is
|
||||
//! only the format-specific half: it finds its block device, sets up the DMA
|
||||
//! bounce buffer, mounts the exFAT engine on it, and hands the mounted volume to
|
||||
//! the shared filesystem harness (library/kernel/file-system-harness), which owns
|
||||
//! everything else — vfs serving, the open-node table, mount registration, the
|
||||
//! exit sweep, durable-on-close. The engine (engine.zig) is the pure,
|
||||
//! host-testable format code; on-disk.zig its byte layout.
|
||||
//!
|
||||
//! This service is a near-clone of fat.zig: the second engine reuses the harness
|
||||
//! wholesale, which is the reuse the storage architecture promised
|
||||
//! (docs/file-system-development/storage-architecture.md). The block data path
|
||||
//! never crosses IPC: a DMA bounce buffer is handed to the block driver by
|
||||
//! physical address, and the engine copies sectors in and out.
|
||||
|
||||
const std = @import("std");
|
||||
const channel = @import("channel");
|
||||
const volume_manager_protocol = @import("volume-manager-protocol");
|
||||
const ipc = @import("ipc");
|
||||
const process = @import("process");
|
||||
const block = @import("block");
|
||||
const memory = @import("memory");
|
||||
const logging = @import("logging");
|
||||
const time = @import("time");
|
||||
const engine = @import("engine.zig");
|
||||
const envelope = @import("envelope");
|
||||
const harness = @import("file-system-harness");
|
||||
|
||||
/// The serving harness, specialized for the exFAT engine. One volume per process.
|
||||
const Harness = harness.Server(engine.FileSystem);
|
||||
|
||||
// The engine's BlockDevice, backed by the `.block` driver plus a DMA bounce
|
||||
// buffer the driver reads/writes by physical address.
|
||||
const IpcBlock = struct {
|
||||
device: block.Device,
|
||||
bounce: memory.DmaRegion, // engine.max_transfer_sectors * 512 bytes
|
||||
|
||||
fn readBlocks(context: *anyopaque, lba: u64, count: u32, buffer: []u8) bool {
|
||||
const self: *IpcBlock = @ptrCast(@alignCast(context));
|
||||
if (count == 0 or count > engine.max_transfer_sectors) return false;
|
||||
const len = count * 512;
|
||||
if (!self.device.read(lba, count, self.bounce.physical)) return false;
|
||||
const source: [*]const u8 = @ptrFromInt(self.bounce.virtual);
|
||||
@memcpy(buffer[0..len], source[0..len]);
|
||||
return true;
|
||||
}
|
||||
fn writeBlocks(context: *anyopaque, lba: u64, count: u32, buffer: []const u8) bool {
|
||||
const self: *IpcBlock = @ptrCast(@alignCast(context));
|
||||
if (count == 0 or count > engine.max_transfer_sectors) return false;
|
||||
const len = count * 512;
|
||||
const destination: [*]u8 = @ptrFromInt(self.bounce.virtual);
|
||||
@memcpy(destination[0..len], buffer[0..len]);
|
||||
if (!self.device.write(lba, count, self.bounce.physical)) return false;
|
||||
device_dirty = true; // a block reached the device; a close will flush it
|
||||
return true;
|
||||
}
|
||||
};
|
||||
|
||||
var ipc_block: IpcBlock = undefined;
|
||||
// Set whenever a block is written, cleared when the device cache is flushed on a
|
||||
// file close — so writes are committed to stable media before a power-off.
|
||||
var device_dirty: bool = false;
|
||||
var filesystem: engine.FileSystem = undefined;
|
||||
/// The volume this exFAT process serves, its id given as argv[1] by the volume
|
||||
/// manager that spawned it. The startup hello names it so the manager returns the
|
||||
/// right volume's channel.
|
||||
var my_volume_id: u64 = 0;
|
||||
|
||||
/// The volume's own mount path, handed in as argv[2] by the volume manager: the
|
||||
/// volume's content id-path (e.g. /volumes/exfat-12345678). Defaults to
|
||||
/// /volumes/exfat only for a bare launch with no argument; the manager always
|
||||
/// passes it. The slice points into the entry block, valid for the process life.
|
||||
var volume_mount_prefix: []const u8 = "/volumes/exfat";
|
||||
|
||||
/// The mounts this volume installs: its own root, plus — only if it is the boot
|
||||
/// volume (it resolves /system/configuration) — the two FHS rewrites, so the
|
||||
/// logger's /system/logs stays decoupled from which volume backs it. Boot-volume
|
||||
/// detection is by content, so it works no matter which volume carries /system.
|
||||
/// bound: mounts one volume installs (its root + the two boot rewrites)
|
||||
/// decided-by: ours
|
||||
/// protects: the mount_specs array
|
||||
/// at-limit: truncate - unreachable today (fixed at 3); more configured mounts
|
||||
/// would need this raised, a deliberate change
|
||||
/// observed-by: a mount silently missing from the harness's mount log
|
||||
const maximum_mounts_per_volume = 4;
|
||||
var mount_specs: [maximum_mounts_per_volume]harness.MountSpec = undefined;
|
||||
|
||||
/// Get this volume's block channel from the volume manager (establishment by
|
||||
/// lineage — `block` is not a registry name). The manager spawned this process,
|
||||
/// confined it to its partition, and answers the hello with the channel; the
|
||||
/// channel is range-confined to this process's badge. Null until the manager has
|
||||
/// the volume ready — this retries.
|
||||
fn acquireVolume() ?block.Device {
|
||||
var attempts: u32 = 0;
|
||||
const vm = while (attempts < 500) : (attempts += 1) {
|
||||
if (channel.openEndpoint("volume-manager")) |handle| break handle;
|
||||
time.sleepMillis(20);
|
||||
} else return null;
|
||||
|
||||
attempts = 0;
|
||||
while (attempts < 500) : (attempts += 1) {
|
||||
var packet: [volume_manager_protocol.message_maximum]u8 = undefined;
|
||||
const framed = volume_manager_protocol.Protocol.encodeRequest(.hello, my_volume_id, .{}, &.{}, &packet) orelse return null;
|
||||
var reply: [volume_manager_protocol.message_maximum]u8 = undefined;
|
||||
const answered = ipc.callCap(vm, framed, &reply, null) catch return null;
|
||||
const status = envelope.statusOf(reply[0..answered.len]) orelse return null;
|
||||
if (status.status != 0) {
|
||||
if (answered.cap) |stray| _ = ipc.close(stray);
|
||||
_ = logging.write("/system/services/exfat: volume manager refused the hello\n");
|
||||
return null;
|
||||
}
|
||||
if (answered.cap) |bus| return .{ .endpoint = bus };
|
||||
// Acked with no channel: the volume is not ready yet — retry.
|
||||
time.sleepMillis(20);
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Durable-on-close: commit the device write cache if any block reached it since
|
||||
/// the last flush. The harness calls this on every close; the dirty check keeps
|
||||
/// it cheap.
|
||||
fn flushIfDirty() void {
|
||||
if (device_dirty) {
|
||||
_ = ipc_block.device.flush();
|
||||
device_dirty = false;
|
||||
}
|
||||
}
|
||||
|
||||
/// exFAT bring-up: find the block device, set up DMA, mount the engine, and hand
|
||||
/// the volume to the harness — or null to retry on the harness's timer.
|
||||
fn exfatBringUp(endpoint: ipc.Handle) ?Harness.Volume {
|
||||
_ = endpoint;
|
||||
const device = acquireVolume() orelse return null;
|
||||
const geometry = device.geometry() orelse {
|
||||
_ = logging.write("/system/services/exfat: block geometry unavailable\n");
|
||||
return null;
|
||||
};
|
||||
// Shareable so the buffer's capability can be attached down the chain, making
|
||||
// its physical addresses reachable under an enforcing IOMMU. No-op otherwise.
|
||||
const bounce = memory.dmaAlloc(engine.max_transfer_sectors * 512, memory.dma_coherent | memory.dma_shareable) orelse return null;
|
||||
if (bounce.handle) |handle| {
|
||||
// Attach, detach, and attach again: the round trip exercises BOTH verbs of
|
||||
// the DMA-window lifecycle through the whole chain on every boot.
|
||||
if (!device.attach(handle)) {
|
||||
_ = logging.write("/system/services/exfat: could not attach the DMA bounce buffer\n");
|
||||
return null;
|
||||
}
|
||||
if (!device.detach(handle)) {
|
||||
_ = logging.write("/system/services/exfat: could not detach the DMA bounce buffer\n");
|
||||
return null;
|
||||
}
|
||||
if (!device.attach(handle)) {
|
||||
_ = logging.write("/system/services/exfat: could not re-attach the DMA bounce buffer\n");
|
||||
return null;
|
||||
}
|
||||
_ = ipc.close(handle); // the binding holds its own reference now
|
||||
}
|
||||
ipc_block = .{ .device = device, .bounce = bounce };
|
||||
|
||||
const block_device = engine.BlockDevice{
|
||||
.context = &ipc_block,
|
||||
.block_size = geometry.block_size,
|
||||
.block_count = geometry.block_count,
|
||||
.readBlocksFn = IpcBlock.readBlocks,
|
||||
.writeBlocksFn = IpcBlock.writeBlocks,
|
||||
};
|
||||
filesystem = engine.FileSystem.mount(block_device) orelse {
|
||||
_ = logging.write("/system/services/exfat: not an exFAT filesystem\n");
|
||||
return null;
|
||||
};
|
||||
std.log.info("mounted exFAT ({d} clusters, {d} sectors/cluster, serial 0x{x})", .{ filesystem.geometry.cluster_count, filesystem.geometry.sectors_per_cluster, filesystem.geometry.volume_serial_number });
|
||||
|
||||
// The volume mounts at its id-path (argv[2]). The boot/system volume — the one
|
||||
// carrying the /system tree — additionally installs the two FHS rewrites, by
|
||||
// CONTENT: it resolves /system/configuration on its own media. A data volume
|
||||
// mounts only at its id-path and never shadows the running system.
|
||||
mount_specs[0] = .{ .prefix = volume_mount_prefix };
|
||||
var mount_count: usize = 1;
|
||||
if (filesystem.resolve("/system/configuration") != null) {
|
||||
std.log.info("volume {d} carries the system tree; backing /system/configuration and /system/logs", .{my_volume_id});
|
||||
mount_specs[1] = .{ .prefix = "/system/configuration", .rewrite = "/system/configuration" };
|
||||
mount_specs[2] = .{ .prefix = "/system/logs", .rewrite = "/system/logs" };
|
||||
mount_count = 3;
|
||||
} else {
|
||||
std.log.info("volume {d} is a data volume; mounted at {s}", .{ my_volume_id, volume_mount_prefix });
|
||||
}
|
||||
return .{ .engine = &filesystem, .mounts = mount_specs[0..mount_count], .flush = flushIfDirty };
|
||||
}
|
||||
|
||||
pub fn main(init: process.Init) void {
|
||||
// The volume manager spawns this process with its volume id as argv[1] and the
|
||||
// volume's mount path (its id-path) as argv[2].
|
||||
if (init.arguments.get(1)) |id| {
|
||||
my_volume_id = std.fmt.parseInt(u64, id, 10) catch 0;
|
||||
}
|
||||
if (init.arguments.get(2)) |prefix| {
|
||||
volume_mount_prefix = prefix;
|
||||
}
|
||||
_ = logging.write("/system/services/exfat: starting, waiting for a block device\n");
|
||||
Harness.run(.{ .bringUp = exfatBringUp });
|
||||
}
|
||||
@@ -0,0 +1,475 @@
|
||||
//! The on-disk layout of an exFAT filesystem — the Main Boot Sector (VBR) and the
|
||||
//! six 32-byte directory-entry types — as `align(1)` extern structs that bit-cast
|
||||
//! straight out of a sector (multi-byte fields little-endian). Pure data, plus the
|
||||
//! three exFAT checksums (boot region, up-case table, directory-entry set), the
|
||||
//! name hash, and the packed timestamp <-> Unix-epoch conversion. Host-testable.
|
||||
//!
|
||||
//! exFAT departs from FAT in three ways this file encodes: geometry lives in a
|
||||
//! MustBeZero-guarded VBR (byte 11 is zero, which is exactly why the FAT prober
|
||||
//! rejects an exFAT volume — it reads a zero bytes-per-sector); a file is a SET of
|
||||
//! entries (a File entry, a Stream Extension, and one or more File Name entries)
|
||||
//! validated by a rotate-right checksum; and names are compared case-folded through
|
||||
//! the volume's own on-disk up-case table (the folding itself lives in the engine,
|
||||
//! which holds the loaded table; the hash it feeds is here).
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
// exFAT's fixed on-disk widths — byte and UTF-16-unit counts the FORMAT defines,
|
||||
// not ceilings danos chooses. Named so the wire-format structs carry no bare
|
||||
// literal lengths; the values are facts of the spec.
|
||||
pub const entry_bytes: usize = 32; // every directory entry
|
||||
const jump_boot_bytes = 3;
|
||||
const filesystem_name_bytes = 8; // "EXFAT "
|
||||
const must_be_zero_bytes = 53; // the FAT-BPB overlap the format holds zero
|
||||
const boot_code_bytes = 390;
|
||||
const volume_label_units = 11;
|
||||
|
||||
// --- the Main Boot Sector (VBR, sector 0) ------------------------------------
|
||||
|
||||
/// The exFAT Main Boot Sector. `must_be_zero` (offset 11..64) overlaps where a
|
||||
/// FAT BPB keeps bytes-per-sector/sectors-per-cluster/etc.; exFAT holds it zero,
|
||||
/// so a FAT prober reading a zero bytes-per-sector rejects the volume — the
|
||||
/// mutual-exclusion the two engines rely on.
|
||||
pub const MainBootSector = extern struct {
|
||||
jump_boot: [jump_boot_bytes]u8, // 0
|
||||
filesystem_name: [filesystem_name_bytes]u8, // 3 "EXFAT "
|
||||
must_be_zero: [must_be_zero_bytes]u8, // 11
|
||||
partition_offset: u64 align(1), // 64 sectors, informational
|
||||
volume_length: u64 align(1), // 72 sectors
|
||||
fat_offset: u32 align(1), // 80 sectors from volume start
|
||||
fat_length: u32 align(1), // 84 sectors, per FAT
|
||||
cluster_heap_offset: u32 align(1), // 88 sectors from volume start
|
||||
cluster_count: u32 align(1), // 92
|
||||
first_cluster_of_root: u32 align(1), // 96
|
||||
volume_serial_number: u32 align(1), // 100
|
||||
filesystem_revision: u16 align(1), // 104
|
||||
volume_flags: u16 align(1), // 106 (skipped by the boot checksum)
|
||||
bytes_per_sector_shift: u8, // 108 9..12
|
||||
sectors_per_cluster_shift: u8, // 109
|
||||
number_of_fats: u8, // 110 1 (2 for TexFAT)
|
||||
drive_select: u8, // 111
|
||||
percent_in_use: u8, // 112 (skipped by the boot checksum)
|
||||
reserved: [7]u8, // 113
|
||||
boot_code: [boot_code_bytes]u8, // 120
|
||||
boot_signature: u16 align(1), // 510 0xAA55
|
||||
};
|
||||
|
||||
// --- directory entries (32 bytes each) ---------------------------------------
|
||||
|
||||
/// Entry-type bytes. The high bit (0x80) is InUse: a type with it clear is not in
|
||||
/// use, and 0x00 ends the directory. Deleting an entry clears bit 7 (0x85 -> 0x05).
|
||||
pub const entry_type_allocation_bitmap: u8 = 0x81;
|
||||
pub const entry_type_upcase_table: u8 = 0x82;
|
||||
pub const entry_type_volume_label: u8 = 0x83;
|
||||
pub const entry_type_file: u8 = 0x85;
|
||||
pub const entry_type_stream_extension: u8 = 0xC0;
|
||||
pub const entry_type_file_name: u8 = 0xC1;
|
||||
pub const entry_type_in_use_bit: u8 = 0x80;
|
||||
pub const entry_type_end_of_directory: u8 = 0x00;
|
||||
|
||||
/// A raw 32-byte entry, for type dispatch before it is reinterpreted as a
|
||||
/// specific entry.
|
||||
pub const RawEntry = extern struct {
|
||||
entry_type: u8,
|
||||
data: [entry_bytes - 1]u8,
|
||||
|
||||
pub fn inUse(self: RawEntry) bool {
|
||||
return self.entry_type & entry_type_in_use_bit != 0;
|
||||
}
|
||||
pub fn isEnd(self: RawEntry) bool {
|
||||
return self.entry_type == entry_type_end_of_directory;
|
||||
}
|
||||
};
|
||||
|
||||
/// 0x81 — the Allocation Bitmap: one bit per cluster (cluster 2 = bit 0), the
|
||||
/// authority for which clusters are free. The deepest departure from FAT, where
|
||||
/// the chain itself was the authority.
|
||||
pub const AllocationBitmapEntry = extern struct {
|
||||
entry_type: u8, // 0 0x81
|
||||
bitmap_flags: u8, // 1
|
||||
reserved: [18]u8, // 2
|
||||
first_cluster: u32 align(1), // 20
|
||||
data_length: u64 align(1), // 24
|
||||
};
|
||||
|
||||
/// 0x82 — the Up-case Table: the on-disk case-fold map (code unit -> uppercase),
|
||||
/// referenced by cluster and validated by `table_checksum`.
|
||||
pub const UpcaseTableEntry = extern struct {
|
||||
entry_type: u8, // 0 0x82
|
||||
reserved1: [3]u8, // 1
|
||||
table_checksum: u32 align(1), // 4
|
||||
reserved2: [12]u8, // 8
|
||||
first_cluster: u32 align(1), // 20
|
||||
data_length: u64 align(1), // 24
|
||||
};
|
||||
|
||||
/// 0x83 — the Volume Label (up to 11 UTF-16 units).
|
||||
pub const VolumeLabelEntry = extern struct {
|
||||
entry_type: u8, // 0 0x83
|
||||
character_count: u8, // 1
|
||||
volume_label: [volume_label_units]u16 align(1), // 2
|
||||
reserved: [8]u8, // 24
|
||||
};
|
||||
|
||||
/// 0x85 — the File entry: the head of a set, carrying the attributes,
|
||||
/// timestamps, the secondary-entry count, and the set checksum.
|
||||
pub const FileEntry = extern struct {
|
||||
entry_type: u8, // 0 0x85
|
||||
secondary_count: u8, // 1 stream (1) + name entries
|
||||
set_checksum: u16 align(1), // 2 over the whole set, skipping these two bytes
|
||||
file_attributes: u16 align(1), // 4
|
||||
reserved1: u16 align(1), // 6
|
||||
create_timestamp: u32 align(1), // 8
|
||||
last_modified_timestamp: u32 align(1), // 12
|
||||
last_accessed_timestamp: u32 align(1), // 16
|
||||
create_10ms: u8, // 20
|
||||
last_modified_10ms: u8, // 21
|
||||
create_utc_offset: u8, // 22
|
||||
last_modified_utc_offset: u8, // 23
|
||||
last_accessed_utc_offset: u8, // 24
|
||||
reserved2: [7]u8, // 25
|
||||
};
|
||||
|
||||
/// 0xC0 — the Stream Extension: the second entry of every file set, carrying the
|
||||
/// name length + hash and the data location (first cluster, sizes, the
|
||||
/// no-FAT-chain flag).
|
||||
pub const StreamExtensionEntry = extern struct {
|
||||
entry_type: u8, // 0 0xC0
|
||||
general_secondary_flags: u8, // 1
|
||||
reserved1: u8, // 2
|
||||
name_length: u8, // 3 UTF-16 units
|
||||
name_hash: u16 align(1), // 4
|
||||
reserved2: u16 align(1), // 6
|
||||
valid_data_length: u64 align(1), // 8
|
||||
reserved3: u32 align(1), // 16
|
||||
first_cluster: u32 align(1), // 20
|
||||
data_length: u64 align(1), // 24
|
||||
};
|
||||
|
||||
/// 0xC1 — a File Name entry: 15 UTF-16 units of the name; a set carries
|
||||
/// ceil(name_length / 15) of them.
|
||||
pub const FileNameEntry = extern struct {
|
||||
entry_type: u8, // 0 0xC1
|
||||
general_secondary_flags: u8, // 1
|
||||
file_name: [name_units_per_entry]u16 align(1), // 2
|
||||
};
|
||||
|
||||
pub const name_units_per_entry: usize = 15;
|
||||
|
||||
// General secondary flags (Stream Extension + File Name entries).
|
||||
pub const secondary_flag_allocation_possible: u8 = 0x01;
|
||||
pub const secondary_flag_no_fat_chain: u8 = 0x02;
|
||||
|
||||
// File attributes (same bit assignments as FAT).
|
||||
pub const attribute_read_only: u16 = 0x0001;
|
||||
pub const attribute_hidden: u16 = 0x0002;
|
||||
pub const attribute_system: u16 = 0x0004;
|
||||
pub const attribute_directory: u16 = 0x0010;
|
||||
pub const attribute_archive: u16 = 0x0020;
|
||||
|
||||
// FAT special cluster values (exFAT's FAT is 32-bit; used only for a fragmented
|
||||
// chain, i.e. when no_fat_chain is clear).
|
||||
pub const first_data_cluster: u32 = 2;
|
||||
pub const end_of_chain: u32 = 0xFFFFFFFF;
|
||||
pub const bad_cluster: u32 = 0xFFFFFFF7;
|
||||
|
||||
pub const boot_signature_offset: usize = 510; // 0x55 0xAA
|
||||
|
||||
// --- geometry ----------------------------------------------------------------
|
||||
|
||||
pub const Geometry = struct {
|
||||
bytes_per_sector: u32,
|
||||
sectors_per_cluster: u32,
|
||||
cluster_count: u32,
|
||||
fat_offset_sectors: u32, // from volume start
|
||||
fat_length_sectors: u32,
|
||||
cluster_heap_offset_sectors: u32, // from volume start
|
||||
first_cluster_of_root: u32,
|
||||
volume_serial_number: u32,
|
||||
volume_length_sectors: u64,
|
||||
number_of_fats: u32,
|
||||
};
|
||||
|
||||
/// Derive the geometry from a Main Boot Sector. Returns null unless it is a
|
||||
/// plausible exFAT VBR: the "EXFAT " name, an all-zero MustBeZero region, the
|
||||
/// 0xAA55 signature, and sane shifts. Accepting ONLY these is what keeps exFAT and
|
||||
/// FAT from ever claiming each other's volumes.
|
||||
pub fn geometryOf(sector: []const u8) ?Geometry {
|
||||
if (sector.len < 512) return null;
|
||||
if (sector[boot_signature_offset] != 0x55 or sector[boot_signature_offset + 1] != 0xAA) return null;
|
||||
const vbr = std.mem.bytesToValue(MainBootSector, sector[0..@sizeOf(MainBootSector)]);
|
||||
if (!std.mem.eql(u8, &vbr.filesystem_name, "EXFAT ")) return null;
|
||||
for (vbr.must_be_zero) |byte| if (byte != 0) return null;
|
||||
if (vbr.bytes_per_sector_shift < 9 or vbr.bytes_per_sector_shift > 12) return null;
|
||||
// The exFAT spec caps a cluster at 2^25 bytes (32 MiB): bytes-per-sector-shift
|
||||
// plus sectors-per-cluster-shift must not exceed 25. Enforcing it here is also
|
||||
// what keeps the engine's u32 cluster-byte arithmetic (sectors_per_cluster *
|
||||
// 512) from overflowing on a crafted VBR off untrusted removable media.
|
||||
if (@as(u16, vbr.bytes_per_sector_shift) + vbr.sectors_per_cluster_shift > 25) return null;
|
||||
// cluster_count is capped at 0xFFFFFFF5 (the spec's ClusterCount maximum), so
|
||||
// cluster_count + first_data_cluster cannot overflow u32 in the bounds checks.
|
||||
if (vbr.number_of_fats == 0 or vbr.cluster_count == 0 or vbr.cluster_count > 0xFFFFFFF5) return null;
|
||||
if (vbr.first_cluster_of_root < first_data_cluster) return null;
|
||||
return .{
|
||||
.bytes_per_sector = @as(u32, 1) << @intCast(vbr.bytes_per_sector_shift),
|
||||
.sectors_per_cluster = @as(u32, 1) << @intCast(vbr.sectors_per_cluster_shift),
|
||||
.cluster_count = vbr.cluster_count,
|
||||
.fat_offset_sectors = vbr.fat_offset,
|
||||
.fat_length_sectors = vbr.fat_length,
|
||||
.cluster_heap_offset_sectors = vbr.cluster_heap_offset,
|
||||
.first_cluster_of_root = vbr.first_cluster_of_root,
|
||||
.volume_serial_number = vbr.volume_serial_number,
|
||||
.volume_length_sectors = vbr.volume_length,
|
||||
.number_of_fats = vbr.number_of_fats,
|
||||
};
|
||||
}
|
||||
|
||||
// --- checksums and the name hash ---------------------------------------------
|
||||
|
||||
/// The directory-entry-SET checksum (a File entry's `set_checksum`), a 16-bit
|
||||
/// rotate-right sum over every byte of the set, skipping the two checksum bytes
|
||||
/// themselves (offset 2..3 of the first entry). `entries` is the whole set:
|
||||
/// (secondary_count + 1) * 32 bytes.
|
||||
pub fn setChecksum(entries: []const u8) u16 {
|
||||
var checksum: u16 = 0;
|
||||
for (entries, 0..) |byte, i| {
|
||||
if (i == 2 or i == 3) continue;
|
||||
checksum = std.math.rotr(u16, checksum, 1) +% byte;
|
||||
}
|
||||
return checksum;
|
||||
}
|
||||
|
||||
/// The up-case-table checksum (an Up-case entry's `table_checksum`), a 32-bit
|
||||
/// rotate-right sum over the table's on-disk bytes.
|
||||
pub fn upcaseChecksum(table_bytes: []const u8) u32 {
|
||||
var checksum: u32 = 0;
|
||||
for (table_bytes) |byte| checksum = std.math.rotr(u32, checksum, 1) +% byte;
|
||||
return checksum;
|
||||
}
|
||||
|
||||
/// The boot-region checksum — the u32 the checksum sector repeats — a 32-bit
|
||||
/// rotate-right sum over the first eleven sectors, skipping VolumeFlags (offset
|
||||
/// 106..107) and PercentInUse (offset 112) of the first sector.
|
||||
pub fn bootChecksum(region: []const u8) u32 {
|
||||
var checksum: u32 = 0;
|
||||
for (region, 0..) |byte, i| {
|
||||
if (i == 106 or i == 107 or i == 112) continue;
|
||||
checksum = std.math.rotr(u32, checksum, 1) +% byte;
|
||||
}
|
||||
return checksum;
|
||||
}
|
||||
|
||||
/// The name hash a Stream entry carries: a 16-bit rotate-right sum over the
|
||||
/// UP-CASED name's bytes (low byte then high byte of each UTF-16 unit). The caller
|
||||
/// up-cases through the volume's table first; a mismatch lets a lookup reject a
|
||||
/// name without reading its File Name entries.
|
||||
pub fn nameHash(upcased: []const u16) u16 {
|
||||
var hash: u16 = 0;
|
||||
for (upcased) |unit| {
|
||||
hash = std.math.rotr(u16, hash, 1) +% @as(u8, @truncate(unit));
|
||||
hash = std.math.rotr(u16, hash, 1) +% @as(u8, @truncate(unit >> 8));
|
||||
}
|
||||
return hash;
|
||||
}
|
||||
|
||||
// --- timestamps --------------------------------------------------------------
|
||||
//
|
||||
// exFAT packs a timestamp into one u32: the high 16 bits are a DOS date
|
||||
// (year-1980 | month | day), the low 16 a DOS time (hour | minute | second/2).
|
||||
// Same field layout as FAT, so the epoch math matches; there is no timezone in
|
||||
// the packed value (a separate UTC-offset byte carries that, which danos leaves
|
||||
// zero = UTC).
|
||||
|
||||
fn isLeapYear(year: u32) bool {
|
||||
return (year % 4 == 0 and year % 100 != 0) or (year % 400 == 0);
|
||||
}
|
||||
|
||||
const days_in_month = [_]u8{ 31, 28, 31, 30, 31, 30, 31, 31, 30, 31, 30, 31 };
|
||||
|
||||
/// Convert a packed exFAT timestamp to Unix epoch seconds (UTC). 0 for unset.
|
||||
pub fn timestampToEpoch(timestamp: u32) u64 {
|
||||
if (timestamp == 0) return 0;
|
||||
const date: u32 = timestamp >> 16;
|
||||
const time: u32 = timestamp & 0xFFFF;
|
||||
const day: u32 = date & 0x1F;
|
||||
const month: u32 = (date >> 5) & 0x0F;
|
||||
const year: u32 = 1980 + (date >> 9);
|
||||
if (month < 1 or month > 12 or day < 1) return 0;
|
||||
const second: u32 = (time & 0x1F) * 2;
|
||||
const minute: u32 = (time >> 5) & 0x3F;
|
||||
const hour: u32 = (time >> 11) & 0x1F;
|
||||
|
||||
var days: u64 = 0;
|
||||
var y: u32 = 1970;
|
||||
while (y < year) : (y += 1) days += if (isLeapYear(y)) 366 else 365;
|
||||
var m: u32 = 1;
|
||||
while (m < month) : (m += 1) {
|
||||
days += days_in_month[m - 1];
|
||||
if (m == 2 and isLeapYear(year)) days += 1;
|
||||
}
|
||||
days += day - 1;
|
||||
return ((days * 24 + hour) * 60 + minute) * 60 + second;
|
||||
}
|
||||
|
||||
/// Convert Unix epoch seconds (UTC) to a packed exFAT timestamp. 0 for epoch 0 or
|
||||
/// any time before 1980 (unrepresentable).
|
||||
pub fn epochToTimestamp(epoch: u64) u32 {
|
||||
if (epoch == 0) return 0;
|
||||
var remaining = epoch;
|
||||
const second: u32 = @intCast(remaining % 60);
|
||||
remaining /= 60;
|
||||
const minute: u32 = @intCast(remaining % 60);
|
||||
remaining /= 60;
|
||||
const hour: u32 = @intCast(remaining % 24);
|
||||
remaining /= 24;
|
||||
var days: u32 = @intCast(remaining);
|
||||
|
||||
var year: u32 = 1970;
|
||||
while (true) {
|
||||
const year_days: u32 = if (isLeapYear(year)) 366 else 365;
|
||||
if (days < year_days) break;
|
||||
days -= year_days;
|
||||
year += 1;
|
||||
}
|
||||
if (year < 1980) return 0;
|
||||
var month: u32 = 1;
|
||||
while (true) {
|
||||
var month_days: u32 = days_in_month[month - 1];
|
||||
if (month == 2 and isLeapYear(year)) month_days += 1;
|
||||
if (days < month_days) break;
|
||||
days -= month_days;
|
||||
month += 1;
|
||||
}
|
||||
const day = days + 1;
|
||||
const date: u32 = ((year - 1980) << 9) | (month << 5) | day;
|
||||
const time: u32 = (hour << 11) | (minute << 5) | (second / 2);
|
||||
return (date << 16) | time;
|
||||
}
|
||||
|
||||
// --- tests -------------------------------------------------------------------
|
||||
|
||||
test "on-disk struct sizes match the specification" {
|
||||
try std.testing.expectEqual(@as(usize, 512), @sizeOf(MainBootSector));
|
||||
try std.testing.expectEqual(@as(usize, 32), @sizeOf(RawEntry));
|
||||
try std.testing.expectEqual(@as(usize, 32), @sizeOf(AllocationBitmapEntry));
|
||||
try std.testing.expectEqual(@as(usize, 32), @sizeOf(UpcaseTableEntry));
|
||||
try std.testing.expectEqual(@as(usize, 32), @sizeOf(VolumeLabelEntry));
|
||||
try std.testing.expectEqual(@as(usize, 32), @sizeOf(FileEntry));
|
||||
try std.testing.expectEqual(@as(usize, 32), @sizeOf(StreamExtensionEntry));
|
||||
try std.testing.expectEqual(@as(usize, 32), @sizeOf(FileNameEntry));
|
||||
}
|
||||
|
||||
test "MainBootSector field offsets" {
|
||||
try std.testing.expectEqual(@as(usize, 3), @offsetOf(MainBootSector, "filesystem_name"));
|
||||
try std.testing.expectEqual(@as(usize, 11), @offsetOf(MainBootSector, "must_be_zero"));
|
||||
try std.testing.expectEqual(@as(usize, 80), @offsetOf(MainBootSector, "fat_offset"));
|
||||
try std.testing.expectEqual(@as(usize, 88), @offsetOf(MainBootSector, "cluster_heap_offset"));
|
||||
try std.testing.expectEqual(@as(usize, 96), @offsetOf(MainBootSector, "first_cluster_of_root"));
|
||||
try std.testing.expectEqual(@as(usize, 106), @offsetOf(MainBootSector, "volume_flags"));
|
||||
try std.testing.expectEqual(@as(usize, 112), @offsetOf(MainBootSector, "percent_in_use"));
|
||||
try std.testing.expectEqual(@as(usize, 510), @offsetOf(MainBootSector, "boot_signature"));
|
||||
// The Stream Extension's data location must sit where the spec places it.
|
||||
try std.testing.expectEqual(@as(usize, 20), @offsetOf(StreamExtensionEntry, "first_cluster"));
|
||||
try std.testing.expectEqual(@as(usize, 24), @offsetOf(StreamExtensionEntry, "data_length"));
|
||||
}
|
||||
|
||||
test "geometryOf accepts exFAT and the MustBeZero guard rejects a FAT-shaped sector" {
|
||||
var sector = [_]u8{0} ** 512;
|
||||
@memcpy(sector[3..11], "EXFAT ");
|
||||
sector[510] = 0x55;
|
||||
sector[511] = 0xAA;
|
||||
// fat_offset=128, fat_length=64, cluster_heap_offset=256, cluster_count=1000,
|
||||
// root cluster=5, bytes/sector=512 (shift 9), sectors/cluster=8 (shift 3), 1 FAT.
|
||||
std.mem.writeInt(u32, sector[80..84], 128, .little);
|
||||
std.mem.writeInt(u32, sector[84..88], 64, .little);
|
||||
std.mem.writeInt(u32, sector[88..92], 256, .little);
|
||||
std.mem.writeInt(u32, sector[92..96], 1000, .little);
|
||||
std.mem.writeInt(u32, sector[96..100], 5, .little);
|
||||
sector[108] = 9; // bytes_per_sector_shift
|
||||
sector[109] = 3; // sectors_per_cluster_shift
|
||||
sector[110] = 1; // number_of_fats
|
||||
const geo = geometryOf(§or) orelse return error.ShouldParse;
|
||||
try std.testing.expectEqual(@as(u32, 512), geo.bytes_per_sector);
|
||||
try std.testing.expectEqual(@as(u32, 8), geo.sectors_per_cluster);
|
||||
try std.testing.expectEqual(@as(u32, 1000), geo.cluster_count);
|
||||
try std.testing.expectEqual(@as(u32, 5), geo.first_cluster_of_root);
|
||||
|
||||
// A non-zero byte in MustBeZero (where a FAT BPB keeps bytes-per-sector) is
|
||||
// rejected — the mutual exclusion between the engines.
|
||||
sector[11] = 0x02;
|
||||
try std.testing.expect(geometryOf(§or) == null);
|
||||
sector[11] = 0;
|
||||
// Wrong name is rejected too.
|
||||
sector[3] = 'F';
|
||||
try std.testing.expect(geometryOf(§or) == null);
|
||||
sector[3] = 'E';
|
||||
}
|
||||
|
||||
test "geometryOf rejects crafted VBRs that would overflow u32 cluster arithmetic" {
|
||||
var sector = [_]u8{0} ** 512;
|
||||
@memcpy(sector[3..11], "EXFAT ");
|
||||
sector[510] = 0x55;
|
||||
sector[511] = 0xAA;
|
||||
std.mem.writeInt(u32, sector[92..96], 1000, .little); // cluster_count
|
||||
std.mem.writeInt(u32, sector[96..100], 5, .little); // root cluster
|
||||
sector[108] = 9; // bytes_per_sector_shift
|
||||
sector[110] = 1; // number_of_fats
|
||||
// A cluster shift past the exFAT ceiling (9 + 17 = 26 > 25) would make
|
||||
// sectors_per_cluster * 512 overflow u32 — rejected.
|
||||
sector[109] = 17;
|
||||
try std.testing.expect(geometryOf(§or) == null);
|
||||
sector[109] = 3; // sane again
|
||||
try std.testing.expect(geometryOf(§or) != null);
|
||||
// cluster_count above the spec maximum (0xFFFFFFF5) would overflow
|
||||
// cluster_count + first_data_cluster in the bounds checks — rejected.
|
||||
std.mem.writeInt(u32, sector[92..96], 0xFFFFFFFF, .little);
|
||||
try std.testing.expect(geometryOf(§or) == null);
|
||||
}
|
||||
|
||||
test "set checksum skips its own two bytes and depends on the rest" {
|
||||
var set = [_]u8{0} ** 64; // a File entry + one secondary
|
||||
set[0] = entry_type_file;
|
||||
set[1] = 1;
|
||||
set[4] = 0x20; // an attribute byte
|
||||
set[40] = 0xAB; // a byte in the secondary entry
|
||||
const base = setChecksum(&set);
|
||||
// Changing the checksum field itself must NOT change the computed checksum.
|
||||
set[2] = 0xFF;
|
||||
set[3] = 0xEE;
|
||||
try std.testing.expectEqual(base, setChecksum(&set));
|
||||
// Changing any other byte MUST change it.
|
||||
set[4] = 0x21;
|
||||
try std.testing.expect(setChecksum(&set) != base);
|
||||
}
|
||||
|
||||
test "name hash is deterministic and order-sensitive" {
|
||||
const readme = [_]u16{ 'R', 'E', 'A', 'D', 'M', 'E' };
|
||||
const different = [_]u16{ 'E', 'R', 'A', 'D', 'M', 'E' };
|
||||
try std.testing.expectEqual(nameHash(&readme), nameHash(&readme));
|
||||
try std.testing.expect(nameHash(&readme) != nameHash(&different));
|
||||
}
|
||||
|
||||
test "boot checksum skips VolumeFlags and PercentInUse" {
|
||||
var region = [_]u8{0} ** 1536; // three 512-byte sectors is enough to exercise the skips
|
||||
region[64] = 0x11;
|
||||
const base = bootChecksum(®ion);
|
||||
for ([_]usize{ 106, 107, 112 }) |skipped| {
|
||||
var copy = region;
|
||||
copy[skipped] = 0xFF;
|
||||
try std.testing.expectEqual(base, bootChecksum(©));
|
||||
}
|
||||
var copy = region;
|
||||
copy[108] = 0xFF; // a non-skipped byte
|
||||
try std.testing.expect(bootChecksum(©) != base);
|
||||
}
|
||||
|
||||
test "exFAT timestamp <-> Unix epoch round trip" {
|
||||
for ([_]u64{ 1_577_836_800, 1_700_000_000, 1_262_304_000, 1_783_971_244 }) |epoch| {
|
||||
try std.testing.expectEqual(epoch, timestampToEpoch(epochToTimestamp(epoch)));
|
||||
}
|
||||
// 1577836800 is 2020-01-01 00:00:00 UTC.
|
||||
const stamp = epochToTimestamp(1_577_836_800);
|
||||
try std.testing.expectEqual(@as(u32, 2020), 1980 + (stamp >> 16 >> 9));
|
||||
try std.testing.expectEqual(@as(u64, 0), timestampToEpoch(0));
|
||||
try std.testing.expectEqual(@as(u32, 0), epochToTimestamp(0));
|
||||
}
|
||||
@@ -1657,3 +1657,21 @@ test "short-name checksum matches the reference vector" {
|
||||
const c = FileSystem.shortChecksum("REDAME TXT".*);
|
||||
try std.testing.expect(a != c);
|
||||
}
|
||||
|
||||
test "the FAT engine rejects an exFAT volume (mutual exclusion at mount)" {
|
||||
const allocator = std.testing.allocator;
|
||||
const bytes = try allocator.alloc(u8, 5000 * sector_size);
|
||||
defer allocator.free(bytes);
|
||||
@memset(bytes, 0);
|
||||
// An exFAT boot sector: the "EXFAT " name and 0x55AA, but MustBeZero (offset
|
||||
// 11, where a FAT BPB keeps bytes-per-sector) stays zero — so this engine's
|
||||
// geometryOf reads a zero bytes-per-sector and rejects it.
|
||||
@memcpy(bytes[3..11], "EXFAT ");
|
||||
bytes[on_disk.boot_signature_offset] = 0x55;
|
||||
bytes[on_disk.boot_signature_offset + 1] = 0xAA;
|
||||
var disk = RamDisk{ .bytes = bytes };
|
||||
try std.testing.expect(FileSystem.mount(disk.device()) == null);
|
||||
// Control: a real FAT16 mounts.
|
||||
formatFat16(bytes);
|
||||
try std.testing.expect(FileSystem.mount(disk.device()) != null);
|
||||
}
|
||||
|
||||
+42
-11
@@ -64,15 +64,24 @@ var filesystem: engine.FileSystem = undefined;
|
||||
/// the right volume's channel.
|
||||
var my_volume_id: u64 = 0;
|
||||
|
||||
/// The prefixes this volume installs: /volumes/usb from the volume root, plus
|
||||
/// the two hierarchy subtrees the boot volume carries (rewrite == prefix), so
|
||||
/// hierarchy paths (the logger's /system/logs) stay decoupled from which volume
|
||||
/// backs them.
|
||||
const fat_mounts = [_]harness.MountSpec{
|
||||
.{ .prefix = "/volumes/usb" },
|
||||
.{ .prefix = "/system/configuration", .rewrite = "/system/configuration" },
|
||||
.{ .prefix = "/system/logs", .rewrite = "/system/logs" },
|
||||
};
|
||||
/// The volume's own mount path, handed in as argv[2] by the volume manager: the
|
||||
/// volume's content id-path (e.g. /volumes/fat-12345678). Defaults to
|
||||
/// /volumes/usb only for a bare launch with no argument; the manager always
|
||||
/// passes it. The slice points into the entry block, valid for the process life.
|
||||
var volume_mount_prefix: []const u8 = "/volumes/usb";
|
||||
|
||||
/// The mounts this volume installs: its own root, plus — only if it is the boot
|
||||
/// volume (it resolves /system/configuration) — the two FHS rewrites, so the
|
||||
/// logger's /system/logs stays decoupled from which volume backs it. Boot-volume
|
||||
/// detection is by content, so it works no matter which volume carries /system.
|
||||
/// bound: mounts one volume installs (its root + the two boot rewrites)
|
||||
/// decided-by: ours
|
||||
/// protects: the mount_specs array
|
||||
/// at-limit: truncate - unreachable today (fixed at 3); more configured mounts
|
||||
/// would need this raised, a deliberate change
|
||||
/// observed-by: a mount silently missing from the harness's mount log
|
||||
const maximum_mounts_per_volume = 4;
|
||||
var mount_specs: [maximum_mounts_per_volume]harness.MountSpec = undefined;
|
||||
|
||||
/// Get this volume's block channel from the volume manager (establishment by
|
||||
/// lineage, communication.md "Establishment: two planes" — `block` is not a
|
||||
@@ -164,14 +173,36 @@ fn fatBringUp(endpoint: ipc.Handle) ?Harness.Volume {
|
||||
};
|
||||
std.log.info("mounted FAT ({s}, {d} clusters, partition lba {d})", .{ @tagName(filesystem.geometry.fat_type), filesystem.geometry.cluster_count, filesystem.base_lba });
|
||||
|
||||
return .{ .engine = &filesystem, .mounts = &fat_mounts, .flush = flushIfDirty };
|
||||
// Every volume mounts at its own id-path (argv[2]). The boot/system volume —
|
||||
// the one carrying the /system tree — ADDITIONALLY installs the two FHS
|
||||
// rewrites, so hierarchy paths (config reads, the logger's persistent
|
||||
// /system/logs) stay decoupled from which volume backs them. Detection is by
|
||||
// CONTENT, not spawn order: a volume is the system volume iff /system/
|
||||
// configuration resolves on its own media. A data volume has no /system, so it
|
||||
// mounts only at its id-path and never shadows the running system's config or
|
||||
// logs with a dead mount.
|
||||
mount_specs[0] = .{ .prefix = volume_mount_prefix };
|
||||
var mount_count: usize = 1;
|
||||
if (filesystem.resolve("/system/configuration") != null) {
|
||||
std.log.info("volume {d} carries the system tree; backing /system/configuration and /system/logs", .{my_volume_id});
|
||||
mount_specs[1] = .{ .prefix = "/system/configuration", .rewrite = "/system/configuration" };
|
||||
mount_specs[2] = .{ .prefix = "/system/logs", .rewrite = "/system/logs" };
|
||||
mount_count = 3;
|
||||
} else {
|
||||
std.log.info("volume {d} is a data volume; mounted at {s}", .{ my_volume_id, volume_mount_prefix });
|
||||
}
|
||||
return .{ .engine = &filesystem, .mounts = mount_specs[0..mount_count], .flush = flushIfDirty };
|
||||
}
|
||||
|
||||
pub fn main(init: process.Init) void {
|
||||
// The volume manager spawns this process with its volume id as argv[1].
|
||||
// The volume manager spawns this process with its volume id as argv[1] and
|
||||
// the volume's mount path (its id-path) as argv[2].
|
||||
if (init.arguments.get(1)) |id| {
|
||||
my_volume_id = std.fmt.parseInt(u64, id, 10) catch 0;
|
||||
}
|
||||
if (init.arguments.get(2)) |prefix| {
|
||||
volume_mount_prefix = prefix;
|
||||
}
|
||||
_ = logging.write("/system/services/fat: starting, waiting for a block device\n");
|
||||
Harness.run(.{ .bringUp = fatBringUp });
|
||||
}
|
||||
|
||||
@@ -10,21 +10,46 @@ pub fn build(b: *std.Build) void {
|
||||
.name = "volume-manager",
|
||||
.root_source_file = b.path("volume-manager.zig"),
|
||||
.imports = &.{
|
||||
"block", "channel", "device-manager-protocol", "driver",
|
||||
"envelope", "ipc", "logging", "memory",
|
||||
"process", "service", "time", "volume-manager-protocol",
|
||||
"block", "channel", "csv", "device-manager-protocol",
|
||||
"driver", "envelope", "file-system", "ipc",
|
||||
"logging", "memory", "process", "service",
|
||||
"time", "volume-manager-protocol",
|
||||
},
|
||||
});
|
||||
b.installArtifact(exe);
|
||||
|
||||
// Standalone `zig build test` for the partition parser; the root build keeps
|
||||
// its aggregate test step.
|
||||
const test_step = b.step("test", "Run the partition-parser unit tests");
|
||||
const tests = b.addTest(.{
|
||||
// Standalone `zig build test` for the parser + mount-map modules; the root
|
||||
// build keeps its aggregate test step.
|
||||
const csv = b.dependency("csv", .{});
|
||||
const test_step = b.step("test", "Run the partition parser + mount-map unit tests");
|
||||
|
||||
const partition_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("partition.zig"),
|
||||
.target = b.resolveTargetQuery(.{}),
|
||||
}),
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(tests).step);
|
||||
test_step.dependOn(&b.addRunArtifact(partition_tests).step);
|
||||
|
||||
// filesystem-map imports csv (and, by path, partition.zig), so its test
|
||||
// module needs csv wired.
|
||||
const filesystem_map_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("filesystem-map.zig"),
|
||||
.target = b.resolveTargetQuery(.{}),
|
||||
.imports = &.{.{ .name = "csv", .module = csv.module("csv") }},
|
||||
}),
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(filesystem_map_tests).step);
|
||||
|
||||
// volume-map imports csv (and, by path, partition.zig) for the id-path
|
||||
// deriver and the volumes.csv override parser.
|
||||
const volume_map_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("volume-map.zig"),
|
||||
.target = b.resolveTargetQuery(.{}),
|
||||
.imports = &.{.{ .name = "csv", .module = csv.module("csv") }},
|
||||
}),
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(volume_map_tests).step);
|
||||
}
|
||||
|
||||
@@ -11,6 +11,8 @@
|
||||
.kernel = .{ .path = "../../../library/kernel" },
|
||||
.device = .{ .path = "../../../library/device" },
|
||||
.protocol = .{ .path = "../../../library/protocol" },
|
||||
// csv parses filesystems.csv / volumes.csv, the mount-map configuration.
|
||||
.csv = .{ .path = "../../../library/csv" },
|
||||
},
|
||||
.paths = .{""},
|
||||
}
|
||||
|
||||
@@ -0,0 +1,121 @@
|
||||
//! filesystem-map — parse `/system/configuration/filesystems.csv` into
|
||||
//! content-signature → service-binary rules, and pick the binary for a probed
|
||||
//! volume's signature. The data-driven replacement for the volume manager's
|
||||
//! hardcoded `filesystem_binary` const: a signature no row matches goes unserved
|
||||
//! (logged), never guessed — the same discipline the device registry uses.
|
||||
//!
|
||||
//! Pure logic: no hardware, no syscalls, no allocator. The `binary` slice points
|
||||
//! into the CSV source, which the manager holds in a static buffer for the life
|
||||
//! of the process (zero-copy), so the source must outlive the rules.
|
||||
//!
|
||||
//! Format: one rule per line, two comma-separated fields, `#` comments (whole-
|
||||
//! line or trailing), blank lines ignored:
|
||||
//!
|
||||
//! signature, binary
|
||||
//!
|
||||
//! `signature` is a filesystem token (`fat`; `exfat` lands with S4); `binary` is
|
||||
//! a full ramdisk path.
|
||||
|
||||
const std = @import("std");
|
||||
const csv = @import("csv");
|
||||
const partition = @import("partition.zig");
|
||||
|
||||
/// One parsed row: a content signature and the service binary that serves it.
|
||||
pub const Rule = struct {
|
||||
kind: partition.FilesystemKind,
|
||||
binary: []const u8,
|
||||
};
|
||||
|
||||
/// How many rules landed, how many non-blank lines were malformed (for the
|
||||
/// manager to log), and whether there were more rules than the buffer could hold.
|
||||
pub const ParseResult = struct { count: usize, malformed: usize, truncated: bool };
|
||||
|
||||
const Line = union(enum) { rule: Rule, ignorable, malformed };
|
||||
|
||||
fn parseLine(line: []const u8) Line {
|
||||
const body = csv.stripComment(line);
|
||||
if (body.len == 0) return .ignorable;
|
||||
var it = csv.fields(body);
|
||||
const sig = it.next() orelse return .malformed;
|
||||
const binary = it.next() orelse return .malformed;
|
||||
if (it.next() != null) return .malformed; // too many columns
|
||||
if (binary.len == 0) return .malformed;
|
||||
const kind = partition.FilesystemKind.fromToken(sig);
|
||||
if (kind == .unknown) return .malformed; // an unrecognised signature token
|
||||
return .{ .rule = .{ .kind = kind, .binary = binary } };
|
||||
}
|
||||
|
||||
/// Parse a whole `filesystems.csv` into `out_rules`. The `binary` slices point
|
||||
/// into `source`, which must outlive them.
|
||||
pub fn parse(source: []const u8, out_rules: []Rule) ParseResult {
|
||||
var result: ParseResult = .{ .count = 0, .malformed = 0, .truncated = false };
|
||||
var lines = std.mem.splitScalar(u8, source, '\n');
|
||||
while (lines.next()) |line| {
|
||||
switch (parseLine(line)) {
|
||||
.ignorable => {},
|
||||
.malformed => result.malformed += 1,
|
||||
.rule => |rule| {
|
||||
if (result.count >= out_rules.len) {
|
||||
result.truncated = true;
|
||||
continue;
|
||||
}
|
||||
out_rules[result.count] = rule;
|
||||
result.count += 1;
|
||||
},
|
||||
}
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
/// The service binary for a probed volume's signature — the first matching row,
|
||||
/// or null (the volume goes unserved, like a device no registry row matches).
|
||||
pub fn match(rules: []const Rule, kind: partition.FilesystemKind) ?[]const u8 {
|
||||
for (rules) |rule| {
|
||||
if (rule.kind == kind) return rule.binary;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
// --- tests -------------------------------------------------------------------
|
||||
|
||||
const testing = std.testing;
|
||||
|
||||
// A fixture-sized rule buffer for the tests, named so the bounds gate (which
|
||||
// flags literal array lengths) stays quiet: this is a test input, not a runtime
|
||||
// ceiling — the real one is maximum_filesystem_rules in the volume manager.
|
||||
const test_rule_slots = 4;
|
||||
|
||||
test "a fat signature maps to its binary; an unmatched signature is null" {
|
||||
const text =
|
||||
\\# signature, binary
|
||||
\\fat, /system/services/fat
|
||||
;
|
||||
var rules: [test_rule_slots]Rule = undefined;
|
||||
const parsed = parse(text, &rules);
|
||||
try testing.expectEqual(@as(usize, 1), parsed.count);
|
||||
try testing.expectEqual(@as(usize, 0), parsed.malformed);
|
||||
try testing.expectEqualStrings("/system/services/fat", match(rules[0..parsed.count], .fat).?);
|
||||
try testing.expect(match(rules[0..parsed.count], .unknown) == null);
|
||||
}
|
||||
|
||||
test "the binary is chosen by content, not hardcoded" {
|
||||
// Point the fat row at a different binary and confirm that binary is chosen —
|
||||
// a constant could not satisfy this, which is the whole point of the map.
|
||||
const text = "fat, /system/services/other-fat\n";
|
||||
var rules: [test_rule_slots]Rule = undefined;
|
||||
const parsed = parse(text, &rules);
|
||||
try testing.expectEqualStrings("/system/services/other-fat", match(rules[0..parsed.count], .fat).?);
|
||||
}
|
||||
|
||||
test "malformed rows are counted, not bound" {
|
||||
const text =
|
||||
\\fat, /system/services/fat
|
||||
\\bogusfs, /system/services/x
|
||||
\\fat,
|
||||
\\fat, /a, /b
|
||||
;
|
||||
var rules: [test_rule_slots]Rule = undefined;
|
||||
const parsed = parse(text, &rules);
|
||||
try testing.expectEqual(@as(usize, 1), parsed.count); // only the first fat row
|
||||
try testing.expectEqual(@as(usize, 3), parsed.malformed); // bad token, empty binary, too many columns
|
||||
}
|
||||
@@ -31,10 +31,11 @@ pub const label_maximum = 36;
|
||||
/// stay distinct, and it drives how the mount path is rendered from the id.
|
||||
pub const Rung = enum(u8) {
|
||||
gpt_guid = 1,
|
||||
filesystem_uuid = 2, // reserved: no non-FAT engine reads a superblock UUID yet
|
||||
filesystem_uuid = 2, // reserved: no engine reads a superblock UUID yet
|
||||
fat_serial = 3,
|
||||
mbr_index = 4,
|
||||
anonymous = 5,
|
||||
exfat_serial = 6, // exFAT's VolumeSerialNumber — content-strong like fat_serial
|
||||
};
|
||||
|
||||
/// A volume's content identity. `key` is the ID — the stable, unique handle the
|
||||
@@ -60,12 +61,29 @@ pub const Identity = struct {
|
||||
}
|
||||
};
|
||||
|
||||
/// One volume the parser found on the device: the block sub-range it occupies
|
||||
/// and its content identity.
|
||||
/// Which filesystem a volume's content is — the key `filesystems.csv` maps to a
|
||||
/// service binary. FAT and exFAT are recognized by their VBRs; content that is
|
||||
/// neither falls back to `.fat`, the volume manager's historical hand-off.
|
||||
pub const FilesystemKind = enum {
|
||||
fat,
|
||||
exfat,
|
||||
unknown,
|
||||
|
||||
pub fn fromToken(token: []const u8) FilesystemKind {
|
||||
if (std.mem.eql(u8, token, "fat")) return .fat;
|
||||
if (std.mem.eql(u8, token, "exfat")) return .exfat;
|
||||
return .unknown;
|
||||
}
|
||||
};
|
||||
|
||||
/// One volume the parser found on the device: the block sub-range it occupies,
|
||||
/// its content identity, and which filesystem its content is (the signature the
|
||||
/// `filesystems.csv` map keys on to pick the service binary).
|
||||
pub const Volume = struct {
|
||||
base_lba: u64,
|
||||
block_count: u64,
|
||||
identity: Identity,
|
||||
signature: FilesystemKind = .fat,
|
||||
};
|
||||
|
||||
/// Read sectors on demand. `context` + `readFn` mirror the FAT engine's
|
||||
@@ -156,39 +174,41 @@ fn setLabelFromUtf16(id: *Identity, name_bytes: []const u8) void {
|
||||
id.label_len = @intCast(out);
|
||||
}
|
||||
|
||||
/// The first GPT volume, or null if LBA 1 is not a valid GPT header or no entry
|
||||
/// validates. The header CRC-32 and the per-entry overflow-safe range check are
|
||||
/// the confinement-safety guards the driver's clamp rests on — the invariant
|
||||
/// firstVolume documents for MBR, extended to untrusted GPT metadata. The
|
||||
/// entry-array CRC is deferred (correctness-only; the range check carries safety).
|
||||
fn gptFirstVolume(reader: SectorReader, device_blocks: u64) ?Volume {
|
||||
/// Append every valid GPT volume to `out` (up to `out.len`), returning the count
|
||||
/// (0 if LBA 1 is not a valid GPT header). The header CRC-32 and the per-entry
|
||||
/// overflow-safe range check are the confinement-safety guards the driver's clamp
|
||||
/// rests on — the invariant documented for MBR, extended to untrusted GPT
|
||||
/// metadata. The entry-array CRC is deferred (correctness-only; the range check
|
||||
/// carries safety).
|
||||
fn gptAllVolumes(reader: SectorReader, device_blocks: u64, out: []Volume) usize {
|
||||
var header: [sector_bytes]u8 = undefined;
|
||||
if (!reader.read(1, &header)) return null;
|
||||
if (!std.mem.eql(u8, header[0..8], gpt_signature)) return null;
|
||||
if (!reader.read(1, &header)) return 0;
|
||||
if (!std.mem.eql(u8, header[0..8], gpt_signature)) return 0;
|
||||
const header_size = std.mem.readInt(u32, header[12..16], .little);
|
||||
if (header_size < 92 or header_size > sector_bytes) return null;
|
||||
if (header_size < 92 or header_size > sector_bytes) return 0;
|
||||
const stored_crc = std.mem.readInt(u32, header[16..20], .little);
|
||||
var check: [sector_bytes]u8 = undefined;
|
||||
@memcpy(check[0..header_size], header[0..header_size]);
|
||||
@memset(check[16..20], 0);
|
||||
if (crc32(check[0..header_size]) != stored_crc) return null;
|
||||
if (crc32(check[0..header_size]) != stored_crc) return 0;
|
||||
|
||||
const entry_lba = std.mem.readInt(u64, header[72..80], .little);
|
||||
const num_entries = std.mem.readInt(u32, header[80..84], .little);
|
||||
const entry_size = std.mem.readInt(u32, header[84..88], .little);
|
||||
if (entry_size != 128 and entry_size != 256 and entry_size != 512) return null;
|
||||
if (entry_lba == 0 or entry_lba >= device_blocks) return null;
|
||||
if (entry_size != 128 and entry_size != 256 and entry_size != 512) return 0;
|
||||
if (entry_lba == 0 or entry_lba >= device_blocks) return 0;
|
||||
|
||||
const scan = @min(num_entries, gpt_entry_scan_maximum);
|
||||
var sector_buf: [sector_bytes]u8 = undefined;
|
||||
var loaded: u64 = std.math.maxInt(u64);
|
||||
var count: usize = 0;
|
||||
var i: u32 = 0;
|
||||
while (i < scan) : (i += 1) {
|
||||
while (i < scan and count < out.len) : (i += 1) {
|
||||
const abs = @as(u64, i) * entry_size;
|
||||
const lba = entry_lba + abs / sector_bytes;
|
||||
const off = @as(usize, @intCast(abs % sector_bytes));
|
||||
if (lba != loaded) {
|
||||
if (!reader.read(lba, §or_buf)) return null;
|
||||
if (!reader.read(lba, §or_buf)) break; // return what we have
|
||||
loaded = lba;
|
||||
}
|
||||
const entry = sector_buf[off..][0..128]; // the fields we read live in the first 128 bytes
|
||||
@@ -208,9 +228,10 @@ fn gptFirstVolume(reader: SectorReader, device_blocks: u64) ?Volume {
|
||||
if (start == 0 or end < start or end >= device_blocks) continue;
|
||||
var id = Identity{ .rung = .gpt_guid, .key = std.mem.readInt(u128, entry[16..32], .little) };
|
||||
setLabelFromUtf16(&id, entry[56..128]);
|
||||
return .{ .base_lba = start, .block_count = end - start + 1, .identity = id };
|
||||
out[count] = .{ .base_lba = start, .block_count = end - start + 1, .identity = id, .signature = signatureAt(reader, start) };
|
||||
count += 1;
|
||||
}
|
||||
return null;
|
||||
return count;
|
||||
}
|
||||
|
||||
/// Trim trailing spaces (FAT labels are space-padded) and copy into the display
|
||||
@@ -243,18 +264,52 @@ fn fatIdentity(reader: SectorReader, start_lba: u64) ?Identity {
|
||||
return id;
|
||||
}
|
||||
|
||||
/// The first volume on the device `reader` addresses, whose whole-device size is
|
||||
/// `device_blocks`, or null if none is found. A GPT disk (protective MBR) is
|
||||
/// handled by GPT, authoritatively — its null is final. Otherwise an MBR with a
|
||||
/// non-empty entry yields that partition's [start, size); otherwise a boot
|
||||
/// signature with no partitions is treated as a bare FAT spanning the device.
|
||||
pub fn firstVolume(reader: SectorReader, device_blocks: u64) ?Volume {
|
||||
/// The exFAT VolumeSerialNumber (offset 100) read from the Main Boot Sector at
|
||||
/// `start_lba` — its content identity, rung `exfat_serial`. Null unless the sector
|
||||
/// is an exFAT VBR (the "EXFAT " name at offset 3 + the 0x55AA signature; the
|
||||
/// name is where a FAT BPB keeps its OEM string, so the two never collide). The
|
||||
/// label lives in a root-directory entry, not the VBR, so it is left empty here.
|
||||
fn exfatIdentity(reader: SectorReader, start_lba: u64) ?Identity {
|
||||
var vbr: [sector_bytes]u8 = undefined;
|
||||
if (!reader.read(start_lba, &vbr)) return null;
|
||||
if (vbr[510] != 0x55 or vbr[511] != 0xAA) return null;
|
||||
if (!std.mem.eql(u8, vbr[3..11], "EXFAT ")) return null;
|
||||
return .{ .rung = .exfat_serial, .key = std.mem.readInt(u32, vbr[100..104], .little) };
|
||||
}
|
||||
|
||||
const Recognized = struct { identity: Identity, signature: FilesystemKind };
|
||||
|
||||
/// Recognize the filesystem at `start_lba` by its VBR: exFAT first (its serial and
|
||||
/// the `.exfat` signature), else FAT (its serial), else unknown content that keeps
|
||||
/// the MBR disk-signature identity and the historical `.fat` hand-off.
|
||||
fn recognize(reader: SectorReader, start_lba: u64, block0: *const [sector_bytes]u8, index: u8) Recognized {
|
||||
if (exfatIdentity(reader, start_lba)) |id| return .{ .identity = id, .signature = .exfat };
|
||||
if (fatIdentity(reader, start_lba)) |id| return .{ .identity = id, .signature = .fat };
|
||||
return .{ .identity = mbrIdentity(block0, index), .signature = .fat };
|
||||
}
|
||||
|
||||
/// The filesystem signature at `start_lba` when the identity is decided elsewhere
|
||||
/// (a GPT partition keeps its GUID identity but still needs its content's kind).
|
||||
fn signatureAt(reader: SectorReader, start_lba: u64) FilesystemKind {
|
||||
return if (exfatIdentity(reader, start_lba) != null) .exfat else .fat;
|
||||
}
|
||||
|
||||
/// Append every volume on the device `reader` addresses, whose whole-device size
|
||||
/// is `device_blocks`, to `out` (up to `out.len`), returning the count. A GPT
|
||||
/// disk (protective MBR) is enumerated by GPT, authoritatively — a zero count is
|
||||
/// final. Otherwise every fitting MBR entry is a volume; a boot signature with no
|
||||
/// partition entries is a bare FAT spanning the whole device. Each volume's
|
||||
/// [start, count) is validated overflow-safe (the confinement invariant the
|
||||
/// driver's clamp rests on), and each prefers its FAT serial identity over the
|
||||
/// disk signature.
|
||||
pub fn allVolumes(reader: SectorReader, device_blocks: u64, out: []Volume) usize {
|
||||
var block0: [sector_bytes]u8 = undefined;
|
||||
if (!reader.read(0, &block0)) return null;
|
||||
if (!hasBootSignature(&block0)) return null;
|
||||
if (isProtectiveMbr(&block0)) return gptFirstVolume(reader, device_blocks);
|
||||
if (!reader.read(0, &block0)) return 0;
|
||||
if (!hasBootSignature(&block0)) return 0;
|
||||
if (isProtectiveMbr(&block0)) return gptAllVolumes(reader, device_blocks, out);
|
||||
var count: usize = 0;
|
||||
var index: u8 = 0;
|
||||
while (index < 4) : (index += 1) {
|
||||
while (index < 4 and count < out.len) : (index += 1) {
|
||||
const entry = block0[446 + @as(usize, index) * 16 ..][0..16];
|
||||
const kind = entry[4];
|
||||
const start = std.mem.readInt(u32, entry[8..12], .little);
|
||||
@@ -267,10 +322,27 @@ pub fn firstVolume(reader: SectorReader, device_blocks: u64) ?Volume {
|
||||
// device (usb-storage.zig resolveTransfer), which only holds because the
|
||||
// range handed down is validated here. The subtraction cannot overflow.
|
||||
if (start > device_blocks or device_blocks - start < size) continue;
|
||||
return .{ .base_lba = start, .block_count = size, .identity = fatIdentity(reader, start) orelse mbrIdentity(&block0, index) };
|
||||
const found = recognize(reader, start, &block0, index);
|
||||
out[count] = .{ .base_lba = start, .block_count = size, .identity = found.identity, .signature = found.signature };
|
||||
count += 1;
|
||||
}
|
||||
// No partition entries: a bare FAT spanning the device.
|
||||
return .{ .base_lba = 0, .block_count = device_blocks, .identity = fatIdentity(reader, 0) orelse mbrIdentity(&block0, 0) };
|
||||
if (count == 0 and out.len > 0) {
|
||||
// No partition entries: a bare FAT or exFAT spanning the device.
|
||||
const found = recognize(reader, 0, &block0, 0);
|
||||
out[0] = .{ .base_lba = 0, .block_count = device_blocks, .identity = found.identity, .signature = found.signature };
|
||||
return 1;
|
||||
}
|
||||
return count;
|
||||
}
|
||||
|
||||
/// firstVolume is allVolumes into a one-element buffer.
|
||||
const one_volume_slot = 1;
|
||||
|
||||
/// The first volume on the device, or null — the single-volume case of
|
||||
/// `allVolumes`, kept for callers that want just one.
|
||||
pub fn firstVolume(reader: SectorReader, device_blocks: u64) ?Volume {
|
||||
var one: [one_volume_slot]Volume = undefined;
|
||||
return if (allVolumes(reader, device_blocks, &one) > 0) one[0] else null;
|
||||
}
|
||||
|
||||
/// A read-only RAM disk over a byte slice of sectors, for the host tests.
|
||||
@@ -322,6 +394,31 @@ test "no boot signature is no volume" {
|
||||
const disk = RamDisk{ .sectors = &block0 };
|
||||
try std.testing.expect(firstVolume(disk.reader(), 65536) == null);
|
||||
}
|
||||
test "a bare exFAT volume is recognized by its VBR, with its serial as the id" {
|
||||
var block0 = [_]u8{0} ** 512;
|
||||
@memcpy(block0[3..11], "EXFAT ");
|
||||
block0[510] = 0x55;
|
||||
block0[511] = 0xAA;
|
||||
std.mem.writeInt(u32, block0[100..104], 0xDA7A0001, .little); // VolumeSerialNumber
|
||||
const disk = RamDisk{ .sectors = &block0 };
|
||||
const v = firstVolume(disk.reader(), 65536).?;
|
||||
try std.testing.expectEqual(FilesystemKind.exfat, v.signature);
|
||||
try std.testing.expectEqual(Rung.exfat_serial, v.identity.rung);
|
||||
try std.testing.expectEqual(@as(u128, 0xDA7A0001), v.identity.key);
|
||||
}
|
||||
test "a FAT VBR is recognized as fat, not exfat — the signatures never collide" {
|
||||
var block0 = [_]u8{0} ** 512;
|
||||
@memcpy(block0[3..11], "MSWIN4.1"); // a FAT OEM name, not "EXFAT "
|
||||
block0[510] = 0x55;
|
||||
block0[511] = 0xAA;
|
||||
std.mem.writeInt(u16, block0[22..24], 16, .little); // fat_size_16 != 0 -> FAT16 shape
|
||||
block0[38] = 0x29; // extended boot signature
|
||||
std.mem.writeInt(u32, block0[39..43], 0x12345678, .little); // volume id
|
||||
const disk = RamDisk{ .sectors = &block0 };
|
||||
const v = firstVolume(disk.reader(), 65536).?;
|
||||
try std.testing.expectEqual(FilesystemKind.fat, v.signature);
|
||||
try std.testing.expectEqual(Rung.fat_serial, v.identity.rung);
|
||||
}
|
||||
|
||||
test "a partition that runs past the device is skipped, not trusted" {
|
||||
var block0 = [_]u8{0} ** 512;
|
||||
@@ -491,3 +588,33 @@ test "GPT with 256-byte entries reads the non-128 offset arithmetic correctly" {
|
||||
try std.testing.expectEqual(Rung.gpt_guid, v.identity.rung);
|
||||
try std.testing.expectEqual(@as(u128, 0xF00D), v.identity.key);
|
||||
}
|
||||
|
||||
// A fixture-sized volume buffer for the multi-volume tests, named so the bounds
|
||||
// gate (which flags literal array lengths) stays quiet: a test input.
|
||||
const test_volume_slots = 4;
|
||||
|
||||
test "allVolumes returns every fitting MBR partition with distinct identities" {
|
||||
var block0 = [_]u8{0} ** 512;
|
||||
block0[510] = 0x55;
|
||||
block0[511] = 0xAA;
|
||||
std.mem.writeInt(u32, block0[440..444], 0xDEADBEEF, .little);
|
||||
// partition 0: start 2048, size 1000
|
||||
block0[446 + 4] = 0x0c;
|
||||
std.mem.writeInt(u32, block0[446 + 8 ..][0..4], 2048, .little);
|
||||
std.mem.writeInt(u32, block0[446 + 12 ..][0..4], 1000, .little);
|
||||
// partition 1: start 4096, size 2000
|
||||
block0[462 + 4] = 0x0c;
|
||||
std.mem.writeInt(u32, block0[462 + 8 ..][0..4], 4096, .little);
|
||||
std.mem.writeInt(u32, block0[462 + 12 ..][0..4], 2000, .little);
|
||||
const disk = RamDisk{ .sectors = &block0 };
|
||||
var vols: [test_volume_slots]Volume = undefined;
|
||||
const n = allVolumes(disk.reader(), 200000, &vols);
|
||||
try std.testing.expectEqual(@as(usize, 2), n); // both partitions, not just the first
|
||||
try std.testing.expectEqual(@as(u64, 2048), vols[0].base_lba);
|
||||
try std.testing.expectEqual(@as(u64, 4096), vols[1].base_lba);
|
||||
// distinct rung-4 identities (no FAT VBR at those LBAs): index 0 vs 1.
|
||||
try std.testing.expectEqual((@as(u128, 0xDEADBEEF) << 8) | 0, vols[0].identity.key);
|
||||
try std.testing.expectEqual((@as(u128, 0xDEADBEEF) << 8) | 1, vols[1].identity.key);
|
||||
// firstVolume (the 1-buffer case) still returns just the first.
|
||||
try std.testing.expectEqual(@as(u64, 2048), firstVolume(disk.reader(), 200000).?.base_lba);
|
||||
}
|
||||
|
||||
@@ -8,10 +8,12 @@
|
||||
//! supervises the filesystems it spawns, exactly as the device manager
|
||||
//! supervises drivers.
|
||||
//!
|
||||
//! This increment (V3b) is the flip: the FAT service stops acquiring its own
|
||||
//! volume and is spawned here instead, confined to its partition, and handed
|
||||
//! its channel over the volume-manager protocol. Single volume for now; the
|
||||
//! mount map (volumes.csv) and multi-volume land next.
|
||||
//! The manager holds a table of adopted storage DEVICES and a table of the
|
||||
//! VOLUMES on them: it adopts every storage device the device-manager tree
|
||||
//! carries, probes each one's whole partition table, and spawns one filesystem
|
||||
//! process per volume — each confined to its partition's badge-scoped block
|
||||
//! range, each supervised with its own budget. A device leaving the tree takes
|
||||
//! its volumes with it.
|
||||
|
||||
const std = @import("std");
|
||||
const channel = @import("channel");
|
||||
@@ -26,60 +28,146 @@ const process = @import("process");
|
||||
const service = @import("service");
|
||||
const time = @import("time");
|
||||
const envelope = @import("envelope");
|
||||
const fs = @import("file-system");
|
||||
const partition = @import("partition.zig");
|
||||
const filesystem_map = @import("filesystem-map.zig");
|
||||
const volume_map = @import("volume-map.zig");
|
||||
|
||||
const Serve = volume_manager_protocol.Protocol.Provider(void);
|
||||
const Invocation = envelope.Invocation;
|
||||
const Answer = envelope.Answer;
|
||||
|
||||
/// The single volume this increment handles: its provider channel, its block
|
||||
/// sub-range, its identity, the id it is addressed by, and the filesystem
|
||||
/// process serving it (0 until spawned; reset on death for respawn).
|
||||
const Volume = struct {
|
||||
storage: block.Device,
|
||||
storage_device_id: u64, // the device-manager id this volume's provider serves
|
||||
base_lba: u64,
|
||||
block_count: u64,
|
||||
identity: partition.Identity,
|
||||
id: u64,
|
||||
filesystem_pid: u32 = 0,
|
||||
/// One adopted storage device: the block channel to its provider (opened once and
|
||||
/// shared — refcounted per confined filesystem via the hello reply) and the
|
||||
/// device-manager id it serves. A device leaving the tree takes its volumes.
|
||||
const StorageDevice = struct {
|
||||
used: bool = false,
|
||||
device_id: u64 = 0,
|
||||
channel: block.Device = undefined,
|
||||
};
|
||||
|
||||
/// The filesystem binary a probed volume is served by. The signature->binary
|
||||
/// map (filesystems.csv) lands with the identity ladder; for now every FAT-shaped
|
||||
/// volume gets the FAT service.
|
||||
const filesystem_binary = "/system/services/fat";
|
||||
const volume_id: u64 = 1;
|
||||
/// One volume: which device serves it, its block sub-range, its content
|
||||
/// identity, the id it is addressed by, the service binary + mount path it was
|
||||
/// spawned with, the filesystem process serving it, and its own supervision
|
||||
/// budget (so one volume's crash loop never touches another's).
|
||||
const Volume = struct {
|
||||
used: bool = false,
|
||||
device_id: u64 = 0,
|
||||
base_lba: u64 = 0,
|
||||
block_count: u64 = 0,
|
||||
identity: partition.Identity = .{ .rung = .anonymous },
|
||||
id: u64 = 0,
|
||||
binary: []const u8 = "",
|
||||
mount_prefix: []const u8 = "",
|
||||
filesystem_pid: u32 = 0,
|
||||
// Per-volume supervision, mirroring the device manager's: a clean exit is not
|
||||
// restarted, a fault restarts with backoff, a fast crash loop gives up.
|
||||
restarts: u32 = 0,
|
||||
spawn_ns: u64 = 0,
|
||||
failed: bool = false,
|
||||
restart_pending: bool = false,
|
||||
restart_due_ns: u64 = 0,
|
||||
};
|
||||
|
||||
// The mount map, read from configuration at boot (the policy home, storage-
|
||||
// architecture.md): filesystems.csv (content signature -> service binary) and
|
||||
// volumes.csv (an optional id -> mount-prefix override). The sources are held
|
||||
// for the process life so the parsed rules' slices into them stay valid.
|
||||
/// bound: bytes of filesystems.csv / volumes.csv the manager reads
|
||||
/// decided-by: ours
|
||||
/// protects: the config source buffers below
|
||||
/// at-limit: truncate - a longer file is cut; a row split by the cut is malformed
|
||||
/// observed-by: the per-file "malformed/truncated" log line
|
||||
const config_source_bytes = 2048;
|
||||
var filesystems_source: [config_source_bytes]u8 = undefined;
|
||||
var volumes_source: [config_source_bytes]u8 = undefined;
|
||||
/// bound: filesystem-map rules held (one per content signature)
|
||||
/// decided-by: ours
|
||||
/// protects: the filesystem_rules table
|
||||
/// at-limit: truncate - extra rows are dropped and the "truncated" note logged
|
||||
/// observed-by: the "truncated" log line
|
||||
const maximum_filesystem_rules = 8;
|
||||
/// bound: volumes.csv override rows held (one per pinned volume id)
|
||||
/// decided-by: ours
|
||||
/// protects: the volume_rules table
|
||||
/// at-limit: truncate - extra rows are dropped and the "truncated" note logged
|
||||
/// observed-by: the "truncated" log line
|
||||
const maximum_volume_rules = 64;
|
||||
var filesystem_rules: [maximum_filesystem_rules]filesystem_map.Rule = undefined;
|
||||
var filesystem_rule_count: usize = 0;
|
||||
var volume_rules: [maximum_volume_rules]volume_map.Override = undefined;
|
||||
var volume_rule_count: usize = 0;
|
||||
/// bound: bytes of a composed /volumes/<id> mount path
|
||||
/// decided-by: ours
|
||||
/// protects: the per-volume mount_prefix buffers below
|
||||
/// at-limit: truncate - bufPrint fails; the volume mounts at a fallback path (logged)
|
||||
/// observed-by: the fallback path in the log
|
||||
const mount_path_maximum = 64;
|
||||
|
||||
/// bound: volumes the manager serves at once
|
||||
/// decided-by: ours
|
||||
/// protects: the volumes table and its per-volume mount-path buffers
|
||||
/// at-limit: truncate - a further partition is left unserved and logged (real
|
||||
/// machines carry a handful of volumes, far under this)
|
||||
/// observed-by: the "volume table full" log line
|
||||
const maximum_volumes = 16;
|
||||
/// bound: storage devices the manager adopts at once
|
||||
/// decided-by: ours
|
||||
/// protects: the devices table
|
||||
/// at-limit: truncate - a further device is left unadopted and logged
|
||||
/// observed-by: the "device table full" log line
|
||||
const maximum_devices = 8;
|
||||
var devices = [_]StorageDevice{.{}} ** maximum_devices;
|
||||
var volumes = [_]Volume{.{}} ** maximum_volumes;
|
||||
/// Each volume's composed default mount path lives in its slot's buffer; a
|
||||
/// volumes.csv override is used in place (a slice into volumes_source, no buffer).
|
||||
var mount_prefix_bufs: [maximum_volumes][mount_path_maximum]u8 = undefined;
|
||||
var next_volume_id: u64 = 1; // monotonic — never reused, so a stale id can't address the wrong child
|
||||
|
||||
var service_endpoint: ipc.Handle = 0;
|
||||
var manager_handle: ?ipc.Handle = null;
|
||||
var bounce: memory.DmaRegion = undefined;
|
||||
var bounce_ready = false;
|
||||
/// The currently-mounted volume, or null while no storage is present. The whole
|
||||
/// removal lifecycle is this field going null and back: the poll sees the
|
||||
/// storage provider leave the device tree (a pulled stick), kills the filesystem
|
||||
/// and clears this; when it returns, the poll re-acquires and re-mounts.
|
||||
var volume: ?Volume = null;
|
||||
var logged_no_volume = false;
|
||||
/// How often the poll checks whether the storage provider is present. Fast
|
||||
/// enough that an unplug unmounts promptly; the poll is a bare device-manager
|
||||
/// enumerate, no channel work, so it is cheap to run continuously.
|
||||
/// How often the poll checks device presence and fires due restarts. Fast enough
|
||||
/// that an unplug unmounts promptly; the poll is a bare device-manager enumerate,
|
||||
/// no channel work, so it is cheap to run continuously.
|
||||
const poll_interval_ms = 500;
|
||||
|
||||
// Filesystem supervision, mirroring the device manager's (device-manager.zig):
|
||||
// a clean exit is not restarted, a fault restarts with backoff, and a fast
|
||||
// crash loop gives up rather than spinning. Without this a faulting filesystem
|
||||
// respawns in a zero-delay loop.
|
||||
// Filesystem supervision, mirroring the device manager's (device-manager.zig).
|
||||
const fast_death_ns: u64 = 2_000_000_000;
|
||||
const crash_loop_cap: u32 = 3;
|
||||
const backoff_base_ms: u64 = 300;
|
||||
var fs_restarts: u32 = 0;
|
||||
var fs_spawn_ns: u64 = 0;
|
||||
var fs_failed = false;
|
||||
/// A fat restart is due at `restart_due_ns`; the poll loop performs it once the
|
||||
/// backoff has elapsed (one timer, folded into the poll — no second timer).
|
||||
var restart_pending = false;
|
||||
var restart_due_ns: u64 = 0;
|
||||
/// bytes to format a u64 volume id as decimal (20 digits fit)
|
||||
const id_decimal_bytes = 24;
|
||||
|
||||
// --- table lookups -----------------------------------------------------------
|
||||
|
||||
fn deviceById(id: u64) ?*StorageDevice {
|
||||
for (&devices) |*d| if (d.used and d.device_id == id) return d;
|
||||
return null;
|
||||
}
|
||||
fn claimDevice() ?*StorageDevice {
|
||||
for (&devices) |*d| if (!d.used) return d;
|
||||
return null;
|
||||
}
|
||||
fn volumeById(id: u64) ?*Volume {
|
||||
for (&volumes) |*v| if (v.used and v.id == id) return v;
|
||||
return null;
|
||||
}
|
||||
fn volumeByPid(pid: u32) ?*Volume {
|
||||
for (&volumes) |*v| if (v.used and v.filesystem_pid == pid) return v;
|
||||
return null;
|
||||
}
|
||||
fn firstUsedVolume() ?*Volume {
|
||||
for (&volumes) |*v| if (v.used) return v;
|
||||
return null;
|
||||
}
|
||||
fn claimVolumeIndex() ?usize {
|
||||
for (&volumes, 0..) |*v, i| if (!v.used) return i;
|
||||
return null;
|
||||
}
|
||||
|
||||
// --- device-manager plumbing -------------------------------------------------
|
||||
|
||||
fn deviceManager() ?ipc.Handle {
|
||||
if (manager_handle) |h| return h;
|
||||
@@ -90,13 +178,12 @@ fn deviceManager() ?ipc.Handle {
|
||||
|
||||
const OpenedStorage = struct { device_id: u64, device: block.Device };
|
||||
|
||||
/// The first mass-storage provider whose block channel actually opens, with its
|
||||
/// device id. A device-manager tree can carry more than one entry of the
|
||||
/// mass-storage identity — a phantom that no driver is bound to answers a
|
||||
/// consumer hello with NO channel — so this tries each and takes the first that
|
||||
/// yields a channel, exactly as a filesystem's own acquisition loop does.
|
||||
/// Called only when there is no volume (an insertion), so the hellos it makes
|
||||
/// are not per-poll churn.
|
||||
/// The first mass-storage provider whose block channel opens and is NOT already
|
||||
/// adopted, with its device id. A device-manager tree can carry more than one
|
||||
/// entry of the mass-storage identity — a phantom that no driver is bound to
|
||||
/// answers a consumer hello with NO channel — so this tries each and takes the
|
||||
/// first that yields a channel. Skips already-adopted devices so a re-poll does
|
||||
/// not re-open a device it already serves.
|
||||
fn openAnyStorage() ?OpenedStorage {
|
||||
const manager = deviceManager() orelse return null;
|
||||
const Entry = device_manager_protocol.ChildEntry;
|
||||
@@ -116,6 +203,7 @@ fn openAnyStorage() ?OpenedStorage {
|
||||
const entry = std.mem.bytesToValue(Entry, tail[index * @sizeOf(Entry) ..][0..@sizeOf(Entry)]);
|
||||
if (entry.device_id == device_manager_protocol.no_device) continue;
|
||||
if ((entry.identity >> 16) & 0xff != 0x08 or (entry.identity >> 8) & 0xff != 0x06) continue;
|
||||
if (deviceById(entry.device_id) != null) continue; // already adopted
|
||||
const exchanged = driver.helloOn(manager, .consumer, entry.device_id, null, true) orelse continue;
|
||||
const provider = exchanged.channel orelse continue; // a phantom / not-yet-bound entry
|
||||
return .{ .device_id = entry.device_id, .device = .{ .endpoint = provider } };
|
||||
@@ -126,7 +214,7 @@ fn openAnyStorage() ?OpenedStorage {
|
||||
|
||||
/// Whether `device_id` is still in the device-manager tree — a bare enumerate,
|
||||
/// no consumer-hello, so it is cheap to call every poll. This is how removal is
|
||||
/// detected: the specific device the mounted volume sits on disappears.
|
||||
/// detected: the specific device a mounted volume sits on disappears.
|
||||
fn isDevicePresent(device_id: u64) bool {
|
||||
const manager = deviceManager() orelse return false;
|
||||
const Entry = device_manager_protocol.ChildEntry;
|
||||
@@ -150,59 +238,94 @@ fn isDevicePresent(device_id: u64) bool {
|
||||
}
|
||||
}
|
||||
|
||||
/// Spawn the filesystem for `v`, confine it to the volume's range, and record
|
||||
/// its pid. The confinement is defined for the fresh pid BEFORE the filesystem
|
||||
/// runs, so its first read is already bounded; the volume manager is the
|
||||
/// confinement controller (it defines the first range on the device).
|
||||
// --- lifecycle ---------------------------------------------------------------
|
||||
|
||||
/// Spawn the filesystem for `v`, confine it to the volume's range on its device's
|
||||
/// channel, and record its pid. The confinement is defined for the fresh pid
|
||||
/// BEFORE the filesystem runs, so its first read is already bounded; the volume
|
||||
/// manager is the confinement controller (it defines the first range on the
|
||||
/// device).
|
||||
fn spawnFilesystem(v: *Volume) void {
|
||||
if (fs_failed) return;
|
||||
const pid = process.spawnSupervised(filesystem_binary, &.{"1"}, service_endpoint) orelse {
|
||||
if (v.failed) return;
|
||||
const dev = deviceById(v.device_id) orelse return; // its device left — poll will clean up
|
||||
var id_str_buf: [id_decimal_bytes]u8 = undefined;
|
||||
const id_str = std.fmt.bufPrint(&id_str_buf, "{d}", .{v.id}) catch "1";
|
||||
const pid = process.spawnSupervised(v.binary, &.{ id_str, v.mount_prefix }, service_endpoint) orelse {
|
||||
_ = logging.write("volume-manager: could not spawn the filesystem; retrying\n");
|
||||
armRestart();
|
||||
armRestart(v);
|
||||
return;
|
||||
};
|
||||
if (!v.storage.defineRange(pid, v.base_lba, v.block_count)) {
|
||||
if (!dev.channel.defineRange(pid, v.base_lba, v.block_count)) {
|
||||
_ = logging.write("volume-manager: could not confine the filesystem to its volume; retrying\n");
|
||||
_ = process.kill(pid);
|
||||
armRestart();
|
||||
armRestart(v);
|
||||
return;
|
||||
}
|
||||
v.filesystem_pid = pid;
|
||||
fs_spawn_ns = time.clock();
|
||||
std.log.info("volume 0x{x} -> {s} (pid {d}), lba {d}, {d} blocks", .{ v.identity.key, filesystem_binary, pid, v.base_lba, v.block_count });
|
||||
v.spawn_ns = time.clock();
|
||||
std.log.info("volume 0x{x} -> {s} (pid {d}), lba {d}, {d} blocks", .{ v.identity.key, v.binary, pid, v.base_lba, v.block_count });
|
||||
}
|
||||
|
||||
/// Schedule a fat restart after backoff; the poll loop performs it once due.
|
||||
fn armRestart() void {
|
||||
const delay = if (fs_restarts == 0) backoff_base_ms else backoff_base_ms << @intCast(@min(fs_restarts - 1, 5));
|
||||
restart_due_ns = time.clock() + delay * 1_000_000;
|
||||
restart_pending = true;
|
||||
/// Schedule a restart for `v` after backoff; the poll loop performs it once due.
|
||||
fn armRestart(v: *Volume) void {
|
||||
const delay = if (v.restarts == 0) backoff_base_ms else backoff_base_ms << @intCast(@min(v.restarts - 1, 5));
|
||||
v.restart_due_ns = time.clock() + delay * 1_000_000;
|
||||
v.restart_pending = true;
|
||||
}
|
||||
|
||||
/// A storage provider just appeared: open its channel, read block 0, parse the
|
||||
/// volume, and spawn its filesystem. On any failure the channel is closed (so a
|
||||
/// present-but-unreadable device does not leak a handle every poll) and `volume`
|
||||
/// stays null — the next poll retries. A fresh medium gets a fresh supervision
|
||||
/// budget.
|
||||
fn bringUpVolume() void {
|
||||
/// Compose a volume's mount path (its id-path `/volumes/<id>`, or a volumes.csv
|
||||
/// override) into its slot's buffer, and return the slice.
|
||||
fn composeMountPrefix(slot: usize, identity: partition.Identity) []const u8 {
|
||||
var id_buf: [volume_map.id_maximum]u8 = undefined;
|
||||
const id = volume_map.idString(identity, &id_buf);
|
||||
return volume_map.overrideFor(volume_rules[0..volume_rule_count], id) orelse
|
||||
(std.fmt.bufPrint(&mount_prefix_bufs[slot], "/volumes/{s}", .{id}) catch "/volumes/unknown");
|
||||
}
|
||||
|
||||
/// Adopt the next present, not-yet-adopted storage device: take its channel,
|
||||
/// probe its whole partition table, and spawn a filesystem per volume it carries.
|
||||
/// Returns true when it consumed a device (so the caller can loop to adopt every
|
||||
/// present device in one tick), false when none remain or the device table is full.
|
||||
///
|
||||
/// A device is adopted exactly once and kept until it leaves the tree — even when
|
||||
/// it carries no volume we can serve, or its geometry cannot be read. Keeping the
|
||||
/// empty/unreadable device adopted (rather than dropping and re-probing) is what
|
||||
/// lets openAnyStorage advance PAST it to the devices behind it; dropping it would
|
||||
/// make openAnyStorage hand back the same unservable device every tick and starve
|
||||
/// the rest. A genuine removal frees the slot (removeDevice); a re-insert gets a
|
||||
/// fresh device id and is probed anew.
|
||||
fn bringUpVolume() bool {
|
||||
if (!bounce_ready) {
|
||||
bounce = memory.dmaAlloc(512, memory.dma_coherent | memory.dma_shareable) orelse return;
|
||||
bounce = memory.dmaAlloc(512, memory.dma_coherent | memory.dma_shareable) orelse return false;
|
||||
bounce_ready = true;
|
||||
}
|
||||
const opened = openAnyStorage() orelse return;
|
||||
const opened = openAnyStorage() orelse return false;
|
||||
const dev = claimDevice() orelse {
|
||||
_ = logging.write("volume-manager: device table full; a storage device is left unadopted\n");
|
||||
_ = ipc.close(opened.device.endpoint);
|
||||
return false;
|
||||
};
|
||||
dev.* = .{ .used = true, .device_id = opened.device_id, .channel = opened.device };
|
||||
// Consume this device's medium_changed events (the second of the removal
|
||||
// lifecycle's two triggers: the device stays in the tree while its medium
|
||||
// leaves — a card reader, an eject). Best effort: a provider that never
|
||||
// publishes the event simply never wakes us, and device-pull is still caught
|
||||
// by the presence poll.
|
||||
_ = opened.device.subscribeMedium(service_endpoint);
|
||||
const device = opened.device;
|
||||
// Attach the read buffer to THIS device (a no-op without an enforcing IOMMU).
|
||||
// The handle is kept, not closed, so it can be re-attached to the next
|
||||
// device after a replug.
|
||||
// The handle is kept, not closed, so it can be re-attached after a replug. A
|
||||
// failed attach or geometry read leaves the device adopted but empty — we just
|
||||
// cannot read it, and the slot still watches it for removal.
|
||||
if (bounce.handle) |handle| {
|
||||
if (!device.attach(handle)) {
|
||||
_ = ipc.close(device.endpoint);
|
||||
return;
|
||||
_ = logging.write("volume-manager: could not attach the read buffer to a storage device; no volume served\n");
|
||||
return true;
|
||||
}
|
||||
}
|
||||
const geometry = device.geometry() orelse {
|
||||
_ = ipc.close(device.endpoint);
|
||||
return;
|
||||
_ = logging.write("volume-manager: could not read a storage device's geometry; no volume served\n");
|
||||
return true;
|
||||
};
|
||||
const ProbeReader = struct {
|
||||
device: block.Device,
|
||||
@@ -216,59 +339,104 @@ fn bringUpVolume() void {
|
||||
};
|
||||
var probe = ProbeReader{ .device = device };
|
||||
const reader = partition.SectorReader{ .context = &probe, .readFn = ProbeReader.readSector };
|
||||
const found = partition.firstVolume(reader, geometry.block_count) orelse {
|
||||
if (!logged_no_volume) {
|
||||
_ = logging.write("volume-manager: storage present but no recognizable volume\n");
|
||||
logged_no_volume = true;
|
||||
}
|
||||
_ = ipc.close(device.endpoint);
|
||||
return;
|
||||
};
|
||||
logged_no_volume = false;
|
||||
fs_restarts = 0;
|
||||
fs_failed = false;
|
||||
restart_pending = false;
|
||||
volume = .{ .storage = device, .storage_device_id = opened.device_id, .base_lba = found.base_lba, .block_count = found.block_count, .identity = found.identity, .id = volume_id };
|
||||
spawnFilesystem(&volume.?);
|
||||
var found: [maximum_volumes]partition.Volume = undefined;
|
||||
const n = partition.allVolumes(reader, geometry.block_count, found[0..]);
|
||||
if (n == 0) {
|
||||
std.log.info("device {d} present but carries no recognizable volume", .{dev.device_id});
|
||||
return true;
|
||||
}
|
||||
for (found[0..n]) |fv| {
|
||||
// Pick the service binary from the volume's content signature. A signature
|
||||
// no filesystems.csv row serves goes unserved (logged), like an unbound
|
||||
// device — the manager does not guess.
|
||||
const binary = filesystem_map.match(filesystem_rules[0..filesystem_rule_count], fv.signature) orelse {
|
||||
_ = logging.write("volume-manager: no filesystem serves this volume's content; unserved\n");
|
||||
continue;
|
||||
};
|
||||
const slot = claimVolumeIndex() orelse {
|
||||
_ = logging.write("volume-manager: volume table full; a volume is left unserved\n");
|
||||
break;
|
||||
};
|
||||
volumes[slot] = .{
|
||||
.used = true,
|
||||
.device_id = dev.device_id,
|
||||
.base_lba = fv.base_lba,
|
||||
.block_count = fv.block_count,
|
||||
.identity = fv.identity,
|
||||
.id = next_volume_id,
|
||||
.binary = binary,
|
||||
.mount_prefix = composeMountPrefix(slot, fv.identity),
|
||||
};
|
||||
next_volume_id += 1;
|
||||
spawnFilesystem(&volumes[slot]);
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
/// The storage provider left the device tree (a pulled stick): kill the
|
||||
/// filesystem so its mounts are retired. Retirement is lazy, not an eager
|
||||
/// death-time sweep — killing the process marks the filesystem's backend
|
||||
/// endpoint dead, and the VFS router drops each mount that endpoint backed on
|
||||
/// the next path resolution under it (that resolve frees the slot and returns
|
||||
/// not_found). Then drop the now-dead channel and clear the volume; the next
|
||||
/// poll that sees storage return re-mounts.
|
||||
fn removeVolume() void {
|
||||
const v = volume orelse return;
|
||||
/// Close a device's channel and free its slot. No volumes are touched (the caller
|
||||
/// ensures none remain, or there never were any).
|
||||
fn dropDevice(dev: *StorageDevice) void {
|
||||
// Free the driver's subscriber slot before the channel closes. On a still-live
|
||||
// channel (a medium eject) this frees the slot; on a dead one (a device pull)
|
||||
// the call fails fast and the exit sweep frees it anyway.
|
||||
_ = dev.channel.unsubscribeMedium();
|
||||
_ = ipc.close(dev.channel.endpoint);
|
||||
dev.* = .{};
|
||||
}
|
||||
|
||||
/// Retire one volume: kill its filesystem so its mounts are retired. Retirement
|
||||
/// is lazy, not an eager death-time sweep — killing the process marks the
|
||||
/// filesystem's backend endpoint dead, and the VFS router drops each mount that
|
||||
/// endpoint backed on the next path resolution under it (that resolve frees the
|
||||
/// slot and returns not_found). Then free the volume slot.
|
||||
fn removeVolumeState(v: *Volume) void {
|
||||
std.log.info("storage for volume {d} removed; unmounting", .{v.id});
|
||||
if (v.filesystem_pid != 0) _ = process.kill(v.filesystem_pid);
|
||||
_ = ipc.close(v.storage.endpoint);
|
||||
volume = null;
|
||||
restart_pending = false;
|
||||
fs_restarts = 0;
|
||||
fs_failed = false;
|
||||
v.* = .{};
|
||||
}
|
||||
|
||||
/// One poll tick. Removal is checked FIRST and supersedes a pending restart: if
|
||||
/// the device is gone there is nothing to restart fat onto, and respawning it
|
||||
/// against the dead channel would just churn until the crash cap. Only once the
|
||||
/// device is confirmed present does a due restart fire.
|
||||
fn pollTick() void {
|
||||
if (volume) |v| {
|
||||
// Serving: watch for the specific device leaving (a pulled stick).
|
||||
if (!isDevicePresent(v.storage_device_id)) {
|
||||
removeVolume();
|
||||
return;
|
||||
}
|
||||
if (restart_pending and time.clock() >= restart_due_ns) {
|
||||
restart_pending = false;
|
||||
spawnFilesystem(&volume.?);
|
||||
}
|
||||
} else {
|
||||
// Idle: try to bring a present storage device up.
|
||||
bringUpVolume();
|
||||
/// A storage device left the tree (a pulled stick): retire every volume it served
|
||||
/// and drop its channel. One removal path, whether the device is pulled cleanly
|
||||
/// or vanishes.
|
||||
fn removeDevice(dev: *StorageDevice) void {
|
||||
for (&volumes) |*v| {
|
||||
if (v.used and v.device_id == dev.device_id) removeVolumeState(v);
|
||||
}
|
||||
dropDevice(dev);
|
||||
}
|
||||
|
||||
/// Whether a device's block channel still answers — a geometry() probe. A storage
|
||||
/// driver that DIED while its device stays in the tree (it crashed; the device
|
||||
/// manager will re-delegate the device to a restarted driver on a FRESH channel)
|
||||
/// leaves a dead channel here, even though isDevicePresent still reports the device
|
||||
/// present. geometry() on the dead endpoint fails fast, so this catches the crash
|
||||
/// that presence-polling alone cannot — the V4 review's open edge.
|
||||
fn channelAlive(dev: *StorageDevice) bool {
|
||||
return dev.channel.geometry() != null;
|
||||
}
|
||||
|
||||
/// One poll tick. Device removal is reconciled FIRST and supersedes a pending
|
||||
/// restart: a volume whose device left (a pull) OR whose driver died on a channel
|
||||
/// that no longer answers is retired before its restart could fire, so nothing
|
||||
/// respawns against a dead channel. Dropping the device frees its slot, so the
|
||||
/// adopt loop below re-adopts the still-present device on the restarted driver's
|
||||
/// fresh channel — the rebuild. Then due restarts fire for present volumes.
|
||||
fn pollTick() void {
|
||||
for (&devices) |*dev| {
|
||||
if (dev.used and (!isDevicePresent(dev.device_id) or !channelAlive(dev))) removeDevice(dev);
|
||||
}
|
||||
for (&volumes) |*v| {
|
||||
if (v.used and v.restart_pending and time.clock() >= v.restart_due_ns) {
|
||||
v.restart_pending = false;
|
||||
spawnFilesystem(v);
|
||||
}
|
||||
}
|
||||
// Adopt every present, not-yet-adopted storage device. Each call consumes at
|
||||
// most one device (openAnyStorage skips the adopted), so the loop terminates
|
||||
// once none remain; the maximum_devices guard is insurance against a logic
|
||||
// slip, never the normal exit.
|
||||
var adopted: usize = 0;
|
||||
while (adopted < maximum_devices and bringUpVolume()) : (adopted += 1) {}
|
||||
}
|
||||
|
||||
/// A filesystem announces itself for the volume it was spawned to serve. Reply
|
||||
@@ -276,29 +444,79 @@ fn pollTick() void {
|
||||
/// badge) as the call's returned capability. No channel means the volume is not
|
||||
/// ready — the filesystem retries.
|
||||
fn onHello(_: void, invocation: Invocation(volume_manager_protocol.Hello), _: Answer(void)) isize {
|
||||
const v = volume orelse return 0; // not probed yet — retryable, no cap
|
||||
if (invocation.target != v.id) return 0; // unknown volume — retryable
|
||||
const v = volumeById(invocation.target) orelse return 0; // not probed yet — retryable, no cap
|
||||
if (invocation.sender != v.filesystem_pid) {
|
||||
// Not the filesystem we spawned for this volume. Refuse: only the
|
||||
// confined filesystem gets the channel.
|
||||
// Not the filesystem we spawned for this volume. Refuse: only the confined
|
||||
// filesystem gets the channel.
|
||||
std.log.info("refused hello for volume {d} from process {d}", .{ invocation.target, invocation.sender });
|
||||
return -envelope.EPERM;
|
||||
}
|
||||
service.replyWithCapability(v.storage.endpoint);
|
||||
const dev = deviceById(v.device_id) orelse return 0; // its device left — retryable
|
||||
service.replyWithCapability(dev.channel.endpoint);
|
||||
std.log.info("handed volume {d} to pid {d}", .{ v.id, invocation.sender });
|
||||
return 0;
|
||||
}
|
||||
|
||||
const handlers = Serve.Handlers{ .hello = onHello };
|
||||
/// Answer a `volumes` query with a mounted volume's descriptor — its id (its
|
||||
/// mount path is /volumes/<id> unless overridden), its actual mount path, and its
|
||||
/// display label. Software keys on the id; a UI shows the label. Returns the first
|
||||
/// mounted volume for now; a full enumerate is a later refinement. Empty reply
|
||||
/// means no volume is mounted.
|
||||
fn onVolumes(_: void, _: Invocation(volume_manager_protocol.Volumes), answer: Answer(void)) isize {
|
||||
const v = firstUsedVolume() orelse return 0;
|
||||
var id_buf: [volume_map.id_maximum]u8 = undefined;
|
||||
const info = volume_manager_protocol.VolumeInfo{
|
||||
.id = volume_map.idString(v.identity, &id_buf),
|
||||
.mount_path = v.mount_prefix,
|
||||
.label = v.identity.labelSlice(),
|
||||
};
|
||||
const encoded = info.encode(answer.tail()) orelse return 0;
|
||||
return @intCast(encoded.len);
|
||||
}
|
||||
|
||||
const handlers = Serve.Handlers{ .hello = onHello, .volumes = onVolumes };
|
||||
|
||||
fn onMessage(message: []const u8, out: []u8, sender: u32, arrived: *ipc.Arrival) usize {
|
||||
// No verb takes a capability up, so the turn closes whatever arrives.
|
||||
return Serve.dispatch({}, handlers, message, sender, arrived.peek(), out);
|
||||
}
|
||||
|
||||
/// Read a config file into `buf`, returning the byte count (0 if missing).
|
||||
fn readConfig(path: []const u8, buf: []u8) usize {
|
||||
var file = fs.open(path, .{}) orelse {
|
||||
std.log.info("volume-manager: {s} missing", .{path});
|
||||
return 0;
|
||||
};
|
||||
defer file.close();
|
||||
var used: usize = 0;
|
||||
while (used < buf.len) {
|
||||
const nn = file.read(buf[used..]) orelse break;
|
||||
if (nn == 0) break;
|
||||
used += nn;
|
||||
}
|
||||
return used;
|
||||
}
|
||||
|
||||
/// Load the mount map from configuration once at boot (mirrors the device
|
||||
/// manager's registry load). A missing or empty filesystems.csv means no volume
|
||||
/// is served; volumes.csv is optional — no rows means every volume takes its
|
||||
/// default /volumes/<id> path.
|
||||
fn loadTables() void {
|
||||
const fs_used = readConfig("/system/configuration/filesystems.csv", &filesystems_source);
|
||||
const fr = filesystem_map.parse(filesystems_source[0..fs_used], &filesystem_rules);
|
||||
filesystem_rule_count = fr.count;
|
||||
if (fr.malformed != 0 or fr.truncated) std.log.info("filesystems.csv: {d} malformed, truncated={}", .{ fr.malformed, fr.truncated });
|
||||
|
||||
const vol_used = readConfig("/system/configuration/volumes.csv", &volumes_source);
|
||||
const vr = volume_map.parse(volumes_source[0..vol_used], &volume_rules);
|
||||
volume_rule_count = vr.count;
|
||||
if (vr.malformed != 0 or vr.truncated) std.log.info("volumes.csv: {d} malformed, truncated={}", .{ vr.malformed, vr.truncated });
|
||||
}
|
||||
|
||||
fn initialise(endpoint: ipc.Handle) bool {
|
||||
service_endpoint = endpoint;
|
||||
_ = logging.write("volume-manager: starting, waiting for a storage device\n");
|
||||
loadTables();
|
||||
_ = process.subscribeExits(endpoint);
|
||||
pollTick();
|
||||
_ = time.timerOnce(endpoint, poll_interval_ms); // the poll runs for the life of the boot
|
||||
@@ -318,23 +536,59 @@ fn onNotification(badge: u64) void {
|
||||
// reclaimed by the driver on the same death; the respawn confines afresh.
|
||||
if (got.isChildExit()) {
|
||||
const dead = got.childProcessId();
|
||||
const v = &(volume orelse return);
|
||||
if (v.filesystem_pid != dead) return;
|
||||
const v = volumeByPid(dead) orelse return;
|
||||
v.filesystem_pid = 0;
|
||||
const reason = process.exitReason(dead) orelse .fault;
|
||||
if (reason == .exited) {
|
||||
std.log.info("filesystem for volume {d} exited cleanly; not restarting", .{v.id});
|
||||
return;
|
||||
}
|
||||
const alive = time.clock() -| fs_spawn_ns;
|
||||
fs_restarts = if (alive < fast_death_ns) fs_restarts + 1 else 1;
|
||||
if (fs_restarts >= crash_loop_cap) {
|
||||
fs_failed = true;
|
||||
const alive = time.clock() -| v.spawn_ns;
|
||||
v.restarts = if (alive < fast_death_ns) v.restarts + 1 else 1;
|
||||
if (v.restarts >= crash_loop_cap) {
|
||||
v.failed = true;
|
||||
std.log.info("filesystem for volume {d} is failing repeatedly; giving up", .{v.id});
|
||||
return;
|
||||
}
|
||||
std.log.info("filesystem for volume {d} died ({s}); restarting", .{ v.id, @tagName(reason) });
|
||||
armRestart();
|
||||
armRestart(v);
|
||||
}
|
||||
}
|
||||
|
||||
fn anyVolumeOn(device_id: u64) bool {
|
||||
for (&volumes) |*v| if (v.used and v.device_id == device_id) return true;
|
||||
return false;
|
||||
}
|
||||
|
||||
/// A storage device published `medium_changed` — the second removal trigger: the
|
||||
/// device stays in the tree while its medium leaves or returns (a card reader, an
|
||||
/// eject). This arrives as a buffered async message, NOT a protocol request, so it
|
||||
/// never reaches `Serve.dispatch` (its event op number collides with the manager's
|
||||
/// own `hello`); it is decoded here by hand. Single-volume scope: the event names
|
||||
/// no device, so `absent` retires every adopted device (its volumes unmount and
|
||||
/// the poll re-adopts the still-present device with its now-empty medium), and
|
||||
/// `present` frees any empty adopted device so the poll re-probes and remounts it.
|
||||
///
|
||||
/// We act on every edge and do NOT dedup on `change_count`. The driver publishes
|
||||
/// exactly once per transition, each with a unique monotonic count, so a count is
|
||||
/// never legitimately re-sent within one subscription — an equality dedup could
|
||||
/// only ever fire spuriously, and it did: `change_count` restarts at 0 in each
|
||||
/// driver instance (usb-storage.zig), so a global "last count" carried across a
|
||||
/// driver restart (S5's own crash-rebuild) mistook the fresh instance's first
|
||||
/// edge for a re-delivery and dropped a real eject, wedging a mount over absent
|
||||
/// media. Both branches are idempotent (a freed device stops matching `dev.used`)
|
||||
/// and the poll reconciles, so reacting to each genuine edge is safe.
|
||||
fn onMediumEvent(payload: []const u8) void {
|
||||
const event = block.decodeMediumChanged(payload) orelse return;
|
||||
if (event.present == 0) {
|
||||
std.log.info("medium left a storage device; unmounting its volume(s)", .{});
|
||||
for (&devices) |*dev| {
|
||||
if (dev.used) removeDevice(dev);
|
||||
}
|
||||
} else {
|
||||
for (&devices) |*dev| {
|
||||
if (dev.used and !anyVolumeOn(dev.device_id)) removeDevice(dev);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -345,5 +599,6 @@ pub fn main(init: process.Init) void {
|
||||
.init = initialise,
|
||||
.on_message = onMessage,
|
||||
.on_notification = onNotification,
|
||||
.on_buffered_message = onMediumEvent,
|
||||
});
|
||||
}
|
||||
|
||||
@@ -0,0 +1,139 @@
|
||||
//! volume-map — render a volume's identity into its stable mount id-string, and
|
||||
//! parse `/system/configuration/volumes.csv` (danos's fstab) into optional
|
||||
//! id → mount-prefix overrides. This is where the id/label split becomes the
|
||||
//! path: a volume's mount point is derived from its content identity (the id),
|
||||
//! never from a port or a label. Two distinct volumes that share a label get
|
||||
//! distinct id-strings automatically; only identical ids (dd-cloned media) can
|
||||
//! collide, which is the narrow case the manager's duplicate policy is for.
|
||||
//!
|
||||
//! `volumes.csv` is an OPTIONAL override: a row `id, mount_prefix` pins a volume
|
||||
//! (by its id-string) to a chosen path. A volume with no row takes its default
|
||||
//! `/volumes/<id>`. The label is display metadata, exposed by the manager's
|
||||
//! `volumes` query, and never appears here.
|
||||
//!
|
||||
//! Pure logic: no syscalls, no allocator. Override slices point into the CSV
|
||||
//! source, which the manager holds in a static buffer for the process life.
|
||||
|
||||
const std = @import("std");
|
||||
const csv = @import("csv");
|
||||
const partition = @import("partition.zig");
|
||||
|
||||
/// bound: bytes of the longest volume id-string the deriver renders
|
||||
/// decided-by: ours
|
||||
/// protects: the caller's id-string buffer
|
||||
/// at-limit: truncate - bufPrint fails and idString returns ""; the volume goes
|
||||
/// unnamed and the manager logs it rather than mounting at an empty path
|
||||
/// observed-by: a volume with an empty id in the `volumes` query / the log
|
||||
pub const id_maximum = 40; // "gpt-" (4) or "uuid-" (5) + 32 hex fits in 40
|
||||
|
||||
/// One parsed override row: a volume id-string and the mount prefix it pins to.
|
||||
pub const Override = struct { id: []const u8, prefix: []const u8 };
|
||||
|
||||
/// How many overrides landed, how many non-blank lines were malformed, and
|
||||
/// whether there were more rows than the buffer could hold.
|
||||
pub const ParseResult = struct { count: usize, malformed: usize, truncated: bool };
|
||||
|
||||
/// Render a volume's identity into its id-string — the content-derived, unique,
|
||||
/// order-independent token whose default mount path is `/volumes/<id>`. The rung
|
||||
/// tags the scheme so ids never collide across rungs; the key is the content id,
|
||||
/// so a moved drive keeps its id (and thus its path).
|
||||
pub fn idString(identity: partition.Identity, buf: []u8) []const u8 {
|
||||
return switch (identity.rung) {
|
||||
.gpt_guid => std.fmt.bufPrint(buf, "gpt-{x:0>32}", .{identity.key}) catch "",
|
||||
.filesystem_uuid => std.fmt.bufPrint(buf, "uuid-{x:0>32}", .{identity.key}) catch "",
|
||||
.fat_serial => std.fmt.bufPrint(buf, "fat-{x:0>8}", .{@as(u32, @truncate(identity.key))}) catch "",
|
||||
.exfat_serial => std.fmt.bufPrint(buf, "exfat-{x:0>8}", .{@as(u32, @truncate(identity.key))}) catch "",
|
||||
.mbr_index => std.fmt.bufPrint(buf, "mbr-{x}-{d}", .{
|
||||
@as(u32, @truncate(identity.key >> 8)),
|
||||
@as(u8, @truncate(identity.key & 0xff)),
|
||||
}) catch "",
|
||||
.anonymous => std.fmt.bufPrint(buf, "anon-{x}", .{identity.key}) catch "",
|
||||
};
|
||||
}
|
||||
|
||||
const Line = union(enum) { override: Override, ignorable, malformed };
|
||||
|
||||
fn parseLine(line: []const u8) Line {
|
||||
const body = csv.stripComment(line);
|
||||
if (body.len == 0) return .ignorable;
|
||||
var it = csv.fields(body);
|
||||
const id = it.next() orelse return .malformed;
|
||||
const prefix = it.next() orelse return .malformed;
|
||||
if (it.next() != null) return .malformed; // too many columns
|
||||
if (id.len == 0 or prefix.len == 0) return .malformed;
|
||||
return .{ .override = .{ .id = id, .prefix = prefix } };
|
||||
}
|
||||
|
||||
/// Parse a whole `volumes.csv` into `out_rules`. The slices point into `source`,
|
||||
/// which must outlive them.
|
||||
pub fn parse(source: []const u8, out_rules: []Override) ParseResult {
|
||||
var result: ParseResult = .{ .count = 0, .malformed = 0, .truncated = false };
|
||||
var lines = std.mem.splitScalar(u8, source, '\n');
|
||||
while (lines.next()) |line| {
|
||||
switch (parseLine(line)) {
|
||||
.ignorable => {},
|
||||
.malformed => result.malformed += 1,
|
||||
.override => |ov| {
|
||||
if (result.count >= out_rules.len) {
|
||||
result.truncated = true;
|
||||
continue;
|
||||
}
|
||||
out_rules[result.count] = ov;
|
||||
result.count += 1;
|
||||
},
|
||||
}
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
/// The override mount prefix for a volume whose id-string is `id`, or null (the
|
||||
/// volume takes its default `/volumes/<id>` path). First matching row wins.
|
||||
pub fn overrideFor(rules: []const Override, id: []const u8) ?[]const u8 {
|
||||
for (rules) |rule| {
|
||||
if (std.mem.eql(u8, rule.id, id)) return rule.prefix;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
// --- tests -------------------------------------------------------------------
|
||||
|
||||
const testing = std.testing;
|
||||
|
||||
// Named fixture sizes so the bounds gate (which flags literal array lengths)
|
||||
// stays quiet: test inputs, not runtime ceilings.
|
||||
const test_override_slots = 4;
|
||||
|
||||
test "idString renders each rung's id token" {
|
||||
var buf: [id_maximum]u8 = undefined;
|
||||
try testing.expectEqualStrings("fat-12345678", idString(.{ .rung = .fat_serial, .key = 0x12345678 }, &buf));
|
||||
try testing.expectEqualStrings("exfat-da7a0001", idString(.{ .rung = .exfat_serial, .key = 0xDA7A0001 }, &buf));
|
||||
try testing.expectEqualStrings("mbr-deadbeef-1", idString(.{ .rung = .mbr_index, .key = (@as(u128, 0xDEADBEEF) << 8) | 1 }, &buf));
|
||||
const guid: u128 = 0x00112233445566778899AABBCCDDEEFF;
|
||||
try testing.expectEqualStrings("gpt-00112233445566778899aabbccddeeff", idString(.{ .rung = .gpt_guid, .key = guid }, &buf));
|
||||
}
|
||||
|
||||
test "overrideFor returns the mapped prefix, else null" {
|
||||
const text =
|
||||
\\# id, mount_prefix
|
||||
\\fat-12345678, /mnt/boot
|
||||
;
|
||||
var rules: [test_override_slots]Override = undefined;
|
||||
const parsed = parse(text, &rules);
|
||||
try testing.expectEqual(@as(usize, 1), parsed.count);
|
||||
try testing.expectEqual(@as(usize, 0), parsed.malformed);
|
||||
try testing.expectEqualStrings("/mnt/boot", overrideFor(rules[0..parsed.count], "fat-12345678").?);
|
||||
try testing.expect(overrideFor(rules[0..parsed.count], "fat-99999999") == null);
|
||||
}
|
||||
|
||||
test "malformed volume rows are counted, not bound" {
|
||||
const text =
|
||||
\\fat-1, /mnt/a
|
||||
\\onlyonecolumn
|
||||
\\fat-2,
|
||||
\\fat-3, /a, /b
|
||||
;
|
||||
var rules: [test_override_slots]Override = undefined;
|
||||
const parsed = parse(text, &rules);
|
||||
try testing.expectEqual(@as(usize, 1), parsed.count); // only the first valid row
|
||||
try testing.expectEqual(@as(usize, 3), parsed.malformed); // one column, empty prefix, too many columns
|
||||
}
|
||||
+140
-5
@@ -179,7 +179,7 @@ CASES = [
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"qemu_extra": ["-device", "intel-iommu,intremap=off"],
|
||||
"expect": r"(?s)(?=.*/system/kernel: iommu online)(?=.*fat: mounted /volumes/usb)(?=.*fat-test: ok)",
|
||||
"expect": r"(?s)(?=.*/system/kernel: iommu online)(?=.*fat: mounted /volumes/fat-12345678)(?=.*fat-test: ok)",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL|DANOS-IOMMU-FAULT"},
|
||||
# DMA + MSI under translation: interrupt-IN reports arrive through translated DMA and
|
||||
# the xHC's MSI/MSI-X still delivers (the 0xFEE00000 interrupt window bypasses second-
|
||||
@@ -215,7 +215,7 @@ CASES = [
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"qemu_extra": ["-device", "amd-iommu,dma-remap=on,intremap=off"],
|
||||
"expect": r"(?s)(?=.*iommu online \(AMD-Vi\))(?=.*fat: mounted /volumes/usb)(?=.*fat-test: ok)",
|
||||
"expect": r"(?s)(?=.*iommu online \(AMD-Vi\))(?=.*fat: mounted /volumes/fat-12345678)(?=.*fat-test: ok)",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL|DANOS-IOMMU-FAULT"},
|
||||
# Port I/O grants: a claimed device's io_port resource lets a driver read/write its
|
||||
# ports (PS/2 status 0x64), gated by the claim; out-of-range/unclaimed is refused.
|
||||
@@ -749,13 +749,26 @@ CASES = [
|
||||
"expect": r"usb-storage: ready[\s\S]*usb-storage: block 0 signature 0x55aa",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# FAT mount end to end: the fat server mounts the boot usb-storage device (the
|
||||
# FAT32 image) into the VFS at /volumes/usb. A fat-test client then lists and reads
|
||||
# FAT32 image) into the VFS at /volumes/fat-12345678. A fat-test client then lists and reads
|
||||
# through the mount — proof of the whole stack: block device -> FAT parse ->
|
||||
# VFS routing -> file read.
|
||||
{"name": "fat-mount",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"expect": r"fat: mounted /volumes/usb[\s\S]*fat-test: ok",
|
||||
"expect": r"fat: mounted /volumes/fat-12345678[\s\S]*fat-test: ok",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# The id-path naming (S2, storage-stack-plan.md). The boot volume mounts at
|
||||
# its CONTENT-derived id-path (/volumes/fat-12345678, from the FAT32 serial
|
||||
# 0x12345678) — never a port name — and keeps its FHS rewrites so /system/logs
|
||||
# persistence still rides the volume. Discrimination: before S2's flip fat
|
||||
# hardcoded /volumes/usb, so the id-path mount line never appears. (Making the
|
||||
# rewrites content-conditional on which volume carries the system is S3.)
|
||||
{"name": "volume-identity-name",
|
||||
"build_case": "fat-mount",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"expect": r"(?s)fat: mounted /volumes/fat-12345678"
|
||||
r"[\s\S]*fat: mounted /system/logs",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# The removal lifecycle (V4, docs/volume-manager-plan.md): pull the boot stick
|
||||
# mid-run. device_del the usb-storage device -> the bus reports the port empty
|
||||
@@ -780,9 +793,45 @@ CASES = [
|
||||
"qmp_sequence": [
|
||||
{"delay": 8, "command": "device_del", "arguments": {"id": "bootstorage"}},
|
||||
],
|
||||
"expect": r"(?s)fat: mounted /volumes/usb"
|
||||
"expect": r"(?s)fat: mounted /volumes/fat-12345678"
|
||||
r"[\s\S]*volume-manager: storage for volume \d+ removed; unmounting",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# S5 medium_changed: the SECOND removal trigger. QMP-eject the MEDIUM (the
|
||||
# block backend, not the device) — the usb-storage device stays in the tree,
|
||||
# but its TEST UNIT READY poll reports not-ready and publishes medium_changed
|
||||
# (absent). The volume manager, now a subscriber, runs the same unmount path as
|
||||
# a device pull. Discrimination: before S5 the manager never subscribed, so the
|
||||
# event reached no one and the mount persisted (device-presence polling cannot
|
||||
# see a medium leave while the device stays). One lifecycle, two triggers.
|
||||
{"name": "volume-medium-change",
|
||||
"build_case": "fat-mount",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"qmp_sequence": [
|
||||
{"delay": 8, "command": "eject", "arguments": {"device": "bootusb", "force": True}},
|
||||
],
|
||||
"expect": r"(?s)fat: mounted /volumes/fat-12345678"
|
||||
r"[\s\S]*usb-storage: medium absent"
|
||||
r"[\s\S]*volume-manager: medium left a storage device; unmounting"
|
||||
r"[\s\S]*volume-manager: storage for volume \d+ removed; unmounting",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# S5 storage-driver-crash rebuild. The device manager (test-storage-restart
|
||||
# mode) kills usb-storage once, ~2s in — after its volume mounted. The device
|
||||
# stays in the tree, so device-presence polling alone would leave fat wedged on
|
||||
# the dead channel; the volume manager's channel-liveness probe (a geometry()
|
||||
# that fails on the dead endpoint) must notice, reap the volume, and rebuild on
|
||||
# the restarted driver's fresh channel — a SECOND mount of the same id-path.
|
||||
# Discrimination: a pre-S5 manager checks only isDevicePresent (still true), so
|
||||
# it never reaps and the second mount never appears (it would restart fat on
|
||||
# the stale channel and crash-loop).
|
||||
{"name": "volume-driver-restart",
|
||||
"build_case": "volume-driver-restart",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"expect": r"(?s)fat: mounted /volumes/fat-12345678"
|
||||
r"[\s\S]*volume-manager: storage for volume \d+ removed; unmounting"
|
||||
r"[\s\S]*fat: mounted /volumes/fat-12345678",
|
||||
"fail": r"failing repeatedly; giving up|\[FAIL\]|DANOS-TEST-RESULT: FAIL"},
|
||||
# Volume-manager discovery + probe (V3a, docs/volume-manager-plan.md). Reuses
|
||||
# the fat-mount kernel build (the default boot now spawns the volume manager
|
||||
# from init.csv). It acquires the mass-storage block channel through the
|
||||
@@ -803,6 +852,64 @@ CASES = [
|
||||
"expect": r"volume-manager: volume 0x0*12345678 -> \S+ \(pid \d+\), lba \d+, \d+ blocks"
|
||||
r"[\s\S]*volume-manager: handed volume \d+ to pid \d+",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# S3 multi-volume: a SECOND usb-storage device (a generated data volume, serial
|
||||
# da7a0001, an empty FAT with no /system) plugged in beside the boot volume.
|
||||
# Proves the volume manager adopts BOTH devices and spawns a confined fat per
|
||||
# volume, each mounted at its own CONTENT id-path (/volumes/fat-<serial>); and
|
||||
# that boot-volume detection is by content — only the volume that carries
|
||||
# /system backs /system/configuration, while the data volume mounts at its
|
||||
# id-path alone. Against the pre-S3 one-device/one-volume manager the data
|
||||
# volume never mounts, so the da7a0001 lookaheads fail (toggle-demonstrated by
|
||||
# checking out the step-2 volume-manager.zig).
|
||||
{"name": "two-volumes",
|
||||
"build_case": "fat-mount",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"data_volume": {"serial": "DA7A0001", "label": "DATAVOL", "size_mib": 64},
|
||||
"expect": r"(?s)(?=.*volume-manager: volume 0x0*12345678 -> )"
|
||||
r"(?=.*volume-manager: volume 0x0*da7a0001 -> )"
|
||||
r"(?=.*fat: mounted /volumes/fat-12345678)"
|
||||
r"(?=.*fat: mounted /volumes/fat-da7a0001)"
|
||||
r"(?=.*carries the system tree)"
|
||||
r"(?=.*data volume; mounted at /volumes/fat-da7a0001)",
|
||||
"fail": r"data volume; mounted at /volumes/fat-12345678|DANOS-TEST-RESULT: FAIL"},
|
||||
# S3 shared-channel multi-volume: ONE usb-storage device carrying an MBR with
|
||||
# TWO FAT partitions (da7a0001 at lba 2048, da7a0002 at lba 83968). allVolumes
|
||||
# walks the table and the manager spawns a confined fat per partition on the
|
||||
# SAME block channel, each clamped to its own LBA range (usb-storage's
|
||||
# per-badge range table) — the path a pair of single-volume sticks (the
|
||||
# two-volumes case) does NOT exercise. The two mount lines sit at two DISTINCT
|
||||
# non-zero base_lbas on one device. Against the pre-uncap allVolumes (S3 step
|
||||
# 2, capped to one partition) only da7a0001 mounts, so the da7a0002 lookaheads
|
||||
# fail.
|
||||
{"name": "partitioned-volume",
|
||||
"build_case": "fat-mount",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"data_volume": {"partitions": [{"serial": "DA7A0001", "size_mib": 40},
|
||||
{"serial": "DA7A0002", "size_mib": 40}]},
|
||||
"expect": r"(?s)(?=.*volume 0x0*da7a0001 -> \S+ \(pid \d+\), lba 2048, )"
|
||||
r"(?=.*volume 0x0*da7a0002 -> \S+ \(pid \d+\), lba 83968, )"
|
||||
r"(?=.*fat: mounted /volumes/fat-da7a0001)"
|
||||
r"(?=.*fat: mounted /volumes/fat-da7a0002)",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# S4 second engine: a bare exFAT data device (serial e0fa0001) attached beside
|
||||
# the FAT boot volume. The volume manager content-routes it to the exFAT
|
||||
# service (not fat), which mounts it at its id-path /volumes/exfat-e0fa0001;
|
||||
# the exfat-test client then reads the seeded HELLO.TXT and mutates through the
|
||||
# mount (mkdir/write/rename/read/remove). Proves the second engine reuses the
|
||||
# shared harness end to end. Fails against pre-S4 (no exfat binary, csv row, or
|
||||
# VBR recognizer — the device would go to fat, which rejects the exFAT VBR).
|
||||
{"name": "exfat-volume",
|
||||
"build_case": "exfat-volume",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"data_volume": {"exfat": True, "serial": "E0FA0001", "size_mib": 48},
|
||||
"expect": r"(?s)(?=.*volume 0x0*e0fa0001 -> /system/services/exfat )"
|
||||
r"(?=.*exfat: mounted /volumes/exfat-e0fa0001)"
|
||||
r"(?=.*exfat-test: read HELLO.TXT ok)"
|
||||
r"(?=.*exfat-test: ok)",
|
||||
"fail": r"exfat-test: FAILED|DANOS-TEST-RESULT: FAIL"},
|
||||
# Phase 2b: mkdir/unlink through the mount. Reuses the fat-mount build — the
|
||||
# fat-test client, after listing, makes a directory, writes+reads a file inside
|
||||
# it, then removes the file, exercising the whole VFS -> fat mutation path.
|
||||
@@ -1409,6 +1516,34 @@ def run_case(arch, case):
|
||||
cmd[cmd.index("-m") + 1] = case["mem"]
|
||||
if case.get("qemu_extra"): # extra qemu args, e.g. -device intel-iommu for the IOMMU case
|
||||
cmd += case["qemu_extra"]
|
||||
# A multi-volume case attaches a second usb-storage device backed by a freshly
|
||||
# GENERATED data volume: a distinct-serial FAT32 with no /system tree, so the
|
||||
# volume manager mounts it at its own id-path and the fat process marks it a
|
||||
# data volume (never a system volume). Regenerated per run — no image is
|
||||
# committed to the tree (the user keeps the boot files copyable, not baked in).
|
||||
if case.get("data_volume"):
|
||||
dv = case["data_volume"]
|
||||
data_img = os.path.join(WORK, "data-volume.img")
|
||||
if dv.get("partitions"):
|
||||
# One device, an MBR with several FAT partitions: several volumes share
|
||||
# ONE block channel, each confined to its own LBA range.
|
||||
gen = [sys.executable, os.path.join(REPO, "tools", "make-partitioned-image.py"), data_img]
|
||||
for part in dv["partitions"]:
|
||||
gen += [part["serial"], str(part.get("size_mib", 40))]
|
||||
elif dv.get("exfat"):
|
||||
# One device, a bare exFAT volume — the second engine's medium.
|
||||
gen = [sys.executable, os.path.join(REPO, "tools", "make-exfat-image.py"),
|
||||
"--serial", dv["serial"], data_img, str(dv.get("size_mib", 48))]
|
||||
else:
|
||||
# One device, one bare FAT volume.
|
||||
gen = [sys.executable, os.path.join(REPO, "tools", "make-fat-image.py"),
|
||||
"--serial", dv["serial"], "--label", dv.get("label", "DATAVOL"),
|
||||
data_img, str(dv.get("size_mib", 64))]
|
||||
subprocess.run(gen, check=True, stdout=subprocess.DEVNULL)
|
||||
cmd += [
|
||||
"-drive", f"if=none,id=datausb,format=raw,file={data_img}",
|
||||
"-device", "usb-storage,bus=xhci.0,port=4,drive=datausb,removable=on,id=datastorage",
|
||||
]
|
||||
# A QMP control socket, always present (additive): how a case's `qmp_after`
|
||||
# hook injects host-side events into the guest mid-run. Kept under a short temp
|
||||
# dir, not WORK: a unix socket path is capped at ~104 bytes (sun_path), and a
|
||||
|
||||
@@ -38,7 +38,7 @@ const time = @import("time");
|
||||
/// A scratch file on the volume, so the node the intruder tries to write through
|
||||
/// is one nothing else reads. (A foreign write that *succeeded* would prove the
|
||||
/// bug — it must not also damage the boot volume proving it.)
|
||||
const held_path = "/volumes/usb/BADGE.TXT";
|
||||
const held_path = "/volumes/fat-12345678/BADGE.TXT";
|
||||
const held_contents = "held";
|
||||
|
||||
fn line(comptime format: []const u8, arguments: anytype) void {
|
||||
@@ -46,12 +46,12 @@ fn line(comptime format: []const u8, arguments: anytype) void {
|
||||
_ = logging.write(std.fmt.bufPrint(&buffer, format, arguments) catch return);
|
||||
}
|
||||
|
||||
/// The fat server mounts /volumes/usb only after the whole USB storage chain is
|
||||
/// The fat server mounts /volumes/fat-12345678 only after the whole USB storage chain is
|
||||
/// up, and both instances race it.
|
||||
fn waitForVolume() bool {
|
||||
var tries: u32 = 0;
|
||||
while (tries < 1400) : (tries += 1) {
|
||||
if (fs.openDirectory("/volumes/usb")) |opened| {
|
||||
if (fs.openDirectory("/volumes/fat-12345678")) |opened| {
|
||||
var directory = opened;
|
||||
directory.close();
|
||||
return true;
|
||||
@@ -79,7 +79,7 @@ pub fn main(init: process.Init) void {
|
||||
|
||||
fn own() void {
|
||||
if (!waitForVolume()) {
|
||||
_ = logging.write("badge-scope-test: FAILED (/volumes/usb never became available)\n");
|
||||
_ = logging.write("badge-scope-test: FAILED (/volumes/fat-12345678 never became available)\n");
|
||||
return;
|
||||
}
|
||||
var held = fs.open(held_path, .{ .create = true, .truncate = true }) orelse {
|
||||
@@ -142,7 +142,7 @@ fn own() void {
|
||||
|
||||
fn intrude(foreign_node: u64, foreign_layer: u32) void {
|
||||
if (!waitForVolume()) {
|
||||
_ = logging.write("badge-scope-test: FAILED (/volumes/usb never became available)\n");
|
||||
_ = logging.write("badge-scope-test: FAILED (/volumes/fat-12345678 never became available)\n");
|
||||
return;
|
||||
}
|
||||
const node_verdict = probeNode(foreign_node);
|
||||
|
||||
@@ -0,0 +1,15 @@
|
||||
//! The exfat-test test fixture as a binary package (docs/build-packages-plan.md):
|
||||
//! this file names the binary and EXACTLY the modules its source imports —
|
||||
//! build-support resolves each name from the domains this zon declares.
|
||||
|
||||
const std = @import("std");
|
||||
const build_support = @import("build-support");
|
||||
|
||||
pub fn build(b: *std.Build) void {
|
||||
const exe = build_support.userBinary(b, .{
|
||||
.name = "exfat-test",
|
||||
.root_source_file = b.path("exfat-test.zig"),
|
||||
.imports = &.{ "file-system", "logging", "process", "time" },
|
||||
});
|
||||
b.installArtifact(exe);
|
||||
}
|
||||
@@ -0,0 +1,14 @@
|
||||
.{
|
||||
.name = .exfat_test,
|
||||
.version = "0.0.0",
|
||||
.fingerprint = 0x77b19e3f7ee43ce3, // Changing this has security and trust implications.
|
||||
.minimum_zig_version = "0.16.0",
|
||||
.dependencies = .{
|
||||
// build-support supplies the shared recipe; kernel is implicit in
|
||||
// every binary (the root shim + link script live there). The rest
|
||||
// are exactly the homes of this binary's declared imports.
|
||||
.@"build-support" = .{ .path = "../../../../build-support" },
|
||||
.kernel = .{ .path = "../../../../library/kernel" },
|
||||
},
|
||||
.paths = .{""},
|
||||
}
|
||||
@@ -0,0 +1,88 @@
|
||||
//! test/system/services/exfat-test — a client that proves the exFAT mount end to
|
||||
//! end: it waits for the exfat server to mount the volume at /volumes/exfat-
|
||||
//! e0fa0001, reads the seeded HELLO.TXT off it, and exercises mkdir / write /
|
||||
//! read / rename / remove through the VFS (which routes the id-path to the exfat
|
||||
//! backend). Shipped in the initial_ramdisk; the `exfat-volume` QEMU case spawns
|
||||
//! it alongside init with an exFAT data device attached beside the FAT boot volume.
|
||||
|
||||
const std = @import("std");
|
||||
const fs = @import("file-system");
|
||||
const process = @import("process");
|
||||
const time = @import("time");
|
||||
const logging = @import("logging");
|
||||
|
||||
// The exFAT data device's content id-path — its VolumeSerialNumber is 0xE0FA0001
|
||||
// (the `exfat-volume` case passes --serial E0FA0001 to make-exfat-image.py).
|
||||
const mount = "/volumes/exfat-e0fa0001";
|
||||
|
||||
fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||
var line: [128]u8 = undefined;
|
||||
_ = logging.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
||||
}
|
||||
|
||||
pub fn main(init: process.Init) void {
|
||||
_ = init;
|
||||
|
||||
// Wait for the exfat server to bring up the USB storage chain and mount.
|
||||
var opened: ?fs.Directory = null;
|
||||
var tries: u32 = 0;
|
||||
while (opened == null and tries < 1400) : (tries += 1) {
|
||||
opened = fs.openDirectory(mount);
|
||||
if (opened == null) time.sleepMillis(50);
|
||||
}
|
||||
var dir = opened orelse {
|
||||
_ = logging.write("exfat-test: " ++ mount ++ " never became available\n");
|
||||
return;
|
||||
};
|
||||
var count: u32 = 0;
|
||||
var entry: fs.Entry = .{};
|
||||
while (dir.next(&entry)) {
|
||||
writeLine("exfat-test: entry '{s}' size={d}\n", .{ entry.name(), entry.size });
|
||||
count += 1;
|
||||
if (count > 32) break;
|
||||
}
|
||||
dir.close();
|
||||
|
||||
// Read the seeded HELLO.TXT (make-exfat-image.py writes "exfat hello danos\n").
|
||||
var read_ok = false;
|
||||
if (fs.open(mount ++ "/HELLO.TXT", .{})) |opened_file| {
|
||||
var file = opened_file;
|
||||
var buf: [32]u8 = undefined;
|
||||
const n = file.read(&buf) orelse 0;
|
||||
file.close();
|
||||
read_ok = std.mem.startsWith(u8, buf[0..n], "exfat hello danos");
|
||||
}
|
||||
if (read_ok) _ = logging.write("exfat-test: read HELLO.TXT ok\n");
|
||||
|
||||
// Mutation through the mount: mkdir, create + write, rename, read back, remove
|
||||
// — proof the write path reaches the engine over a real device.
|
||||
var mut_ok = false;
|
||||
if (fs.makeDirectory(mount ++ "/TESTDIR")) {
|
||||
var wrote = false;
|
||||
if (fs.open(mount ++ "/TESTDIR/W.TXT", .{ .create = true, .truncate = true })) |created| {
|
||||
var f = created;
|
||||
wrote = (f.writeAll("exfat-mutation-ok") orelse 0) == "exfat-mutation-ok".len;
|
||||
f.close();
|
||||
}
|
||||
const renamed = fs.rename(mount ++ "/TESTDIR/W.TXT", mount ++ "/TESTDIR/R.TXT");
|
||||
var readback = false;
|
||||
if (fs.open(mount ++ "/TESTDIR/R.TXT", .{})) |reopened| {
|
||||
var f = reopened;
|
||||
var buf: [32]u8 = undefined;
|
||||
const got = f.read(&buf) orelse 0;
|
||||
f.close();
|
||||
readback = std.mem.eql(u8, buf[0..got], "exfat-mutation-ok");
|
||||
}
|
||||
const removed = fs.remove(mount ++ "/TESTDIR/R.TXT");
|
||||
mut_ok = wrote and renamed and readback and removed;
|
||||
}
|
||||
if (mut_ok) _ = logging.write("exfat-test: mutations ok\n");
|
||||
|
||||
if (read_ok and mut_ok) {
|
||||
while (true) {
|
||||
_ = logging.write("exfat-test: ok\n");
|
||||
time.sleepMillis(1000);
|
||||
}
|
||||
}
|
||||
writeLine("exfat-test: FAILED (read={} mutations={})\n", .{ read_ok, mut_ok });
|
||||
}
|
||||
@@ -1,6 +1,6 @@
|
||||
//! test/system/services/fat-test — a client that proves the FAT mount end to end:
|
||||
//! it waits for the fat server to mount the USB volume at /volumes/usb, lists the
|
||||
//! root directory through the VFS (which routes /volumes/usb to the fat backend), and
|
||||
//! it waits for the fat server to mount the USB volume at /volumes/fat-12345678, lists the
|
||||
//! root directory through the VFS (which routes /volumes/fat-12345678 to the fat backend), and
|
||||
//! reads a known file off it. Shipped in the initial_ramdisk; the `fat-mount`
|
||||
//! kernel test spawns it alongside init.
|
||||
|
||||
@@ -18,16 +18,16 @@ fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||
pub fn main(init: process.Init) void {
|
||||
_ = init;
|
||||
|
||||
// Wait for /volumes/usb to be mounted — the fat server races us at boot (it must
|
||||
// Wait for /volumes/fat-12345678 to be mounted — the fat server races us at boot (it must
|
||||
// bring up the whole USB storage chain first).
|
||||
var opened: ?fs.Directory = null;
|
||||
var tries: u32 = 0;
|
||||
while (opened == null and tries < 1400) : (tries += 1) {
|
||||
opened = fs.openDirectory("/volumes/usb");
|
||||
opened = fs.openDirectory("/volumes/fat-12345678");
|
||||
if (opened == null) time.sleepMillis(50);
|
||||
}
|
||||
var dir = opened orelse {
|
||||
_ = logging.write("fat-test: /volumes/usb never became available\n");
|
||||
_ = logging.write("fat-test: /volumes/fat-12345678 never became available\n");
|
||||
return;
|
||||
};
|
||||
|
||||
@@ -43,56 +43,56 @@ pub fn main(init: process.Init) void {
|
||||
|
||||
// Read a known file off the boot volume through the mount (best effort): the
|
||||
// kernel image is an ELF, so its first bytes are the ELF magic.
|
||||
if (fs.open("/volumes/usb/system/kernel", .{})) |opened_file| {
|
||||
if (fs.open("/volumes/fat-12345678/system/kernel", .{})) |opened_file| {
|
||||
var file = opened_file;
|
||||
var magic: [4]u8 = undefined;
|
||||
const n = file.read(&magic) orelse 0;
|
||||
file.close();
|
||||
if (n == 4 and magic[0] == 0x7F and magic[1] == 'E' and magic[2] == 'L' and magic[3] == 'F') {
|
||||
_ = logging.write("fat-test: read /volumes/usb/system/kernel ELF magic ok\n");
|
||||
_ = logging.write("fat-test: read /volumes/fat-12345678/system/kernel ELF magic ok\n");
|
||||
} else {
|
||||
writeLine("fat-test: /volumes/usb/system/kernel read {d} bytes (not ELF magic)\n", .{n});
|
||||
writeLine("fat-test: /volumes/fat-12345678/system/kernel read {d} bytes (not ELF magic)\n", .{n});
|
||||
}
|
||||
}
|
||||
|
||||
// Exercise directory + file mutation through the mount: mkdir, create a file
|
||||
// inside it, read it back, then remove it — proof mkdir/unlink reach the engine.
|
||||
if (fs.makeDirectory("/volumes/usb/TESTDIR")) {
|
||||
if (fs.makeDirectory("/volumes/fat-12345678/TESTDIR")) {
|
||||
var wrote = false;
|
||||
if (fs.open("/volumes/usb/TESTDIR/HELLO.TXT", .{ .create = true, .truncate = true })) |created| {
|
||||
if (fs.open("/volumes/fat-12345678/TESTDIR/HELLO.TXT", .{ .create = true, .truncate = true })) |created| {
|
||||
var f = created;
|
||||
wrote = (f.writeAll("mutation-ok") orelse 0) == "mutation-ok".len;
|
||||
f.close();
|
||||
}
|
||||
// The created file carries a real modification time (stamped from the RTC).
|
||||
var mtime_ok = false;
|
||||
if (fs.attributes("/volumes/usb/TESTDIR/HELLO.TXT")) |attrs| {
|
||||
if (fs.attributes("/volumes/fat-12345678/TESTDIR/HELLO.TXT")) |attrs| {
|
||||
writeLine("fat-test: mtime {d}\n", .{attrs.mtime});
|
||||
mtime_ok = attrs.mtime > 1_577_836_800; // after 2020-01-01
|
||||
}
|
||||
if (mtime_ok) _ = logging.write("fat-test: mtime ok\n");
|
||||
|
||||
// Rename it, then read from the new name and confirm the old name is gone.
|
||||
const renamed = fs.rename("/volumes/usb/TESTDIR/HELLO.TXT", "/volumes/usb/TESTDIR/RENAMED.TXT");
|
||||
const old_gone = !fs.exists("/volumes/usb/TESTDIR/HELLO.TXT");
|
||||
const renamed = fs.rename("/volumes/fat-12345678/TESTDIR/HELLO.TXT", "/volumes/fat-12345678/TESTDIR/RENAMED.TXT");
|
||||
const old_gone = !fs.exists("/volumes/fat-12345678/TESTDIR/HELLO.TXT");
|
||||
if (renamed and old_gone) _ = logging.write("fat-test: rename ok\n");
|
||||
var readback = false;
|
||||
if (fs.open("/volumes/usb/TESTDIR/RENAMED.TXT", .{})) |reopened| {
|
||||
if (fs.open("/volumes/fat-12345678/TESTDIR/RENAMED.TXT", .{})) |reopened| {
|
||||
var f = reopened;
|
||||
var buf: [16]u8 = undefined;
|
||||
const got = f.read(&buf) orelse 0;
|
||||
f.close();
|
||||
readback = std.mem.eql(u8, buf[0..got], "mutation-ok");
|
||||
}
|
||||
const removed = fs.remove("/volumes/usb/TESTDIR/RENAMED.TXT");
|
||||
const gone = !fs.exists("/volumes/usb/TESTDIR/RENAMED.TXT");
|
||||
const removed = fs.remove("/volumes/fat-12345678/TESTDIR/RENAMED.TXT");
|
||||
const gone = !fs.exists("/volumes/fat-12345678/TESTDIR/RENAMED.TXT");
|
||||
if (wrote and mtime_ok and renamed and old_gone and readback and removed and gone) {
|
||||
_ = logging.write("fat-test: mutations ok\n");
|
||||
} else {
|
||||
writeLine("fat-test: mutations FAILED (wrote={} mtime={} renamed={} oldgone={} read={} removed={} gone={})\n", .{ wrote, mtime_ok, renamed, old_gone, readback, removed, gone });
|
||||
}
|
||||
} else {
|
||||
_ = logging.write("fat-test: mkdir /volumes/usb/TESTDIR failed\n");
|
||||
_ = logging.write("fat-test: mkdir /volumes/fat-12345678/TESTDIR failed\n");
|
||||
}
|
||||
|
||||
if (count > 0) {
|
||||
|
||||
@@ -77,7 +77,7 @@ fn park() void {
|
||||
var parked: ?fs.File = null;
|
||||
var tries: u32 = 0;
|
||||
while (parked == null and tries < 1000) : (tries += 1) {
|
||||
parked = fs.open("/volumes/usb/parked", .{ .create = true });
|
||||
parked = fs.open("/volumes/fat-12345678/parked", .{ .create = true });
|
||||
if (parked == null) time.sleepMillis(20);
|
||||
}
|
||||
if (parked == null) {
|
||||
@@ -92,16 +92,16 @@ fn park() void {
|
||||
// "parked" marker, which fails the vfs-client-death case: before the
|
||||
// ownership gate existed, any process could unmount any prefix, and this
|
||||
// fixture would have deleted the volume out from under the whole boot.
|
||||
if (fs.fsUnmount("/volumes/usb")) {
|
||||
if (fs.fsUnmount("/volumes/fat-12345678")) {
|
||||
_ = logging.write("vfstest: foreign unmount was ALLOWED\n");
|
||||
return;
|
||||
}
|
||||
if (fs.open("/volumes/usb/parked", .{})) |resolved| {
|
||||
if (fs.open("/volumes/fat-12345678/parked", .{})) |resolved| {
|
||||
var verification = resolved;
|
||||
verification.close(); // the park below must be the client's ONLY open
|
||||
// handle — the kernel test string-matches "released 1 handle(s)".
|
||||
} else {
|
||||
_ = logging.write("vfstest: /volumes/usb gone after refused unmount\n");
|
||||
_ = logging.write("vfstest: /volumes/fat-12345678 gone after refused unmount\n");
|
||||
return;
|
||||
}
|
||||
_ = logging.write("vfstest: foreign unmount refused\n");
|
||||
|
||||
@@ -194,7 +194,6 @@ system/kernel/process.zig:write_buffer
|
||||
system/kernel/scheduler.zig:ipc_maximum_handles
|
||||
system/kernel/scheduler.zig:maximum_space_mappings
|
||||
system/kernel/vfs.zig:maximum_directories
|
||||
system/kernel/vfs.zig:maximum_mounts
|
||||
system/kernel/vfs.zig:maximum_prefix
|
||||
system/kernel/vfs.zig:maximum_rewrite
|
||||
system/services/acpi/acpi.zig:blocks
|
||||
|
||||
@@ -0,0 +1,307 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Format a real exFAT image from scratch — the danos exFAT test volume.
|
||||
|
||||
Pure Python 3 standard library (no mkfs.exfat / mtools). It writes a valid exFAT
|
||||
filesystem — a Main Boot Sector + its boot-region checksum + a backup region, the
|
||||
32-bit FAT, an allocation bitmap, an up-case table (with its checksum), and a root
|
||||
directory whose entry sets a real exFAT reader (and the danos exfat engine) mount
|
||||
and walk. Mirrors tools/make-fat-image.py in spirit.
|
||||
|
||||
make-exfat-image.py [--serial <hex>] [--label <name>] <out.img> <size-MiB>
|
||||
make-exfat-image.py --verify <out.img>
|
||||
|
||||
The image seeds one file, HELLO.TXT, so a mount can be proven by reading it.
|
||||
"""
|
||||
|
||||
import struct
|
||||
import sys
|
||||
|
||||
SECTOR = 512
|
||||
UPCASE_UNITS = 256 # a-z -> A-Z, the rest identity; covers ASCII names
|
||||
|
||||
|
||||
def align_up(value, to):
|
||||
return (value + to - 1) // to * to
|
||||
|
||||
|
||||
def rotr16(v):
|
||||
return ((v >> 1) | (v << 15)) & 0xFFFF
|
||||
|
||||
|
||||
def rotr32(v):
|
||||
return ((v >> 1) | (v << 31)) & 0xFFFFFFFF
|
||||
|
||||
|
||||
def boot_checksum(region):
|
||||
"""32-bit rotate-right sum over the boot region, skipping VolumeFlags
|
||||
(106,107) and PercentInUse (112) of the first sector."""
|
||||
checksum = 0
|
||||
for i, byte in enumerate(region):
|
||||
if i in (106, 107, 112):
|
||||
continue
|
||||
checksum = (rotr32(checksum) + byte) & 0xFFFFFFFF
|
||||
return checksum
|
||||
|
||||
|
||||
def upcase_checksum(table_bytes):
|
||||
checksum = 0
|
||||
for byte in table_bytes:
|
||||
checksum = (rotr32(checksum) + byte) & 0xFFFFFFFF
|
||||
return checksum
|
||||
|
||||
|
||||
def set_checksum(entries):
|
||||
"""16-bit rotate-right sum over a directory-entry set, skipping its own two
|
||||
checksum bytes (offset 2..3 of the first entry)."""
|
||||
checksum = 0
|
||||
for i, byte in enumerate(entries):
|
||||
if i in (2, 3):
|
||||
continue
|
||||
checksum = (rotr16(checksum) + byte) & 0xFFFF
|
||||
return checksum
|
||||
|
||||
|
||||
def name_hash(upcased_units):
|
||||
h = 0
|
||||
for unit in upcased_units:
|
||||
h = (rotr16(h) + (unit & 0xFF)) & 0xFFFF
|
||||
h = (rotr16(h) + (unit >> 8)) & 0xFFFF
|
||||
return h
|
||||
|
||||
|
||||
def ascii_upper(unit):
|
||||
return unit - ord("a") + ord("A") if ord("a") <= unit <= ord("z") else unit
|
||||
|
||||
|
||||
def solve_geometry(total_sectors, spc):
|
||||
"""Solve for cluster count / FAT length / heap offset that fit. The FAT sits
|
||||
after the main + backup boot regions (24 sectors)."""
|
||||
fat_offset = 24
|
||||
fat_length = 1
|
||||
while True:
|
||||
heap_offset = align_up(fat_offset + fat_length, spc)
|
||||
cluster_count = (total_sectors - heap_offset) // spc
|
||||
needed = ((cluster_count + 2) * 4 + SECTOR - 1) // SECTOR
|
||||
if needed <= fat_length:
|
||||
return cluster_count, fat_offset, fat_length, heap_offset
|
||||
fat_length = needed
|
||||
|
||||
|
||||
class ExfatImage:
|
||||
def __init__(self, size_mib, volume_id=0x1234ABCD, label="DANOS"):
|
||||
self.total_sectors = size_mib * 1024 * 1024 // SECTOR
|
||||
self.spc = 8 # 4 KiB clusters
|
||||
self.volume_id = volume_id & 0xFFFFFFFF
|
||||
self.label = label
|
||||
self.cluster_count, self.fat_offset, self.fat_length, self.heap_offset = solve_geometry(self.total_sectors, self.spc)
|
||||
if self.cluster_count < 16:
|
||||
sys.exit(f"error: image too small for exFAT ({self.cluster_count} clusters)")
|
||||
self.cluster_bytes = self.spc * SECTOR
|
||||
# Layout: the allocation bitmap (as many clusters as it needs — one per
|
||||
# 8*cluster_bytes clusters of the volume), then the up-case table, the root
|
||||
# directory, and the seeded file. A single-cluster bitmap (small volumes,
|
||||
# e.g. the 48 MiB fixture) puts root at cluster 4, as before.
|
||||
self.bitmap_bytes = (self.cluster_count + 7) // 8
|
||||
self.bitmap_clusters = (self.bitmap_bytes + self.cluster_bytes - 1) // self.cluster_bytes
|
||||
self.bitmap_cluster = 2
|
||||
self.upcase_cluster = self.bitmap_cluster + self.bitmap_clusters
|
||||
self.root_cluster = self.upcase_cluster + 1
|
||||
self.hello_cluster = self.root_cluster + 1
|
||||
self.image = bytearray(self.total_sectors * SECTOR)
|
||||
|
||||
def cluster_offset(self, cluster):
|
||||
return (self.heap_offset + (cluster - 2) * self.spc) * SECTOR
|
||||
|
||||
def set_fat(self, cluster, value):
|
||||
struct.pack_into("<I", self.image, self.fat_offset * SECTOR + cluster * 4, value)
|
||||
|
||||
def mark_allocated(self, cluster):
|
||||
bit = cluster - 2
|
||||
pos = self.cluster_offset(2) + bit // 8
|
||||
self.image[pos] |= 1 << (bit % 8)
|
||||
|
||||
def main_boot_sector(self):
|
||||
sector = bytearray(SECTOR)
|
||||
sector[0:3] = b"\xEB\x76\x90" # jump boot
|
||||
sector[3:11] = b"EXFAT " # filesystem name
|
||||
# 11..64 MustBeZero (already zero)
|
||||
struct.pack_into("<Q", sector, 72, self.total_sectors) # volume length
|
||||
struct.pack_into("<I", sector, 80, self.fat_offset) # fat offset
|
||||
struct.pack_into("<I", sector, 84, self.fat_length) # fat length
|
||||
struct.pack_into("<I", sector, 88, self.heap_offset) # cluster heap offset
|
||||
struct.pack_into("<I", sector, 92, self.cluster_count) # cluster count
|
||||
struct.pack_into("<I", sector, 96, self.root_cluster) # first cluster of root
|
||||
struct.pack_into("<I", sector, 100, self.volume_id) # volume serial number
|
||||
struct.pack_into("<H", sector, 104, 0x0100) # filesystem revision 1.0
|
||||
sector[108] = 9 # bytes per sector shift (512)
|
||||
sector[109] = self.spc.bit_length() - 1 # sectors per cluster shift
|
||||
sector[110] = 1 # number of FATs
|
||||
sector[111] = 0x80 # drive select
|
||||
sector[112] = 0xFF # percent in use (unknown)
|
||||
sector[510] = 0x55
|
||||
sector[511] = 0xAA
|
||||
return sector
|
||||
|
||||
def build(self):
|
||||
# Main boot region (sectors 0..11): VBR, eight extended boot sectors, OEM
|
||||
# parameters, reserved, then the checksum sector.
|
||||
vbr = self.main_boot_sector()
|
||||
self.image[0:SECTOR] = vbr
|
||||
for s in range(1, 9): # extended boot sectors carry the 0xAA550000 signature
|
||||
struct.pack_into("<I", self.image, s * SECTOR + 508, 0xAA550000)
|
||||
# sectors 9 (OEM) and 10 (reserved) stay zero
|
||||
region = bytes(self.image[0 : 11 * SECTOR])
|
||||
checksum = boot_checksum(region)
|
||||
for i in range(SECTOR // 4):
|
||||
struct.pack_into("<I", self.image, 11 * SECTOR + i * 4, checksum)
|
||||
# Backup boot region (sectors 12..23) is a copy of 0..11.
|
||||
self.image[12 * SECTOR : 24 * SECTOR] = self.image[0 : 12 * SECTOR]
|
||||
|
||||
# FAT: reserved entries, then a single-cluster chain per metadata object,
|
||||
# except the bitmap which spans self.bitmap_clusters (a real FAT chain).
|
||||
self.set_fat(0, 0xFFFFFFF8)
|
||||
self.set_fat(1, 0xFFFFFFFF)
|
||||
used = []
|
||||
for i in range(self.bitmap_clusters):
|
||||
cluster = self.bitmap_cluster + i
|
||||
self.set_fat(cluster, 0xFFFFFFFF if i == self.bitmap_clusters - 1 else cluster + 1)
|
||||
used.append(cluster)
|
||||
for cluster in (self.upcase_cluster, self.root_cluster, self.hello_cluster):
|
||||
self.set_fat(cluster, 0xFFFFFFFF)
|
||||
used.append(cluster)
|
||||
|
||||
# Allocation bitmap: every metadata/file cluster in use.
|
||||
for cluster in used:
|
||||
self.mark_allocated(cluster)
|
||||
|
||||
# Up-case table: 256 explicit units, a-z -> A-Z.
|
||||
upcase = bytearray(UPCASE_UNITS * 2)
|
||||
for i in range(UPCASE_UNITS):
|
||||
struct.pack_into("<H", upcase, i * 2, ascii_upper(i))
|
||||
off = self.cluster_offset(self.upcase_cluster)
|
||||
self.image[off : off + len(upcase)] = upcase
|
||||
table_checksum = upcase_checksum(upcase)
|
||||
|
||||
# Seed file HELLO.TXT (contiguous, one cluster).
|
||||
content = b"exfat hello danos\n"
|
||||
off = self.cluster_offset(self.hello_cluster)
|
||||
self.image[off : off + len(content)] = content
|
||||
|
||||
# Root directory: bitmap, up-case, volume label, HELLO set.
|
||||
root = self.cluster_offset(self.root_cluster)
|
||||
# 0x81 Allocation Bitmap
|
||||
struct.pack_into("<BBB", self.image, root, 0x81, 0, 0)
|
||||
struct.pack_into("<I", self.image, root + 20, self.bitmap_cluster)
|
||||
struct.pack_into("<Q", self.image, root + 24, self.bitmap_bytes)
|
||||
# 0x82 Up-case Table
|
||||
struct.pack_into("<B", self.image, root + 32, 0x82)
|
||||
struct.pack_into("<I", self.image, root + 32 + 4, table_checksum)
|
||||
struct.pack_into("<I", self.image, root + 32 + 20, self.upcase_cluster)
|
||||
struct.pack_into("<Q", self.image, root + 32 + 24, UPCASE_UNITS * 2)
|
||||
# 0x83 Volume Label
|
||||
label_units = [ord(c) for c in self.label[:11]]
|
||||
struct.pack_into("<BB", self.image, root + 64, 0x83, len(label_units))
|
||||
for i, u in enumerate(label_units):
|
||||
struct.pack_into("<H", self.image, root + 64 + 2 + i * 2, u)
|
||||
# HELLO.TXT set: File (0x85) + Stream (0xC0) + Name (0xC1)
|
||||
name = "HELLO.TXT"
|
||||
self.write_file_set(root + 96, name, first_cluster=self.hello_cluster, length=len(content))
|
||||
|
||||
def write_file_set(self, offset, name, first_cluster, length):
|
||||
entries = bytearray(32 * 3)
|
||||
# File entry
|
||||
entries[0] = 0x85
|
||||
entries[1] = 2 # stream + one name entry
|
||||
struct.pack_into("<H", entries, 4, 0x20) # attributes: archive
|
||||
# Stream entry
|
||||
entries[32 + 0] = 0xC0
|
||||
entries[32 + 1] = 0x01 | 0x02 # allocation possible + no FAT chain (contiguous)
|
||||
entries[32 + 3] = len(name)
|
||||
upname = [ascii_upper(ord(c)) for c in name]
|
||||
struct.pack_into("<H", entries, 32 + 4, name_hash(upname))
|
||||
struct.pack_into("<Q", entries, 32 + 8, length) # valid data length
|
||||
struct.pack_into("<I", entries, 32 + 20, first_cluster)
|
||||
struct.pack_into("<Q", entries, 32 + 24, length) # data length
|
||||
# File Name entry
|
||||
entries[64 + 0] = 0xC1
|
||||
for i, c in enumerate(name):
|
||||
struct.pack_into("<H", entries, 64 + 2 + i * 2, ord(c))
|
||||
struct.pack_into("<H", entries, 2, set_checksum(entries))
|
||||
self.image[offset : offset + len(entries)] = entries
|
||||
|
||||
def serialize(self):
|
||||
self.build()
|
||||
return bytes(self.image)
|
||||
|
||||
|
||||
def verify(path):
|
||||
with open(path, "rb") as handle:
|
||||
data = handle.read()
|
||||
if len(data) < 512 or data[510] != 0x55 or data[511] != 0xAA:
|
||||
sys.exit("verify: missing 0x55AA boot signature")
|
||||
if data[3:11] != b"EXFAT ":
|
||||
sys.exit("verify: not an exFAT boot sector")
|
||||
if any(data[11:64]):
|
||||
sys.exit("verify: MustBeZero region is not zero")
|
||||
fat_offset = struct.unpack_from("<I", data, 80)[0]
|
||||
heap_offset = struct.unpack_from("<I", data, 88)[0]
|
||||
cluster_count = struct.unpack_from("<I", data, 92)[0]
|
||||
root_cluster = struct.unpack_from("<I", data, 96)[0]
|
||||
spc = 1 << data[109]
|
||||
# Boot checksum sector 11 must match a fresh checksum over sectors 0..10.
|
||||
expected = boot_checksum(data[0 : 11 * SECTOR])
|
||||
got = struct.unpack_from("<I", data, 11 * SECTOR)[0]
|
||||
if expected != got:
|
||||
sys.exit(f"verify: boot checksum mismatch (0x{got:08X} != 0x{expected:08X})")
|
||||
# Walk the root directory for the HELLO.TXT set and check its checksum.
|
||||
root = (heap_offset + (root_cluster - 2) * spc) * SECTOR
|
||||
found = False
|
||||
for i in range(spc * SECTOR // 32):
|
||||
entry = root + i * 32
|
||||
if data[entry] == 0x00:
|
||||
break
|
||||
if data[entry] == 0x85:
|
||||
secondary = data[entry + 1]
|
||||
total = (secondary + 1) * 32
|
||||
stored = struct.unpack_from("<H", data, entry + 2)[0]
|
||||
if set_checksum(data[entry : entry + total]) != stored:
|
||||
sys.exit("verify: a file set checksum is wrong")
|
||||
found = True
|
||||
if not found:
|
||||
sys.exit("verify: no file set in the root directory")
|
||||
print(f"make-exfat-image: {path} OK "
|
||||
f"({cluster_count} clusters of {spc * SECTOR} bytes, fat@{fat_offset}, heap@{heap_offset})")
|
||||
|
||||
|
||||
def main(argv):
|
||||
if len(argv) == 3 and argv[1] == "--verify":
|
||||
verify(argv[2])
|
||||
return 0
|
||||
argv = list(argv)
|
||||
volume_id = 0x1234ABCD
|
||||
label = "DANOS"
|
||||
i = 1
|
||||
while i < len(argv):
|
||||
if argv[i] == "--serial" and i + 1 < len(argv):
|
||||
volume_id = int(argv[i + 1], 16)
|
||||
del argv[i : i + 2]
|
||||
elif argv[i] == "--label" and i + 1 < len(argv):
|
||||
label = argv[i + 1]
|
||||
del argv[i : i + 2]
|
||||
else:
|
||||
i += 1
|
||||
if len(argv) != 3:
|
||||
sys.exit("usage: make-exfat-image.py [--serial <hex>] [--label <name>] <out.img> <size-MiB>\n"
|
||||
" make-exfat-image.py --verify <out.img>")
|
||||
out_path = argv[1]
|
||||
size_mib = int(argv[2])
|
||||
image = ExfatImage(size_mib, volume_id, label)
|
||||
with open(out_path, "wb") as handle:
|
||||
handle.write(image.serialize())
|
||||
print(f"make-exfat-image: wrote {out_path} "
|
||||
f"({size_mib} MiB exFAT, {image.cluster_count} clusters, serial 0x{image.volume_id:08X})")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main(sys.argv))
|
||||
+31
-7
@@ -47,8 +47,14 @@ def fat32_geometry(total_sectors):
|
||||
|
||||
|
||||
class Fat32Image:
|
||||
def __init__(self, total_sectors):
|
||||
def __init__(self, total_sectors, volume_id=0x12345678, label="DANOS"):
|
||||
self.total_sectors = total_sectors
|
||||
# The FAT volume serial (its content identity — the /volumes/fat-<id>
|
||||
# mount path danos derives from it) and the display label. A second image
|
||||
# needs a distinct serial so its id-path does not collide with the boot
|
||||
# volume's.
|
||||
self.volume_id = volume_id & 0xFFFFFFFF
|
||||
self.label = label
|
||||
self.fat_size, self.cluster_count = fat32_geometry(total_sectors)
|
||||
if self.cluster_count < 65525:
|
||||
sys.exit(f"error: image too small for FAT32 ({self.cluster_count} clusters "
|
||||
@@ -146,8 +152,8 @@ class Fat32Image:
|
||||
0x80, # drive number
|
||||
0, # reserved
|
||||
0x29, # extended boot signature
|
||||
0x12345678, # volume id
|
||||
b"DANOS ", # volume label
|
||||
self.volume_id, # volume id
|
||||
self.label.encode("ascii", "replace")[:11].ljust(11, b" "), # volume label
|
||||
b"FAT32 ", # filesystem type
|
||||
)
|
||||
sector[510] = 0x55
|
||||
@@ -276,9 +282,9 @@ def build_tree(pairs):
|
||||
return root
|
||||
|
||||
|
||||
def build(out_path, size_mib, pairs):
|
||||
def build(out_path, size_mib, pairs, volume_id=0x12345678, label="DANOS"):
|
||||
total_sectors = size_mib * 1024 * 1024 // SECTOR
|
||||
image = Fat32Image(total_sectors)
|
||||
image = Fat32Image(total_sectors, volume_id, label)
|
||||
tree = build_tree(pairs)
|
||||
write_directory(image, 2, tree, 0, True)
|
||||
with open(out_path, "wb") as handle:
|
||||
@@ -350,14 +356,32 @@ def main(argv):
|
||||
if len(argv) == 3 and argv[1] == "--verify":
|
||||
verify(argv[2])
|
||||
return 0
|
||||
# Optional flags ahead of the positionals: --serial <hex> sets the FAT volume
|
||||
# id (the /volumes/fat-<id> content identity), --label <name> its display
|
||||
# label. A second FAT image passes a distinct --serial so its id-path cannot
|
||||
# collide with the boot volume's.
|
||||
argv = list(argv)
|
||||
volume_id = 0x12345678
|
||||
label = "DANOS"
|
||||
i = 1
|
||||
while i < len(argv):
|
||||
if argv[i] == "--serial" and i + 1 < len(argv):
|
||||
volume_id = int(argv[i + 1], 16)
|
||||
del argv[i:i + 2]
|
||||
elif argv[i] == "--label" and i + 1 < len(argv):
|
||||
label = argv[i + 1]
|
||||
del argv[i:i + 2]
|
||||
else:
|
||||
i += 1
|
||||
if len(argv) < 3 or (len(argv) - 3) % 2 != 0:
|
||||
sys.exit("usage: make-fat-image.py <out.img> <size-MiB> [<dest> <host>]...\n"
|
||||
sys.exit("usage: make-fat-image.py [--serial <hex>] [--label <name>] "
|
||||
"<out.img> <size-MiB> [<dest> <host>]...\n"
|
||||
" make-fat-image.py --verify <out.img>")
|
||||
out_path = argv[1]
|
||||
size_mib = int(argv[2])
|
||||
rest = argv[3:]
|
||||
pairs = [(rest[i], rest[i + 1]) for i in range(0, len(rest), 2)]
|
||||
build(out_path, size_mib, pairs)
|
||||
build(out_path, size_mib, pairs, volume_id, label)
|
||||
return 0
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,88 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Assemble an MBR-partitioned disk image from N FAT32 partitions — the danos
|
||||
multi-volume test disk.
|
||||
|
||||
Pure Python 3 stdlib (no mtools / parted). Each partition is a real FAT32
|
||||
filesystem produced by make-fat-image.py, laid out behind a classic MBR so the
|
||||
danos partition prober (partition.allVolumes) walks the table and the volume
|
||||
manager spawns one confined filesystem per partition — several volumes sharing
|
||||
ONE block channel, each clamped to its own LBA range. That shared-channel,
|
||||
per-partition path is what a single stick with two partitions exercises and a
|
||||
pair of single-volume sticks does not.
|
||||
|
||||
make-partitioned-image.py <out.img> [<serial-hex> <size-MiB>]...
|
||||
|
||||
Each partition is an empty FAT32 with the given volume serial (its /volumes/
|
||||
fat-<serial> content id). Partitions are 1-MiB aligned; the MBR marks each
|
||||
type 0x0C (FAT32 LBA). At most four (an MBR holds four primaries).
|
||||
"""
|
||||
|
||||
import os
|
||||
import struct
|
||||
import subprocess
|
||||
import sys
|
||||
import tempfile
|
||||
|
||||
SECTOR = 512
|
||||
ALIGN = 2048 # sectors (1 MiB) — standard partition alignment, and the gap the MBR sits in
|
||||
MBR_TYPE_FAT32_LBA = 0x0C
|
||||
MAX_PRIMARY_PARTITIONS = 4
|
||||
HERE = os.path.dirname(os.path.abspath(__file__))
|
||||
|
||||
|
||||
def align_up(sectors, to=ALIGN):
|
||||
return (sectors + to - 1) // to * to
|
||||
|
||||
|
||||
def main(argv):
|
||||
if len(argv) < 4 or (len(argv) - 2) % 2 != 0:
|
||||
sys.exit("usage: make-partitioned-image.py <out.img> [<serial-hex> <size-MiB>]...")
|
||||
out_path = argv[1]
|
||||
specs = [(argv[i], int(argv[i + 1])) for i in range(2, len(argv), 2)]
|
||||
if len(specs) > MAX_PRIMARY_PARTITIONS:
|
||||
sys.exit(f"error: an MBR holds at most {MAX_PRIMARY_PARTITIONS} primary partitions")
|
||||
|
||||
# Generate each partition's FAT32 image, then place it at its aligned start.
|
||||
partitions = [] # (start_sector, sector_count, bytes)
|
||||
cursor = ALIGN # leave the first 1 MiB for the MBR + alignment gap
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
for idx, (serial, size_mib) in enumerate(specs):
|
||||
part_path = os.path.join(tmp, f"p{idx}.img")
|
||||
subprocess.run(
|
||||
[sys.executable, os.path.join(HERE, "make-fat-image.py"),
|
||||
"--serial", serial, "--label", f"DATA{idx}",
|
||||
part_path, str(size_mib)],
|
||||
check=True, stdout=subprocess.DEVNULL)
|
||||
with open(part_path, "rb") as handle:
|
||||
data = handle.read()
|
||||
count = len(data) // SECTOR
|
||||
partitions.append((cursor, count, data))
|
||||
cursor = align_up(cursor + count)
|
||||
|
||||
total_sectors = cursor
|
||||
disk = bytearray(total_sectors * SECTOR)
|
||||
# The MBR: a disk signature, one partition entry per FAT partition, 0x55AA.
|
||||
# No boot code (this disk is data, never booted); danos's mount() sees the
|
||||
# signature but no BPB at LBA 0 and takes the MBR-walk path.
|
||||
struct.pack_into("<I", disk, 440, 0x0D05DA05) # arbitrary but fixed disk signature
|
||||
for idx, (start, count, data) in enumerate(partitions):
|
||||
entry = 446 + idx * 16
|
||||
disk[entry + 0] = 0x00 # not bootable
|
||||
disk[entry + 1:entry + 4] = b"\xFE\xFF\xFF" # CHS start (LBA-aware tools ignore)
|
||||
disk[entry + 4] = MBR_TYPE_FAT32_LBA
|
||||
disk[entry + 5:entry + 8] = b"\xFE\xFF\xFF" # CHS end
|
||||
struct.pack_into("<I", disk, entry + 8, start) # start LBA
|
||||
struct.pack_into("<I", disk, entry + 12, count) # sector count
|
||||
disk[start * SECTOR:start * SECTOR + len(data)] = data
|
||||
disk[510] = 0x55
|
||||
disk[511] = 0xAA
|
||||
|
||||
with open(out_path, "wb") as handle:
|
||||
handle.write(disk)
|
||||
print(f"make-partitioned-image: wrote {out_path} "
|
||||
f"({total_sectors * SECTOR // (1024 * 1024)} MiB, {len(partitions)} partitions)")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main(sys.argv))
|
||||
Reference in New Issue
Block a user