Compare commits
24
Commits
6d4992ae02
..
main
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
4037e746aa | ||
|
|
4dfb5012c0 | ||
|
|
061eb7c004 | ||
|
|
5dc966838a | ||
|
|
700452dc4e | ||
|
|
0faa0fd21b | ||
|
|
c47215821c | ||
|
|
6ddb08091d | ||
|
|
77b64229c2 | ||
|
|
90906bcefe | ||
|
|
2c2745e9e5 | ||
|
|
2a6d604577 | ||
|
|
e81a4e6f1d | ||
|
|
e240341bfb | ||
|
|
62eb2a748a | ||
|
|
56bd2e7678 | ||
|
|
0b25cd2c94 | ||
|
|
d59279422e | ||
|
|
bf9f8560c6 | ||
|
|
da7dcce64e | ||
|
|
9750db14da | ||
|
|
b2a5a0a3c6 | ||
|
|
7efe7b72d8 | ||
|
|
d4b544d66b |
@@ -122,6 +122,7 @@ fn driverArtifact(comptime package: []const u8, comptime artifact: []const u8) S
|
|||||||
/// (docs/build-packages-plan.md).
|
/// (docs/build-packages-plan.md).
|
||||||
const production_ship = [_]ShipRow{
|
const production_ship = [_]ShipRow{
|
||||||
service("fat"),
|
service("fat"),
|
||||||
|
service("exfat"),
|
||||||
service("display"),
|
service("display"),
|
||||||
service("display-demo"),
|
service("display-demo"),
|
||||||
service("device-manager"),
|
service("device-manager"),
|
||||||
@@ -332,6 +333,7 @@ pub fn build(b: *std.Build) void {
|
|||||||
if (test_case != null) for ([_][]const u8{
|
if (test_case != null) for ([_][]const u8{
|
||||||
"vfs-test", // the user-space VFS round-trip client
|
"vfs-test", // the user-space VFS round-trip client
|
||||||
"fat-test",
|
"fat-test",
|
||||||
|
"exfat-test", // the exFAT mount round-trip client (S4)
|
||||||
"badge-scope-test", // the guessable-id probe: a second process names the first's node and layer
|
"badge-scope-test", // the guessable-id probe: a second process names the first's node and layer
|
||||||
"shared-memory-server",
|
"shared-memory-server",
|
||||||
"shared-memory-client",
|
"shared-memory-client",
|
||||||
@@ -447,6 +449,7 @@ pub fn build(b: *std.Build) void {
|
|||||||
csv_library,
|
csv_library,
|
||||||
xkeyboard_config_library,
|
xkeyboard_config_library,
|
||||||
b.dependency("fat", .{}),
|
b.dependency("fat", .{}),
|
||||||
|
b.dependency("exfat", .{}),
|
||||||
b.dependency("volume-manager", .{}),
|
b.dependency("volume-manager", .{}),
|
||||||
b.dependency("display", .{}),
|
b.dependency("display", .{}),
|
||||||
b.dependency("ps2-bus", .{}),
|
b.dependency("ps2-bus", .{}),
|
||||||
|
|||||||
@@ -46,6 +46,7 @@
|
|||||||
.@"pci-bus" = .{ .path = "system/drivers/pci-bus" },
|
.@"pci-bus" = .{ .path = "system/drivers/pci-bus" },
|
||||||
.init = .{ .path = "system/services/init" },
|
.init = .{ .path = "system/services/init" },
|
||||||
.fat = .{ .path = "system/services/fat" },
|
.fat = .{ .path = "system/services/fat" },
|
||||||
|
.exfat = .{ .path = "system/services/exfat" },
|
||||||
.display = .{ .path = "system/services/display" },
|
.display = .{ .path = "system/services/display" },
|
||||||
.@"display-demo" = .{ .path = "system/services/display-demo" },
|
.@"display-demo" = .{ .path = "system/services/display-demo" },
|
||||||
.@"device-manager" = .{ .path = "system/services/device-manager" },
|
.@"device-manager" = .{ .path = "system/services/device-manager" },
|
||||||
@@ -64,6 +65,7 @@
|
|||||||
.@"virtio-gpu" = .{ .path = "system/drivers/virtio-gpu" },
|
.@"virtio-gpu" = .{ .path = "system/drivers/virtio-gpu" },
|
||||||
.@"vfs-test" = .{ .path = "test/system/services/vfs-test", .lazy = true },
|
.@"vfs-test" = .{ .path = "test/system/services/vfs-test", .lazy = true },
|
||||||
.@"fat-test" = .{ .path = "test/system/services/fat-test", .lazy = true },
|
.@"fat-test" = .{ .path = "test/system/services/fat-test", .lazy = true },
|
||||||
|
.@"exfat-test" = .{ .path = "test/system/services/exfat-test", .lazy = true },
|
||||||
.@"badge-scope-test" = .{ .path = "test/system/services/badge-scope-test", .lazy = true },
|
.@"badge-scope-test" = .{ .path = "test/system/services/badge-scope-test", .lazy = true },
|
||||||
.@"shared-memory-server" = .{ .path = "test/system/services/shared-memory-server", .lazy = true },
|
.@"shared-memory-server" = .{ .path = "test/system/services/shared-memory-server", .lazy = true },
|
||||||
.@"shared-memory-client" = .{ .path = "test/system/services/shared-memory-client", .lazy = true },
|
.@"shared-memory-client" = .{ .path = "test/system/services/shared-memory-client", .lazy = true },
|
||||||
|
|||||||
@@ -11,15 +11,29 @@
|
|||||||
> identity ladder (GPT GUID + name, FAT serial + label, MBR), and the mount map:
|
> identity ladder (GPT GUID + name, FAT serial + label, MBR), and the mount map:
|
||||||
> `filesystems.csv` (signature → binary) + `volumes.csv` (identity → optional
|
> `filesystems.csv` (signature → binary) + `volumes.csv` (identity → optional
|
||||||
> override), a volume's mount path IS its content id (`/volumes/<id>`), with the
|
> override), a volume's mount path IS its content id (`/volumes/<id>`), with the
|
||||||
> label as display metadata a `volumes` query returns. **Still pending**: the
|
> label as display metadata a `volumes` query returns. Multi-volume is **built**:
|
||||||
> `filesystem UUID` rung (needs a non-FAT engine), multi-volume (one FAT volume
|
> the manager adopts every storage device, probes each device's whole partition
|
||||||
> today; fat's boot rewrites are unconditional until S3 makes them
|
> table, and spawns one range-confined FAT per volume — several volumes across
|
||||||
> content-conditional), the volume manager *consuming* `medium_changed` (removal
|
> several devices, or several partitions sharing one device's channel — each at
|
||||||
> is detected by device-presence polling; the event is published but only a
|
> its own `/volumes/<id>` path with its own supervision. The boot volume is
|
||||||
> card-reader medium change needs the subscription), and the remount-on-replug
|
> identified by **content** (a volume backs `/system/configuration` + `/system/logs`
|
||||||
> end-to-end (the logic is in place; QEMU can't re-present the boot-controller
|
> only when it resolves `/system/configuration` on its own media), so it works as
|
||||||
> device, so it is bench-verified). A few markers below are left where a duty is
|
> any partition of any device. exFAT is **built** as a second engine
|
||||||
> still pending.
|
> (`system/services/exfat`): full read + write, directories, rename, and on-disk
|
||||||
|
> up-case folding, reusing `library/kernel/file-system-harness` wholesale — the
|
||||||
|
> reuse claim, proven — and a volume routes to fat or exfat by its VBR, at an
|
||||||
|
> `exfat-<serial>` id-path. Removal is robust to all three triggers now: a
|
||||||
|
> pulled device (presence polling), a medium that leaves while its device stays
|
||||||
|
> (the volume manager CONSUMES `medium_changed`), and a storage driver that
|
||||||
|
> crashes while its device stays present (a channel-liveness `geometry()` probe
|
||||||
|
> reaps the volume and rebuilds it on the restarted driver's fresh channel). The
|
||||||
|
> re-adopt-and-remount path is QEMU-proven by the driver-crash rebuild; a physical
|
||||||
|
> unplug/replug exercises the same path but is bench-pending (QEMU cannot
|
||||||
|
> re-present a usb-storage `device_add`). **Still pending**: the `filesystem UUID`
|
||||||
|
> rung (ext-family superblocks, which need such an engine); and arbitration when
|
||||||
|
> two volumes both resolve the boot markers (S3 mounts both and logs each claim;
|
||||||
|
> picking one is deferred). A few
|
||||||
|
> markers below are left where a duty is still pending.
|
||||||
|
|
||||||
## The model
|
## The model
|
||||||
|
|
||||||
@@ -121,12 +135,16 @@ its own mounts with the kernel; its write cache lives inside the process, so a
|
|||||||
write error is observed by the code that owns the volume and surfaces on the
|
write error is observed by the code that owns the volume and surfaces on the
|
||||||
owning channel (the anti-fsyncgate rule — never a system-wide dirty pool).
|
owning channel (the anti-fsyncgate rule — never a system-wide dirty pool).
|
||||||
*(Built:)* fat receives its mount path as `argv[2]` from the volume manager (the
|
*(Built:)* fat receives its mount path as `argv[2]` from the volume manager (the
|
||||||
volume's id-path, e.g. `/volumes/fat-12345678`) and mounts its root there, plus
|
volume's id-path, e.g. `/volumes/fat-12345678`) and mounts its root there. It
|
||||||
the two `/system` hierarchy rewrites it installs in place (unconditional this
|
installs the two `/system` hierarchy rewrites (`/system/configuration`,
|
||||||
increment; S3 makes them content-conditional across volumes). It no longer
|
`/system/logs`) only when it is the boot volume — decided by **content**: it
|
||||||
|
resolves `/system/configuration` on its own media at mount, so a data volume
|
||||||
|
mounts at its id-path alone and never shadows the running system. It no longer
|
||||||
self-acquires a volume — the V3b flip made it receive its volume id and block
|
self-acquires a volume — the V3b flip made it receive its volume id and block
|
||||||
channel from the volume manager, consistent with "it never discovers devices"
|
channel from the volume manager, consistent with "it never discovers devices"
|
||||||
above.
|
above. Because several volumes now serve at once, no filesystem binds a shared
|
||||||
|
service name; clients reach each through the kernel mount table (`fs_resolve`
|
||||||
|
routes by prefix to the backing endpoint).
|
||||||
|
|
||||||
**Kernel** (mechanism only): the mount table routes paths to backend
|
**Kernel** (mechanism only): the mount table routes paths to backend
|
||||||
endpoints — resolve and redirect, never data. Remount-replace is the restart
|
endpoints — resolve and redirect, never data. Remount-replace is the restart
|
||||||
@@ -172,25 +190,26 @@ surprise-removal path — kill the filesystem process, retire its mounts,
|
|||||||
respawn on return. No half-alive states, no `remount-ro`, no mounts that
|
respawn on return. No half-alive states, no `remount-ro`, no mounts that
|
||||||
error forever (Plan 9's dead-server wart).
|
error forever (Plan 9's dead-server wart).
|
||||||
|
|
||||||
The path has **two triggers, one lifecycle**: the *device* leaving (the
|
The path folds **three triggers into one lifecycle**: the *device* leaving (a
|
||||||
storage driver dies — channel death, the table below), and the *medium*
|
pulled stick — presence polling); the *medium* leaving while the device stays
|
||||||
leaving while the device stays (an SD card pulled from its reader, an ATAPI
|
(an SD card pulled from its reader, an ATAPI tray opened, a USB card reader);
|
||||||
tray opened — including USB card readers today). The second trigger is the
|
and a storage *driver crashing* while its device stays in the tree. The second
|
||||||
pushed `medium_changed` event on the block protocol — published today from a
|
trigger is the pushed `medium_changed` event on the block protocol, published
|
||||||
TEST UNIT READY poll; still *planned* is the volume manager *consuming* it
|
from a TEST UNIT READY poll — the volume manager now **consumes** it (subscribed
|
||||||
(today removal is driven only by device-presence polling) and translating the
|
per device), running the same kill-retire path and re-probing on medium return,
|
||||||
transport's native signal (SCSI UNIT ATTENTION, AHCI PxSSTS, NVMe
|
so a swapped card is never served with the previous card's filesystem state. The
|
||||||
namespace-change AER) in place of the poll. On the event the volume manager
|
third is caught by a channel-liveness `geometry()` probe: presence polling alone
|
||||||
runs the same kill-retire path, then re-probes on medium return exactly as on
|
sees the device still present, but the channel is dead, so the manager reaps the
|
||||||
device return. Without it, a swapped card would be served with the previous
|
volume and rebuilds it on the restarted driver's fresh channel. Still *planned*
|
||||||
card's filesystem state.
|
is translating the transport's native signal (SCSI UNIT ATTENTION, AHCI PxSSTS,
|
||||||
|
NVMe namespace-change AER) in place of the presence poll.
|
||||||
|
|
||||||
| Layer | Observes | Must do | Guarantees |
|
| Layer | Observes | Must do | Guarantees |
|
||||||
|---|---|---|---|
|
|---|---|---|---|
|
||||||
| Bus driver | port/hub status change | tear down the device's slots (children first, recursively — built, hot-plug matrix), report `child_removed` per interface | the device tree is honest within one reconcile tick |
|
| Bus driver | port/hub status change | tear down the device's slots (children first, recursively — built, hot-plug matrix), report `child_removed` per interface | the device tree is honest within one reconcile tick |
|
||||||
| Device manager | `child_removed` / reporter death | prune the child; **reap the bound driver** (built) — the storage driver for that stick dies now, not never | no zombie storage processes; re-report rebinds |
|
| Device manager | `child_removed` / reporter death | prune the child; **reap the bound driver** (built) — the storage driver for that stick dies now, not never | no zombie storage processes; re-report rebinds |
|
||||||
| Storage driver | its own death (it IS the removed device's driver) | nothing — dying is its removal handling; DMA/IOMMU/claims release mechanically at death | in-flight transfers fail visibly to callers, never hang |
|
| Storage driver | its own death (it IS the removed device's driver) | nothing — dying is its removal handling; DMA/IOMMU/claims release mechanically at death | in-flight transfers fail visibly to callers, never hang |
|
||||||
| Volume manager *(removal built; remount bench-pending)* | the storage device leaving the device-manager tree (poll) | kill the filesystem service of that device's volume; its kernel mounts retire | one removal path; mounts never dangle; log persistence stops *cleanly* |
|
| Volume manager *(built)* | a device leaving the tree (poll), a `medium_changed` event, or a dead channel under a still-present device (a crashed driver — `geometry()` liveness probe) | kill that volume's filesystem service (its mounts retire), then re-adopt + remount on return or on the restarted driver's fresh channel | one removal path for all three triggers; mounts never dangle; the manager never serves from behind a dead channel |
|
||||||
| Filesystem service | its block channel dies (`EPEER`) mid-operation, or it is killed by the volume manager | if it observes the death first: flush nothing (the medium is gone), answer in-flight requests with errors, exit; dirty write-back data is **lost and said to be lost** | the unflushed write-back window is dropped on a surprise yank — danos writes no on-disk dirty/clean-shutdown marker today; the process never serves from behind a dead channel |
|
| Filesystem service | its block channel dies (`EPEER`) mid-operation, or it is killed by the volume manager | if it observes the death first: flush nothing (the medium is gone), answer in-flight requests with errors, exit; dirty write-back data is **lost and said to be lost** | the unflushed write-back window is dropped on a surprise yank — danos writes no on-disk dirty/clean-shutdown marker today; the process never serves from behind a dead channel |
|
||||||
| Kernel | backend endpoint death | lazy mount-slot sweep on next resolve (built); ownership-gated `fs_unmount` (built, V0) | resolution under a dead mount is `not_found`, not a hang |
|
| Kernel | backend endpoint death | lazy mount-slot sweep on next resolve (built); ownership-gated `fs_unmount` (built, V0) | resolution under a dead mount is `not_found`, not a hang |
|
||||||
| Application | `not_found` / error on paths under the vanished mount | its own error handling — the contract is honest absence, identical to the path never existing | no operation blocks forever on removed media |
|
| Application | `not_found` / error on paths under the vanished mount | its own error handling — the contract is honest absence, identical to the path never existing | no operation blocks forever on removed media |
|
||||||
|
|||||||
@@ -104,12 +104,19 @@ and /system/logs), closing the two-sticks question honestly.
|
|||||||
**Filesystems (per volume, one process).** The proven unit everywhere from
|
**Filesystems (per volume, one process).** The proven unit everywhere from
|
||||||
Plan 9's `dossrv` to Minix to Fuchsia: block-client + engine + file-protocol
|
Plan 9's `dossrv` to Minix to Fuchsia: block-client + engine + file-protocol
|
||||||
provider in one binary, one process per volume (9front practice; per-volume
|
provider in one binary, one process per volume (9front practice; per-volume
|
||||||
fault isolation is what our supervision makes cheap). fat's shell becomes a
|
fault isolation is what our supervision makes cheap). fat's shell became a
|
||||||
shared *filesystem harness* library before a second engine is written; a
|
shared *filesystem harness* library (`library/kernel/file-system-harness`), and
|
||||||
partition walk is added in the volume manager (`partition.zig`) — the engine's
|
the second engine — **exFAT**, `system/services/exfat` — now reuses it wholesale:
|
||||||
own MBR walk currently remains alongside it; write caching stays
|
the reuse this design promised, proven. exfat is nothing but the exFAT engine +
|
||||||
inside the process (the anti-fsyncgate rule). Each mounts its prefixes into the
|
a near-clone of fat's thin service, full read + write + directories + rename +
|
||||||
kernel mount table itself, exactly as today.
|
on-disk up-case folding, differing only in the format it wraps. A partition walk
|
||||||
|
lives in the volume manager (`partition.zig`), which recognizes fat vs exFAT by
|
||||||
|
VBR and routes each to its engine; write caching stays inside the process (the
|
||||||
|
anti-fsyncgate rule). Each mounts its prefixes into the kernel mount table
|
||||||
|
itself. Two surface limits are shared across both engines and are the vfs
|
||||||
|
layer's, not an exFAT shortcut: file offsets are u32 (a 4 GiB addressable cap),
|
||||||
|
and file names are ASCII bytes (a non-ASCII unit becomes `?`) — teaching the vfs
|
||||||
|
name layer UTF-8 is a separate cross-cutting change.
|
||||||
|
|
||||||
**Kernel: two small changes only.** `fs_unmount` gains ownership (only the
|
**Kernel: two small changes only.** `fs_unmount` gains ownership (only the
|
||||||
mounting endpoint's holder may unmount — possession-is-capability, consistent
|
mounting endpoint's holder may unmount — possession-is-capability, consistent
|
||||||
@@ -145,9 +152,13 @@ matrix-proven shape; genuinely open.
|
|||||||
|
|
||||||
**The pressure points, honestly:**
|
**The pressure points, honestly:**
|
||||||
|
|
||||||
1. **Multi-volume providers are reserved, not implemented.** The volume
|
1. **Multi-volume is built; multi-namespace-per-provider is untried.** The
|
||||||
manager flow assumes one provider, one volume; NVMe namespaces make
|
volume manager adopts every device and spawns one range-confined FAT per
|
||||||
endpoint-per-volume real work with hardware demanding it.
|
partition — several volumes across several devices, or several partitions
|
||||||
|
sharing one device's channel, both proven on USB. What is untried is a single
|
||||||
|
provider exposing several volumes as *namespaces* (NVMe): the endpoint and
|
||||||
|
per-badge range machinery generalizes, but no such driver exists yet to
|
||||||
|
exercise it.
|
||||||
2. **The current transport will bottleneck NVMe.** Synchronous call/reply,
|
2. **The current transport will bottleneck NVMe.** Synchronous call/reply,
|
||||||
one operation in flight, one bounce buffer — fine for a USB2 stick,
|
one operation in flight, one bounce buffer — fine for a USB2 stick,
|
||||||
forfeits an NVMe drive's queue depth and per-queue MSI-X. Correctness
|
forfeits an NVMe drive's queue depth and per-queue MSI-X. Correctness
|
||||||
@@ -198,18 +209,23 @@ matrix-proven shape; genuinely open.
|
|||||||
names it as the 256-byte ceiling's unlock — Fuchsia's FIFO+VMO is the
|
names it as the 256-byte ceiling's unlock — Fuchsia's FIFO+VMO is the
|
||||||
precedent); format-level crash honesty (a Power-Safe-style journaling or COW
|
precedent); format-level crash honesty (a Power-Safe-style journaling or COW
|
||||||
filesystem) once danos outgrows FAT; per-process namespaces.
|
filesystem) once danos outgrows FAT; per-process namespaces.
|
||||||
7. **The media-presence event** (settled in principle; lands with the volume
|
7. **The media-presence event** (the consuming half is BUILT; the
|
||||||
manager): the block protocol gains a pushed event — `medium_changed`, with
|
transport-native signal stays future): the block protocol carries a pushed
|
||||||
present/absent and a change counter — produced by the storage driver from
|
event — `medium_changed`, with present/absent and a change counter —
|
||||||
its transport's native signal (SCSI UNIT ATTENTION / TEST UNIT READY for
|
produced today by the storage driver from a TEST UNIT READY poll (the
|
||||||
USB and ATAPI, PxSSTS for AHCI, namespace-change AER for NVMe) and
|
transport's native signal — SCSI UNIT ATTENTION, PxSSTS for AHCI,
|
||||||
consumed by the volume manager, which runs the SAME kill-retire-remount
|
namespace-change AER for NVMe — is the future refinement in place of the
|
||||||
path it runs on channel death — one lifecycle, two triggers. The driver
|
poll) and now **consumed** by the volume manager, which subscribes per
|
||||||
reports presence, never content; a pushed event carries no capability,
|
device and runs the SAME kill-retire-remount path it runs on channel death.
|
||||||
which the kernel already guarantees. The device staying while its medium
|
The driver reports presence, never content; a pushed event carries no
|
||||||
leaves is the one removable-media case the channel-death trigger cannot
|
capability, which the kernel already guarantees. The device staying while
|
||||||
see; without this event a swapped SD card would be served with the old
|
its medium leaves is the one removable-media case the channel-death trigger
|
||||||
card's filesystem state.
|
cannot see; without this event a swapped SD card would be served with the
|
||||||
|
old card's filesystem state. A THIRD trigger closes the last gap — a
|
||||||
|
storage driver that *crashes* while its device stays present: channel death
|
||||||
|
there is invisible to presence polling, so the volume manager probes channel
|
||||||
|
liveness (`geometry()`) each tick and reaps-then-rebuilds the volume on the
|
||||||
|
restarted driver's fresh channel. One lifecycle, three triggers.
|
||||||
8. **Volume identity, and the mount map as danos's fstab** (settled). The
|
8. **Volume identity, and the mount map as danos's fstab** (settled). The
|
||||||
lesson is Linux's own history: fstab keyed on `/dev/sda1` for years and
|
lesson is Linux's own history: fstab keyed on `/dev/sda1` for years and
|
||||||
broke whenever a drive changed ports or enumeration order; `UUID=` entries
|
broke whenever a drive changed ports or enumeration order; `UUID=` entries
|
||||||
@@ -233,13 +249,16 @@ matrix-proven shape; genuinely open.
|
|||||||
Consequences, each mechanical once identity keys the map: **moving a drive
|
Consequences, each mechanical once identity keys the map: **moving a drive
|
||||||
to a different port changes nothing** — same identity, same mount point,
|
to a different port changes nothing** — same identity, same mount point,
|
||||||
whether USB port, hub depth, SATA port, or a stick that left as USB and
|
whether USB port, hub depth, SATA port, or a stick that left as USB and
|
||||||
returned in a SATA dock; **replug remounts at the same path** (a map lookup
|
returned in a SATA dock; **replug remounts at the same path** (the id-path is
|
||||||
once the map exists; today's single volume re-probes and remounts at the
|
content-derived, so a volume returns to `/volumes/<id>` wherever it reappears;
|
||||||
fixed prefix, and remount-on-replug is bench-verified, not QEMU-tested); **the boot volume** is the
|
the re-adopt+remount code path is QEMU-proven by the driver-crash rebuild, but
|
||||||
recorded identity of the volume carrying `/system/configuration`, findable
|
remount on a *physical* replug end-to-end is bench-pending, not QEMU-testable,
|
||||||
on any port; and **duplicate identity is a policy case, not a surprise** —
|
because QEMU can't re-present the boot-controller device); **the boot volume** is the volume
|
||||||
two cloned sticks at once: first keeps the mapped name, second mounts
|
that resolves `/system/configuration` on its own media, findable on any port or
|
||||||
suffixed and is logged loudly, never silently shadowed. Unknown identities
|
partition; and **duplicate identity is a known S4 gap** — two cloned sticks
|
||||||
|
share one content id, so today they collide on `/volumes/<id>` (the kernel
|
||||||
|
remount-replaces; the last wins) and each boot-volume claim is logged loudly.
|
||||||
|
Distinguishing them with a suffix is arbitration, deferred to S4. Unknown identities
|
||||||
mount under a derived name (sanitized label, else generated) at
|
mount under a derived name (sanitized label, else generated) at
|
||||||
`/volumes/<name>` — the hierarchy's documented home for attached media,
|
`/volumes/<name>` — the hierarchy's documented home for attached media,
|
||||||
which stands: `/system` is what danos IS; attached media is what it isn't.
|
which stands: `/system` is what danos IS; attached media is what it isn't.
|
||||||
|
|||||||
@@ -7,12 +7,25 @@
|
|||||||
//! `runtime.dma.alloc`), so whole sectors move without crossing the IPC size
|
//! `runtime.dma.alloc`), so whole sectors move without crossing the IPC size
|
||||||
//! limit — the same handoff usb-storage uses toward the controller.
|
//! limit — the same handoff usb-storage uses toward the controller.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
const envelope = @import("envelope");
|
const envelope = @import("envelope");
|
||||||
const ipc = @import("ipc");
|
const ipc = @import("ipc");
|
||||||
const block_protocol = @import("block-protocol");
|
const block_protocol = @import("block-protocol");
|
||||||
|
|
||||||
const Protocol = block_protocol.Protocol;
|
const Protocol = block_protocol.Protocol;
|
||||||
|
|
||||||
|
/// The medium_changed event payload, re-exported so a consumer decodes it without
|
||||||
|
/// reaching into the wire-format module.
|
||||||
|
pub const MediumChanged = block_protocol.MediumChanged;
|
||||||
|
|
||||||
|
/// Decode a medium_changed event from a buffered-message payload a subscriber
|
||||||
|
/// received (a `Received.isMessage` wake). Null if the bytes are too short to be
|
||||||
|
/// one — a caller ignores anything that is not a well-formed event.
|
||||||
|
pub fn decodeMediumChanged(payload: []const u8) ?MediumChanged {
|
||||||
|
if (payload.len < envelope.prefix_size + @sizeOf(MediumChanged)) return null;
|
||||||
|
return std.mem.bytesToValue(MediumChanged, payload[envelope.prefix_size..][0..@sizeOf(MediumChanged)]);
|
||||||
|
}
|
||||||
|
|
||||||
pub const Geometry = struct { block_size: u32, block_count: u64 };
|
pub const Geometry = struct { block_size: u32, block_count: u64 };
|
||||||
|
|
||||||
pub const Device = struct {
|
pub const Device = struct {
|
||||||
@@ -88,6 +101,30 @@ pub const Device = struct {
|
|||||||
if (status.status != 0) return null;
|
if (status.status != 0) return null;
|
||||||
return reply[0..answer.len];
|
return reply[0..answer.len];
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Subscribe `subscriber` (an endpoint) to this device's medium_changed
|
||||||
|
/// events: the reserved `subscribe` verb carries the subscriber's endpoint as
|
||||||
|
/// the capability, and the driver then ipc.sends each medium transition to it.
|
||||||
|
pub fn subscribeMedium(self: Device, subscriber: ipc.Handle) bool {
|
||||||
|
var packet: [block_protocol.message_maximum]u8 = undefined;
|
||||||
|
const framed = envelope.encodeSubscribe(0, &packet) orelse return false; // interest 0: every event (block has one)
|
||||||
|
var reply: [block_protocol.message_maximum]u8 = undefined;
|
||||||
|
const answer = ipc.callCap(self.endpoint, framed, &reply, subscriber) catch return false;
|
||||||
|
const status = envelope.statusOf(reply[0..answer.len]) orelse return false;
|
||||||
|
return status.status == 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Unsubscribe from this device's medium_changed events. Call before closing
|
||||||
|
/// the channel so the driver's bounded subscriber table frees the slot rather
|
||||||
|
/// than holding a dead endpoint until an exit sweep notices.
|
||||||
|
pub fn unsubscribeMedium(self: Device) bool {
|
||||||
|
var packet: [block_protocol.message_maximum]u8 = undefined;
|
||||||
|
const framed = envelope.encodeUnsubscribe(&packet) orelse return false;
|
||||||
|
var reply: [block_protocol.message_maximum]u8 = undefined;
|
||||||
|
const answer = ipc.callCap(self.endpoint, framed, &reply, null) catch return false;
|
||||||
|
const status = envelope.statusOf(reply[0..answer.len]) orelse return false;
|
||||||
|
return status.status == 0;
|
||||||
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
// There is deliberately no open-by-name here: `block` is not a registry name.
|
// There is deliberately no open-by-name here: `block` is not a registry name.
|
||||||
|
|||||||
@@ -61,10 +61,14 @@ pub fn Server(comptime Engine: type) type {
|
|||||||
/// caller does the filesystem-specific bring-up (find the block
|
/// caller does the filesystem-specific bring-up (find the block
|
||||||
/// device, set up DMA, mount the engine) and returns a `Volume`.
|
/// device, set up DMA, mount the engine) and returns a `Volume`.
|
||||||
bringUp: *const fn (endpoint: ipc.Handle) ?Volume,
|
bringUp: *const fn (endpoint: ipc.Handle) ?Volume,
|
||||||
/// The vfs contract name to bind. A filesystem serving one volume
|
/// A contract name to bind under /protocol, or null to bind none. In
|
||||||
/// binds "vfs" today; the volume-manager era hands each per-volume
|
/// the volume-manager era every filesystem is a per-volume process and
|
||||||
/// process its own establishment and this fades.
|
/// clients reach it through the kernel mount table — fs_resolve routes
|
||||||
service_name: ?[]const u8 = "vfs",
|
/// a path to its backing endpoint by prefix — so no filesystem binds a
|
||||||
|
/// shared name. Two volumes would collide on one: the second's bind is
|
||||||
|
/// refused and service.run would exit, so its volume never mounts. The
|
||||||
|
/// endpoint still serves as the mount backend without a name.
|
||||||
|
service_name: ?[]const u8 = null,
|
||||||
};
|
};
|
||||||
|
|
||||||
// --- the harness's own state, one set per instantiation ---------------
|
// --- the harness's own state, one set per instantiation ---------------
|
||||||
|
|||||||
@@ -67,6 +67,15 @@ pub const Callbacks = struct {
|
|||||||
/// A notification that is not a signal — a subscribed exit event, a bound
|
/// A notification that is not a signal — a subscribed exit event, a bound
|
||||||
/// IRQ, a timer landing. The raw badge; decode with the ipc helpers.
|
/// IRQ, a timer landing. The raw badge; decode with the ipc helpers.
|
||||||
on_notification: ?*const fn (badge: u64) void = null,
|
on_notification: ?*const fn (badge: u64) void = null,
|
||||||
|
/// A buffered async message (`Received.isMessage`): a pushed event from a
|
||||||
|
/// provider this service subscribed to, its payload in the receive buffer.
|
||||||
|
/// Unlike `on_message`, it never goes through the protocol dispatch — so an
|
||||||
|
/// event whose reserved op number collides with one of this service's own
|
||||||
|
/// verbs (a `block` `medium_changed` reaching the volume manager, whose own
|
||||||
|
/// protocol numbers `hello` the same) is decoded by hand here, not
|
||||||
|
/// mis-dispatched. Default null: the badge alone still reaches
|
||||||
|
/// `on_notification`, exactly as before this callback existed.
|
||||||
|
on_buffered_message: ?*const fn (message: []const u8) void = null,
|
||||||
/// The reload signal. Default: ignored.
|
/// The reload signal. Default: ignored.
|
||||||
on_reload: ?*const fn () void = null,
|
on_reload: ?*const fn () void = null,
|
||||||
/// The terminate signal, called before the loop returns. The clean exit is
|
/// The terminate signal, called before the loop returns. The clean exit is
|
||||||
@@ -372,6 +381,13 @@ pub fn run(comptime maximum_message: usize, callbacks: Callbacks) void {
|
|||||||
if (got.isChildExit()) {
|
if (got.isChildExit()) {
|
||||||
if (callbacks.subscribers) |subscribers| subscribers.forget(got.childProcessId());
|
if (callbacks.subscribers) |subscribers| subscribers.forget(got.childProcessId());
|
||||||
}
|
}
|
||||||
|
// A buffered async message (a pushed event) carries a payload; hand it
|
||||||
|
// to the service that asked for it. The badge still reaches
|
||||||
|
// on_notification below, so a coalesced timer/exit riding the same wake
|
||||||
|
// is not lost — and a service without this callback is unchanged.
|
||||||
|
if (got.isMessage()) {
|
||||||
|
if (callbacks.on_buffered_message) |onBuffered| onBuffered(receive[0..got.len]);
|
||||||
|
}
|
||||||
if (callbacks.on_notification) |onNotification| onNotification(got.badge);
|
if (callbacks.on_notification) |onNotification| onNotification(got.badge);
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -5,3 +5,4 @@
|
|||||||
#
|
#
|
||||||
# signature, binary
|
# signature, binary
|
||||||
fat, /system/services/fat
|
fat, /system/services/fat
|
||||||
|
exfat, /system/services/exfat
|
||||||
|
|||||||
|
@@ -79,6 +79,7 @@
|
|||||||
# 'kernel' as the supervisor. Nothing else changes: the binary must still match.
|
# 'kernel' as the supervisor. Nothing else changes: the binary must still match.
|
||||||
/system/services/input, kernel, bind, input
|
/system/services/input, kernel, bind, input
|
||||||
/system/services/device-manager, kernel, bind, device-manager
|
/system/services/device-manager, kernel, bind, device-manager
|
||||||
|
/system/services/volume-manager, kernel, bind, volume-manager
|
||||||
/system/services/fat, kernel, bind, vfs
|
/system/services/fat, kernel, bind, vfs
|
||||||
/system/services/display, kernel, bind, display
|
/system/services/display, kernel, bind, display
|
||||||
/system/services/discovery, kernel, bind, power
|
/system/services/discovery, kernel, bind, power
|
||||||
@@ -102,9 +103,14 @@
|
|||||||
# own endpoint (the mouse-listener thread opens /protocol/display like any other
|
# own endpoint (the mouse-listener thread opens /protocol/display like any other
|
||||||
# client — threads share no handles), and the input stream that moves the cursor.
|
# client — threads share no handles), and the input stream that moves the cursor.
|
||||||
/system/services/fat, /system/services/volume-manager, open, volume-manager
|
/system/services/fat, /system/services/volume-manager, open, volume-manager
|
||||||
|
# exfat reaches the volume manager the same way — the second engine, same lineage.
|
||||||
|
/system/services/exfat, /system/services/volume-manager, open, volume-manager
|
||||||
# The volume manager reaches the device manager to be routed to each storage
|
# The volume manager reaches the device manager to be routed to each storage
|
||||||
# provider's block channel, then confines a filesystem to each volume.
|
# provider's block channel, then confines a filesystem to each volume.
|
||||||
/system/services/volume-manager, /system/services/init, open, device-manager
|
/system/services/volume-manager, /system/services/init, open, device-manager
|
||||||
|
# ...and again under the kernel supervisor for the manual-tree drills (S5's
|
||||||
|
# volume-driver-restart spawns the volume manager directly, not via init).
|
||||||
|
/system/services/volume-manager, kernel, open, device-manager
|
||||||
/system/services/display, /system/services/init, open, scanout
|
/system/services/display, /system/services/init, open, scanout
|
||||||
/system/services/display, /system/services/init, open, display
|
/system/services/display, /system/services/init, open, display
|
||||||
/system/services/display, /system/services/init, open, input
|
/system/services/display, /system/services/init, open, input
|
||||||
|
|||||||
|
Can't render this file because it contains an unexpected character in line 12 and column 15.
|
@@ -227,6 +227,10 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void {
|
|||||||
usbStorageTest(boot_information);
|
usbStorageTest(boot_information);
|
||||||
} else if (eql(case, "fat-mount")) {
|
} else if (eql(case, "fat-mount")) {
|
||||||
fatMountTest(boot_information);
|
fatMountTest(boot_information);
|
||||||
|
} else if (eql(case, "exfat-volume")) {
|
||||||
|
exfatVolumeTest(boot_information);
|
||||||
|
} else if (eql(case, "volume-driver-restart")) {
|
||||||
|
volumeDriverRestartTest(boot_information);
|
||||||
} else if (eql(case, "device-list")) {
|
} else if (eql(case, "device-list")) {
|
||||||
deviceListTest(boot_information);
|
deviceListTest(boot_information);
|
||||||
} else if (eql(case, "pci-scan")) {
|
} else if (eql(case, "pci-scan")) {
|
||||||
@@ -3036,6 +3040,32 @@ fn fatMountTest(boot_information: *const BootInformation) void {
|
|||||||
result();
|
result();
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// The exFAT mount chain (S4): boot the full tree, which brings up the USB storage
|
||||||
|
/// chain. The harness attaches a SECOND device — a data-only exFAT volume — beside
|
||||||
|
/// the FAT boot volume, so the volume manager spawns the exfat service for it
|
||||||
|
/// (content-routed, its id-path /volumes/exfat-<serial>). Then spawn exfat-test,
|
||||||
|
/// which reads the seeded file and mutates through the mount. The reuse of the
|
||||||
|
/// shared harness by a second engine is proven end to end here.
|
||||||
|
fn exfatVolumeTest(boot_information: *const BootInformation) void {
|
||||||
|
log("DANOS-TEST-BEGIN: exfat-volume\n", .{});
|
||||||
|
if (boot_information.initial_ramdisk_len == 0) {
|
||||||
|
check("bootloader handed over the initial_ramdisk", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
const ramdisk = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||||
|
const rd = initial_ramdisk.Reader.init(ramdisk) orelse {
|
||||||
|
check("initial_ramdisk image is valid", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
process.setInitialRamdisk(ramdisk);
|
||||||
|
const init_ok = if (process.spawnBundled("/system/services/init")) true else |_| false;
|
||||||
|
check("init spawned (boots the tree, incl. the volume manager)", init_ok);
|
||||||
|
check("exfat-test client spawned", spawnNamed(rd, "exfat-test"));
|
||||||
|
result();
|
||||||
|
}
|
||||||
|
|
||||||
/// Per-sender range confinement (V2a, docs/volume-manager-plan.md): the fixture
|
/// Per-sender range confinement (V2a, docs/volume-manager-plan.md): the fixture
|
||||||
/// acquires the block channel, confines ITSELF to a sub-range, and asserts it
|
/// acquires the block channel, confines ITSELF to a sub-range, and asserts it
|
||||||
/// cannot read past that range or widen it. Boots init in REGISTRY-ONLY mode
|
/// cannot read past that range or widen it. Boots init in REGISTRY-ONLY mode
|
||||||
@@ -3657,6 +3687,44 @@ fn displayReattachTest(boot_information: *const BootInformation) void {
|
|||||||
while (true) scheduler.yield();
|
while (true) scheduler.yield();
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Storage-driver-crash rebuild (S5): the device manager runs in
|
||||||
|
/// "test-storage-restart" mode and kills the usb-storage driver once, a moment
|
||||||
|
/// after its volume has mounted. The driver's device stays in the tree, so the
|
||||||
|
/// volume manager's presence poll alone would miss the death and leave fat wedged
|
||||||
|
/// on a dead channel; its channel-liveness probe must notice, reap the volume, and
|
||||||
|
/// rebuild on the restarted driver's fresh channel — a SECOND mount of the same
|
||||||
|
/// id-path is the proof. (A pre-S5 manager, checking only device presence, never
|
||||||
|
/// reaps, so the second mount never appears.)
|
||||||
|
fn volumeDriverRestartTest(boot_information: *const BootInformation) void {
|
||||||
|
log("DANOS-TEST-BEGIN: volume-driver-restart\n", .{});
|
||||||
|
if (boot_information.initial_ramdisk_len == 0) {
|
||||||
|
check("bootloader handed over an initial_ramdisk", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||||
|
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||||
|
check("initial_ramdisk image is valid", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
process.setInitialRamdisk(image);
|
||||||
|
_ = spawnRegistry(rd);
|
||||||
|
var manager: u32 = 0;
|
||||||
|
var i: u32 = 0;
|
||||||
|
while (i < rd.count) : (i += 1) {
|
||||||
|
const item = rd.entry(i) orelse continue;
|
||||||
|
if (!eql(initial_ramdisk.basename(item.name), "device-manager")) continue;
|
||||||
|
manager = process.spawnProcessSupervised(item.blob, 4, &.{ item.name, "test-storage-restart" }, scheduler.currentId(), null) catch 0;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
check("device-manager spawned (test-storage-restart mode)", manager != 0);
|
||||||
|
check("volume-manager spawned", spawnNamed(rd, "volume-manager"));
|
||||||
|
check("fat-test client spawned", spawnNamed(rd, "fat-test"));
|
||||||
|
scheduler.setPriority(1); // below the tree, so it runs
|
||||||
|
while (true) scheduler.yield();
|
||||||
|
}
|
||||||
|
|
||||||
/// Process arguments, end to end: spawn args-echo bare (its argv[0] is the
|
/// Process arguments, end to end: spawn args-echo bare (its argv[0] is the
|
||||||
/// initial-ramdisk name). Instance 1 sees argc == 1 and respawns itself through
|
/// initial-ramdisk name). Instance 1 sees argc == 1 and respawns itself through
|
||||||
/// `system_spawn` with the extra arguments "alpha beta-42" — the syscall argument
|
/// `system_spawn` with the extra arguments "alpha beta-42" — the syscall argument
|
||||||
|
|||||||
+25
-6
@@ -55,7 +55,17 @@ fn tokenIndex(t: u64) u64 {
|
|||||||
|
|
||||||
// --- the mount table ---------------------------------------------------------
|
// --- the mount table ---------------------------------------------------------
|
||||||
|
|
||||||
pub const maximum_mounts = 8;
|
/// bound: prefixes mounted in the kernel VFS table at once
|
||||||
|
/// decided-by: ours
|
||||||
|
/// protects: the `mounts` table below
|
||||||
|
/// at-limit: refuse - installMount returns false and mountBackend propagates it;
|
||||||
|
/// the mounting filesystem's harness logs "could not mount <prefix>" and the
|
||||||
|
/// mount simply does not exist (no silent success). Budget: the initrd's
|
||||||
|
/// top-level dirs (/system, /test) plus one id-path mount per volume and the
|
||||||
|
/// system volume's two FHS rewrites — a few over the volume manager's
|
||||||
|
/// maximum_volumes (16); 32 leaves headroom.
|
||||||
|
/// observed-by: the harness "file-system: could not mount <prefix>" ring line
|
||||||
|
pub const maximum_mounts = 32;
|
||||||
const maximum_prefix = 64;
|
const maximum_prefix = 64;
|
||||||
const maximum_rewrite = 32;
|
const maximum_rewrite = 32;
|
||||||
|
|
||||||
@@ -156,7 +166,10 @@ pub fn setInitialRamdisk(image: []const u8) void {
|
|||||||
for (directories[0..directory_count], 0..) |*d, index| {
|
for (directories[0..directory_count], 0..) |*d, index| {
|
||||||
const parent = parentOf(d.slice());
|
const parent = parentOf(d.slice());
|
||||||
d.parent = directoryIndex(parent) orelse index;
|
d.parent = directoryIndex(parent) orelse index;
|
||||||
if (parent.len == 1) installMount(d.slice(), .kernel_initrd, null, "");
|
// Boot-time install of one mount per top-level initrd dir (/system, /test):
|
||||||
|
// provably few, far under maximum_mounts, so a full table here is
|
||||||
|
// impossible — but discard the result explicitly rather than assume it.
|
||||||
|
if (parent.len == 1) _ = installMount(d.slice(), .kernel_initrd, null, "");
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -167,7 +180,12 @@ fn directoryIndex(path: []const u8) ?usize {
|
|||||||
return null;
|
return null;
|
||||||
}
|
}
|
||||||
|
|
||||||
fn installMount(prefix: []const u8, kind: MountKind, backend: ?*ipc.Endpoint, rewrite: []const u8) void {
|
/// Install (or remount-replace) a prefix. Returns false when the table is full
|
||||||
|
/// and no slot could be claimed — the caller must surface that, never report a
|
||||||
|
/// dropped mount as success. A remount of an already-mounted prefix reuses its
|
||||||
|
/// slot and always succeeds; a /protocol remount is refused-as-noop (returns
|
||||||
|
/// true: the first mount stands, nothing is dropped).
|
||||||
|
fn installMount(prefix: []const u8, kind: MountKind, backend: ?*ipc.Endpoint, rewrite: []const u8) bool {
|
||||||
// Remount replaces: a restarted backend re-mounts its prefix.
|
// Remount replaces: a restarted backend re-mounts its prefix.
|
||||||
var slot: ?*Mount = null;
|
var slot: ?*Mount = null;
|
||||||
for (&mounts) |*m| {
|
for (&mounts) |*m| {
|
||||||
@@ -176,19 +194,20 @@ fn installMount(prefix: []const u8, kind: MountKind, backend: ?*ipc.Endpoint, re
|
|||||||
// restarted FAT retakes /volumes/usb; letting it retake /protocol
|
// restarted FAT retakes /volumes/usb; letting it retake /protocol
|
||||||
// would hand the whole naming layer to whoever asked second.
|
// would hand the whole naming layer to whoever asked second.
|
||||||
// First mount wins, and init (PID 1) is always first.
|
// First mount wins, and init (PID 1) is always first.
|
||||||
if (std.mem.eql(u8, prefix, protocol_root)) return;
|
if (std.mem.eql(u8, prefix, protocol_root)) return true;
|
||||||
if (m.backend) |old| ipc.dropRef(old);
|
if (m.backend) |old| ipc.dropRef(old);
|
||||||
slot = m;
|
slot = m;
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
if (slot == null and !m.used) slot = m;
|
if (slot == null and !m.used) slot = m;
|
||||||
}
|
}
|
||||||
const m = slot orelse return;
|
const m = slot orelse return false;
|
||||||
m.* = .{ .used = true, .kind = kind, .backend = backend };
|
m.* = .{ .used = true, .kind = kind, .backend = backend };
|
||||||
@memcpy(m.prefix[0..prefix.len], prefix);
|
@memcpy(m.prefix[0..prefix.len], prefix);
|
||||||
m.prefix_len = prefix.len;
|
m.prefix_len = prefix.len;
|
||||||
@memcpy(m.rewrite[0..rewrite.len], rewrite);
|
@memcpy(m.rewrite[0..rewrite.len], rewrite);
|
||||||
m.rewrite_len = rewrite.len;
|
m.rewrite_len = rewrite.len;
|
||||||
|
return true;
|
||||||
}
|
}
|
||||||
|
|
||||||
// --- resolve -----------------------------------------------------------------
|
// --- resolve -----------------------------------------------------------------
|
||||||
@@ -406,7 +425,7 @@ pub fn mountBackend(prefix: []const u8, backend: *ipc.Endpoint, rewrite: []const
|
|||||||
if (!isInitrdCarveOut(prefix)) return false;
|
if (!isInitrdCarveOut(prefix)) return false;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
installMount(prefix, .backend, backend, rewrite);
|
if (!installMount(prefix, .backend, backend, rewrite)) return false; // table full
|
||||||
for (&mounts) |*m| {
|
for (&mounts) |*m| {
|
||||||
if (m.used and std.mem.eql(u8, m.prefixSlice(), prefix)) m.owner = owner;
|
if (m.used and std.mem.eql(u8, m.prefixSlice(), prefix)) m.owner = owner;
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -170,6 +170,8 @@ var test_usb_killed = false;
|
|||||||
var test_pci_restart_mode = false;
|
var test_pci_restart_mode = false;
|
||||||
var test_scanout_restart_mode = false;
|
var test_scanout_restart_mode = false;
|
||||||
var test_scanout_killed = false;
|
var test_scanout_killed = false;
|
||||||
|
var test_storage_restart_mode = false;
|
||||||
|
var test_storage_killed = false;
|
||||||
var test_kill_pid: u32 = 0;
|
var test_kill_pid: u32 = 0;
|
||||||
var test_kill_due_ns: u64 = 0;
|
var test_kill_due_ns: u64 = 0;
|
||||||
|
|
||||||
@@ -600,6 +602,16 @@ fn onHello(_: void, invocation: Invocation(device_manager_protocol.Hello), _: An
|
|||||||
test_kill_due_ns = time.clock() + 1_500_000_000;
|
test_kill_due_ns = time.clock() + 1_500_000_000;
|
||||||
_ = time.timerOnce(manager_endpoint, 1600);
|
_ = time.timerOnce(manager_endpoint, 1600);
|
||||||
}
|
}
|
||||||
|
// Storage-driver-crash drill (S5): once, a moment after usb-storage hellos —
|
||||||
|
// long enough that its volume has mounted — kill it. The manager re-delegates
|
||||||
|
// the still-present device to a restarted driver on a fresh channel; the volume
|
||||||
|
// manager's channel-liveness probe must notice the dead channel and rebuild.
|
||||||
|
if (test_storage_restart_mode and !test_storage_killed and std.mem.eql(u8, driver.name(), "/system/drivers/usb-storage")) {
|
||||||
|
test_storage_killed = true;
|
||||||
|
test_kill_pid = invocation.sender;
|
||||||
|
test_kill_due_ns = time.clock() + 2_000_000_000; // after the ~0.6s mount
|
||||||
|
_ = time.timerOnce(manager_endpoint, 2100);
|
||||||
|
}
|
||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -763,6 +775,7 @@ pub fn main(init: process.Init) void {
|
|||||||
test_usb_restart_mode = std.mem.eql(u8, mode, "test-usb-restart");
|
test_usb_restart_mode = std.mem.eql(u8, mode, "test-usb-restart");
|
||||||
test_pci_restart_mode = std.mem.eql(u8, mode, "test-pci-restart");
|
test_pci_restart_mode = std.mem.eql(u8, mode, "test-pci-restart");
|
||||||
test_scanout_restart_mode = std.mem.eql(u8, mode, "test-scanout-restart");
|
test_scanout_restart_mode = std.mem.eql(u8, mode, "test-scanout-restart");
|
||||||
|
test_storage_restart_mode = std.mem.eql(u8, mode, "test-storage-restart");
|
||||||
}
|
}
|
||||||
service.run(device_manager_protocol.message_maximum, .{
|
service.run(device_manager_protocol.message_maximum, .{
|
||||||
.service = "device-manager",
|
.service = "device-manager",
|
||||||
|
|||||||
@@ -0,0 +1,34 @@
|
|||||||
|
//! The exfat service as a binary package (docs/build-packages-plan.md):
|
||||||
|
//! this file names the binary and EXACTLY the modules its source imports —
|
||||||
|
//! build-support resolves each name from the domains this zon declares.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const build_support = @import("build-support");
|
||||||
|
|
||||||
|
pub fn build(b: *std.Build) void {
|
||||||
|
const exe = build_support.userBinary(b, .{
|
||||||
|
.name = "exfat",
|
||||||
|
.root_source_file = b.path("exfat.zig"),
|
||||||
|
.imports = &.{
|
||||||
|
"block", "channel", "envelope", "file-system-harness",
|
||||||
|
"ipc", "logging", "memory", "process",
|
||||||
|
"time", "volume-manager-protocol",
|
||||||
|
},
|
||||||
|
});
|
||||||
|
b.installArtifact(exe);
|
||||||
|
|
||||||
|
// Standalone `zig build test`; the root aggregate depends on this step.
|
||||||
|
const test_step = b.step("test", "Run the exfat unit tests");
|
||||||
|
for ([_][]const u8{
|
||||||
|
"on-disk.zig", // exFAT on-disk struct sizes + geometry + checksums
|
||||||
|
"engine.zig", // exFAT read/write over a RAM-backed image
|
||||||
|
}) |test_root| {
|
||||||
|
const unit_tests = b.addTest(.{
|
||||||
|
.root_module = b.createModule(.{
|
||||||
|
.root_source_file = b.path(test_root),
|
||||||
|
.target = b.resolveTargetQuery(.{}),
|
||||||
|
}),
|
||||||
|
});
|
||||||
|
test_step.dependOn(&b.addRunArtifact(unit_tests).step);
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,16 @@
|
|||||||
|
.{
|
||||||
|
.name = .exfat,
|
||||||
|
.version = "0.0.0",
|
||||||
|
.fingerprint = 0x5eafdf02d20f93dd, // Changing this has security and trust implications.
|
||||||
|
.minimum_zig_version = "0.16.0",
|
||||||
|
.dependencies = .{
|
||||||
|
// build-support supplies the shared recipe; kernel is implicit in
|
||||||
|
// every binary (the root shim + link script live there). The rest
|
||||||
|
// are exactly the homes of this binary's declared imports.
|
||||||
|
.@"build-support" = .{ .path = "../../../build-support" },
|
||||||
|
.kernel = .{ .path = "../../../library/kernel" },
|
||||||
|
.device = .{ .path = "../../../library/device" },
|
||||||
|
.protocol = .{ .path = "../../../library/protocol" },
|
||||||
|
},
|
||||||
|
.paths = .{""},
|
||||||
|
}
|
||||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,200 @@
|
|||||||
|
//! system/services/exfat — the exFAT filesystem service. Like fat.zig, this is
|
||||||
|
//! only the format-specific half: it finds its block device, sets up the DMA
|
||||||
|
//! bounce buffer, mounts the exFAT engine on it, and hands the mounted volume to
|
||||||
|
//! the shared filesystem harness (library/kernel/file-system-harness), which owns
|
||||||
|
//! everything else — vfs serving, the open-node table, mount registration, the
|
||||||
|
//! exit sweep, durable-on-close. The engine (engine.zig) is the pure,
|
||||||
|
//! host-testable format code; on-disk.zig its byte layout.
|
||||||
|
//!
|
||||||
|
//! This service is a near-clone of fat.zig: the second engine reuses the harness
|
||||||
|
//! wholesale, which is the reuse the storage architecture promised
|
||||||
|
//! (docs/file-system-development/storage-architecture.md). The block data path
|
||||||
|
//! never crosses IPC: a DMA bounce buffer is handed to the block driver by
|
||||||
|
//! physical address, and the engine copies sectors in and out.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const channel = @import("channel");
|
||||||
|
const volume_manager_protocol = @import("volume-manager-protocol");
|
||||||
|
const ipc = @import("ipc");
|
||||||
|
const process = @import("process");
|
||||||
|
const block = @import("block");
|
||||||
|
const memory = @import("memory");
|
||||||
|
const logging = @import("logging");
|
||||||
|
const time = @import("time");
|
||||||
|
const engine = @import("engine.zig");
|
||||||
|
const envelope = @import("envelope");
|
||||||
|
const harness = @import("file-system-harness");
|
||||||
|
|
||||||
|
/// The serving harness, specialized for the exFAT engine. One volume per process.
|
||||||
|
const Harness = harness.Server(engine.FileSystem);
|
||||||
|
|
||||||
|
// The engine's BlockDevice, backed by the `.block` driver plus a DMA bounce
|
||||||
|
// buffer the driver reads/writes by physical address.
|
||||||
|
const IpcBlock = struct {
|
||||||
|
device: block.Device,
|
||||||
|
bounce: memory.DmaRegion, // engine.max_transfer_sectors * 512 bytes
|
||||||
|
|
||||||
|
fn readBlocks(context: *anyopaque, lba: u64, count: u32, buffer: []u8) bool {
|
||||||
|
const self: *IpcBlock = @ptrCast(@alignCast(context));
|
||||||
|
if (count == 0 or count > engine.max_transfer_sectors) return false;
|
||||||
|
const len = count * 512;
|
||||||
|
if (!self.device.read(lba, count, self.bounce.physical)) return false;
|
||||||
|
const source: [*]const u8 = @ptrFromInt(self.bounce.virtual);
|
||||||
|
@memcpy(buffer[0..len], source[0..len]);
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
fn writeBlocks(context: *anyopaque, lba: u64, count: u32, buffer: []const u8) bool {
|
||||||
|
const self: *IpcBlock = @ptrCast(@alignCast(context));
|
||||||
|
if (count == 0 or count > engine.max_transfer_sectors) return false;
|
||||||
|
const len = count * 512;
|
||||||
|
const destination: [*]u8 = @ptrFromInt(self.bounce.virtual);
|
||||||
|
@memcpy(destination[0..len], buffer[0..len]);
|
||||||
|
if (!self.device.write(lba, count, self.bounce.physical)) return false;
|
||||||
|
device_dirty = true; // a block reached the device; a close will flush it
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
var ipc_block: IpcBlock = undefined;
|
||||||
|
// Set whenever a block is written, cleared when the device cache is flushed on a
|
||||||
|
// file close — so writes are committed to stable media before a power-off.
|
||||||
|
var device_dirty: bool = false;
|
||||||
|
var filesystem: engine.FileSystem = undefined;
|
||||||
|
/// The volume this exFAT process serves, its id given as argv[1] by the volume
|
||||||
|
/// manager that spawned it. The startup hello names it so the manager returns the
|
||||||
|
/// right volume's channel.
|
||||||
|
var my_volume_id: u64 = 0;
|
||||||
|
|
||||||
|
/// The volume's own mount path, handed in as argv[2] by the volume manager: the
|
||||||
|
/// volume's content id-path (e.g. /volumes/exfat-12345678). Defaults to
|
||||||
|
/// /volumes/exfat only for a bare launch with no argument; the manager always
|
||||||
|
/// passes it. The slice points into the entry block, valid for the process life.
|
||||||
|
var volume_mount_prefix: []const u8 = "/volumes/exfat";
|
||||||
|
|
||||||
|
/// The mounts this volume installs: its own root, plus — only if it is the boot
|
||||||
|
/// volume (it resolves /system/configuration) — the two FHS rewrites, so the
|
||||||
|
/// logger's /system/logs stays decoupled from which volume backs it. Boot-volume
|
||||||
|
/// detection is by content, so it works no matter which volume carries /system.
|
||||||
|
/// bound: mounts one volume installs (its root + the two boot rewrites)
|
||||||
|
/// decided-by: ours
|
||||||
|
/// protects: the mount_specs array
|
||||||
|
/// at-limit: truncate - unreachable today (fixed at 3); more configured mounts
|
||||||
|
/// would need this raised, a deliberate change
|
||||||
|
/// observed-by: a mount silently missing from the harness's mount log
|
||||||
|
const maximum_mounts_per_volume = 4;
|
||||||
|
var mount_specs: [maximum_mounts_per_volume]harness.MountSpec = undefined;
|
||||||
|
|
||||||
|
/// Get this volume's block channel from the volume manager (establishment by
|
||||||
|
/// lineage — `block` is not a registry name). The manager spawned this process,
|
||||||
|
/// confined it to its partition, and answers the hello with the channel; the
|
||||||
|
/// channel is range-confined to this process's badge. Null until the manager has
|
||||||
|
/// the volume ready — this retries.
|
||||||
|
fn acquireVolume() ?block.Device {
|
||||||
|
var attempts: u32 = 0;
|
||||||
|
const vm = while (attempts < 500) : (attempts += 1) {
|
||||||
|
if (channel.openEndpoint("volume-manager")) |handle| break handle;
|
||||||
|
time.sleepMillis(20);
|
||||||
|
} else return null;
|
||||||
|
|
||||||
|
attempts = 0;
|
||||||
|
while (attempts < 500) : (attempts += 1) {
|
||||||
|
var packet: [volume_manager_protocol.message_maximum]u8 = undefined;
|
||||||
|
const framed = volume_manager_protocol.Protocol.encodeRequest(.hello, my_volume_id, .{}, &.{}, &packet) orelse return null;
|
||||||
|
var reply: [volume_manager_protocol.message_maximum]u8 = undefined;
|
||||||
|
const answered = ipc.callCap(vm, framed, &reply, null) catch return null;
|
||||||
|
const status = envelope.statusOf(reply[0..answered.len]) orelse return null;
|
||||||
|
if (status.status != 0) {
|
||||||
|
if (answered.cap) |stray| _ = ipc.close(stray);
|
||||||
|
_ = logging.write("/system/services/exfat: volume manager refused the hello\n");
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
if (answered.cap) |bus| return .{ .endpoint = bus };
|
||||||
|
// Acked with no channel: the volume is not ready yet — retry.
|
||||||
|
time.sleepMillis(20);
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Durable-on-close: commit the device write cache if any block reached it since
|
||||||
|
/// the last flush. The harness calls this on every close; the dirty check keeps
|
||||||
|
/// it cheap.
|
||||||
|
fn flushIfDirty() void {
|
||||||
|
if (device_dirty) {
|
||||||
|
_ = ipc_block.device.flush();
|
||||||
|
device_dirty = false;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// exFAT bring-up: find the block device, set up DMA, mount the engine, and hand
|
||||||
|
/// the volume to the harness — or null to retry on the harness's timer.
|
||||||
|
fn exfatBringUp(endpoint: ipc.Handle) ?Harness.Volume {
|
||||||
|
_ = endpoint;
|
||||||
|
const device = acquireVolume() orelse return null;
|
||||||
|
const geometry = device.geometry() orelse {
|
||||||
|
_ = logging.write("/system/services/exfat: block geometry unavailable\n");
|
||||||
|
return null;
|
||||||
|
};
|
||||||
|
// Shareable so the buffer's capability can be attached down the chain, making
|
||||||
|
// its physical addresses reachable under an enforcing IOMMU. No-op otherwise.
|
||||||
|
const bounce = memory.dmaAlloc(engine.max_transfer_sectors * 512, memory.dma_coherent | memory.dma_shareable) orelse return null;
|
||||||
|
if (bounce.handle) |handle| {
|
||||||
|
// Attach, detach, and attach again: the round trip exercises BOTH verbs of
|
||||||
|
// the DMA-window lifecycle through the whole chain on every boot.
|
||||||
|
if (!device.attach(handle)) {
|
||||||
|
_ = logging.write("/system/services/exfat: could not attach the DMA bounce buffer\n");
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
if (!device.detach(handle)) {
|
||||||
|
_ = logging.write("/system/services/exfat: could not detach the DMA bounce buffer\n");
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
if (!device.attach(handle)) {
|
||||||
|
_ = logging.write("/system/services/exfat: could not re-attach the DMA bounce buffer\n");
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
_ = ipc.close(handle); // the binding holds its own reference now
|
||||||
|
}
|
||||||
|
ipc_block = .{ .device = device, .bounce = bounce };
|
||||||
|
|
||||||
|
const block_device = engine.BlockDevice{
|
||||||
|
.context = &ipc_block,
|
||||||
|
.block_size = geometry.block_size,
|
||||||
|
.block_count = geometry.block_count,
|
||||||
|
.readBlocksFn = IpcBlock.readBlocks,
|
||||||
|
.writeBlocksFn = IpcBlock.writeBlocks,
|
||||||
|
};
|
||||||
|
filesystem = engine.FileSystem.mount(block_device) orelse {
|
||||||
|
_ = logging.write("/system/services/exfat: not an exFAT filesystem\n");
|
||||||
|
return null;
|
||||||
|
};
|
||||||
|
std.log.info("mounted exFAT ({d} clusters, {d} sectors/cluster, serial 0x{x})", .{ filesystem.geometry.cluster_count, filesystem.geometry.sectors_per_cluster, filesystem.geometry.volume_serial_number });
|
||||||
|
|
||||||
|
// The volume mounts at its id-path (argv[2]). The boot/system volume — the one
|
||||||
|
// carrying the /system tree — additionally installs the two FHS rewrites, by
|
||||||
|
// CONTENT: it resolves /system/configuration on its own media. A data volume
|
||||||
|
// mounts only at its id-path and never shadows the running system.
|
||||||
|
mount_specs[0] = .{ .prefix = volume_mount_prefix };
|
||||||
|
var mount_count: usize = 1;
|
||||||
|
if (filesystem.resolve("/system/configuration") != null) {
|
||||||
|
std.log.info("volume {d} carries the system tree; backing /system/configuration and /system/logs", .{my_volume_id});
|
||||||
|
mount_specs[1] = .{ .prefix = "/system/configuration", .rewrite = "/system/configuration" };
|
||||||
|
mount_specs[2] = .{ .prefix = "/system/logs", .rewrite = "/system/logs" };
|
||||||
|
mount_count = 3;
|
||||||
|
} else {
|
||||||
|
std.log.info("volume {d} is a data volume; mounted at {s}", .{ my_volume_id, volume_mount_prefix });
|
||||||
|
}
|
||||||
|
return .{ .engine = &filesystem, .mounts = mount_specs[0..mount_count], .flush = flushIfDirty };
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn main(init: process.Init) void {
|
||||||
|
// The volume manager spawns this process with its volume id as argv[1] and the
|
||||||
|
// volume's mount path (its id-path) as argv[2].
|
||||||
|
if (init.arguments.get(1)) |id| {
|
||||||
|
my_volume_id = std.fmt.parseInt(u64, id, 10) catch 0;
|
||||||
|
}
|
||||||
|
if (init.arguments.get(2)) |prefix| {
|
||||||
|
volume_mount_prefix = prefix;
|
||||||
|
}
|
||||||
|
_ = logging.write("/system/services/exfat: starting, waiting for a block device\n");
|
||||||
|
Harness.run(.{ .bringUp = exfatBringUp });
|
||||||
|
}
|
||||||
@@ -0,0 +1,475 @@
|
|||||||
|
//! The on-disk layout of an exFAT filesystem — the Main Boot Sector (VBR) and the
|
||||||
|
//! six 32-byte directory-entry types — as `align(1)` extern structs that bit-cast
|
||||||
|
//! straight out of a sector (multi-byte fields little-endian). Pure data, plus the
|
||||||
|
//! three exFAT checksums (boot region, up-case table, directory-entry set), the
|
||||||
|
//! name hash, and the packed timestamp <-> Unix-epoch conversion. Host-testable.
|
||||||
|
//!
|
||||||
|
//! exFAT departs from FAT in three ways this file encodes: geometry lives in a
|
||||||
|
//! MustBeZero-guarded VBR (byte 11 is zero, which is exactly why the FAT prober
|
||||||
|
//! rejects an exFAT volume — it reads a zero bytes-per-sector); a file is a SET of
|
||||||
|
//! entries (a File entry, a Stream Extension, and one or more File Name entries)
|
||||||
|
//! validated by a rotate-right checksum; and names are compared case-folded through
|
||||||
|
//! the volume's own on-disk up-case table (the folding itself lives in the engine,
|
||||||
|
//! which holds the loaded table; the hash it feeds is here).
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
|
||||||
|
// exFAT's fixed on-disk widths — byte and UTF-16-unit counts the FORMAT defines,
|
||||||
|
// not ceilings danos chooses. Named so the wire-format structs carry no bare
|
||||||
|
// literal lengths; the values are facts of the spec.
|
||||||
|
pub const entry_bytes: usize = 32; // every directory entry
|
||||||
|
const jump_boot_bytes = 3;
|
||||||
|
const filesystem_name_bytes = 8; // "EXFAT "
|
||||||
|
const must_be_zero_bytes = 53; // the FAT-BPB overlap the format holds zero
|
||||||
|
const boot_code_bytes = 390;
|
||||||
|
const volume_label_units = 11;
|
||||||
|
|
||||||
|
// --- the Main Boot Sector (VBR, sector 0) ------------------------------------
|
||||||
|
|
||||||
|
/// The exFAT Main Boot Sector. `must_be_zero` (offset 11..64) overlaps where a
|
||||||
|
/// FAT BPB keeps bytes-per-sector/sectors-per-cluster/etc.; exFAT holds it zero,
|
||||||
|
/// so a FAT prober reading a zero bytes-per-sector rejects the volume — the
|
||||||
|
/// mutual-exclusion the two engines rely on.
|
||||||
|
pub const MainBootSector = extern struct {
|
||||||
|
jump_boot: [jump_boot_bytes]u8, // 0
|
||||||
|
filesystem_name: [filesystem_name_bytes]u8, // 3 "EXFAT "
|
||||||
|
must_be_zero: [must_be_zero_bytes]u8, // 11
|
||||||
|
partition_offset: u64 align(1), // 64 sectors, informational
|
||||||
|
volume_length: u64 align(1), // 72 sectors
|
||||||
|
fat_offset: u32 align(1), // 80 sectors from volume start
|
||||||
|
fat_length: u32 align(1), // 84 sectors, per FAT
|
||||||
|
cluster_heap_offset: u32 align(1), // 88 sectors from volume start
|
||||||
|
cluster_count: u32 align(1), // 92
|
||||||
|
first_cluster_of_root: u32 align(1), // 96
|
||||||
|
volume_serial_number: u32 align(1), // 100
|
||||||
|
filesystem_revision: u16 align(1), // 104
|
||||||
|
volume_flags: u16 align(1), // 106 (skipped by the boot checksum)
|
||||||
|
bytes_per_sector_shift: u8, // 108 9..12
|
||||||
|
sectors_per_cluster_shift: u8, // 109
|
||||||
|
number_of_fats: u8, // 110 1 (2 for TexFAT)
|
||||||
|
drive_select: u8, // 111
|
||||||
|
percent_in_use: u8, // 112 (skipped by the boot checksum)
|
||||||
|
reserved: [7]u8, // 113
|
||||||
|
boot_code: [boot_code_bytes]u8, // 120
|
||||||
|
boot_signature: u16 align(1), // 510 0xAA55
|
||||||
|
};
|
||||||
|
|
||||||
|
// --- directory entries (32 bytes each) ---------------------------------------
|
||||||
|
|
||||||
|
/// Entry-type bytes. The high bit (0x80) is InUse: a type with it clear is not in
|
||||||
|
/// use, and 0x00 ends the directory. Deleting an entry clears bit 7 (0x85 -> 0x05).
|
||||||
|
pub const entry_type_allocation_bitmap: u8 = 0x81;
|
||||||
|
pub const entry_type_upcase_table: u8 = 0x82;
|
||||||
|
pub const entry_type_volume_label: u8 = 0x83;
|
||||||
|
pub const entry_type_file: u8 = 0x85;
|
||||||
|
pub const entry_type_stream_extension: u8 = 0xC0;
|
||||||
|
pub const entry_type_file_name: u8 = 0xC1;
|
||||||
|
pub const entry_type_in_use_bit: u8 = 0x80;
|
||||||
|
pub const entry_type_end_of_directory: u8 = 0x00;
|
||||||
|
|
||||||
|
/// A raw 32-byte entry, for type dispatch before it is reinterpreted as a
|
||||||
|
/// specific entry.
|
||||||
|
pub const RawEntry = extern struct {
|
||||||
|
entry_type: u8,
|
||||||
|
data: [entry_bytes - 1]u8,
|
||||||
|
|
||||||
|
pub fn inUse(self: RawEntry) bool {
|
||||||
|
return self.entry_type & entry_type_in_use_bit != 0;
|
||||||
|
}
|
||||||
|
pub fn isEnd(self: RawEntry) bool {
|
||||||
|
return self.entry_type == entry_type_end_of_directory;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// 0x81 — the Allocation Bitmap: one bit per cluster (cluster 2 = bit 0), the
|
||||||
|
/// authority for which clusters are free. The deepest departure from FAT, where
|
||||||
|
/// the chain itself was the authority.
|
||||||
|
pub const AllocationBitmapEntry = extern struct {
|
||||||
|
entry_type: u8, // 0 0x81
|
||||||
|
bitmap_flags: u8, // 1
|
||||||
|
reserved: [18]u8, // 2
|
||||||
|
first_cluster: u32 align(1), // 20
|
||||||
|
data_length: u64 align(1), // 24
|
||||||
|
};
|
||||||
|
|
||||||
|
/// 0x82 — the Up-case Table: the on-disk case-fold map (code unit -> uppercase),
|
||||||
|
/// referenced by cluster and validated by `table_checksum`.
|
||||||
|
pub const UpcaseTableEntry = extern struct {
|
||||||
|
entry_type: u8, // 0 0x82
|
||||||
|
reserved1: [3]u8, // 1
|
||||||
|
table_checksum: u32 align(1), // 4
|
||||||
|
reserved2: [12]u8, // 8
|
||||||
|
first_cluster: u32 align(1), // 20
|
||||||
|
data_length: u64 align(1), // 24
|
||||||
|
};
|
||||||
|
|
||||||
|
/// 0x83 — the Volume Label (up to 11 UTF-16 units).
|
||||||
|
pub const VolumeLabelEntry = extern struct {
|
||||||
|
entry_type: u8, // 0 0x83
|
||||||
|
character_count: u8, // 1
|
||||||
|
volume_label: [volume_label_units]u16 align(1), // 2
|
||||||
|
reserved: [8]u8, // 24
|
||||||
|
};
|
||||||
|
|
||||||
|
/// 0x85 — the File entry: the head of a set, carrying the attributes,
|
||||||
|
/// timestamps, the secondary-entry count, and the set checksum.
|
||||||
|
pub const FileEntry = extern struct {
|
||||||
|
entry_type: u8, // 0 0x85
|
||||||
|
secondary_count: u8, // 1 stream (1) + name entries
|
||||||
|
set_checksum: u16 align(1), // 2 over the whole set, skipping these two bytes
|
||||||
|
file_attributes: u16 align(1), // 4
|
||||||
|
reserved1: u16 align(1), // 6
|
||||||
|
create_timestamp: u32 align(1), // 8
|
||||||
|
last_modified_timestamp: u32 align(1), // 12
|
||||||
|
last_accessed_timestamp: u32 align(1), // 16
|
||||||
|
create_10ms: u8, // 20
|
||||||
|
last_modified_10ms: u8, // 21
|
||||||
|
create_utc_offset: u8, // 22
|
||||||
|
last_modified_utc_offset: u8, // 23
|
||||||
|
last_accessed_utc_offset: u8, // 24
|
||||||
|
reserved2: [7]u8, // 25
|
||||||
|
};
|
||||||
|
|
||||||
|
/// 0xC0 — the Stream Extension: the second entry of every file set, carrying the
|
||||||
|
/// name length + hash and the data location (first cluster, sizes, the
|
||||||
|
/// no-FAT-chain flag).
|
||||||
|
pub const StreamExtensionEntry = extern struct {
|
||||||
|
entry_type: u8, // 0 0xC0
|
||||||
|
general_secondary_flags: u8, // 1
|
||||||
|
reserved1: u8, // 2
|
||||||
|
name_length: u8, // 3 UTF-16 units
|
||||||
|
name_hash: u16 align(1), // 4
|
||||||
|
reserved2: u16 align(1), // 6
|
||||||
|
valid_data_length: u64 align(1), // 8
|
||||||
|
reserved3: u32 align(1), // 16
|
||||||
|
first_cluster: u32 align(1), // 20
|
||||||
|
data_length: u64 align(1), // 24
|
||||||
|
};
|
||||||
|
|
||||||
|
/// 0xC1 — a File Name entry: 15 UTF-16 units of the name; a set carries
|
||||||
|
/// ceil(name_length / 15) of them.
|
||||||
|
pub const FileNameEntry = extern struct {
|
||||||
|
entry_type: u8, // 0 0xC1
|
||||||
|
general_secondary_flags: u8, // 1
|
||||||
|
file_name: [name_units_per_entry]u16 align(1), // 2
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const name_units_per_entry: usize = 15;
|
||||||
|
|
||||||
|
// General secondary flags (Stream Extension + File Name entries).
|
||||||
|
pub const secondary_flag_allocation_possible: u8 = 0x01;
|
||||||
|
pub const secondary_flag_no_fat_chain: u8 = 0x02;
|
||||||
|
|
||||||
|
// File attributes (same bit assignments as FAT).
|
||||||
|
pub const attribute_read_only: u16 = 0x0001;
|
||||||
|
pub const attribute_hidden: u16 = 0x0002;
|
||||||
|
pub const attribute_system: u16 = 0x0004;
|
||||||
|
pub const attribute_directory: u16 = 0x0010;
|
||||||
|
pub const attribute_archive: u16 = 0x0020;
|
||||||
|
|
||||||
|
// FAT special cluster values (exFAT's FAT is 32-bit; used only for a fragmented
|
||||||
|
// chain, i.e. when no_fat_chain is clear).
|
||||||
|
pub const first_data_cluster: u32 = 2;
|
||||||
|
pub const end_of_chain: u32 = 0xFFFFFFFF;
|
||||||
|
pub const bad_cluster: u32 = 0xFFFFFFF7;
|
||||||
|
|
||||||
|
pub const boot_signature_offset: usize = 510; // 0x55 0xAA
|
||||||
|
|
||||||
|
// --- geometry ----------------------------------------------------------------
|
||||||
|
|
||||||
|
pub const Geometry = struct {
|
||||||
|
bytes_per_sector: u32,
|
||||||
|
sectors_per_cluster: u32,
|
||||||
|
cluster_count: u32,
|
||||||
|
fat_offset_sectors: u32, // from volume start
|
||||||
|
fat_length_sectors: u32,
|
||||||
|
cluster_heap_offset_sectors: u32, // from volume start
|
||||||
|
first_cluster_of_root: u32,
|
||||||
|
volume_serial_number: u32,
|
||||||
|
volume_length_sectors: u64,
|
||||||
|
number_of_fats: u32,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Derive the geometry from a Main Boot Sector. Returns null unless it is a
|
||||||
|
/// plausible exFAT VBR: the "EXFAT " name, an all-zero MustBeZero region, the
|
||||||
|
/// 0xAA55 signature, and sane shifts. Accepting ONLY these is what keeps exFAT and
|
||||||
|
/// FAT from ever claiming each other's volumes.
|
||||||
|
pub fn geometryOf(sector: []const u8) ?Geometry {
|
||||||
|
if (sector.len < 512) return null;
|
||||||
|
if (sector[boot_signature_offset] != 0x55 or sector[boot_signature_offset + 1] != 0xAA) return null;
|
||||||
|
const vbr = std.mem.bytesToValue(MainBootSector, sector[0..@sizeOf(MainBootSector)]);
|
||||||
|
if (!std.mem.eql(u8, &vbr.filesystem_name, "EXFAT ")) return null;
|
||||||
|
for (vbr.must_be_zero) |byte| if (byte != 0) return null;
|
||||||
|
if (vbr.bytes_per_sector_shift < 9 or vbr.bytes_per_sector_shift > 12) return null;
|
||||||
|
// The exFAT spec caps a cluster at 2^25 bytes (32 MiB): bytes-per-sector-shift
|
||||||
|
// plus sectors-per-cluster-shift must not exceed 25. Enforcing it here is also
|
||||||
|
// what keeps the engine's u32 cluster-byte arithmetic (sectors_per_cluster *
|
||||||
|
// 512) from overflowing on a crafted VBR off untrusted removable media.
|
||||||
|
if (@as(u16, vbr.bytes_per_sector_shift) + vbr.sectors_per_cluster_shift > 25) return null;
|
||||||
|
// cluster_count is capped at 0xFFFFFFF5 (the spec's ClusterCount maximum), so
|
||||||
|
// cluster_count + first_data_cluster cannot overflow u32 in the bounds checks.
|
||||||
|
if (vbr.number_of_fats == 0 or vbr.cluster_count == 0 or vbr.cluster_count > 0xFFFFFFF5) return null;
|
||||||
|
if (vbr.first_cluster_of_root < first_data_cluster) return null;
|
||||||
|
return .{
|
||||||
|
.bytes_per_sector = @as(u32, 1) << @intCast(vbr.bytes_per_sector_shift),
|
||||||
|
.sectors_per_cluster = @as(u32, 1) << @intCast(vbr.sectors_per_cluster_shift),
|
||||||
|
.cluster_count = vbr.cluster_count,
|
||||||
|
.fat_offset_sectors = vbr.fat_offset,
|
||||||
|
.fat_length_sectors = vbr.fat_length,
|
||||||
|
.cluster_heap_offset_sectors = vbr.cluster_heap_offset,
|
||||||
|
.first_cluster_of_root = vbr.first_cluster_of_root,
|
||||||
|
.volume_serial_number = vbr.volume_serial_number,
|
||||||
|
.volume_length_sectors = vbr.volume_length,
|
||||||
|
.number_of_fats = vbr.number_of_fats,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- checksums and the name hash ---------------------------------------------
|
||||||
|
|
||||||
|
/// The directory-entry-SET checksum (a File entry's `set_checksum`), a 16-bit
|
||||||
|
/// rotate-right sum over every byte of the set, skipping the two checksum bytes
|
||||||
|
/// themselves (offset 2..3 of the first entry). `entries` is the whole set:
|
||||||
|
/// (secondary_count + 1) * 32 bytes.
|
||||||
|
pub fn setChecksum(entries: []const u8) u16 {
|
||||||
|
var checksum: u16 = 0;
|
||||||
|
for (entries, 0..) |byte, i| {
|
||||||
|
if (i == 2 or i == 3) continue;
|
||||||
|
checksum = std.math.rotr(u16, checksum, 1) +% byte;
|
||||||
|
}
|
||||||
|
return checksum;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The up-case-table checksum (an Up-case entry's `table_checksum`), a 32-bit
|
||||||
|
/// rotate-right sum over the table's on-disk bytes.
|
||||||
|
pub fn upcaseChecksum(table_bytes: []const u8) u32 {
|
||||||
|
var checksum: u32 = 0;
|
||||||
|
for (table_bytes) |byte| checksum = std.math.rotr(u32, checksum, 1) +% byte;
|
||||||
|
return checksum;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The boot-region checksum — the u32 the checksum sector repeats — a 32-bit
|
||||||
|
/// rotate-right sum over the first eleven sectors, skipping VolumeFlags (offset
|
||||||
|
/// 106..107) and PercentInUse (offset 112) of the first sector.
|
||||||
|
pub fn bootChecksum(region: []const u8) u32 {
|
||||||
|
var checksum: u32 = 0;
|
||||||
|
for (region, 0..) |byte, i| {
|
||||||
|
if (i == 106 or i == 107 or i == 112) continue;
|
||||||
|
checksum = std.math.rotr(u32, checksum, 1) +% byte;
|
||||||
|
}
|
||||||
|
return checksum;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The name hash a Stream entry carries: a 16-bit rotate-right sum over the
|
||||||
|
/// UP-CASED name's bytes (low byte then high byte of each UTF-16 unit). The caller
|
||||||
|
/// up-cases through the volume's table first; a mismatch lets a lookup reject a
|
||||||
|
/// name without reading its File Name entries.
|
||||||
|
pub fn nameHash(upcased: []const u16) u16 {
|
||||||
|
var hash: u16 = 0;
|
||||||
|
for (upcased) |unit| {
|
||||||
|
hash = std.math.rotr(u16, hash, 1) +% @as(u8, @truncate(unit));
|
||||||
|
hash = std.math.rotr(u16, hash, 1) +% @as(u8, @truncate(unit >> 8));
|
||||||
|
}
|
||||||
|
return hash;
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- timestamps --------------------------------------------------------------
|
||||||
|
//
|
||||||
|
// exFAT packs a timestamp into one u32: the high 16 bits are a DOS date
|
||||||
|
// (year-1980 | month | day), the low 16 a DOS time (hour | minute | second/2).
|
||||||
|
// Same field layout as FAT, so the epoch math matches; there is no timezone in
|
||||||
|
// the packed value (a separate UTC-offset byte carries that, which danos leaves
|
||||||
|
// zero = UTC).
|
||||||
|
|
||||||
|
fn isLeapYear(year: u32) bool {
|
||||||
|
return (year % 4 == 0 and year % 100 != 0) or (year % 400 == 0);
|
||||||
|
}
|
||||||
|
|
||||||
|
const days_in_month = [_]u8{ 31, 28, 31, 30, 31, 30, 31, 31, 30, 31, 30, 31 };
|
||||||
|
|
||||||
|
/// Convert a packed exFAT timestamp to Unix epoch seconds (UTC). 0 for unset.
|
||||||
|
pub fn timestampToEpoch(timestamp: u32) u64 {
|
||||||
|
if (timestamp == 0) return 0;
|
||||||
|
const date: u32 = timestamp >> 16;
|
||||||
|
const time: u32 = timestamp & 0xFFFF;
|
||||||
|
const day: u32 = date & 0x1F;
|
||||||
|
const month: u32 = (date >> 5) & 0x0F;
|
||||||
|
const year: u32 = 1980 + (date >> 9);
|
||||||
|
if (month < 1 or month > 12 or day < 1) return 0;
|
||||||
|
const second: u32 = (time & 0x1F) * 2;
|
||||||
|
const minute: u32 = (time >> 5) & 0x3F;
|
||||||
|
const hour: u32 = (time >> 11) & 0x1F;
|
||||||
|
|
||||||
|
var days: u64 = 0;
|
||||||
|
var y: u32 = 1970;
|
||||||
|
while (y < year) : (y += 1) days += if (isLeapYear(y)) 366 else 365;
|
||||||
|
var m: u32 = 1;
|
||||||
|
while (m < month) : (m += 1) {
|
||||||
|
days += days_in_month[m - 1];
|
||||||
|
if (m == 2 and isLeapYear(year)) days += 1;
|
||||||
|
}
|
||||||
|
days += day - 1;
|
||||||
|
return ((days * 24 + hour) * 60 + minute) * 60 + second;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Convert Unix epoch seconds (UTC) to a packed exFAT timestamp. 0 for epoch 0 or
|
||||||
|
/// any time before 1980 (unrepresentable).
|
||||||
|
pub fn epochToTimestamp(epoch: u64) u32 {
|
||||||
|
if (epoch == 0) return 0;
|
||||||
|
var remaining = epoch;
|
||||||
|
const second: u32 = @intCast(remaining % 60);
|
||||||
|
remaining /= 60;
|
||||||
|
const minute: u32 = @intCast(remaining % 60);
|
||||||
|
remaining /= 60;
|
||||||
|
const hour: u32 = @intCast(remaining % 24);
|
||||||
|
remaining /= 24;
|
||||||
|
var days: u32 = @intCast(remaining);
|
||||||
|
|
||||||
|
var year: u32 = 1970;
|
||||||
|
while (true) {
|
||||||
|
const year_days: u32 = if (isLeapYear(year)) 366 else 365;
|
||||||
|
if (days < year_days) break;
|
||||||
|
days -= year_days;
|
||||||
|
year += 1;
|
||||||
|
}
|
||||||
|
if (year < 1980) return 0;
|
||||||
|
var month: u32 = 1;
|
||||||
|
while (true) {
|
||||||
|
var month_days: u32 = days_in_month[month - 1];
|
||||||
|
if (month == 2 and isLeapYear(year)) month_days += 1;
|
||||||
|
if (days < month_days) break;
|
||||||
|
days -= month_days;
|
||||||
|
month += 1;
|
||||||
|
}
|
||||||
|
const day = days + 1;
|
||||||
|
const date: u32 = ((year - 1980) << 9) | (month << 5) | day;
|
||||||
|
const time: u32 = (hour << 11) | (minute << 5) | (second / 2);
|
||||||
|
return (date << 16) | time;
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- tests -------------------------------------------------------------------
|
||||||
|
|
||||||
|
test "on-disk struct sizes match the specification" {
|
||||||
|
try std.testing.expectEqual(@as(usize, 512), @sizeOf(MainBootSector));
|
||||||
|
try std.testing.expectEqual(@as(usize, 32), @sizeOf(RawEntry));
|
||||||
|
try std.testing.expectEqual(@as(usize, 32), @sizeOf(AllocationBitmapEntry));
|
||||||
|
try std.testing.expectEqual(@as(usize, 32), @sizeOf(UpcaseTableEntry));
|
||||||
|
try std.testing.expectEqual(@as(usize, 32), @sizeOf(VolumeLabelEntry));
|
||||||
|
try std.testing.expectEqual(@as(usize, 32), @sizeOf(FileEntry));
|
||||||
|
try std.testing.expectEqual(@as(usize, 32), @sizeOf(StreamExtensionEntry));
|
||||||
|
try std.testing.expectEqual(@as(usize, 32), @sizeOf(FileNameEntry));
|
||||||
|
}
|
||||||
|
|
||||||
|
test "MainBootSector field offsets" {
|
||||||
|
try std.testing.expectEqual(@as(usize, 3), @offsetOf(MainBootSector, "filesystem_name"));
|
||||||
|
try std.testing.expectEqual(@as(usize, 11), @offsetOf(MainBootSector, "must_be_zero"));
|
||||||
|
try std.testing.expectEqual(@as(usize, 80), @offsetOf(MainBootSector, "fat_offset"));
|
||||||
|
try std.testing.expectEqual(@as(usize, 88), @offsetOf(MainBootSector, "cluster_heap_offset"));
|
||||||
|
try std.testing.expectEqual(@as(usize, 96), @offsetOf(MainBootSector, "first_cluster_of_root"));
|
||||||
|
try std.testing.expectEqual(@as(usize, 106), @offsetOf(MainBootSector, "volume_flags"));
|
||||||
|
try std.testing.expectEqual(@as(usize, 112), @offsetOf(MainBootSector, "percent_in_use"));
|
||||||
|
try std.testing.expectEqual(@as(usize, 510), @offsetOf(MainBootSector, "boot_signature"));
|
||||||
|
// The Stream Extension's data location must sit where the spec places it.
|
||||||
|
try std.testing.expectEqual(@as(usize, 20), @offsetOf(StreamExtensionEntry, "first_cluster"));
|
||||||
|
try std.testing.expectEqual(@as(usize, 24), @offsetOf(StreamExtensionEntry, "data_length"));
|
||||||
|
}
|
||||||
|
|
||||||
|
test "geometryOf accepts exFAT and the MustBeZero guard rejects a FAT-shaped sector" {
|
||||||
|
var sector = [_]u8{0} ** 512;
|
||||||
|
@memcpy(sector[3..11], "EXFAT ");
|
||||||
|
sector[510] = 0x55;
|
||||||
|
sector[511] = 0xAA;
|
||||||
|
// fat_offset=128, fat_length=64, cluster_heap_offset=256, cluster_count=1000,
|
||||||
|
// root cluster=5, bytes/sector=512 (shift 9), sectors/cluster=8 (shift 3), 1 FAT.
|
||||||
|
std.mem.writeInt(u32, sector[80..84], 128, .little);
|
||||||
|
std.mem.writeInt(u32, sector[84..88], 64, .little);
|
||||||
|
std.mem.writeInt(u32, sector[88..92], 256, .little);
|
||||||
|
std.mem.writeInt(u32, sector[92..96], 1000, .little);
|
||||||
|
std.mem.writeInt(u32, sector[96..100], 5, .little);
|
||||||
|
sector[108] = 9; // bytes_per_sector_shift
|
||||||
|
sector[109] = 3; // sectors_per_cluster_shift
|
||||||
|
sector[110] = 1; // number_of_fats
|
||||||
|
const geo = geometryOf(§or) orelse return error.ShouldParse;
|
||||||
|
try std.testing.expectEqual(@as(u32, 512), geo.bytes_per_sector);
|
||||||
|
try std.testing.expectEqual(@as(u32, 8), geo.sectors_per_cluster);
|
||||||
|
try std.testing.expectEqual(@as(u32, 1000), geo.cluster_count);
|
||||||
|
try std.testing.expectEqual(@as(u32, 5), geo.first_cluster_of_root);
|
||||||
|
|
||||||
|
// A non-zero byte in MustBeZero (where a FAT BPB keeps bytes-per-sector) is
|
||||||
|
// rejected — the mutual exclusion between the engines.
|
||||||
|
sector[11] = 0x02;
|
||||||
|
try std.testing.expect(geometryOf(§or) == null);
|
||||||
|
sector[11] = 0;
|
||||||
|
// Wrong name is rejected too.
|
||||||
|
sector[3] = 'F';
|
||||||
|
try std.testing.expect(geometryOf(§or) == null);
|
||||||
|
sector[3] = 'E';
|
||||||
|
}
|
||||||
|
|
||||||
|
test "geometryOf rejects crafted VBRs that would overflow u32 cluster arithmetic" {
|
||||||
|
var sector = [_]u8{0} ** 512;
|
||||||
|
@memcpy(sector[3..11], "EXFAT ");
|
||||||
|
sector[510] = 0x55;
|
||||||
|
sector[511] = 0xAA;
|
||||||
|
std.mem.writeInt(u32, sector[92..96], 1000, .little); // cluster_count
|
||||||
|
std.mem.writeInt(u32, sector[96..100], 5, .little); // root cluster
|
||||||
|
sector[108] = 9; // bytes_per_sector_shift
|
||||||
|
sector[110] = 1; // number_of_fats
|
||||||
|
// A cluster shift past the exFAT ceiling (9 + 17 = 26 > 25) would make
|
||||||
|
// sectors_per_cluster * 512 overflow u32 — rejected.
|
||||||
|
sector[109] = 17;
|
||||||
|
try std.testing.expect(geometryOf(§or) == null);
|
||||||
|
sector[109] = 3; // sane again
|
||||||
|
try std.testing.expect(geometryOf(§or) != null);
|
||||||
|
// cluster_count above the spec maximum (0xFFFFFFF5) would overflow
|
||||||
|
// cluster_count + first_data_cluster in the bounds checks — rejected.
|
||||||
|
std.mem.writeInt(u32, sector[92..96], 0xFFFFFFFF, .little);
|
||||||
|
try std.testing.expect(geometryOf(§or) == null);
|
||||||
|
}
|
||||||
|
|
||||||
|
test "set checksum skips its own two bytes and depends on the rest" {
|
||||||
|
var set = [_]u8{0} ** 64; // a File entry + one secondary
|
||||||
|
set[0] = entry_type_file;
|
||||||
|
set[1] = 1;
|
||||||
|
set[4] = 0x20; // an attribute byte
|
||||||
|
set[40] = 0xAB; // a byte in the secondary entry
|
||||||
|
const base = setChecksum(&set);
|
||||||
|
// Changing the checksum field itself must NOT change the computed checksum.
|
||||||
|
set[2] = 0xFF;
|
||||||
|
set[3] = 0xEE;
|
||||||
|
try std.testing.expectEqual(base, setChecksum(&set));
|
||||||
|
// Changing any other byte MUST change it.
|
||||||
|
set[4] = 0x21;
|
||||||
|
try std.testing.expect(setChecksum(&set) != base);
|
||||||
|
}
|
||||||
|
|
||||||
|
test "name hash is deterministic and order-sensitive" {
|
||||||
|
const readme = [_]u16{ 'R', 'E', 'A', 'D', 'M', 'E' };
|
||||||
|
const different = [_]u16{ 'E', 'R', 'A', 'D', 'M', 'E' };
|
||||||
|
try std.testing.expectEqual(nameHash(&readme), nameHash(&readme));
|
||||||
|
try std.testing.expect(nameHash(&readme) != nameHash(&different));
|
||||||
|
}
|
||||||
|
|
||||||
|
test "boot checksum skips VolumeFlags and PercentInUse" {
|
||||||
|
var region = [_]u8{0} ** 1536; // three 512-byte sectors is enough to exercise the skips
|
||||||
|
region[64] = 0x11;
|
||||||
|
const base = bootChecksum(®ion);
|
||||||
|
for ([_]usize{ 106, 107, 112 }) |skipped| {
|
||||||
|
var copy = region;
|
||||||
|
copy[skipped] = 0xFF;
|
||||||
|
try std.testing.expectEqual(base, bootChecksum(©));
|
||||||
|
}
|
||||||
|
var copy = region;
|
||||||
|
copy[108] = 0xFF; // a non-skipped byte
|
||||||
|
try std.testing.expect(bootChecksum(©) != base);
|
||||||
|
}
|
||||||
|
|
||||||
|
test "exFAT timestamp <-> Unix epoch round trip" {
|
||||||
|
for ([_]u64{ 1_577_836_800, 1_700_000_000, 1_262_304_000, 1_783_971_244 }) |epoch| {
|
||||||
|
try std.testing.expectEqual(epoch, timestampToEpoch(epochToTimestamp(epoch)));
|
||||||
|
}
|
||||||
|
// 1577836800 is 2020-01-01 00:00:00 UTC.
|
||||||
|
const stamp = epochToTimestamp(1_577_836_800);
|
||||||
|
try std.testing.expectEqual(@as(u32, 2020), 1980 + (stamp >> 16 >> 9));
|
||||||
|
try std.testing.expectEqual(@as(u64, 0), timestampToEpoch(0));
|
||||||
|
try std.testing.expectEqual(@as(u32, 0), epochToTimestamp(0));
|
||||||
|
}
|
||||||
@@ -1657,3 +1657,21 @@ test "short-name checksum matches the reference vector" {
|
|||||||
const c = FileSystem.shortChecksum("REDAME TXT".*);
|
const c = FileSystem.shortChecksum("REDAME TXT".*);
|
||||||
try std.testing.expect(a != c);
|
try std.testing.expect(a != c);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
test "the FAT engine rejects an exFAT volume (mutual exclusion at mount)" {
|
||||||
|
const allocator = std.testing.allocator;
|
||||||
|
const bytes = try allocator.alloc(u8, 5000 * sector_size);
|
||||||
|
defer allocator.free(bytes);
|
||||||
|
@memset(bytes, 0);
|
||||||
|
// An exFAT boot sector: the "EXFAT " name and 0x55AA, but MustBeZero (offset
|
||||||
|
// 11, where a FAT BPB keeps bytes-per-sector) stays zero — so this engine's
|
||||||
|
// geometryOf reads a zero bytes-per-sector and rejects it.
|
||||||
|
@memcpy(bytes[3..11], "EXFAT ");
|
||||||
|
bytes[on_disk.boot_signature_offset] = 0x55;
|
||||||
|
bytes[on_disk.boot_signature_offset + 1] = 0xAA;
|
||||||
|
var disk = RamDisk{ .bytes = bytes };
|
||||||
|
try std.testing.expect(FileSystem.mount(disk.device()) == null);
|
||||||
|
// Control: a real FAT16 mounts.
|
||||||
|
formatFat16(bytes);
|
||||||
|
try std.testing.expect(FileSystem.mount(disk.device()) != null);
|
||||||
|
}
|
||||||
|
|||||||
@@ -173,16 +173,25 @@ fn fatBringUp(endpoint: ipc.Handle) ?Harness.Volume {
|
|||||||
};
|
};
|
||||||
std.log.info("mounted FAT ({s}, {d} clusters, partition lba {d})", .{ @tagName(filesystem.geometry.fat_type), filesystem.geometry.cluster_count, filesystem.base_lba });
|
std.log.info("mounted FAT ({s}, {d} clusters, partition lba {d})", .{ @tagName(filesystem.geometry.fat_type), filesystem.geometry.cluster_count, filesystem.base_lba });
|
||||||
|
|
||||||
// The volume mounts at its id-path (argv[2]), plus the two FHS rewrites so
|
// Every volume mounts at its own id-path (argv[2]). The boot/system volume —
|
||||||
// hierarchy paths (the logger's /system/logs) stay decoupled from which
|
// the one carrying the /system tree — ADDITIONALLY installs the two FHS
|
||||||
// volume backs them. This single-volume increment's one volume IS the boot
|
// rewrites, so hierarchy paths (config reads, the logger's persistent
|
||||||
// volume, so it installs both unconditionally; S3 (multi-volume) makes the
|
// /system/logs) stay decoupled from which volume backs them. Detection is by
|
||||||
// rewrites content-conditional — installed only by whichever volume carries
|
// CONTENT, not spawn order: a volume is the system volume iff /system/
|
||||||
// the system, decided by content, not order.
|
// configuration resolves on its own media. A data volume has no /system, so it
|
||||||
|
// mounts only at its id-path and never shadows the running system's config or
|
||||||
|
// logs with a dead mount.
|
||||||
mount_specs[0] = .{ .prefix = volume_mount_prefix };
|
mount_specs[0] = .{ .prefix = volume_mount_prefix };
|
||||||
|
var mount_count: usize = 1;
|
||||||
|
if (filesystem.resolve("/system/configuration") != null) {
|
||||||
|
std.log.info("volume {d} carries the system tree; backing /system/configuration and /system/logs", .{my_volume_id});
|
||||||
mount_specs[1] = .{ .prefix = "/system/configuration", .rewrite = "/system/configuration" };
|
mount_specs[1] = .{ .prefix = "/system/configuration", .rewrite = "/system/configuration" };
|
||||||
mount_specs[2] = .{ .prefix = "/system/logs", .rewrite = "/system/logs" };
|
mount_specs[2] = .{ .prefix = "/system/logs", .rewrite = "/system/logs" };
|
||||||
return .{ .engine = &filesystem, .mounts = mount_specs[0..3], .flush = flushIfDirty };
|
mount_count = 3;
|
||||||
|
} else {
|
||||||
|
std.log.info("volume {d} is a data volume; mounted at {s}", .{ my_volume_id, volume_mount_prefix });
|
||||||
|
}
|
||||||
|
return .{ .engine = &filesystem, .mounts = mount_specs[0..mount_count], .flush = flushIfDirty };
|
||||||
}
|
}
|
||||||
|
|
||||||
pub fn main(init: process.Init) void {
|
pub fn main(init: process.Init) void {
|
||||||
|
|||||||
@@ -31,10 +31,11 @@ pub const label_maximum = 36;
|
|||||||
/// stay distinct, and it drives how the mount path is rendered from the id.
|
/// stay distinct, and it drives how the mount path is rendered from the id.
|
||||||
pub const Rung = enum(u8) {
|
pub const Rung = enum(u8) {
|
||||||
gpt_guid = 1,
|
gpt_guid = 1,
|
||||||
filesystem_uuid = 2, // reserved: no non-FAT engine reads a superblock UUID yet
|
filesystem_uuid = 2, // reserved: no engine reads a superblock UUID yet
|
||||||
fat_serial = 3,
|
fat_serial = 3,
|
||||||
mbr_index = 4,
|
mbr_index = 4,
|
||||||
anonymous = 5,
|
anonymous = 5,
|
||||||
|
exfat_serial = 6, // exFAT's VolumeSerialNumber — content-strong like fat_serial
|
||||||
};
|
};
|
||||||
|
|
||||||
/// A volume's content identity. `key` is the ID — the stable, unique handle the
|
/// A volume's content identity. `key` is the ID — the stable, unique handle the
|
||||||
@@ -61,15 +62,16 @@ pub const Identity = struct {
|
|||||||
};
|
};
|
||||||
|
|
||||||
/// Which filesystem a volume's content is — the key `filesystems.csv` maps to a
|
/// Which filesystem a volume's content is — the key `filesystems.csv` maps to a
|
||||||
/// service binary. Today only FAT is recognized (S4 adds exFAT with a real VBR
|
/// service binary. FAT and exFAT are recognized by their VBRs; content that is
|
||||||
/// recognizer); until then every probed volume is `.fat`, matching the volume
|
/// neither falls back to `.fat`, the volume manager's historical hand-off.
|
||||||
/// manager's historical hand-off of everything to the FAT service.
|
|
||||||
pub const FilesystemKind = enum {
|
pub const FilesystemKind = enum {
|
||||||
fat,
|
fat,
|
||||||
|
exfat,
|
||||||
unknown,
|
unknown,
|
||||||
|
|
||||||
pub fn fromToken(token: []const u8) FilesystemKind {
|
pub fn fromToken(token: []const u8) FilesystemKind {
|
||||||
if (std.mem.eql(u8, token, "fat")) return .fat;
|
if (std.mem.eql(u8, token, "fat")) return .fat;
|
||||||
|
if (std.mem.eql(u8, token, "exfat")) return .exfat;
|
||||||
return .unknown;
|
return .unknown;
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
@@ -172,39 +174,41 @@ fn setLabelFromUtf16(id: *Identity, name_bytes: []const u8) void {
|
|||||||
id.label_len = @intCast(out);
|
id.label_len = @intCast(out);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// The first GPT volume, or null if LBA 1 is not a valid GPT header or no entry
|
/// Append every valid GPT volume to `out` (up to `out.len`), returning the count
|
||||||
/// validates. The header CRC-32 and the per-entry overflow-safe range check are
|
/// (0 if LBA 1 is not a valid GPT header). The header CRC-32 and the per-entry
|
||||||
/// the confinement-safety guards the driver's clamp rests on — the invariant
|
/// overflow-safe range check are the confinement-safety guards the driver's clamp
|
||||||
/// firstVolume documents for MBR, extended to untrusted GPT metadata. The
|
/// rests on — the invariant documented for MBR, extended to untrusted GPT
|
||||||
/// entry-array CRC is deferred (correctness-only; the range check carries safety).
|
/// metadata. The entry-array CRC is deferred (correctness-only; the range check
|
||||||
fn gptFirstVolume(reader: SectorReader, device_blocks: u64) ?Volume {
|
/// carries safety).
|
||||||
|
fn gptAllVolumes(reader: SectorReader, device_blocks: u64, out: []Volume) usize {
|
||||||
var header: [sector_bytes]u8 = undefined;
|
var header: [sector_bytes]u8 = undefined;
|
||||||
if (!reader.read(1, &header)) return null;
|
if (!reader.read(1, &header)) return 0;
|
||||||
if (!std.mem.eql(u8, header[0..8], gpt_signature)) return null;
|
if (!std.mem.eql(u8, header[0..8], gpt_signature)) return 0;
|
||||||
const header_size = std.mem.readInt(u32, header[12..16], .little);
|
const header_size = std.mem.readInt(u32, header[12..16], .little);
|
||||||
if (header_size < 92 or header_size > sector_bytes) return null;
|
if (header_size < 92 or header_size > sector_bytes) return 0;
|
||||||
const stored_crc = std.mem.readInt(u32, header[16..20], .little);
|
const stored_crc = std.mem.readInt(u32, header[16..20], .little);
|
||||||
var check: [sector_bytes]u8 = undefined;
|
var check: [sector_bytes]u8 = undefined;
|
||||||
@memcpy(check[0..header_size], header[0..header_size]);
|
@memcpy(check[0..header_size], header[0..header_size]);
|
||||||
@memset(check[16..20], 0);
|
@memset(check[16..20], 0);
|
||||||
if (crc32(check[0..header_size]) != stored_crc) return null;
|
if (crc32(check[0..header_size]) != stored_crc) return 0;
|
||||||
|
|
||||||
const entry_lba = std.mem.readInt(u64, header[72..80], .little);
|
const entry_lba = std.mem.readInt(u64, header[72..80], .little);
|
||||||
const num_entries = std.mem.readInt(u32, header[80..84], .little);
|
const num_entries = std.mem.readInt(u32, header[80..84], .little);
|
||||||
const entry_size = std.mem.readInt(u32, header[84..88], .little);
|
const entry_size = std.mem.readInt(u32, header[84..88], .little);
|
||||||
if (entry_size != 128 and entry_size != 256 and entry_size != 512) return null;
|
if (entry_size != 128 and entry_size != 256 and entry_size != 512) return 0;
|
||||||
if (entry_lba == 0 or entry_lba >= device_blocks) return null;
|
if (entry_lba == 0 or entry_lba >= device_blocks) return 0;
|
||||||
|
|
||||||
const scan = @min(num_entries, gpt_entry_scan_maximum);
|
const scan = @min(num_entries, gpt_entry_scan_maximum);
|
||||||
var sector_buf: [sector_bytes]u8 = undefined;
|
var sector_buf: [sector_bytes]u8 = undefined;
|
||||||
var loaded: u64 = std.math.maxInt(u64);
|
var loaded: u64 = std.math.maxInt(u64);
|
||||||
|
var count: usize = 0;
|
||||||
var i: u32 = 0;
|
var i: u32 = 0;
|
||||||
while (i < scan) : (i += 1) {
|
while (i < scan and count < out.len) : (i += 1) {
|
||||||
const abs = @as(u64, i) * entry_size;
|
const abs = @as(u64, i) * entry_size;
|
||||||
const lba = entry_lba + abs / sector_bytes;
|
const lba = entry_lba + abs / sector_bytes;
|
||||||
const off = @as(usize, @intCast(abs % sector_bytes));
|
const off = @as(usize, @intCast(abs % sector_bytes));
|
||||||
if (lba != loaded) {
|
if (lba != loaded) {
|
||||||
if (!reader.read(lba, §or_buf)) return null;
|
if (!reader.read(lba, §or_buf)) break; // return what we have
|
||||||
loaded = lba;
|
loaded = lba;
|
||||||
}
|
}
|
||||||
const entry = sector_buf[off..][0..128]; // the fields we read live in the first 128 bytes
|
const entry = sector_buf[off..][0..128]; // the fields we read live in the first 128 bytes
|
||||||
@@ -224,9 +228,10 @@ fn gptFirstVolume(reader: SectorReader, device_blocks: u64) ?Volume {
|
|||||||
if (start == 0 or end < start or end >= device_blocks) continue;
|
if (start == 0 or end < start or end >= device_blocks) continue;
|
||||||
var id = Identity{ .rung = .gpt_guid, .key = std.mem.readInt(u128, entry[16..32], .little) };
|
var id = Identity{ .rung = .gpt_guid, .key = std.mem.readInt(u128, entry[16..32], .little) };
|
||||||
setLabelFromUtf16(&id, entry[56..128]);
|
setLabelFromUtf16(&id, entry[56..128]);
|
||||||
return .{ .base_lba = start, .block_count = end - start + 1, .identity = id };
|
out[count] = .{ .base_lba = start, .block_count = end - start + 1, .identity = id, .signature = signatureAt(reader, start) };
|
||||||
|
count += 1;
|
||||||
}
|
}
|
||||||
return null;
|
return count;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Trim trailing spaces (FAT labels are space-padded) and copy into the display
|
/// Trim trailing spaces (FAT labels are space-padded) and copy into the display
|
||||||
@@ -259,18 +264,52 @@ fn fatIdentity(reader: SectorReader, start_lba: u64) ?Identity {
|
|||||||
return id;
|
return id;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// The first volume on the device `reader` addresses, whose whole-device size is
|
/// The exFAT VolumeSerialNumber (offset 100) read from the Main Boot Sector at
|
||||||
/// `device_blocks`, or null if none is found. A GPT disk (protective MBR) is
|
/// `start_lba` — its content identity, rung `exfat_serial`. Null unless the sector
|
||||||
/// handled by GPT, authoritatively — its null is final. Otherwise an MBR with a
|
/// is an exFAT VBR (the "EXFAT " name at offset 3 + the 0x55AA signature; the
|
||||||
/// non-empty entry yields that partition's [start, size); otherwise a boot
|
/// name is where a FAT BPB keeps its OEM string, so the two never collide). The
|
||||||
/// signature with no partitions is treated as a bare FAT spanning the device.
|
/// label lives in a root-directory entry, not the VBR, so it is left empty here.
|
||||||
pub fn firstVolume(reader: SectorReader, device_blocks: u64) ?Volume {
|
fn exfatIdentity(reader: SectorReader, start_lba: u64) ?Identity {
|
||||||
|
var vbr: [sector_bytes]u8 = undefined;
|
||||||
|
if (!reader.read(start_lba, &vbr)) return null;
|
||||||
|
if (vbr[510] != 0x55 or vbr[511] != 0xAA) return null;
|
||||||
|
if (!std.mem.eql(u8, vbr[3..11], "EXFAT ")) return null;
|
||||||
|
return .{ .rung = .exfat_serial, .key = std.mem.readInt(u32, vbr[100..104], .little) };
|
||||||
|
}
|
||||||
|
|
||||||
|
const Recognized = struct { identity: Identity, signature: FilesystemKind };
|
||||||
|
|
||||||
|
/// Recognize the filesystem at `start_lba` by its VBR: exFAT first (its serial and
|
||||||
|
/// the `.exfat` signature), else FAT (its serial), else unknown content that keeps
|
||||||
|
/// the MBR disk-signature identity and the historical `.fat` hand-off.
|
||||||
|
fn recognize(reader: SectorReader, start_lba: u64, block0: *const [sector_bytes]u8, index: u8) Recognized {
|
||||||
|
if (exfatIdentity(reader, start_lba)) |id| return .{ .identity = id, .signature = .exfat };
|
||||||
|
if (fatIdentity(reader, start_lba)) |id| return .{ .identity = id, .signature = .fat };
|
||||||
|
return .{ .identity = mbrIdentity(block0, index), .signature = .fat };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The filesystem signature at `start_lba` when the identity is decided elsewhere
|
||||||
|
/// (a GPT partition keeps its GUID identity but still needs its content's kind).
|
||||||
|
fn signatureAt(reader: SectorReader, start_lba: u64) FilesystemKind {
|
||||||
|
return if (exfatIdentity(reader, start_lba) != null) .exfat else .fat;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Append every volume on the device `reader` addresses, whose whole-device size
|
||||||
|
/// is `device_blocks`, to `out` (up to `out.len`), returning the count. A GPT
|
||||||
|
/// disk (protective MBR) is enumerated by GPT, authoritatively — a zero count is
|
||||||
|
/// final. Otherwise every fitting MBR entry is a volume; a boot signature with no
|
||||||
|
/// partition entries is a bare FAT spanning the whole device. Each volume's
|
||||||
|
/// [start, count) is validated overflow-safe (the confinement invariant the
|
||||||
|
/// driver's clamp rests on), and each prefers its FAT serial identity over the
|
||||||
|
/// disk signature.
|
||||||
|
pub fn allVolumes(reader: SectorReader, device_blocks: u64, out: []Volume) usize {
|
||||||
var block0: [sector_bytes]u8 = undefined;
|
var block0: [sector_bytes]u8 = undefined;
|
||||||
if (!reader.read(0, &block0)) return null;
|
if (!reader.read(0, &block0)) return 0;
|
||||||
if (!hasBootSignature(&block0)) return null;
|
if (!hasBootSignature(&block0)) return 0;
|
||||||
if (isProtectiveMbr(&block0)) return gptFirstVolume(reader, device_blocks);
|
if (isProtectiveMbr(&block0)) return gptAllVolumes(reader, device_blocks, out);
|
||||||
|
var count: usize = 0;
|
||||||
var index: u8 = 0;
|
var index: u8 = 0;
|
||||||
while (index < 4) : (index += 1) {
|
while (index < 4 and count < out.len) : (index += 1) {
|
||||||
const entry = block0[446 + @as(usize, index) * 16 ..][0..16];
|
const entry = block0[446 + @as(usize, index) * 16 ..][0..16];
|
||||||
const kind = entry[4];
|
const kind = entry[4];
|
||||||
const start = std.mem.readInt(u32, entry[8..12], .little);
|
const start = std.mem.readInt(u32, entry[8..12], .little);
|
||||||
@@ -283,10 +322,27 @@ pub fn firstVolume(reader: SectorReader, device_blocks: u64) ?Volume {
|
|||||||
// device (usb-storage.zig resolveTransfer), which only holds because the
|
// device (usb-storage.zig resolveTransfer), which only holds because the
|
||||||
// range handed down is validated here. The subtraction cannot overflow.
|
// range handed down is validated here. The subtraction cannot overflow.
|
||||||
if (start > device_blocks or device_blocks - start < size) continue;
|
if (start > device_blocks or device_blocks - start < size) continue;
|
||||||
return .{ .base_lba = start, .block_count = size, .identity = fatIdentity(reader, start) orelse mbrIdentity(&block0, index) };
|
const found = recognize(reader, start, &block0, index);
|
||||||
|
out[count] = .{ .base_lba = start, .block_count = size, .identity = found.identity, .signature = found.signature };
|
||||||
|
count += 1;
|
||||||
}
|
}
|
||||||
// No partition entries: a bare FAT spanning the device.
|
if (count == 0 and out.len > 0) {
|
||||||
return .{ .base_lba = 0, .block_count = device_blocks, .identity = fatIdentity(reader, 0) orelse mbrIdentity(&block0, 0) };
|
// No partition entries: a bare FAT or exFAT spanning the device.
|
||||||
|
const found = recognize(reader, 0, &block0, 0);
|
||||||
|
out[0] = .{ .base_lba = 0, .block_count = device_blocks, .identity = found.identity, .signature = found.signature };
|
||||||
|
return 1;
|
||||||
|
}
|
||||||
|
return count;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// firstVolume is allVolumes into a one-element buffer.
|
||||||
|
const one_volume_slot = 1;
|
||||||
|
|
||||||
|
/// The first volume on the device, or null — the single-volume case of
|
||||||
|
/// `allVolumes`, kept for callers that want just one.
|
||||||
|
pub fn firstVolume(reader: SectorReader, device_blocks: u64) ?Volume {
|
||||||
|
var one: [one_volume_slot]Volume = undefined;
|
||||||
|
return if (allVolumes(reader, device_blocks, &one) > 0) one[0] else null;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// A read-only RAM disk over a byte slice of sectors, for the host tests.
|
/// A read-only RAM disk over a byte slice of sectors, for the host tests.
|
||||||
@@ -338,6 +394,31 @@ test "no boot signature is no volume" {
|
|||||||
const disk = RamDisk{ .sectors = &block0 };
|
const disk = RamDisk{ .sectors = &block0 };
|
||||||
try std.testing.expect(firstVolume(disk.reader(), 65536) == null);
|
try std.testing.expect(firstVolume(disk.reader(), 65536) == null);
|
||||||
}
|
}
|
||||||
|
test "a bare exFAT volume is recognized by its VBR, with its serial as the id" {
|
||||||
|
var block0 = [_]u8{0} ** 512;
|
||||||
|
@memcpy(block0[3..11], "EXFAT ");
|
||||||
|
block0[510] = 0x55;
|
||||||
|
block0[511] = 0xAA;
|
||||||
|
std.mem.writeInt(u32, block0[100..104], 0xDA7A0001, .little); // VolumeSerialNumber
|
||||||
|
const disk = RamDisk{ .sectors = &block0 };
|
||||||
|
const v = firstVolume(disk.reader(), 65536).?;
|
||||||
|
try std.testing.expectEqual(FilesystemKind.exfat, v.signature);
|
||||||
|
try std.testing.expectEqual(Rung.exfat_serial, v.identity.rung);
|
||||||
|
try std.testing.expectEqual(@as(u128, 0xDA7A0001), v.identity.key);
|
||||||
|
}
|
||||||
|
test "a FAT VBR is recognized as fat, not exfat — the signatures never collide" {
|
||||||
|
var block0 = [_]u8{0} ** 512;
|
||||||
|
@memcpy(block0[3..11], "MSWIN4.1"); // a FAT OEM name, not "EXFAT "
|
||||||
|
block0[510] = 0x55;
|
||||||
|
block0[511] = 0xAA;
|
||||||
|
std.mem.writeInt(u16, block0[22..24], 16, .little); // fat_size_16 != 0 -> FAT16 shape
|
||||||
|
block0[38] = 0x29; // extended boot signature
|
||||||
|
std.mem.writeInt(u32, block0[39..43], 0x12345678, .little); // volume id
|
||||||
|
const disk = RamDisk{ .sectors = &block0 };
|
||||||
|
const v = firstVolume(disk.reader(), 65536).?;
|
||||||
|
try std.testing.expectEqual(FilesystemKind.fat, v.signature);
|
||||||
|
try std.testing.expectEqual(Rung.fat_serial, v.identity.rung);
|
||||||
|
}
|
||||||
|
|
||||||
test "a partition that runs past the device is skipped, not trusted" {
|
test "a partition that runs past the device is skipped, not trusted" {
|
||||||
var block0 = [_]u8{0} ** 512;
|
var block0 = [_]u8{0} ** 512;
|
||||||
@@ -507,3 +588,33 @@ test "GPT with 256-byte entries reads the non-128 offset arithmetic correctly" {
|
|||||||
try std.testing.expectEqual(Rung.gpt_guid, v.identity.rung);
|
try std.testing.expectEqual(Rung.gpt_guid, v.identity.rung);
|
||||||
try std.testing.expectEqual(@as(u128, 0xF00D), v.identity.key);
|
try std.testing.expectEqual(@as(u128, 0xF00D), v.identity.key);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// A fixture-sized volume buffer for the multi-volume tests, named so the bounds
|
||||||
|
// gate (which flags literal array lengths) stays quiet: a test input.
|
||||||
|
const test_volume_slots = 4;
|
||||||
|
|
||||||
|
test "allVolumes returns every fitting MBR partition with distinct identities" {
|
||||||
|
var block0 = [_]u8{0} ** 512;
|
||||||
|
block0[510] = 0x55;
|
||||||
|
block0[511] = 0xAA;
|
||||||
|
std.mem.writeInt(u32, block0[440..444], 0xDEADBEEF, .little);
|
||||||
|
// partition 0: start 2048, size 1000
|
||||||
|
block0[446 + 4] = 0x0c;
|
||||||
|
std.mem.writeInt(u32, block0[446 + 8 ..][0..4], 2048, .little);
|
||||||
|
std.mem.writeInt(u32, block0[446 + 12 ..][0..4], 1000, .little);
|
||||||
|
// partition 1: start 4096, size 2000
|
||||||
|
block0[462 + 4] = 0x0c;
|
||||||
|
std.mem.writeInt(u32, block0[462 + 8 ..][0..4], 4096, .little);
|
||||||
|
std.mem.writeInt(u32, block0[462 + 12 ..][0..4], 2000, .little);
|
||||||
|
const disk = RamDisk{ .sectors = &block0 };
|
||||||
|
var vols: [test_volume_slots]Volume = undefined;
|
||||||
|
const n = allVolumes(disk.reader(), 200000, &vols);
|
||||||
|
try std.testing.expectEqual(@as(usize, 2), n); // both partitions, not just the first
|
||||||
|
try std.testing.expectEqual(@as(u64, 2048), vols[0].base_lba);
|
||||||
|
try std.testing.expectEqual(@as(u64, 4096), vols[1].base_lba);
|
||||||
|
// distinct rung-4 identities (no FAT VBR at those LBAs): index 0 vs 1.
|
||||||
|
try std.testing.expectEqual((@as(u128, 0xDEADBEEF) << 8) | 0, vols[0].identity.key);
|
||||||
|
try std.testing.expectEqual((@as(u128, 0xDEADBEEF) << 8) | 1, vols[1].identity.key);
|
||||||
|
// firstVolume (the 1-buffer case) still returns just the first.
|
||||||
|
try std.testing.expectEqual(@as(u64, 2048), firstVolume(disk.reader(), 200000).?.base_lba);
|
||||||
|
}
|
||||||
|
|||||||
@@ -8,10 +8,12 @@
|
|||||||
//! supervises the filesystems it spawns, exactly as the device manager
|
//! supervises the filesystems it spawns, exactly as the device manager
|
||||||
//! supervises drivers.
|
//! supervises drivers.
|
||||||
//!
|
//!
|
||||||
//! This increment (V3b) is the flip: the FAT service stops acquiring its own
|
//! The manager holds a table of adopted storage DEVICES and a table of the
|
||||||
//! volume and is spawned here instead, confined to its partition, and handed
|
//! VOLUMES on them: it adopts every storage device the device-manager tree
|
||||||
//! its channel over the volume-manager protocol. Single volume for now; the
|
//! carries, probes each one's whole partition table, and spawns one filesystem
|
||||||
//! mount map (volumes.csv) and multi-volume land next.
|
//! process per volume — each confined to its partition's badge-scoped block
|
||||||
|
//! range, each supervised with its own budget. A device leaving the tree takes
|
||||||
|
//! its volumes with it.
|
||||||
|
|
||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
const channel = @import("channel");
|
const channel = @import("channel");
|
||||||
@@ -35,22 +37,37 @@ const Serve = volume_manager_protocol.Protocol.Provider(void);
|
|||||||
const Invocation = envelope.Invocation;
|
const Invocation = envelope.Invocation;
|
||||||
const Answer = envelope.Answer;
|
const Answer = envelope.Answer;
|
||||||
|
|
||||||
/// The single volume this increment handles: its provider channel, its block
|
/// One adopted storage device: the block channel to its provider (opened once and
|
||||||
/// sub-range, its identity, the id it is addressed by, and the filesystem
|
/// shared — refcounted per confined filesystem via the hello reply) and the
|
||||||
/// process serving it (0 until spawned; reset on death for respawn).
|
/// device-manager id it serves. A device leaving the tree takes its volumes.
|
||||||
const Volume = struct {
|
const StorageDevice = struct {
|
||||||
storage: block.Device,
|
used: bool = false,
|
||||||
storage_device_id: u64, // the device-manager id this volume's provider serves
|
device_id: u64 = 0,
|
||||||
base_lba: u64,
|
channel: block.Device = undefined,
|
||||||
block_count: u64,
|
|
||||||
identity: partition.Identity,
|
|
||||||
id: u64,
|
|
||||||
binary: []const u8, // the service binary, from filesystems.csv by signature
|
|
||||||
mount_prefix: []const u8, // the volume-root mount path (its id-path, or a volumes.csv override)
|
|
||||||
filesystem_pid: u32 = 0,
|
|
||||||
};
|
};
|
||||||
|
|
||||||
const volume_id: u64 = 1;
|
/// One volume: which device serves it, its block sub-range, its content
|
||||||
|
/// identity, the id it is addressed by, the service binary + mount path it was
|
||||||
|
/// spawned with, the filesystem process serving it, and its own supervision
|
||||||
|
/// budget (so one volume's crash loop never touches another's).
|
||||||
|
const Volume = struct {
|
||||||
|
used: bool = false,
|
||||||
|
device_id: u64 = 0,
|
||||||
|
base_lba: u64 = 0,
|
||||||
|
block_count: u64 = 0,
|
||||||
|
identity: partition.Identity = .{ .rung = .anonymous },
|
||||||
|
id: u64 = 0,
|
||||||
|
binary: []const u8 = "",
|
||||||
|
mount_prefix: []const u8 = "",
|
||||||
|
filesystem_pid: u32 = 0,
|
||||||
|
// Per-volume supervision, mirroring the device manager's: a clean exit is not
|
||||||
|
// restarted, a fault restarts with backoff, a fast crash loop gives up.
|
||||||
|
restarts: u32 = 0,
|
||||||
|
spawn_ns: u64 = 0,
|
||||||
|
failed: bool = false,
|
||||||
|
restart_pending: bool = false,
|
||||||
|
restart_due_ns: u64 = 0,
|
||||||
|
};
|
||||||
|
|
||||||
// The mount map, read from configuration at boot (the policy home, storage-
|
// The mount map, read from configuration at boot (the policy home, storage-
|
||||||
// architecture.md): filesystems.csv (content signature -> service binary) and
|
// architecture.md): filesystems.csv (content signature -> service binary) and
|
||||||
@@ -80,47 +97,77 @@ var filesystem_rules: [maximum_filesystem_rules]filesystem_map.Rule = undefined;
|
|||||||
var filesystem_rule_count: usize = 0;
|
var filesystem_rule_count: usize = 0;
|
||||||
var volume_rules: [maximum_volume_rules]volume_map.Override = undefined;
|
var volume_rules: [maximum_volume_rules]volume_map.Override = undefined;
|
||||||
var volume_rule_count: usize = 0;
|
var volume_rule_count: usize = 0;
|
||||||
/// The composed default mount path (/volumes/<id>) for the current volume; a
|
|
||||||
/// volumes.csv override is used in place and needs no buffer (it is already a
|
|
||||||
/// slice into volumes_source). One buffer suffices while the manager serves one
|
|
||||||
/// volume (multi-volume gives each its own in S3).
|
|
||||||
/// bound: bytes of a composed /volumes/<id> mount path
|
/// bound: bytes of a composed /volumes/<id> mount path
|
||||||
/// decided-by: ours
|
/// decided-by: ours
|
||||||
/// protects: the mount_prefix_buf below
|
/// protects: the per-volume mount_prefix buffers below
|
||||||
/// at-limit: truncate - bufPrint fails; the volume mounts at a fallback path (logged)
|
/// at-limit: truncate - bufPrint fails; the volume mounts at a fallback path (logged)
|
||||||
/// observed-by: the fallback path in the log
|
/// observed-by: the fallback path in the log
|
||||||
const mount_path_maximum = 64;
|
const mount_path_maximum = 64;
|
||||||
var mount_prefix_buf: [mount_path_maximum]u8 = undefined;
|
|
||||||
|
/// bound: volumes the manager serves at once
|
||||||
|
/// decided-by: ours
|
||||||
|
/// protects: the volumes table and its per-volume mount-path buffers
|
||||||
|
/// at-limit: truncate - a further partition is left unserved and logged (real
|
||||||
|
/// machines carry a handful of volumes, far under this)
|
||||||
|
/// observed-by: the "volume table full" log line
|
||||||
|
const maximum_volumes = 16;
|
||||||
|
/// bound: storage devices the manager adopts at once
|
||||||
|
/// decided-by: ours
|
||||||
|
/// protects: the devices table
|
||||||
|
/// at-limit: truncate - a further device is left unadopted and logged
|
||||||
|
/// observed-by: the "device table full" log line
|
||||||
|
const maximum_devices = 8;
|
||||||
|
var devices = [_]StorageDevice{.{}} ** maximum_devices;
|
||||||
|
var volumes = [_]Volume{.{}} ** maximum_volumes;
|
||||||
|
/// Each volume's composed default mount path lives in its slot's buffer; a
|
||||||
|
/// volumes.csv override is used in place (a slice into volumes_source, no buffer).
|
||||||
|
var mount_prefix_bufs: [maximum_volumes][mount_path_maximum]u8 = undefined;
|
||||||
|
var next_volume_id: u64 = 1; // monotonic — never reused, so a stale id can't address the wrong child
|
||||||
|
|
||||||
var service_endpoint: ipc.Handle = 0;
|
var service_endpoint: ipc.Handle = 0;
|
||||||
var manager_handle: ?ipc.Handle = null;
|
var manager_handle: ?ipc.Handle = null;
|
||||||
var bounce: memory.DmaRegion = undefined;
|
var bounce: memory.DmaRegion = undefined;
|
||||||
var bounce_ready = false;
|
var bounce_ready = false;
|
||||||
/// The currently-mounted volume, or null while no storage is present. The whole
|
/// How often the poll checks device presence and fires due restarts. Fast enough
|
||||||
/// removal lifecycle is this field going null and back: the poll sees the
|
/// that an unplug unmounts promptly; the poll is a bare device-manager enumerate,
|
||||||
/// storage provider leave the device tree (a pulled stick), kills the filesystem
|
/// no channel work, so it is cheap to run continuously.
|
||||||
/// and clears this; when it returns, the poll re-acquires and re-mounts.
|
|
||||||
var volume: ?Volume = null;
|
|
||||||
var logged_no_volume = false;
|
|
||||||
/// How often the poll checks whether the storage provider is present. Fast
|
|
||||||
/// enough that an unplug unmounts promptly; the poll is a bare device-manager
|
|
||||||
/// enumerate, no channel work, so it is cheap to run continuously.
|
|
||||||
const poll_interval_ms = 500;
|
const poll_interval_ms = 500;
|
||||||
|
|
||||||
// Filesystem supervision, mirroring the device manager's (device-manager.zig):
|
// Filesystem supervision, mirroring the device manager's (device-manager.zig).
|
||||||
// a clean exit is not restarted, a fault restarts with backoff, and a fast
|
|
||||||
// crash loop gives up rather than spinning. Without this a faulting filesystem
|
|
||||||
// respawns in a zero-delay loop.
|
|
||||||
const fast_death_ns: u64 = 2_000_000_000;
|
const fast_death_ns: u64 = 2_000_000_000;
|
||||||
const crash_loop_cap: u32 = 3;
|
const crash_loop_cap: u32 = 3;
|
||||||
const backoff_base_ms: u64 = 300;
|
const backoff_base_ms: u64 = 300;
|
||||||
var fs_restarts: u32 = 0;
|
/// bytes to format a u64 volume id as decimal (20 digits fit)
|
||||||
var fs_spawn_ns: u64 = 0;
|
const id_decimal_bytes = 24;
|
||||||
var fs_failed = false;
|
|
||||||
/// A fat restart is due at `restart_due_ns`; the poll loop performs it once the
|
// --- table lookups -----------------------------------------------------------
|
||||||
/// backoff has elapsed (one timer, folded into the poll — no second timer).
|
|
||||||
var restart_pending = false;
|
fn deviceById(id: u64) ?*StorageDevice {
|
||||||
var restart_due_ns: u64 = 0;
|
for (&devices) |*d| if (d.used and d.device_id == id) return d;
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
fn claimDevice() ?*StorageDevice {
|
||||||
|
for (&devices) |*d| if (!d.used) return d;
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
fn volumeById(id: u64) ?*Volume {
|
||||||
|
for (&volumes) |*v| if (v.used and v.id == id) return v;
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
fn volumeByPid(pid: u32) ?*Volume {
|
||||||
|
for (&volumes) |*v| if (v.used and v.filesystem_pid == pid) return v;
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
fn firstUsedVolume() ?*Volume {
|
||||||
|
for (&volumes) |*v| if (v.used) return v;
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
fn claimVolumeIndex() ?usize {
|
||||||
|
for (&volumes, 0..) |*v, i| if (!v.used) return i;
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- device-manager plumbing -------------------------------------------------
|
||||||
|
|
||||||
fn deviceManager() ?ipc.Handle {
|
fn deviceManager() ?ipc.Handle {
|
||||||
if (manager_handle) |h| return h;
|
if (manager_handle) |h| return h;
|
||||||
@@ -131,13 +178,12 @@ fn deviceManager() ?ipc.Handle {
|
|||||||
|
|
||||||
const OpenedStorage = struct { device_id: u64, device: block.Device };
|
const OpenedStorage = struct { device_id: u64, device: block.Device };
|
||||||
|
|
||||||
/// The first mass-storage provider whose block channel actually opens, with its
|
/// The first mass-storage provider whose block channel opens and is NOT already
|
||||||
/// device id. A device-manager tree can carry more than one entry of the
|
/// adopted, with its device id. A device-manager tree can carry more than one
|
||||||
/// mass-storage identity — a phantom that no driver is bound to answers a
|
/// entry of the mass-storage identity — a phantom that no driver is bound to
|
||||||
/// consumer hello with NO channel — so this tries each and takes the first that
|
/// answers a consumer hello with NO channel — so this tries each and takes the
|
||||||
/// yields a channel, exactly as a filesystem's own acquisition loop does.
|
/// first that yields a channel. Skips already-adopted devices so a re-poll does
|
||||||
/// Called only when there is no volume (an insertion), so the hellos it makes
|
/// not re-open a device it already serves.
|
||||||
/// are not per-poll churn.
|
|
||||||
fn openAnyStorage() ?OpenedStorage {
|
fn openAnyStorage() ?OpenedStorage {
|
||||||
const manager = deviceManager() orelse return null;
|
const manager = deviceManager() orelse return null;
|
||||||
const Entry = device_manager_protocol.ChildEntry;
|
const Entry = device_manager_protocol.ChildEntry;
|
||||||
@@ -157,6 +203,7 @@ fn openAnyStorage() ?OpenedStorage {
|
|||||||
const entry = std.mem.bytesToValue(Entry, tail[index * @sizeOf(Entry) ..][0..@sizeOf(Entry)]);
|
const entry = std.mem.bytesToValue(Entry, tail[index * @sizeOf(Entry) ..][0..@sizeOf(Entry)]);
|
||||||
if (entry.device_id == device_manager_protocol.no_device) continue;
|
if (entry.device_id == device_manager_protocol.no_device) continue;
|
||||||
if ((entry.identity >> 16) & 0xff != 0x08 or (entry.identity >> 8) & 0xff != 0x06) continue;
|
if ((entry.identity >> 16) & 0xff != 0x08 or (entry.identity >> 8) & 0xff != 0x06) continue;
|
||||||
|
if (deviceById(entry.device_id) != null) continue; // already adopted
|
||||||
const exchanged = driver.helloOn(manager, .consumer, entry.device_id, null, true) orelse continue;
|
const exchanged = driver.helloOn(manager, .consumer, entry.device_id, null, true) orelse continue;
|
||||||
const provider = exchanged.channel orelse continue; // a phantom / not-yet-bound entry
|
const provider = exchanged.channel orelse continue; // a phantom / not-yet-bound entry
|
||||||
return .{ .device_id = entry.device_id, .device = .{ .endpoint = provider } };
|
return .{ .device_id = entry.device_id, .device = .{ .endpoint = provider } };
|
||||||
@@ -167,7 +214,7 @@ fn openAnyStorage() ?OpenedStorage {
|
|||||||
|
|
||||||
/// Whether `device_id` is still in the device-manager tree — a bare enumerate,
|
/// Whether `device_id` is still in the device-manager tree — a bare enumerate,
|
||||||
/// no consumer-hello, so it is cheap to call every poll. This is how removal is
|
/// no consumer-hello, so it is cheap to call every poll. This is how removal is
|
||||||
/// detected: the specific device the mounted volume sits on disappears.
|
/// detected: the specific device a mounted volume sits on disappears.
|
||||||
fn isDevicePresent(device_id: u64) bool {
|
fn isDevicePresent(device_id: u64) bool {
|
||||||
const manager = deviceManager() orelse return false;
|
const manager = deviceManager() orelse return false;
|
||||||
const Entry = device_manager_protocol.ChildEntry;
|
const Entry = device_manager_protocol.ChildEntry;
|
||||||
@@ -191,59 +238,94 @@ fn isDevicePresent(device_id: u64) bool {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Spawn the filesystem for `v`, confine it to the volume's range, and record
|
// --- lifecycle ---------------------------------------------------------------
|
||||||
/// its pid. The confinement is defined for the fresh pid BEFORE the filesystem
|
|
||||||
/// runs, so its first read is already bounded; the volume manager is the
|
/// Spawn the filesystem for `v`, confine it to the volume's range on its device's
|
||||||
/// confinement controller (it defines the first range on the device).
|
/// channel, and record its pid. The confinement is defined for the fresh pid
|
||||||
|
/// BEFORE the filesystem runs, so its first read is already bounded; the volume
|
||||||
|
/// manager is the confinement controller (it defines the first range on the
|
||||||
|
/// device).
|
||||||
fn spawnFilesystem(v: *Volume) void {
|
fn spawnFilesystem(v: *Volume) void {
|
||||||
if (fs_failed) return;
|
if (v.failed) return;
|
||||||
const pid = process.spawnSupervised(v.binary, &.{ "1", v.mount_prefix }, service_endpoint) orelse {
|
const dev = deviceById(v.device_id) orelse return; // its device left — poll will clean up
|
||||||
|
var id_str_buf: [id_decimal_bytes]u8 = undefined;
|
||||||
|
const id_str = std.fmt.bufPrint(&id_str_buf, "{d}", .{v.id}) catch "1";
|
||||||
|
const pid = process.spawnSupervised(v.binary, &.{ id_str, v.mount_prefix }, service_endpoint) orelse {
|
||||||
_ = logging.write("volume-manager: could not spawn the filesystem; retrying\n");
|
_ = logging.write("volume-manager: could not spawn the filesystem; retrying\n");
|
||||||
armRestart();
|
armRestart(v);
|
||||||
return;
|
return;
|
||||||
};
|
};
|
||||||
if (!v.storage.defineRange(pid, v.base_lba, v.block_count)) {
|
if (!dev.channel.defineRange(pid, v.base_lba, v.block_count)) {
|
||||||
_ = logging.write("volume-manager: could not confine the filesystem to its volume; retrying\n");
|
_ = logging.write("volume-manager: could not confine the filesystem to its volume; retrying\n");
|
||||||
_ = process.kill(pid);
|
_ = process.kill(pid);
|
||||||
armRestart();
|
armRestart(v);
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
v.filesystem_pid = pid;
|
v.filesystem_pid = pid;
|
||||||
fs_spawn_ns = time.clock();
|
v.spawn_ns = time.clock();
|
||||||
std.log.info("volume 0x{x} -> {s} (pid {d}), lba {d}, {d} blocks", .{ v.identity.key, v.binary, pid, v.base_lba, v.block_count });
|
std.log.info("volume 0x{x} -> {s} (pid {d}), lba {d}, {d} blocks", .{ v.identity.key, v.binary, pid, v.base_lba, v.block_count });
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Schedule a fat restart after backoff; the poll loop performs it once due.
|
/// Schedule a restart for `v` after backoff; the poll loop performs it once due.
|
||||||
fn armRestart() void {
|
fn armRestart(v: *Volume) void {
|
||||||
const delay = if (fs_restarts == 0) backoff_base_ms else backoff_base_ms << @intCast(@min(fs_restarts - 1, 5));
|
const delay = if (v.restarts == 0) backoff_base_ms else backoff_base_ms << @intCast(@min(v.restarts - 1, 5));
|
||||||
restart_due_ns = time.clock() + delay * 1_000_000;
|
v.restart_due_ns = time.clock() + delay * 1_000_000;
|
||||||
restart_pending = true;
|
v.restart_pending = true;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// A storage provider just appeared: open its channel, read block 0, parse the
|
/// Compose a volume's mount path (its id-path `/volumes/<id>`, or a volumes.csv
|
||||||
/// volume, and spawn its filesystem. On any failure the channel is closed (so a
|
/// override) into its slot's buffer, and return the slice.
|
||||||
/// present-but-unreadable device does not leak a handle every poll) and `volume`
|
fn composeMountPrefix(slot: usize, identity: partition.Identity) []const u8 {
|
||||||
/// stays null — the next poll retries. A fresh medium gets a fresh supervision
|
var id_buf: [volume_map.id_maximum]u8 = undefined;
|
||||||
/// budget.
|
const id = volume_map.idString(identity, &id_buf);
|
||||||
fn bringUpVolume() void {
|
return volume_map.overrideFor(volume_rules[0..volume_rule_count], id) orelse
|
||||||
|
(std.fmt.bufPrint(&mount_prefix_bufs[slot], "/volumes/{s}", .{id}) catch "/volumes/unknown");
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Adopt the next present, not-yet-adopted storage device: take its channel,
|
||||||
|
/// probe its whole partition table, and spawn a filesystem per volume it carries.
|
||||||
|
/// Returns true when it consumed a device (so the caller can loop to adopt every
|
||||||
|
/// present device in one tick), false when none remain or the device table is full.
|
||||||
|
///
|
||||||
|
/// A device is adopted exactly once and kept until it leaves the tree — even when
|
||||||
|
/// it carries no volume we can serve, or its geometry cannot be read. Keeping the
|
||||||
|
/// empty/unreadable device adopted (rather than dropping and re-probing) is what
|
||||||
|
/// lets openAnyStorage advance PAST it to the devices behind it; dropping it would
|
||||||
|
/// make openAnyStorage hand back the same unservable device every tick and starve
|
||||||
|
/// the rest. A genuine removal frees the slot (removeDevice); a re-insert gets a
|
||||||
|
/// fresh device id and is probed anew.
|
||||||
|
fn bringUpVolume() bool {
|
||||||
if (!bounce_ready) {
|
if (!bounce_ready) {
|
||||||
bounce = memory.dmaAlloc(512, memory.dma_coherent | memory.dma_shareable) orelse return;
|
bounce = memory.dmaAlloc(512, memory.dma_coherent | memory.dma_shareable) orelse return false;
|
||||||
bounce_ready = true;
|
bounce_ready = true;
|
||||||
}
|
}
|
||||||
const opened = openAnyStorage() orelse return;
|
const opened = openAnyStorage() orelse return false;
|
||||||
|
const dev = claimDevice() orelse {
|
||||||
|
_ = logging.write("volume-manager: device table full; a storage device is left unadopted\n");
|
||||||
|
_ = ipc.close(opened.device.endpoint);
|
||||||
|
return false;
|
||||||
|
};
|
||||||
|
dev.* = .{ .used = true, .device_id = opened.device_id, .channel = opened.device };
|
||||||
|
// Consume this device's medium_changed events (the second of the removal
|
||||||
|
// lifecycle's two triggers: the device stays in the tree while its medium
|
||||||
|
// leaves — a card reader, an eject). Best effort: a provider that never
|
||||||
|
// publishes the event simply never wakes us, and device-pull is still caught
|
||||||
|
// by the presence poll.
|
||||||
|
_ = opened.device.subscribeMedium(service_endpoint);
|
||||||
const device = opened.device;
|
const device = opened.device;
|
||||||
// Attach the read buffer to THIS device (a no-op without an enforcing IOMMU).
|
// Attach the read buffer to THIS device (a no-op without an enforcing IOMMU).
|
||||||
// The handle is kept, not closed, so it can be re-attached to the next
|
// The handle is kept, not closed, so it can be re-attached after a replug. A
|
||||||
// device after a replug.
|
// failed attach or geometry read leaves the device adopted but empty — we just
|
||||||
|
// cannot read it, and the slot still watches it for removal.
|
||||||
if (bounce.handle) |handle| {
|
if (bounce.handle) |handle| {
|
||||||
if (!device.attach(handle)) {
|
if (!device.attach(handle)) {
|
||||||
_ = ipc.close(device.endpoint);
|
_ = logging.write("volume-manager: could not attach the read buffer to a storage device; no volume served\n");
|
||||||
return;
|
return true;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
const geometry = device.geometry() orelse {
|
const geometry = device.geometry() orelse {
|
||||||
_ = ipc.close(device.endpoint);
|
_ = logging.write("volume-manager: could not read a storage device's geometry; no volume served\n");
|
||||||
return;
|
return true;
|
||||||
};
|
};
|
||||||
const ProbeReader = struct {
|
const ProbeReader = struct {
|
||||||
device: block.Device,
|
device: block.Device,
|
||||||
@@ -257,77 +339,104 @@ fn bringUpVolume() void {
|
|||||||
};
|
};
|
||||||
var probe = ProbeReader{ .device = device };
|
var probe = ProbeReader{ .device = device };
|
||||||
const reader = partition.SectorReader{ .context = &probe, .readFn = ProbeReader.readSector };
|
const reader = partition.SectorReader{ .context = &probe, .readFn = ProbeReader.readSector };
|
||||||
const found = partition.firstVolume(reader, geometry.block_count) orelse {
|
var found: [maximum_volumes]partition.Volume = undefined;
|
||||||
if (!logged_no_volume) {
|
const n = partition.allVolumes(reader, geometry.block_count, found[0..]);
|
||||||
_ = logging.write("volume-manager: storage present but no recognizable volume\n");
|
if (n == 0) {
|
||||||
logged_no_volume = true;
|
std.log.info("device {d} present but carries no recognizable volume", .{dev.device_id});
|
||||||
|
return true;
|
||||||
}
|
}
|
||||||
_ = ipc.close(device.endpoint);
|
for (found[0..n]) |fv| {
|
||||||
return;
|
|
||||||
};
|
|
||||||
// Pick the service binary from the volume's content signature. A signature
|
// Pick the service binary from the volume's content signature. A signature
|
||||||
// no filesystems.csv row serves goes unserved (logged), like an unbound
|
// no filesystems.csv row serves goes unserved (logged), like an unbound
|
||||||
// device — the manager does not guess.
|
// device — the manager does not guess.
|
||||||
const binary = filesystem_map.match(filesystem_rules[0..filesystem_rule_count], found.signature) orelse {
|
const binary = filesystem_map.match(filesystem_rules[0..filesystem_rule_count], fv.signature) orelse {
|
||||||
if (!logged_no_volume) {
|
|
||||||
_ = logging.write("volume-manager: no filesystem serves this volume's content; unserved\n");
|
_ = logging.write("volume-manager: no filesystem serves this volume's content; unserved\n");
|
||||||
logged_no_volume = true;
|
continue;
|
||||||
}
|
|
||||||
_ = ipc.close(device.endpoint);
|
|
||||||
return;
|
|
||||||
};
|
};
|
||||||
// The mount path is the volume's identity id (/volumes/<id>), or a
|
const slot = claimVolumeIndex() orelse {
|
||||||
// volumes.csv override pinning it to a chosen path. The id is content-derived,
|
_ = logging.write("volume-manager: volume table full; a volume is left unserved\n");
|
||||||
// so the path is stable and never a port or a label.
|
break;
|
||||||
var id_buf: [volume_map.id_maximum]u8 = undefined;
|
};
|
||||||
const id = volume_map.idString(found.identity, &id_buf);
|
volumes[slot] = .{
|
||||||
const mount_prefix = volume_map.overrideFor(volume_rules[0..volume_rule_count], id) orelse
|
.used = true,
|
||||||
(std.fmt.bufPrint(&mount_prefix_buf, "/volumes/{s}", .{id}) catch "/volumes/unknown");
|
.device_id = dev.device_id,
|
||||||
logged_no_volume = false;
|
.base_lba = fv.base_lba,
|
||||||
fs_restarts = 0;
|
.block_count = fv.block_count,
|
||||||
fs_failed = false;
|
.identity = fv.identity,
|
||||||
restart_pending = false;
|
.id = next_volume_id,
|
||||||
volume = .{ .storage = device, .storage_device_id = opened.device_id, .base_lba = found.base_lba, .block_count = found.block_count, .identity = found.identity, .id = volume_id, .binary = binary, .mount_prefix = mount_prefix };
|
.binary = binary,
|
||||||
spawnFilesystem(&volume.?);
|
.mount_prefix = composeMountPrefix(slot, fv.identity),
|
||||||
|
};
|
||||||
|
next_volume_id += 1;
|
||||||
|
spawnFilesystem(&volumes[slot]);
|
||||||
|
}
|
||||||
|
return true;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// The storage provider left the device tree (a pulled stick): kill the
|
/// Close a device's channel and free its slot. No volumes are touched (the caller
|
||||||
/// filesystem so its mounts are retired. Retirement is lazy, not an eager
|
/// ensures none remain, or there never were any).
|
||||||
/// death-time sweep — killing the process marks the filesystem's backend
|
fn dropDevice(dev: *StorageDevice) void {
|
||||||
/// endpoint dead, and the VFS router drops each mount that endpoint backed on
|
// Free the driver's subscriber slot before the channel closes. On a still-live
|
||||||
/// the next path resolution under it (that resolve frees the slot and returns
|
// channel (a medium eject) this frees the slot; on a dead one (a device pull)
|
||||||
/// not_found). Then drop the now-dead channel and clear the volume; the next
|
// the call fails fast and the exit sweep frees it anyway.
|
||||||
/// poll that sees storage return re-mounts.
|
_ = dev.channel.unsubscribeMedium();
|
||||||
fn removeVolume() void {
|
_ = ipc.close(dev.channel.endpoint);
|
||||||
const v = volume orelse return;
|
dev.* = .{};
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Retire one volume: kill its filesystem so its mounts are retired. Retirement
|
||||||
|
/// is lazy, not an eager death-time sweep — killing the process marks the
|
||||||
|
/// filesystem's backend endpoint dead, and the VFS router drops each mount that
|
||||||
|
/// endpoint backed on the next path resolution under it (that resolve frees the
|
||||||
|
/// slot and returns not_found). Then free the volume slot.
|
||||||
|
fn removeVolumeState(v: *Volume) void {
|
||||||
std.log.info("storage for volume {d} removed; unmounting", .{v.id});
|
std.log.info("storage for volume {d} removed; unmounting", .{v.id});
|
||||||
if (v.filesystem_pid != 0) _ = process.kill(v.filesystem_pid);
|
if (v.filesystem_pid != 0) _ = process.kill(v.filesystem_pid);
|
||||||
_ = ipc.close(v.storage.endpoint);
|
v.* = .{};
|
||||||
volume = null;
|
|
||||||
restart_pending = false;
|
|
||||||
fs_restarts = 0;
|
|
||||||
fs_failed = false;
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// One poll tick. Removal is checked FIRST and supersedes a pending restart: if
|
/// A storage device left the tree (a pulled stick): retire every volume it served
|
||||||
/// the device is gone there is nothing to restart fat onto, and respawning it
|
/// and drop its channel. One removal path, whether the device is pulled cleanly
|
||||||
/// against the dead channel would just churn until the crash cap. Only once the
|
/// or vanishes.
|
||||||
/// device is confirmed present does a due restart fire.
|
fn removeDevice(dev: *StorageDevice) void {
|
||||||
|
for (&volumes) |*v| {
|
||||||
|
if (v.used and v.device_id == dev.device_id) removeVolumeState(v);
|
||||||
|
}
|
||||||
|
dropDevice(dev);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether a device's block channel still answers — a geometry() probe. A storage
|
||||||
|
/// driver that DIED while its device stays in the tree (it crashed; the device
|
||||||
|
/// manager will re-delegate the device to a restarted driver on a FRESH channel)
|
||||||
|
/// leaves a dead channel here, even though isDevicePresent still reports the device
|
||||||
|
/// present. geometry() on the dead endpoint fails fast, so this catches the crash
|
||||||
|
/// that presence-polling alone cannot — the V4 review's open edge.
|
||||||
|
fn channelAlive(dev: *StorageDevice) bool {
|
||||||
|
return dev.channel.geometry() != null;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// One poll tick. Device removal is reconciled FIRST and supersedes a pending
|
||||||
|
/// restart: a volume whose device left (a pull) OR whose driver died on a channel
|
||||||
|
/// that no longer answers is retired before its restart could fire, so nothing
|
||||||
|
/// respawns against a dead channel. Dropping the device frees its slot, so the
|
||||||
|
/// adopt loop below re-adopts the still-present device on the restarted driver's
|
||||||
|
/// fresh channel — the rebuild. Then due restarts fire for present volumes.
|
||||||
fn pollTick() void {
|
fn pollTick() void {
|
||||||
if (volume) |v| {
|
for (&devices) |*dev| {
|
||||||
// Serving: watch for the specific device leaving (a pulled stick).
|
if (dev.used and (!isDevicePresent(dev.device_id) or !channelAlive(dev))) removeDevice(dev);
|
||||||
if (!isDevicePresent(v.storage_device_id)) {
|
|
||||||
removeVolume();
|
|
||||||
return;
|
|
||||||
}
|
}
|
||||||
if (restart_pending and time.clock() >= restart_due_ns) {
|
for (&volumes) |*v| {
|
||||||
restart_pending = false;
|
if (v.used and v.restart_pending and time.clock() >= v.restart_due_ns) {
|
||||||
spawnFilesystem(&volume.?);
|
v.restart_pending = false;
|
||||||
|
spawnFilesystem(v);
|
||||||
}
|
}
|
||||||
} else {
|
|
||||||
// Idle: try to bring a present storage device up.
|
|
||||||
bringUpVolume();
|
|
||||||
}
|
}
|
||||||
|
// Adopt every present, not-yet-adopted storage device. Each call consumes at
|
||||||
|
// most one device (openAnyStorage skips the adopted), so the loop terminates
|
||||||
|
// once none remain; the maximum_devices guard is insurance against a logic
|
||||||
|
// slip, never the normal exit.
|
||||||
|
var adopted: usize = 0;
|
||||||
|
while (adopted < maximum_devices and bringUpVolume()) : (adopted += 1) {}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// A filesystem announces itself for the volume it was spawned to serve. Reply
|
/// A filesystem announces itself for the volume it was spawned to serve. Reply
|
||||||
@@ -335,26 +444,26 @@ fn pollTick() void {
|
|||||||
/// badge) as the call's returned capability. No channel means the volume is not
|
/// badge) as the call's returned capability. No channel means the volume is not
|
||||||
/// ready — the filesystem retries.
|
/// ready — the filesystem retries.
|
||||||
fn onHello(_: void, invocation: Invocation(volume_manager_protocol.Hello), _: Answer(void)) isize {
|
fn onHello(_: void, invocation: Invocation(volume_manager_protocol.Hello), _: Answer(void)) isize {
|
||||||
const v = volume orelse return 0; // not probed yet — retryable, no cap
|
const v = volumeById(invocation.target) orelse return 0; // not probed yet — retryable, no cap
|
||||||
if (invocation.target != v.id) return 0; // unknown volume — retryable
|
|
||||||
if (invocation.sender != v.filesystem_pid) {
|
if (invocation.sender != v.filesystem_pid) {
|
||||||
// Not the filesystem we spawned for this volume. Refuse: only the
|
// Not the filesystem we spawned for this volume. Refuse: only the confined
|
||||||
// confined filesystem gets the channel.
|
// filesystem gets the channel.
|
||||||
std.log.info("refused hello for volume {d} from process {d}", .{ invocation.target, invocation.sender });
|
std.log.info("refused hello for volume {d} from process {d}", .{ invocation.target, invocation.sender });
|
||||||
return -envelope.EPERM;
|
return -envelope.EPERM;
|
||||||
}
|
}
|
||||||
service.replyWithCapability(v.storage.endpoint);
|
const dev = deviceById(v.device_id) orelse return 0; // its device left — retryable
|
||||||
|
service.replyWithCapability(dev.channel.endpoint);
|
||||||
std.log.info("handed volume {d} to pid {d}", .{ v.id, invocation.sender });
|
std.log.info("handed volume {d} to pid {d}", .{ v.id, invocation.sender });
|
||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Answer a `volumes` query with the mounted volume's descriptor — its id (its
|
/// Answer a `volumes` query with a mounted volume's descriptor — its id (its
|
||||||
/// mount path is /volumes/<id> unless overridden), its actual mount path, and
|
/// mount path is /volumes/<id> unless overridden), its actual mount path, and its
|
||||||
/// its display label. This is how a shell or file manager reads a volume's
|
/// display label. Software keys on the id; a UI shows the label. Returns the first
|
||||||
/// friendly name: software keys on the id, a UI shows the label. An empty reply
|
/// mounted volume for now; a full enumerate is a later refinement. Empty reply
|
||||||
/// means no volume is mounted.
|
/// means no volume is mounted.
|
||||||
fn onVolumes(_: void, _: Invocation(volume_manager_protocol.Volumes), answer: Answer(void)) isize {
|
fn onVolumes(_: void, _: Invocation(volume_manager_protocol.Volumes), answer: Answer(void)) isize {
|
||||||
const v = volume orelse return 0;
|
const v = firstUsedVolume() orelse return 0;
|
||||||
var id_buf: [volume_map.id_maximum]u8 = undefined;
|
var id_buf: [volume_map.id_maximum]u8 = undefined;
|
||||||
const info = volume_manager_protocol.VolumeInfo{
|
const info = volume_manager_protocol.VolumeInfo{
|
||||||
.id = volume_map.idString(v.identity, &id_buf),
|
.id = volume_map.idString(v.identity, &id_buf),
|
||||||
@@ -381,9 +490,9 @@ fn readConfig(path: []const u8, buf: []u8) usize {
|
|||||||
defer file.close();
|
defer file.close();
|
||||||
var used: usize = 0;
|
var used: usize = 0;
|
||||||
while (used < buf.len) {
|
while (used < buf.len) {
|
||||||
const n = file.read(buf[used..]) orelse break;
|
const nn = file.read(buf[used..]) orelse break;
|
||||||
if (n == 0) break;
|
if (nn == 0) break;
|
||||||
used += n;
|
used += nn;
|
||||||
}
|
}
|
||||||
return used;
|
return used;
|
||||||
}
|
}
|
||||||
@@ -427,23 +536,59 @@ fn onNotification(badge: u64) void {
|
|||||||
// reclaimed by the driver on the same death; the respawn confines afresh.
|
// reclaimed by the driver on the same death; the respawn confines afresh.
|
||||||
if (got.isChildExit()) {
|
if (got.isChildExit()) {
|
||||||
const dead = got.childProcessId();
|
const dead = got.childProcessId();
|
||||||
const v = &(volume orelse return);
|
const v = volumeByPid(dead) orelse return;
|
||||||
if (v.filesystem_pid != dead) return;
|
|
||||||
v.filesystem_pid = 0;
|
v.filesystem_pid = 0;
|
||||||
const reason = process.exitReason(dead) orelse .fault;
|
const reason = process.exitReason(dead) orelse .fault;
|
||||||
if (reason == .exited) {
|
if (reason == .exited) {
|
||||||
std.log.info("filesystem for volume {d} exited cleanly; not restarting", .{v.id});
|
std.log.info("filesystem for volume {d} exited cleanly; not restarting", .{v.id});
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
const alive = time.clock() -| fs_spawn_ns;
|
const alive = time.clock() -| v.spawn_ns;
|
||||||
fs_restarts = if (alive < fast_death_ns) fs_restarts + 1 else 1;
|
v.restarts = if (alive < fast_death_ns) v.restarts + 1 else 1;
|
||||||
if (fs_restarts >= crash_loop_cap) {
|
if (v.restarts >= crash_loop_cap) {
|
||||||
fs_failed = true;
|
v.failed = true;
|
||||||
std.log.info("filesystem for volume {d} is failing repeatedly; giving up", .{v.id});
|
std.log.info("filesystem for volume {d} is failing repeatedly; giving up", .{v.id});
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
std.log.info("filesystem for volume {d} died ({s}); restarting", .{ v.id, @tagName(reason) });
|
std.log.info("filesystem for volume {d} died ({s}); restarting", .{ v.id, @tagName(reason) });
|
||||||
armRestart();
|
armRestart(v);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn anyVolumeOn(device_id: u64) bool {
|
||||||
|
for (&volumes) |*v| if (v.used and v.device_id == device_id) return true;
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A storage device published `medium_changed` — the second removal trigger: the
|
||||||
|
/// device stays in the tree while its medium leaves or returns (a card reader, an
|
||||||
|
/// eject). This arrives as a buffered async message, NOT a protocol request, so it
|
||||||
|
/// never reaches `Serve.dispatch` (its event op number collides with the manager's
|
||||||
|
/// own `hello`); it is decoded here by hand. Single-volume scope: the event names
|
||||||
|
/// no device, so `absent` retires every adopted device (its volumes unmount and
|
||||||
|
/// the poll re-adopts the still-present device with its now-empty medium), and
|
||||||
|
/// `present` frees any empty adopted device so the poll re-probes and remounts it.
|
||||||
|
///
|
||||||
|
/// We act on every edge and do NOT dedup on `change_count`. The driver publishes
|
||||||
|
/// exactly once per transition, each with a unique monotonic count, so a count is
|
||||||
|
/// never legitimately re-sent within one subscription — an equality dedup could
|
||||||
|
/// only ever fire spuriously, and it did: `change_count` restarts at 0 in each
|
||||||
|
/// driver instance (usb-storage.zig), so a global "last count" carried across a
|
||||||
|
/// driver restart (S5's own crash-rebuild) mistook the fresh instance's first
|
||||||
|
/// edge for a re-delivery and dropped a real eject, wedging a mount over absent
|
||||||
|
/// media. Both branches are idempotent (a freed device stops matching `dev.used`)
|
||||||
|
/// and the poll reconciles, so reacting to each genuine edge is safe.
|
||||||
|
fn onMediumEvent(payload: []const u8) void {
|
||||||
|
const event = block.decodeMediumChanged(payload) orelse return;
|
||||||
|
if (event.present == 0) {
|
||||||
|
std.log.info("medium left a storage device; unmounting its volume(s)", .{});
|
||||||
|
for (&devices) |*dev| {
|
||||||
|
if (dev.used) removeDevice(dev);
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
for (&devices) |*dev| {
|
||||||
|
if (dev.used and !anyVolumeOn(dev.device_id)) removeDevice(dev);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -454,5 +599,6 @@ pub fn main(init: process.Init) void {
|
|||||||
.init = initialise,
|
.init = initialise,
|
||||||
.on_message = onMessage,
|
.on_message = onMessage,
|
||||||
.on_notification = onNotification,
|
.on_notification = onNotification,
|
||||||
|
.on_buffered_message = onMediumEvent,
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -42,6 +42,7 @@ pub fn idString(identity: partition.Identity, buf: []u8) []const u8 {
|
|||||||
.gpt_guid => std.fmt.bufPrint(buf, "gpt-{x:0>32}", .{identity.key}) catch "",
|
.gpt_guid => std.fmt.bufPrint(buf, "gpt-{x:0>32}", .{identity.key}) catch "",
|
||||||
.filesystem_uuid => std.fmt.bufPrint(buf, "uuid-{x:0>32}", .{identity.key}) catch "",
|
.filesystem_uuid => std.fmt.bufPrint(buf, "uuid-{x:0>32}", .{identity.key}) catch "",
|
||||||
.fat_serial => std.fmt.bufPrint(buf, "fat-{x:0>8}", .{@as(u32, @truncate(identity.key))}) catch "",
|
.fat_serial => std.fmt.bufPrint(buf, "fat-{x:0>8}", .{@as(u32, @truncate(identity.key))}) catch "",
|
||||||
|
.exfat_serial => std.fmt.bufPrint(buf, "exfat-{x:0>8}", .{@as(u32, @truncate(identity.key))}) catch "",
|
||||||
.mbr_index => std.fmt.bufPrint(buf, "mbr-{x}-{d}", .{
|
.mbr_index => std.fmt.bufPrint(buf, "mbr-{x}-{d}", .{
|
||||||
@as(u32, @truncate(identity.key >> 8)),
|
@as(u32, @truncate(identity.key >> 8)),
|
||||||
@as(u8, @truncate(identity.key & 0xff)),
|
@as(u8, @truncate(identity.key & 0xff)),
|
||||||
@@ -105,6 +106,7 @@ const test_override_slots = 4;
|
|||||||
test "idString renders each rung's id token" {
|
test "idString renders each rung's id token" {
|
||||||
var buf: [id_maximum]u8 = undefined;
|
var buf: [id_maximum]u8 = undefined;
|
||||||
try testing.expectEqualStrings("fat-12345678", idString(.{ .rung = .fat_serial, .key = 0x12345678 }, &buf));
|
try testing.expectEqualStrings("fat-12345678", idString(.{ .rung = .fat_serial, .key = 0x12345678 }, &buf));
|
||||||
|
try testing.expectEqualStrings("exfat-da7a0001", idString(.{ .rung = .exfat_serial, .key = 0xDA7A0001 }, &buf));
|
||||||
try testing.expectEqualStrings("mbr-deadbeef-1", idString(.{ .rung = .mbr_index, .key = (@as(u128, 0xDEADBEEF) << 8) | 1 }, &buf));
|
try testing.expectEqualStrings("mbr-deadbeef-1", idString(.{ .rung = .mbr_index, .key = (@as(u128, 0xDEADBEEF) << 8) | 1 }, &buf));
|
||||||
const guid: u128 = 0x00112233445566778899AABBCCDDEEFF;
|
const guid: u128 = 0x00112233445566778899AABBCCDDEEFF;
|
||||||
try testing.expectEqualStrings("gpt-00112233445566778899aabbccddeeff", idString(.{ .rung = .gpt_guid, .key = guid }, &buf));
|
try testing.expectEqualStrings("gpt-00112233445566778899aabbccddeeff", idString(.{ .rung = .gpt_guid, .key = guid }, &buf));
|
||||||
|
|||||||
@@ -796,6 +796,42 @@ CASES = [
|
|||||||
"expect": r"(?s)fat: mounted /volumes/fat-12345678"
|
"expect": r"(?s)fat: mounted /volumes/fat-12345678"
|
||||||
r"[\s\S]*volume-manager: storage for volume \d+ removed; unmounting",
|
r"[\s\S]*volume-manager: storage for volume \d+ removed; unmounting",
|
||||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
|
# S5 medium_changed: the SECOND removal trigger. QMP-eject the MEDIUM (the
|
||||||
|
# block backend, not the device) — the usb-storage device stays in the tree,
|
||||||
|
# but its TEST UNIT READY poll reports not-ready and publishes medium_changed
|
||||||
|
# (absent). The volume manager, now a subscriber, runs the same unmount path as
|
||||||
|
# a device pull. Discrimination: before S5 the manager never subscribed, so the
|
||||||
|
# event reached no one and the mount persisted (device-presence polling cannot
|
||||||
|
# see a medium leave while the device stays). One lifecycle, two triggers.
|
||||||
|
{"name": "volume-medium-change",
|
||||||
|
"build_case": "fat-mount",
|
||||||
|
"smp": 4,
|
||||||
|
"timeout": 150,
|
||||||
|
"qmp_sequence": [
|
||||||
|
{"delay": 8, "command": "eject", "arguments": {"device": "bootusb", "force": True}},
|
||||||
|
],
|
||||||
|
"expect": r"(?s)fat: mounted /volumes/fat-12345678"
|
||||||
|
r"[\s\S]*usb-storage: medium absent"
|
||||||
|
r"[\s\S]*volume-manager: medium left a storage device; unmounting"
|
||||||
|
r"[\s\S]*volume-manager: storage for volume \d+ removed; unmounting",
|
||||||
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
|
# S5 storage-driver-crash rebuild. The device manager (test-storage-restart
|
||||||
|
# mode) kills usb-storage once, ~2s in — after its volume mounted. The device
|
||||||
|
# stays in the tree, so device-presence polling alone would leave fat wedged on
|
||||||
|
# the dead channel; the volume manager's channel-liveness probe (a geometry()
|
||||||
|
# that fails on the dead endpoint) must notice, reap the volume, and rebuild on
|
||||||
|
# the restarted driver's fresh channel — a SECOND mount of the same id-path.
|
||||||
|
# Discrimination: a pre-S5 manager checks only isDevicePresent (still true), so
|
||||||
|
# it never reaps and the second mount never appears (it would restart fat on
|
||||||
|
# the stale channel and crash-loop).
|
||||||
|
{"name": "volume-driver-restart",
|
||||||
|
"build_case": "volume-driver-restart",
|
||||||
|
"smp": 4,
|
||||||
|
"timeout": 150,
|
||||||
|
"expect": r"(?s)fat: mounted /volumes/fat-12345678"
|
||||||
|
r"[\s\S]*volume-manager: storage for volume \d+ removed; unmounting"
|
||||||
|
r"[\s\S]*fat: mounted /volumes/fat-12345678",
|
||||||
|
"fail": r"failing repeatedly; giving up|\[FAIL\]|DANOS-TEST-RESULT: FAIL"},
|
||||||
# Volume-manager discovery + probe (V3a, docs/volume-manager-plan.md). Reuses
|
# Volume-manager discovery + probe (V3a, docs/volume-manager-plan.md). Reuses
|
||||||
# the fat-mount kernel build (the default boot now spawns the volume manager
|
# the fat-mount kernel build (the default boot now spawns the volume manager
|
||||||
# from init.csv). It acquires the mass-storage block channel through the
|
# from init.csv). It acquires the mass-storage block channel through the
|
||||||
@@ -816,6 +852,64 @@ CASES = [
|
|||||||
"expect": r"volume-manager: volume 0x0*12345678 -> \S+ \(pid \d+\), lba \d+, \d+ blocks"
|
"expect": r"volume-manager: volume 0x0*12345678 -> \S+ \(pid \d+\), lba \d+, \d+ blocks"
|
||||||
r"[\s\S]*volume-manager: handed volume \d+ to pid \d+",
|
r"[\s\S]*volume-manager: handed volume \d+ to pid \d+",
|
||||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
|
# S3 multi-volume: a SECOND usb-storage device (a generated data volume, serial
|
||||||
|
# da7a0001, an empty FAT with no /system) plugged in beside the boot volume.
|
||||||
|
# Proves the volume manager adopts BOTH devices and spawns a confined fat per
|
||||||
|
# volume, each mounted at its own CONTENT id-path (/volumes/fat-<serial>); and
|
||||||
|
# that boot-volume detection is by content — only the volume that carries
|
||||||
|
# /system backs /system/configuration, while the data volume mounts at its
|
||||||
|
# id-path alone. Against the pre-S3 one-device/one-volume manager the data
|
||||||
|
# volume never mounts, so the da7a0001 lookaheads fail (toggle-demonstrated by
|
||||||
|
# checking out the step-2 volume-manager.zig).
|
||||||
|
{"name": "two-volumes",
|
||||||
|
"build_case": "fat-mount",
|
||||||
|
"smp": 4,
|
||||||
|
"timeout": 150,
|
||||||
|
"data_volume": {"serial": "DA7A0001", "label": "DATAVOL", "size_mib": 64},
|
||||||
|
"expect": r"(?s)(?=.*volume-manager: volume 0x0*12345678 -> )"
|
||||||
|
r"(?=.*volume-manager: volume 0x0*da7a0001 -> )"
|
||||||
|
r"(?=.*fat: mounted /volumes/fat-12345678)"
|
||||||
|
r"(?=.*fat: mounted /volumes/fat-da7a0001)"
|
||||||
|
r"(?=.*carries the system tree)"
|
||||||
|
r"(?=.*data volume; mounted at /volumes/fat-da7a0001)",
|
||||||
|
"fail": r"data volume; mounted at /volumes/fat-12345678|DANOS-TEST-RESULT: FAIL"},
|
||||||
|
# S3 shared-channel multi-volume: ONE usb-storage device carrying an MBR with
|
||||||
|
# TWO FAT partitions (da7a0001 at lba 2048, da7a0002 at lba 83968). allVolumes
|
||||||
|
# walks the table and the manager spawns a confined fat per partition on the
|
||||||
|
# SAME block channel, each clamped to its own LBA range (usb-storage's
|
||||||
|
# per-badge range table) — the path a pair of single-volume sticks (the
|
||||||
|
# two-volumes case) does NOT exercise. The two mount lines sit at two DISTINCT
|
||||||
|
# non-zero base_lbas on one device. Against the pre-uncap allVolumes (S3 step
|
||||||
|
# 2, capped to one partition) only da7a0001 mounts, so the da7a0002 lookaheads
|
||||||
|
# fail.
|
||||||
|
{"name": "partitioned-volume",
|
||||||
|
"build_case": "fat-mount",
|
||||||
|
"smp": 4,
|
||||||
|
"timeout": 150,
|
||||||
|
"data_volume": {"partitions": [{"serial": "DA7A0001", "size_mib": 40},
|
||||||
|
{"serial": "DA7A0002", "size_mib": 40}]},
|
||||||
|
"expect": r"(?s)(?=.*volume 0x0*da7a0001 -> \S+ \(pid \d+\), lba 2048, )"
|
||||||
|
r"(?=.*volume 0x0*da7a0002 -> \S+ \(pid \d+\), lba 83968, )"
|
||||||
|
r"(?=.*fat: mounted /volumes/fat-da7a0001)"
|
||||||
|
r"(?=.*fat: mounted /volumes/fat-da7a0002)",
|
||||||
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
|
# S4 second engine: a bare exFAT data device (serial e0fa0001) attached beside
|
||||||
|
# the FAT boot volume. The volume manager content-routes it to the exFAT
|
||||||
|
# service (not fat), which mounts it at its id-path /volumes/exfat-e0fa0001;
|
||||||
|
# the exfat-test client then reads the seeded HELLO.TXT and mutates through the
|
||||||
|
# mount (mkdir/write/rename/read/remove). Proves the second engine reuses the
|
||||||
|
# shared harness end to end. Fails against pre-S4 (no exfat binary, csv row, or
|
||||||
|
# VBR recognizer — the device would go to fat, which rejects the exFAT VBR).
|
||||||
|
{"name": "exfat-volume",
|
||||||
|
"build_case": "exfat-volume",
|
||||||
|
"smp": 4,
|
||||||
|
"timeout": 150,
|
||||||
|
"data_volume": {"exfat": True, "serial": "E0FA0001", "size_mib": 48},
|
||||||
|
"expect": r"(?s)(?=.*volume 0x0*e0fa0001 -> /system/services/exfat )"
|
||||||
|
r"(?=.*exfat: mounted /volumes/exfat-e0fa0001)"
|
||||||
|
r"(?=.*exfat-test: read HELLO.TXT ok)"
|
||||||
|
r"(?=.*exfat-test: ok)",
|
||||||
|
"fail": r"exfat-test: FAILED|DANOS-TEST-RESULT: FAIL"},
|
||||||
# Phase 2b: mkdir/unlink through the mount. Reuses the fat-mount build — the
|
# Phase 2b: mkdir/unlink through the mount. Reuses the fat-mount build — the
|
||||||
# fat-test client, after listing, makes a directory, writes+reads a file inside
|
# fat-test client, after listing, makes a directory, writes+reads a file inside
|
||||||
# it, then removes the file, exercising the whole VFS -> fat mutation path.
|
# it, then removes the file, exercising the whole VFS -> fat mutation path.
|
||||||
@@ -1422,6 +1516,34 @@ def run_case(arch, case):
|
|||||||
cmd[cmd.index("-m") + 1] = case["mem"]
|
cmd[cmd.index("-m") + 1] = case["mem"]
|
||||||
if case.get("qemu_extra"): # extra qemu args, e.g. -device intel-iommu for the IOMMU case
|
if case.get("qemu_extra"): # extra qemu args, e.g. -device intel-iommu for the IOMMU case
|
||||||
cmd += case["qemu_extra"]
|
cmd += case["qemu_extra"]
|
||||||
|
# A multi-volume case attaches a second usb-storage device backed by a freshly
|
||||||
|
# GENERATED data volume: a distinct-serial FAT32 with no /system tree, so the
|
||||||
|
# volume manager mounts it at its own id-path and the fat process marks it a
|
||||||
|
# data volume (never a system volume). Regenerated per run — no image is
|
||||||
|
# committed to the tree (the user keeps the boot files copyable, not baked in).
|
||||||
|
if case.get("data_volume"):
|
||||||
|
dv = case["data_volume"]
|
||||||
|
data_img = os.path.join(WORK, "data-volume.img")
|
||||||
|
if dv.get("partitions"):
|
||||||
|
# One device, an MBR with several FAT partitions: several volumes share
|
||||||
|
# ONE block channel, each confined to its own LBA range.
|
||||||
|
gen = [sys.executable, os.path.join(REPO, "tools", "make-partitioned-image.py"), data_img]
|
||||||
|
for part in dv["partitions"]:
|
||||||
|
gen += [part["serial"], str(part.get("size_mib", 40))]
|
||||||
|
elif dv.get("exfat"):
|
||||||
|
# One device, a bare exFAT volume — the second engine's medium.
|
||||||
|
gen = [sys.executable, os.path.join(REPO, "tools", "make-exfat-image.py"),
|
||||||
|
"--serial", dv["serial"], data_img, str(dv.get("size_mib", 48))]
|
||||||
|
else:
|
||||||
|
# One device, one bare FAT volume.
|
||||||
|
gen = [sys.executable, os.path.join(REPO, "tools", "make-fat-image.py"),
|
||||||
|
"--serial", dv["serial"], "--label", dv.get("label", "DATAVOL"),
|
||||||
|
data_img, str(dv.get("size_mib", 64))]
|
||||||
|
subprocess.run(gen, check=True, stdout=subprocess.DEVNULL)
|
||||||
|
cmd += [
|
||||||
|
"-drive", f"if=none,id=datausb,format=raw,file={data_img}",
|
||||||
|
"-device", "usb-storage,bus=xhci.0,port=4,drive=datausb,removable=on,id=datastorage",
|
||||||
|
]
|
||||||
# A QMP control socket, always present (additive): how a case's `qmp_after`
|
# A QMP control socket, always present (additive): how a case's `qmp_after`
|
||||||
# hook injects host-side events into the guest mid-run. Kept under a short temp
|
# hook injects host-side events into the guest mid-run. Kept under a short temp
|
||||||
# dir, not WORK: a unix socket path is capped at ~104 bytes (sun_path), and a
|
# dir, not WORK: a unix socket path is capped at ~104 bytes (sun_path), and a
|
||||||
|
|||||||
@@ -0,0 +1,15 @@
|
|||||||
|
//! The exfat-test test fixture as a binary package (docs/build-packages-plan.md):
|
||||||
|
//! this file names the binary and EXACTLY the modules its source imports —
|
||||||
|
//! build-support resolves each name from the domains this zon declares.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const build_support = @import("build-support");
|
||||||
|
|
||||||
|
pub fn build(b: *std.Build) void {
|
||||||
|
const exe = build_support.userBinary(b, .{
|
||||||
|
.name = "exfat-test",
|
||||||
|
.root_source_file = b.path("exfat-test.zig"),
|
||||||
|
.imports = &.{ "file-system", "logging", "process", "time" },
|
||||||
|
});
|
||||||
|
b.installArtifact(exe);
|
||||||
|
}
|
||||||
@@ -0,0 +1,14 @@
|
|||||||
|
.{
|
||||||
|
.name = .exfat_test,
|
||||||
|
.version = "0.0.0",
|
||||||
|
.fingerprint = 0x77b19e3f7ee43ce3, // Changing this has security and trust implications.
|
||||||
|
.minimum_zig_version = "0.16.0",
|
||||||
|
.dependencies = .{
|
||||||
|
// build-support supplies the shared recipe; kernel is implicit in
|
||||||
|
// every binary (the root shim + link script live there). The rest
|
||||||
|
// are exactly the homes of this binary's declared imports.
|
||||||
|
.@"build-support" = .{ .path = "../../../../build-support" },
|
||||||
|
.kernel = .{ .path = "../../../../library/kernel" },
|
||||||
|
},
|
||||||
|
.paths = .{""},
|
||||||
|
}
|
||||||
@@ -0,0 +1,88 @@
|
|||||||
|
//! test/system/services/exfat-test — a client that proves the exFAT mount end to
|
||||||
|
//! end: it waits for the exfat server to mount the volume at /volumes/exfat-
|
||||||
|
//! e0fa0001, reads the seeded HELLO.TXT off it, and exercises mkdir / write /
|
||||||
|
//! read / rename / remove through the VFS (which routes the id-path to the exfat
|
||||||
|
//! backend). Shipped in the initial_ramdisk; the `exfat-volume` QEMU case spawns
|
||||||
|
//! it alongside init with an exFAT data device attached beside the FAT boot volume.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const fs = @import("file-system");
|
||||||
|
const process = @import("process");
|
||||||
|
const time = @import("time");
|
||||||
|
const logging = @import("logging");
|
||||||
|
|
||||||
|
// The exFAT data device's content id-path — its VolumeSerialNumber is 0xE0FA0001
|
||||||
|
// (the `exfat-volume` case passes --serial E0FA0001 to make-exfat-image.py).
|
||||||
|
const mount = "/volumes/exfat-e0fa0001";
|
||||||
|
|
||||||
|
fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||||
|
var line: [128]u8 = undefined;
|
||||||
|
_ = logging.write(std.fmt.bufPrint(&line, fmt, arguments) catch return);
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn main(init: process.Init) void {
|
||||||
|
_ = init;
|
||||||
|
|
||||||
|
// Wait for the exfat server to bring up the USB storage chain and mount.
|
||||||
|
var opened: ?fs.Directory = null;
|
||||||
|
var tries: u32 = 0;
|
||||||
|
while (opened == null and tries < 1400) : (tries += 1) {
|
||||||
|
opened = fs.openDirectory(mount);
|
||||||
|
if (opened == null) time.sleepMillis(50);
|
||||||
|
}
|
||||||
|
var dir = opened orelse {
|
||||||
|
_ = logging.write("exfat-test: " ++ mount ++ " never became available\n");
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
var count: u32 = 0;
|
||||||
|
var entry: fs.Entry = .{};
|
||||||
|
while (dir.next(&entry)) {
|
||||||
|
writeLine("exfat-test: entry '{s}' size={d}\n", .{ entry.name(), entry.size });
|
||||||
|
count += 1;
|
||||||
|
if (count > 32) break;
|
||||||
|
}
|
||||||
|
dir.close();
|
||||||
|
|
||||||
|
// Read the seeded HELLO.TXT (make-exfat-image.py writes "exfat hello danos\n").
|
||||||
|
var read_ok = false;
|
||||||
|
if (fs.open(mount ++ "/HELLO.TXT", .{})) |opened_file| {
|
||||||
|
var file = opened_file;
|
||||||
|
var buf: [32]u8 = undefined;
|
||||||
|
const n = file.read(&buf) orelse 0;
|
||||||
|
file.close();
|
||||||
|
read_ok = std.mem.startsWith(u8, buf[0..n], "exfat hello danos");
|
||||||
|
}
|
||||||
|
if (read_ok) _ = logging.write("exfat-test: read HELLO.TXT ok\n");
|
||||||
|
|
||||||
|
// Mutation through the mount: mkdir, create + write, rename, read back, remove
|
||||||
|
// — proof the write path reaches the engine over a real device.
|
||||||
|
var mut_ok = false;
|
||||||
|
if (fs.makeDirectory(mount ++ "/TESTDIR")) {
|
||||||
|
var wrote = false;
|
||||||
|
if (fs.open(mount ++ "/TESTDIR/W.TXT", .{ .create = true, .truncate = true })) |created| {
|
||||||
|
var f = created;
|
||||||
|
wrote = (f.writeAll("exfat-mutation-ok") orelse 0) == "exfat-mutation-ok".len;
|
||||||
|
f.close();
|
||||||
|
}
|
||||||
|
const renamed = fs.rename(mount ++ "/TESTDIR/W.TXT", mount ++ "/TESTDIR/R.TXT");
|
||||||
|
var readback = false;
|
||||||
|
if (fs.open(mount ++ "/TESTDIR/R.TXT", .{})) |reopened| {
|
||||||
|
var f = reopened;
|
||||||
|
var buf: [32]u8 = undefined;
|
||||||
|
const got = f.read(&buf) orelse 0;
|
||||||
|
f.close();
|
||||||
|
readback = std.mem.eql(u8, buf[0..got], "exfat-mutation-ok");
|
||||||
|
}
|
||||||
|
const removed = fs.remove(mount ++ "/TESTDIR/R.TXT");
|
||||||
|
mut_ok = wrote and renamed and readback and removed;
|
||||||
|
}
|
||||||
|
if (mut_ok) _ = logging.write("exfat-test: mutations ok\n");
|
||||||
|
|
||||||
|
if (read_ok and mut_ok) {
|
||||||
|
while (true) {
|
||||||
|
_ = logging.write("exfat-test: ok\n");
|
||||||
|
time.sleepMillis(1000);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
writeLine("exfat-test: FAILED (read={} mutations={})\n", .{ read_ok, mut_ok });
|
||||||
|
}
|
||||||
@@ -194,7 +194,6 @@ system/kernel/process.zig:write_buffer
|
|||||||
system/kernel/scheduler.zig:ipc_maximum_handles
|
system/kernel/scheduler.zig:ipc_maximum_handles
|
||||||
system/kernel/scheduler.zig:maximum_space_mappings
|
system/kernel/scheduler.zig:maximum_space_mappings
|
||||||
system/kernel/vfs.zig:maximum_directories
|
system/kernel/vfs.zig:maximum_directories
|
||||||
system/kernel/vfs.zig:maximum_mounts
|
|
||||||
system/kernel/vfs.zig:maximum_prefix
|
system/kernel/vfs.zig:maximum_prefix
|
||||||
system/kernel/vfs.zig:maximum_rewrite
|
system/kernel/vfs.zig:maximum_rewrite
|
||||||
system/services/acpi/acpi.zig:blocks
|
system/services/acpi/acpi.zig:blocks
|
||||||
|
|||||||
@@ -0,0 +1,307 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""Format a real exFAT image from scratch — the danos exFAT test volume.
|
||||||
|
|
||||||
|
Pure Python 3 standard library (no mkfs.exfat / mtools). It writes a valid exFAT
|
||||||
|
filesystem — a Main Boot Sector + its boot-region checksum + a backup region, the
|
||||||
|
32-bit FAT, an allocation bitmap, an up-case table (with its checksum), and a root
|
||||||
|
directory whose entry sets a real exFAT reader (and the danos exfat engine) mount
|
||||||
|
and walk. Mirrors tools/make-fat-image.py in spirit.
|
||||||
|
|
||||||
|
make-exfat-image.py [--serial <hex>] [--label <name>] <out.img> <size-MiB>
|
||||||
|
make-exfat-image.py --verify <out.img>
|
||||||
|
|
||||||
|
The image seeds one file, HELLO.TXT, so a mount can be proven by reading it.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import struct
|
||||||
|
import sys
|
||||||
|
|
||||||
|
SECTOR = 512
|
||||||
|
UPCASE_UNITS = 256 # a-z -> A-Z, the rest identity; covers ASCII names
|
||||||
|
|
||||||
|
|
||||||
|
def align_up(value, to):
|
||||||
|
return (value + to - 1) // to * to
|
||||||
|
|
||||||
|
|
||||||
|
def rotr16(v):
|
||||||
|
return ((v >> 1) | (v << 15)) & 0xFFFF
|
||||||
|
|
||||||
|
|
||||||
|
def rotr32(v):
|
||||||
|
return ((v >> 1) | (v << 31)) & 0xFFFFFFFF
|
||||||
|
|
||||||
|
|
||||||
|
def boot_checksum(region):
|
||||||
|
"""32-bit rotate-right sum over the boot region, skipping VolumeFlags
|
||||||
|
(106,107) and PercentInUse (112) of the first sector."""
|
||||||
|
checksum = 0
|
||||||
|
for i, byte in enumerate(region):
|
||||||
|
if i in (106, 107, 112):
|
||||||
|
continue
|
||||||
|
checksum = (rotr32(checksum) + byte) & 0xFFFFFFFF
|
||||||
|
return checksum
|
||||||
|
|
||||||
|
|
||||||
|
def upcase_checksum(table_bytes):
|
||||||
|
checksum = 0
|
||||||
|
for byte in table_bytes:
|
||||||
|
checksum = (rotr32(checksum) + byte) & 0xFFFFFFFF
|
||||||
|
return checksum
|
||||||
|
|
||||||
|
|
||||||
|
def set_checksum(entries):
|
||||||
|
"""16-bit rotate-right sum over a directory-entry set, skipping its own two
|
||||||
|
checksum bytes (offset 2..3 of the first entry)."""
|
||||||
|
checksum = 0
|
||||||
|
for i, byte in enumerate(entries):
|
||||||
|
if i in (2, 3):
|
||||||
|
continue
|
||||||
|
checksum = (rotr16(checksum) + byte) & 0xFFFF
|
||||||
|
return checksum
|
||||||
|
|
||||||
|
|
||||||
|
def name_hash(upcased_units):
|
||||||
|
h = 0
|
||||||
|
for unit in upcased_units:
|
||||||
|
h = (rotr16(h) + (unit & 0xFF)) & 0xFFFF
|
||||||
|
h = (rotr16(h) + (unit >> 8)) & 0xFFFF
|
||||||
|
return h
|
||||||
|
|
||||||
|
|
||||||
|
def ascii_upper(unit):
|
||||||
|
return unit - ord("a") + ord("A") if ord("a") <= unit <= ord("z") else unit
|
||||||
|
|
||||||
|
|
||||||
|
def solve_geometry(total_sectors, spc):
|
||||||
|
"""Solve for cluster count / FAT length / heap offset that fit. The FAT sits
|
||||||
|
after the main + backup boot regions (24 sectors)."""
|
||||||
|
fat_offset = 24
|
||||||
|
fat_length = 1
|
||||||
|
while True:
|
||||||
|
heap_offset = align_up(fat_offset + fat_length, spc)
|
||||||
|
cluster_count = (total_sectors - heap_offset) // spc
|
||||||
|
needed = ((cluster_count + 2) * 4 + SECTOR - 1) // SECTOR
|
||||||
|
if needed <= fat_length:
|
||||||
|
return cluster_count, fat_offset, fat_length, heap_offset
|
||||||
|
fat_length = needed
|
||||||
|
|
||||||
|
|
||||||
|
class ExfatImage:
|
||||||
|
def __init__(self, size_mib, volume_id=0x1234ABCD, label="DANOS"):
|
||||||
|
self.total_sectors = size_mib * 1024 * 1024 // SECTOR
|
||||||
|
self.spc = 8 # 4 KiB clusters
|
||||||
|
self.volume_id = volume_id & 0xFFFFFFFF
|
||||||
|
self.label = label
|
||||||
|
self.cluster_count, self.fat_offset, self.fat_length, self.heap_offset = solve_geometry(self.total_sectors, self.spc)
|
||||||
|
if self.cluster_count < 16:
|
||||||
|
sys.exit(f"error: image too small for exFAT ({self.cluster_count} clusters)")
|
||||||
|
self.cluster_bytes = self.spc * SECTOR
|
||||||
|
# Layout: the allocation bitmap (as many clusters as it needs — one per
|
||||||
|
# 8*cluster_bytes clusters of the volume), then the up-case table, the root
|
||||||
|
# directory, and the seeded file. A single-cluster bitmap (small volumes,
|
||||||
|
# e.g. the 48 MiB fixture) puts root at cluster 4, as before.
|
||||||
|
self.bitmap_bytes = (self.cluster_count + 7) // 8
|
||||||
|
self.bitmap_clusters = (self.bitmap_bytes + self.cluster_bytes - 1) // self.cluster_bytes
|
||||||
|
self.bitmap_cluster = 2
|
||||||
|
self.upcase_cluster = self.bitmap_cluster + self.bitmap_clusters
|
||||||
|
self.root_cluster = self.upcase_cluster + 1
|
||||||
|
self.hello_cluster = self.root_cluster + 1
|
||||||
|
self.image = bytearray(self.total_sectors * SECTOR)
|
||||||
|
|
||||||
|
def cluster_offset(self, cluster):
|
||||||
|
return (self.heap_offset + (cluster - 2) * self.spc) * SECTOR
|
||||||
|
|
||||||
|
def set_fat(self, cluster, value):
|
||||||
|
struct.pack_into("<I", self.image, self.fat_offset * SECTOR + cluster * 4, value)
|
||||||
|
|
||||||
|
def mark_allocated(self, cluster):
|
||||||
|
bit = cluster - 2
|
||||||
|
pos = self.cluster_offset(2) + bit // 8
|
||||||
|
self.image[pos] |= 1 << (bit % 8)
|
||||||
|
|
||||||
|
def main_boot_sector(self):
|
||||||
|
sector = bytearray(SECTOR)
|
||||||
|
sector[0:3] = b"\xEB\x76\x90" # jump boot
|
||||||
|
sector[3:11] = b"EXFAT " # filesystem name
|
||||||
|
# 11..64 MustBeZero (already zero)
|
||||||
|
struct.pack_into("<Q", sector, 72, self.total_sectors) # volume length
|
||||||
|
struct.pack_into("<I", sector, 80, self.fat_offset) # fat offset
|
||||||
|
struct.pack_into("<I", sector, 84, self.fat_length) # fat length
|
||||||
|
struct.pack_into("<I", sector, 88, self.heap_offset) # cluster heap offset
|
||||||
|
struct.pack_into("<I", sector, 92, self.cluster_count) # cluster count
|
||||||
|
struct.pack_into("<I", sector, 96, self.root_cluster) # first cluster of root
|
||||||
|
struct.pack_into("<I", sector, 100, self.volume_id) # volume serial number
|
||||||
|
struct.pack_into("<H", sector, 104, 0x0100) # filesystem revision 1.0
|
||||||
|
sector[108] = 9 # bytes per sector shift (512)
|
||||||
|
sector[109] = self.spc.bit_length() - 1 # sectors per cluster shift
|
||||||
|
sector[110] = 1 # number of FATs
|
||||||
|
sector[111] = 0x80 # drive select
|
||||||
|
sector[112] = 0xFF # percent in use (unknown)
|
||||||
|
sector[510] = 0x55
|
||||||
|
sector[511] = 0xAA
|
||||||
|
return sector
|
||||||
|
|
||||||
|
def build(self):
|
||||||
|
# Main boot region (sectors 0..11): VBR, eight extended boot sectors, OEM
|
||||||
|
# parameters, reserved, then the checksum sector.
|
||||||
|
vbr = self.main_boot_sector()
|
||||||
|
self.image[0:SECTOR] = vbr
|
||||||
|
for s in range(1, 9): # extended boot sectors carry the 0xAA550000 signature
|
||||||
|
struct.pack_into("<I", self.image, s * SECTOR + 508, 0xAA550000)
|
||||||
|
# sectors 9 (OEM) and 10 (reserved) stay zero
|
||||||
|
region = bytes(self.image[0 : 11 * SECTOR])
|
||||||
|
checksum = boot_checksum(region)
|
||||||
|
for i in range(SECTOR // 4):
|
||||||
|
struct.pack_into("<I", self.image, 11 * SECTOR + i * 4, checksum)
|
||||||
|
# Backup boot region (sectors 12..23) is a copy of 0..11.
|
||||||
|
self.image[12 * SECTOR : 24 * SECTOR] = self.image[0 : 12 * SECTOR]
|
||||||
|
|
||||||
|
# FAT: reserved entries, then a single-cluster chain per metadata object,
|
||||||
|
# except the bitmap which spans self.bitmap_clusters (a real FAT chain).
|
||||||
|
self.set_fat(0, 0xFFFFFFF8)
|
||||||
|
self.set_fat(1, 0xFFFFFFFF)
|
||||||
|
used = []
|
||||||
|
for i in range(self.bitmap_clusters):
|
||||||
|
cluster = self.bitmap_cluster + i
|
||||||
|
self.set_fat(cluster, 0xFFFFFFFF if i == self.bitmap_clusters - 1 else cluster + 1)
|
||||||
|
used.append(cluster)
|
||||||
|
for cluster in (self.upcase_cluster, self.root_cluster, self.hello_cluster):
|
||||||
|
self.set_fat(cluster, 0xFFFFFFFF)
|
||||||
|
used.append(cluster)
|
||||||
|
|
||||||
|
# Allocation bitmap: every metadata/file cluster in use.
|
||||||
|
for cluster in used:
|
||||||
|
self.mark_allocated(cluster)
|
||||||
|
|
||||||
|
# Up-case table: 256 explicit units, a-z -> A-Z.
|
||||||
|
upcase = bytearray(UPCASE_UNITS * 2)
|
||||||
|
for i in range(UPCASE_UNITS):
|
||||||
|
struct.pack_into("<H", upcase, i * 2, ascii_upper(i))
|
||||||
|
off = self.cluster_offset(self.upcase_cluster)
|
||||||
|
self.image[off : off + len(upcase)] = upcase
|
||||||
|
table_checksum = upcase_checksum(upcase)
|
||||||
|
|
||||||
|
# Seed file HELLO.TXT (contiguous, one cluster).
|
||||||
|
content = b"exfat hello danos\n"
|
||||||
|
off = self.cluster_offset(self.hello_cluster)
|
||||||
|
self.image[off : off + len(content)] = content
|
||||||
|
|
||||||
|
# Root directory: bitmap, up-case, volume label, HELLO set.
|
||||||
|
root = self.cluster_offset(self.root_cluster)
|
||||||
|
# 0x81 Allocation Bitmap
|
||||||
|
struct.pack_into("<BBB", self.image, root, 0x81, 0, 0)
|
||||||
|
struct.pack_into("<I", self.image, root + 20, self.bitmap_cluster)
|
||||||
|
struct.pack_into("<Q", self.image, root + 24, self.bitmap_bytes)
|
||||||
|
# 0x82 Up-case Table
|
||||||
|
struct.pack_into("<B", self.image, root + 32, 0x82)
|
||||||
|
struct.pack_into("<I", self.image, root + 32 + 4, table_checksum)
|
||||||
|
struct.pack_into("<I", self.image, root + 32 + 20, self.upcase_cluster)
|
||||||
|
struct.pack_into("<Q", self.image, root + 32 + 24, UPCASE_UNITS * 2)
|
||||||
|
# 0x83 Volume Label
|
||||||
|
label_units = [ord(c) for c in self.label[:11]]
|
||||||
|
struct.pack_into("<BB", self.image, root + 64, 0x83, len(label_units))
|
||||||
|
for i, u in enumerate(label_units):
|
||||||
|
struct.pack_into("<H", self.image, root + 64 + 2 + i * 2, u)
|
||||||
|
# HELLO.TXT set: File (0x85) + Stream (0xC0) + Name (0xC1)
|
||||||
|
name = "HELLO.TXT"
|
||||||
|
self.write_file_set(root + 96, name, first_cluster=self.hello_cluster, length=len(content))
|
||||||
|
|
||||||
|
def write_file_set(self, offset, name, first_cluster, length):
|
||||||
|
entries = bytearray(32 * 3)
|
||||||
|
# File entry
|
||||||
|
entries[0] = 0x85
|
||||||
|
entries[1] = 2 # stream + one name entry
|
||||||
|
struct.pack_into("<H", entries, 4, 0x20) # attributes: archive
|
||||||
|
# Stream entry
|
||||||
|
entries[32 + 0] = 0xC0
|
||||||
|
entries[32 + 1] = 0x01 | 0x02 # allocation possible + no FAT chain (contiguous)
|
||||||
|
entries[32 + 3] = len(name)
|
||||||
|
upname = [ascii_upper(ord(c)) for c in name]
|
||||||
|
struct.pack_into("<H", entries, 32 + 4, name_hash(upname))
|
||||||
|
struct.pack_into("<Q", entries, 32 + 8, length) # valid data length
|
||||||
|
struct.pack_into("<I", entries, 32 + 20, first_cluster)
|
||||||
|
struct.pack_into("<Q", entries, 32 + 24, length) # data length
|
||||||
|
# File Name entry
|
||||||
|
entries[64 + 0] = 0xC1
|
||||||
|
for i, c in enumerate(name):
|
||||||
|
struct.pack_into("<H", entries, 64 + 2 + i * 2, ord(c))
|
||||||
|
struct.pack_into("<H", entries, 2, set_checksum(entries))
|
||||||
|
self.image[offset : offset + len(entries)] = entries
|
||||||
|
|
||||||
|
def serialize(self):
|
||||||
|
self.build()
|
||||||
|
return bytes(self.image)
|
||||||
|
|
||||||
|
|
||||||
|
def verify(path):
|
||||||
|
with open(path, "rb") as handle:
|
||||||
|
data = handle.read()
|
||||||
|
if len(data) < 512 or data[510] != 0x55 or data[511] != 0xAA:
|
||||||
|
sys.exit("verify: missing 0x55AA boot signature")
|
||||||
|
if data[3:11] != b"EXFAT ":
|
||||||
|
sys.exit("verify: not an exFAT boot sector")
|
||||||
|
if any(data[11:64]):
|
||||||
|
sys.exit("verify: MustBeZero region is not zero")
|
||||||
|
fat_offset = struct.unpack_from("<I", data, 80)[0]
|
||||||
|
heap_offset = struct.unpack_from("<I", data, 88)[0]
|
||||||
|
cluster_count = struct.unpack_from("<I", data, 92)[0]
|
||||||
|
root_cluster = struct.unpack_from("<I", data, 96)[0]
|
||||||
|
spc = 1 << data[109]
|
||||||
|
# Boot checksum sector 11 must match a fresh checksum over sectors 0..10.
|
||||||
|
expected = boot_checksum(data[0 : 11 * SECTOR])
|
||||||
|
got = struct.unpack_from("<I", data, 11 * SECTOR)[0]
|
||||||
|
if expected != got:
|
||||||
|
sys.exit(f"verify: boot checksum mismatch (0x{got:08X} != 0x{expected:08X})")
|
||||||
|
# Walk the root directory for the HELLO.TXT set and check its checksum.
|
||||||
|
root = (heap_offset + (root_cluster - 2) * spc) * SECTOR
|
||||||
|
found = False
|
||||||
|
for i in range(spc * SECTOR // 32):
|
||||||
|
entry = root + i * 32
|
||||||
|
if data[entry] == 0x00:
|
||||||
|
break
|
||||||
|
if data[entry] == 0x85:
|
||||||
|
secondary = data[entry + 1]
|
||||||
|
total = (secondary + 1) * 32
|
||||||
|
stored = struct.unpack_from("<H", data, entry + 2)[0]
|
||||||
|
if set_checksum(data[entry : entry + total]) != stored:
|
||||||
|
sys.exit("verify: a file set checksum is wrong")
|
||||||
|
found = True
|
||||||
|
if not found:
|
||||||
|
sys.exit("verify: no file set in the root directory")
|
||||||
|
print(f"make-exfat-image: {path} OK "
|
||||||
|
f"({cluster_count} clusters of {spc * SECTOR} bytes, fat@{fat_offset}, heap@{heap_offset})")
|
||||||
|
|
||||||
|
|
||||||
|
def main(argv):
|
||||||
|
if len(argv) == 3 and argv[1] == "--verify":
|
||||||
|
verify(argv[2])
|
||||||
|
return 0
|
||||||
|
argv = list(argv)
|
||||||
|
volume_id = 0x1234ABCD
|
||||||
|
label = "DANOS"
|
||||||
|
i = 1
|
||||||
|
while i < len(argv):
|
||||||
|
if argv[i] == "--serial" and i + 1 < len(argv):
|
||||||
|
volume_id = int(argv[i + 1], 16)
|
||||||
|
del argv[i : i + 2]
|
||||||
|
elif argv[i] == "--label" and i + 1 < len(argv):
|
||||||
|
label = argv[i + 1]
|
||||||
|
del argv[i : i + 2]
|
||||||
|
else:
|
||||||
|
i += 1
|
||||||
|
if len(argv) != 3:
|
||||||
|
sys.exit("usage: make-exfat-image.py [--serial <hex>] [--label <name>] <out.img> <size-MiB>\n"
|
||||||
|
" make-exfat-image.py --verify <out.img>")
|
||||||
|
out_path = argv[1]
|
||||||
|
size_mib = int(argv[2])
|
||||||
|
image = ExfatImage(size_mib, volume_id, label)
|
||||||
|
with open(out_path, "wb") as handle:
|
||||||
|
handle.write(image.serialize())
|
||||||
|
print(f"make-exfat-image: wrote {out_path} "
|
||||||
|
f"({size_mib} MiB exFAT, {image.cluster_count} clusters, serial 0x{image.volume_id:08X})")
|
||||||
|
return 0
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
sys.exit(main(sys.argv))
|
||||||
+31
-7
@@ -47,8 +47,14 @@ def fat32_geometry(total_sectors):
|
|||||||
|
|
||||||
|
|
||||||
class Fat32Image:
|
class Fat32Image:
|
||||||
def __init__(self, total_sectors):
|
def __init__(self, total_sectors, volume_id=0x12345678, label="DANOS"):
|
||||||
self.total_sectors = total_sectors
|
self.total_sectors = total_sectors
|
||||||
|
# The FAT volume serial (its content identity — the /volumes/fat-<id>
|
||||||
|
# mount path danos derives from it) and the display label. A second image
|
||||||
|
# needs a distinct serial so its id-path does not collide with the boot
|
||||||
|
# volume's.
|
||||||
|
self.volume_id = volume_id & 0xFFFFFFFF
|
||||||
|
self.label = label
|
||||||
self.fat_size, self.cluster_count = fat32_geometry(total_sectors)
|
self.fat_size, self.cluster_count = fat32_geometry(total_sectors)
|
||||||
if self.cluster_count < 65525:
|
if self.cluster_count < 65525:
|
||||||
sys.exit(f"error: image too small for FAT32 ({self.cluster_count} clusters "
|
sys.exit(f"error: image too small for FAT32 ({self.cluster_count} clusters "
|
||||||
@@ -146,8 +152,8 @@ class Fat32Image:
|
|||||||
0x80, # drive number
|
0x80, # drive number
|
||||||
0, # reserved
|
0, # reserved
|
||||||
0x29, # extended boot signature
|
0x29, # extended boot signature
|
||||||
0x12345678, # volume id
|
self.volume_id, # volume id
|
||||||
b"DANOS ", # volume label
|
self.label.encode("ascii", "replace")[:11].ljust(11, b" "), # volume label
|
||||||
b"FAT32 ", # filesystem type
|
b"FAT32 ", # filesystem type
|
||||||
)
|
)
|
||||||
sector[510] = 0x55
|
sector[510] = 0x55
|
||||||
@@ -276,9 +282,9 @@ def build_tree(pairs):
|
|||||||
return root
|
return root
|
||||||
|
|
||||||
|
|
||||||
def build(out_path, size_mib, pairs):
|
def build(out_path, size_mib, pairs, volume_id=0x12345678, label="DANOS"):
|
||||||
total_sectors = size_mib * 1024 * 1024 // SECTOR
|
total_sectors = size_mib * 1024 * 1024 // SECTOR
|
||||||
image = Fat32Image(total_sectors)
|
image = Fat32Image(total_sectors, volume_id, label)
|
||||||
tree = build_tree(pairs)
|
tree = build_tree(pairs)
|
||||||
write_directory(image, 2, tree, 0, True)
|
write_directory(image, 2, tree, 0, True)
|
||||||
with open(out_path, "wb") as handle:
|
with open(out_path, "wb") as handle:
|
||||||
@@ -350,14 +356,32 @@ def main(argv):
|
|||||||
if len(argv) == 3 and argv[1] == "--verify":
|
if len(argv) == 3 and argv[1] == "--verify":
|
||||||
verify(argv[2])
|
verify(argv[2])
|
||||||
return 0
|
return 0
|
||||||
|
# Optional flags ahead of the positionals: --serial <hex> sets the FAT volume
|
||||||
|
# id (the /volumes/fat-<id> content identity), --label <name> its display
|
||||||
|
# label. A second FAT image passes a distinct --serial so its id-path cannot
|
||||||
|
# collide with the boot volume's.
|
||||||
|
argv = list(argv)
|
||||||
|
volume_id = 0x12345678
|
||||||
|
label = "DANOS"
|
||||||
|
i = 1
|
||||||
|
while i < len(argv):
|
||||||
|
if argv[i] == "--serial" and i + 1 < len(argv):
|
||||||
|
volume_id = int(argv[i + 1], 16)
|
||||||
|
del argv[i:i + 2]
|
||||||
|
elif argv[i] == "--label" and i + 1 < len(argv):
|
||||||
|
label = argv[i + 1]
|
||||||
|
del argv[i:i + 2]
|
||||||
|
else:
|
||||||
|
i += 1
|
||||||
if len(argv) < 3 or (len(argv) - 3) % 2 != 0:
|
if len(argv) < 3 or (len(argv) - 3) % 2 != 0:
|
||||||
sys.exit("usage: make-fat-image.py <out.img> <size-MiB> [<dest> <host>]...\n"
|
sys.exit("usage: make-fat-image.py [--serial <hex>] [--label <name>] "
|
||||||
|
"<out.img> <size-MiB> [<dest> <host>]...\n"
|
||||||
" make-fat-image.py --verify <out.img>")
|
" make-fat-image.py --verify <out.img>")
|
||||||
out_path = argv[1]
|
out_path = argv[1]
|
||||||
size_mib = int(argv[2])
|
size_mib = int(argv[2])
|
||||||
rest = argv[3:]
|
rest = argv[3:]
|
||||||
pairs = [(rest[i], rest[i + 1]) for i in range(0, len(rest), 2)]
|
pairs = [(rest[i], rest[i + 1]) for i in range(0, len(rest), 2)]
|
||||||
build(out_path, size_mib, pairs)
|
build(out_path, size_mib, pairs, volume_id, label)
|
||||||
return 0
|
return 0
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,88 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""Assemble an MBR-partitioned disk image from N FAT32 partitions — the danos
|
||||||
|
multi-volume test disk.
|
||||||
|
|
||||||
|
Pure Python 3 stdlib (no mtools / parted). Each partition is a real FAT32
|
||||||
|
filesystem produced by make-fat-image.py, laid out behind a classic MBR so the
|
||||||
|
danos partition prober (partition.allVolumes) walks the table and the volume
|
||||||
|
manager spawns one confined filesystem per partition — several volumes sharing
|
||||||
|
ONE block channel, each clamped to its own LBA range. That shared-channel,
|
||||||
|
per-partition path is what a single stick with two partitions exercises and a
|
||||||
|
pair of single-volume sticks does not.
|
||||||
|
|
||||||
|
make-partitioned-image.py <out.img> [<serial-hex> <size-MiB>]...
|
||||||
|
|
||||||
|
Each partition is an empty FAT32 with the given volume serial (its /volumes/
|
||||||
|
fat-<serial> content id). Partitions are 1-MiB aligned; the MBR marks each
|
||||||
|
type 0x0C (FAT32 LBA). At most four (an MBR holds four primaries).
|
||||||
|
"""
|
||||||
|
|
||||||
|
import os
|
||||||
|
import struct
|
||||||
|
import subprocess
|
||||||
|
import sys
|
||||||
|
import tempfile
|
||||||
|
|
||||||
|
SECTOR = 512
|
||||||
|
ALIGN = 2048 # sectors (1 MiB) — standard partition alignment, and the gap the MBR sits in
|
||||||
|
MBR_TYPE_FAT32_LBA = 0x0C
|
||||||
|
MAX_PRIMARY_PARTITIONS = 4
|
||||||
|
HERE = os.path.dirname(os.path.abspath(__file__))
|
||||||
|
|
||||||
|
|
||||||
|
def align_up(sectors, to=ALIGN):
|
||||||
|
return (sectors + to - 1) // to * to
|
||||||
|
|
||||||
|
|
||||||
|
def main(argv):
|
||||||
|
if len(argv) < 4 or (len(argv) - 2) % 2 != 0:
|
||||||
|
sys.exit("usage: make-partitioned-image.py <out.img> [<serial-hex> <size-MiB>]...")
|
||||||
|
out_path = argv[1]
|
||||||
|
specs = [(argv[i], int(argv[i + 1])) for i in range(2, len(argv), 2)]
|
||||||
|
if len(specs) > MAX_PRIMARY_PARTITIONS:
|
||||||
|
sys.exit(f"error: an MBR holds at most {MAX_PRIMARY_PARTITIONS} primary partitions")
|
||||||
|
|
||||||
|
# Generate each partition's FAT32 image, then place it at its aligned start.
|
||||||
|
partitions = [] # (start_sector, sector_count, bytes)
|
||||||
|
cursor = ALIGN # leave the first 1 MiB for the MBR + alignment gap
|
||||||
|
with tempfile.TemporaryDirectory() as tmp:
|
||||||
|
for idx, (serial, size_mib) in enumerate(specs):
|
||||||
|
part_path = os.path.join(tmp, f"p{idx}.img")
|
||||||
|
subprocess.run(
|
||||||
|
[sys.executable, os.path.join(HERE, "make-fat-image.py"),
|
||||||
|
"--serial", serial, "--label", f"DATA{idx}",
|
||||||
|
part_path, str(size_mib)],
|
||||||
|
check=True, stdout=subprocess.DEVNULL)
|
||||||
|
with open(part_path, "rb") as handle:
|
||||||
|
data = handle.read()
|
||||||
|
count = len(data) // SECTOR
|
||||||
|
partitions.append((cursor, count, data))
|
||||||
|
cursor = align_up(cursor + count)
|
||||||
|
|
||||||
|
total_sectors = cursor
|
||||||
|
disk = bytearray(total_sectors * SECTOR)
|
||||||
|
# The MBR: a disk signature, one partition entry per FAT partition, 0x55AA.
|
||||||
|
# No boot code (this disk is data, never booted); danos's mount() sees the
|
||||||
|
# signature but no BPB at LBA 0 and takes the MBR-walk path.
|
||||||
|
struct.pack_into("<I", disk, 440, 0x0D05DA05) # arbitrary but fixed disk signature
|
||||||
|
for idx, (start, count, data) in enumerate(partitions):
|
||||||
|
entry = 446 + idx * 16
|
||||||
|
disk[entry + 0] = 0x00 # not bootable
|
||||||
|
disk[entry + 1:entry + 4] = b"\xFE\xFF\xFF" # CHS start (LBA-aware tools ignore)
|
||||||
|
disk[entry + 4] = MBR_TYPE_FAT32_LBA
|
||||||
|
disk[entry + 5:entry + 8] = b"\xFE\xFF\xFF" # CHS end
|
||||||
|
struct.pack_into("<I", disk, entry + 8, start) # start LBA
|
||||||
|
struct.pack_into("<I", disk, entry + 12, count) # sector count
|
||||||
|
disk[start * SECTOR:start * SECTOR + len(data)] = data
|
||||||
|
disk[510] = 0x55
|
||||||
|
disk[511] = 0xAA
|
||||||
|
|
||||||
|
with open(out_path, "wb") as handle:
|
||||||
|
handle.write(disk)
|
||||||
|
print(f"make-partitioned-image: wrote {out_path} "
|
||||||
|
f"({total_sectors * SECTOR // (1024 * 1024)} MiB, {len(partitions)} partitions)")
|
||||||
|
return 0
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
sys.exit(main(sys.argv))
|
||||||
Reference in New Issue
Block a user