Compare commits
21
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
68e65803eb | ||
|
|
af47d41989 | ||
|
|
5e89b111cf | ||
|
|
e3ec9fa668 | ||
|
|
9e67a74232 | ||
|
|
b9058fe020 | ||
|
|
7c6ed2ca09 | ||
|
|
a67a7015bf | ||
|
|
301bdcaf5b | ||
|
|
d56b1b81c0 | ||
|
|
bc67771bfd | ||
|
|
89d4592777 | ||
|
|
7af65697cc | ||
|
|
73fbbd3922 | ||
|
|
c37402891a | ||
|
|
f1bdce25e0 | ||
|
|
7ea54a84e5 | ||
|
|
d63a008148 | ||
|
|
b59f981c58 | ||
|
|
60b41c0e82 | ||
|
|
451abba000 |
@@ -125,6 +125,7 @@ const production_ship = [_]ShipRow{
|
||||
service("display"),
|
||||
service("display-demo"),
|
||||
service("device-manager"),
|
||||
service("volume-manager"),
|
||||
service("input"),
|
||||
service("logger"),
|
||||
driver("pci-bus"),
|
||||
@@ -343,6 +344,7 @@ pub fn build(b: *std.Build) void {
|
||||
"protocol-denied-test", // restriction stage one: an ungranted open answers as absence
|
||||
"protocol-conformance-test", // the reserved verbs, asked of every provider the boot bound
|
||||
"device-authority-test", // the attacker: a process handed no device, asserting what it cannot do
|
||||
"block-range-test", // confines itself to a block sub-range, then proves it cannot cross or widen it
|
||||
}) |fixture| {
|
||||
const package = b.lazyDependency(fixture, .{}) orelse
|
||||
@panic("a test fixture package is missing under test/system/services");
|
||||
|
||||
@@ -49,6 +49,7 @@
|
||||
.display = .{ .path = "system/services/display" },
|
||||
.@"display-demo" = .{ .path = "system/services/display-demo" },
|
||||
.@"device-manager" = .{ .path = "system/services/device-manager" },
|
||||
.@"volume-manager" = .{ .path = "system/services/volume-manager" },
|
||||
.input = .{ .path = "system/services/input" },
|
||||
.logger = .{ .path = "system/services/logger" },
|
||||
// The discovery pair and the /test fixtures are lazy: only what a
|
||||
@@ -80,6 +81,7 @@
|
||||
.@"protocol-denied-test" = .{ .path = "test/system/services/protocol-denied-test", .lazy = true },
|
||||
.@"protocol-conformance-test" = .{ .path = "test/system/services/protocol-conformance-test", .lazy = true },
|
||||
.@"device-authority-test" = .{ .path = "test/system/services/device-authority-test", .lazy = true },
|
||||
.@"block-range-test" = .{ .path = "test/system/services/block-range-test", .lazy = true },
|
||||
// See `zig fetch --save <url>` for a command-line interface for adding dependencies.
|
||||
//.example = .{
|
||||
// // When updating this field to a new URL, be sure to delete the corresponding
|
||||
|
||||
@@ -1,13 +1,20 @@
|
||||
# The storage architecture: layers, boundaries, responsibilities
|
||||
|
||||
> **Status:** the layered model below is the settled design
|
||||
> ([storage-design-rationale.md](storage-design-rationale.md) records how it
|
||||
> was reached and what the surveyed systems taught). The data path — vfs
|
||||
> protocol, kernel mount routing, the FAT service, the block protocol,
|
||||
> usb-storage — is **built**. The volume manager, the driver's range
|
||||
> mechanism, per-volume filesystem spawning, and the removal lifecycle are
|
||||
> **planned**; until they land, the FAT service performs volume-manager duties
|
||||
> itself (marked below). This document is the reference for both states.
|
||||
> ([storage-design-rationale.md](storage-design-rationale.md) records how it was
|
||||
> reached, and [volume-manager-plan.md](../volume-manager-plan.md) how it was
|
||||
> built). **Built** (the volume-manager track, V0–V4): the data path, the driver
|
||||
> range confinement (per-sender clamp + the confinement gate), the `medium_changed`
|
||||
> presence event, the volume manager itself — it probes the partition table,
|
||||
> confines each filesystem to its partition, spawns one filesystem per volume, and
|
||||
> supervises it — and the removal half of the lifecycle (a pulled stick unmounts).
|
||||
> **Still pending**: the fuller identity ladder and the `volumes.csv` mount map,
|
||||
> multi-volume (one FAT volume today), the volume manager *consuming*
|
||||
> `medium_changed` (removal is detected by device-presence polling; the event is
|
||||
> published but only a card-reader medium change needs the subscription), and the
|
||||
> remount-on-replug end-to-end (the logic is in place; QEMU can't re-present the
|
||||
> boot-controller device, so it is bench-verified). A few markers below are left
|
||||
> where a duty is still pending.
|
||||
|
||||
## The model
|
||||
|
||||
@@ -47,7 +54,7 @@ Three kinds of boundary, deliberately different:
|
||||
- **Control-plane relationships** sit beside the data path, never on it. Two
|
||||
supervisors, one per layer: the **device manager** wires and revives the
|
||||
device layers (bus and storage drivers — devices only); the **volume
|
||||
manager** *(planned)* wires and revives the volume layer (filesystem
|
||||
manager** *(built)* wires and revives the volume layer (filesystem
|
||||
services). Neither touches steady-state I/O.
|
||||
|
||||
## Who does what
|
||||
@@ -71,19 +78,21 @@ whole disk plus a polite base offset would let a buggy or compromised
|
||||
filesystem scribble the neighboring partition — the same authority-overshoot
|
||||
the device-authority track eliminated for MMIO and DMA.
|
||||
|
||||
**Volume manager** *(planned; today the FAT service squats on these duties)*:
|
||||
the policy home of the volume layer, one service, supervised by init. It
|
||||
subscribes to the device manager's child events; when a storage provider
|
||||
appears it consumer-hellos for the block channel, reads the partition table
|
||||
and the first blocks itself (**it** is the prober), consults its
|
||||
configuration, defines sub-ranges on the driver, spawns the matching
|
||||
filesystem service per volume with that volume's channel, supervises it, and
|
||||
decides mount placement. Its tables are CSV configuration, read by it (the
|
||||
policy), enforced by nobody else:
|
||||
**Volume manager** *(built; `system/services/volume-manager`)*: the policy home
|
||||
of the volume layer, one service, supervised by init. It watches the device
|
||||
manager's tree for a storage provider; when one appears it consumer-hellos for
|
||||
the block channel, reads the partition table and the first blocks itself
|
||||
(**it** is the prober), defines the volume's sub-range on the driver, spawns the
|
||||
matching filesystem service confined to that range, and supervises it (backoff,
|
||||
crash-loop cap). *(Pending)*: it decides mount placement from `volumes.csv` and
|
||||
picks the filesystem binary from `filesystems.csv` — today it hands every
|
||||
FAT-shaped volume to the FAT service and the FAT service carries hardcoded mount
|
||||
prefixes. Those tables are CSV configuration, read by it (the policy), enforced
|
||||
by nobody else:
|
||||
|
||||
- `filesystems.csv` — content signature → filesystem binary. Adding a
|
||||
filesystem adds a row.
|
||||
- `volumes.csv` — the mount map, danos's fstab: **volume identity → mount
|
||||
- `filesystems.csv` *(pending)* — content signature → filesystem binary. Adding
|
||||
a filesystem adds a row.
|
||||
- `volumes.csv` *(pending)* — the mount map, danos's fstab: **volume identity → mount
|
||||
prefix**, keyed on content identity and never on port, path, or arrival
|
||||
order (the lesson of Linux's `/dev/sda1`-era fstab, which broke on every
|
||||
port move until `UUID=` replaced it). Identity is read off the medium by
|
||||
@@ -166,9 +175,9 @@ be served with the previous card's filesystem state.
|
||||
| Bus driver | port/hub status change | tear down the device's slots (children first, recursively — built, hot-plug matrix), report `child_removed` per interface | the device tree is honest within one reconcile tick |
|
||||
| Device manager | `child_removed` / reporter death | prune the child; **reap the bound driver** (built) — the storage driver for that stick dies now, not never | no zombie storage processes; re-report rebinds |
|
||||
| Storage driver | its own death (it IS the removed device's driver) | nothing — dying is its removal handling; DMA/IOMMU/claims release mechanically at death | in-flight transfers fail visibly to callers, never hang |
|
||||
| Volume manager *(planned)* | the storage provider's channel death / the manager's child events | kill each filesystem service of that device's volumes; retire their kernel mounts; remember the volume identity | one removal path; mounts never dangle; log persistence stops *cleanly* |
|
||||
| Volume manager *(removal built; remount bench-pending)* | the storage device leaving the device-manager tree (poll) | kill the filesystem service of that device's volume; its kernel mounts retire | one removal path; mounts never dangle; log persistence stops *cleanly* |
|
||||
| Filesystem service | its block channel dies (`EPEER`) mid-operation, or it is killed by the volume manager | if it observes the death first: flush nothing (the medium is gone), answer in-flight requests with errors, exit; dirty write-back data is **lost and said to be lost** | the FAT dirty flag on disk marks the unclean removal; the process never serves from behind a dead channel |
|
||||
| Kernel | backend endpoint death | lazy mount-slot sweep on next resolve (built); *(planned)* ownership-gated `fs_unmount` | resolution under a dead mount is `not_found`, not a hang |
|
||||
| Kernel | backend endpoint death | lazy mount-slot sweep on next resolve (built); ownership-gated `fs_unmount` (built, V0) | resolution under a dead mount is `not_found`, not a hang |
|
||||
| Application | `not_found` / error on paths under the vanished mount | its own error handling — the contract is honest absence, identical to the path never existing | no operation blocks forever on removed media |
|
||||
|
||||
**On return** the same table runs upward in reverse: the bus re-enumerates and
|
||||
|
||||
@@ -172,11 +172,21 @@ matrix-proven shape; genuinely open.
|
||||
3. **One filesystem process per volume** (fat's binary becomes "the FAT
|
||||
implementation", spawned per FAT volume). Recommendation: yes — it extends
|
||||
recompile-and-restart-live to filesystems and isolates corrupt media.
|
||||
4. **Sub-range addressing**: `target` ids on the storage endpoint versus one
|
||||
endpoint per volume handed out by the driver. Endpoint-per-volume matches
|
||||
the establishment-plane machinery (a channel per party, caps at
|
||||
establishment) and keeps per-client badge scoping simple. Recommendation:
|
||||
endpoint per volume.
|
||||
4. **Sub-range addressing — SETTLED as per-sender confinement at the
|
||||
provider, one serving endpoint.** The deciding argument is precedent: the
|
||||
xHCI bus already serves every class driver on one endpoint with authority
|
||||
scoped by the kernel-stamped badge (the per-client device-token table) —
|
||||
that IS danos's provider pattern, and per-badge range confinement is the
|
||||
same pattern applied to blocks. The volume manager sets each filesystem
|
||||
process's range on the driver; the driver clamps AND translates every
|
||||
transfer by the sender's range, so filesystems address volume-relative
|
||||
LBAs from 0 and the FAT engine's `base_lba` is deleted rather than moved.
|
||||
The enforcement point (the clamp at the provider, never in the consumer)
|
||||
is what carries the security property; endpoint-per-volume would deliver
|
||||
the same property only by inventing a multi-endpoint harness the pattern
|
||||
does not need. It stays available as a future refactor if a multi-endpoint
|
||||
harness ever exists for other reasons; the wire contract is identical
|
||||
either way.
|
||||
5. **`fs_unmount` ownership** — a defect fix more than a decision.
|
||||
6. **Later, kept open**: the shm-ring data plane (communication.md already
|
||||
names it as the 256-byte ceiling's unlock — Fuchsia's FIFO+VMO is the
|
||||
|
||||
@@ -0,0 +1,124 @@
|
||||
# The volume manager: the plan
|
||||
|
||||
*2026-08-09. Executes the settled design in
|
||||
[storage-architecture.md](file-system-development/storage-architecture.md) and
|
||||
[storage-design-rationale.md](file-system-development/storage-design-rationale.md)
|
||||
(decisions 1–8). Track discipline as always: one commit per coherent step, suite
|
||||
green at phase boundaries, every new test shown to fail against the old
|
||||
behavior, one QEMU suite at a time, work on main.*
|
||||
|
||||
**No open decisions.** Decision 4 is settled in the rationale as per-sender
|
||||
range confinement at the provider on one serving endpoint — the badge-scoped
|
||||
provider pattern the xHCI bus already uses, applied to blocks. The volume
|
||||
manager sets each filesystem process's range; the driver clamps and translates
|
||||
every transfer by the sender's kernel-stamped badge; filesystems address
|
||||
volume-relative LBAs from 0 and the FAT engine's `base_lba` is deleted rather
|
||||
than moved. Every other decision the phases below execute is recorded in the
|
||||
rationale (decisions 1–8); nothing in this plan waits on a choice.
|
||||
|
||||
## V0 — `fs_unmount` ownership (the defect fix)
|
||||
|
||||
The kernel records the mounting process on each mount slot; `fs_unmount` is
|
||||
refused for any caller but the owner (the `/protocol` special case stays).
|
||||
Death cleanup is unaffected (the lazy sweep is not an unmount). Discrimination:
|
||||
a fixture unmounting a prefix it does not own must be refused — fails against
|
||||
today's kernel, which lets any process unmount anything.
|
||||
|
||||
## V1 — the filesystem harness (extraction, no behavior change)
|
||||
|
||||
fat's 434-line shell becomes `library/file-system/harness` (name per
|
||||
convention): establishment, the badge-scoped open-node table, the nine
|
||||
vfs-protocol handlers, mount registration, bring-up/teardown. fat becomes
|
||||
engine + on-disk + a thin main wiring the harness. Suite green is the gate;
|
||||
nothing observable changes. This lands FIRST so every later phase touches the
|
||||
harness once, not fat and the harness both.
|
||||
|
||||
## V2 — the driver mechanism: ranges, confinement, presence
|
||||
|
||||
- Block protocol additions (appended; numbering holds): `define_range`
|
||||
(volume-manager-only in practice — see grants), and the pushed
|
||||
`medium_changed` event (present/absent + change counter).
|
||||
- usb-storage: per-badge range table (clamp + translate per sender), range
|
||||
definitions from the volume manager, and presence: a slow idle-time
|
||||
TEST UNIT READY poll plus sense-key inspection on failed transfers, emitting
|
||||
`medium_changed` on transitions. QEMU test lever: `eject` /
|
||||
`blockdev-remove-medium` against a `removable=on` usb-storage device; if
|
||||
QEMU's model refuses, the fallback drill is device_del/add of the whole
|
||||
stick (the H-matrix already proves that path) and presence gets its real
|
||||
test on the bench with a card reader.
|
||||
- Discrimination: a fixture transferring outside its assigned range must be
|
||||
refused; fails against a driver without the clamp.
|
||||
|
||||
**Sequencing (as executed).** V2a landed the range clamp + its discrimination
|
||||
fixture (block-range) — the security mechanism is testable in isolation. V2b
|
||||
adds `medium_changed` and makes usb-storage a publisher that emits it on
|
||||
presence transitions, but its END-TO-END test (eject → medium_changed →
|
||||
unmount/remount) lands in V4 with the real consumer, the volume manager —
|
||||
rather than a throwaway subscriber fixture V3 would immediately replace. Same
|
||||
work, no duplicated scaffolding.
|
||||
|
||||
## V3 — the volume manager service
|
||||
|
||||
New binary `system/services/volume-manager`, spawned by init, serving the
|
||||
(genuinely singular) name `volume-manager`. Duties, all moved OUT of fat:
|
||||
|
||||
- subscribe to the device manager; consumer-hello each storage provider for
|
||||
its block channel;
|
||||
- probe: partition table walk (MBR now, GPT next — the walk LEAVES the FAT
|
||||
engine) and content identity (the decision-8 ladder: GPT GUID → fs UUID →
|
||||
FAT serial+label → MBR signature+index → anonymous);
|
||||
- configuration: `filesystems.csv` (signature → filesystem binary) and
|
||||
`volumes.csv` (identity → mount prefix; the fstab). Boot volume identity
|
||||
recorded at first sight of `/system/configuration`;
|
||||
- define ranges on the driver; spawn one filesystem process per volume
|
||||
(argv: volume id); answer each filesystem's startup hello with its volume
|
||||
channel (the reply-capability path, same as the device manager's);
|
||||
- supervise: hello/mount deadline, crash-loop cap, reap on removal.
|
||||
|
||||
fat sheds `acquireVolume` and its device-manager grant; filesystem binaries
|
||||
get `open volume-manager` only — a filesystem cannot acquire, only be given.
|
||||
Grants move with the code in the same commits.
|
||||
|
||||
**Sequencing (as executed).** V3a (discovery+probe) and V3b (the flip: spawn +
|
||||
confine + hand over the channel, single volume) landed the core. The V3c items —
|
||||
the `volumes.csv` mount map, the fuller identity ladder (FAT serial, GPT GUID),
|
||||
and multi-volume spawning — mainly serve the MULTI-volume drills (two-partitions,
|
||||
clone-policy). The user's goal is the single-volume boot-stick removal lifecycle,
|
||||
so V4's removal lifecycle runs next on the single volume, and the multi-volume
|
||||
work + its drills become a documented follow-on (V3c/multi-volume). fat keeps its
|
||||
hardcoded mount prefixes until the mount map lands.
|
||||
|
||||
## V4 — the removal lifecycle, end to end
|
||||
|
||||
Two triggers, one path: storage-channel death and `medium_changed(absent)`
|
||||
both drive kill-the-filesystem-process + retire-its-mounts; return (device
|
||||
re-report or `medium_changed(present)`) drives re-probe → respawn → remount
|
||||
at the identity's prefix. The logger gains resume patience (retry flushes on
|
||||
the fat cadence, never abandon) and the ring-wrap gap marker. QEMU cases:
|
||||
|
||||
- `volume-replug`: yank the boot stick mid-run, replug, assert remount at the
|
||||
same prefixes and the logger appending to the SAME boot-stamp tree with the
|
||||
gap marked. Discrimination: against pre-V4, fat wedges (`mounted` forever)
|
||||
and no remount happens.
|
||||
- `volume-two-partitions`: a two-partition FAT image → two volumes, two
|
||||
filesystem processes, two mounts from one stick; yank once, both die; return
|
||||
once, both remount. Proves multi-volume and the reap breadth. (New image
|
||||
fixture beside make-fat-image.py.)
|
||||
- `volume-clone-policy`: two sticks with identical FAT serials — first keeps
|
||||
the mapped name, second mounts suffixed, loudly logged.
|
||||
- The **lifecycle conformance drill**, parameterized by filesystem: mount,
|
||||
serve, yank mid-write, verify honest loss (dirty flag set, gap said),
|
||||
replug, remount. FAT is implementation #1; the drill is the definition of
|
||||
"danos supports filesystem X".
|
||||
|
||||
## V5 — close-out
|
||||
|
||||
Full suite green; the architecture doc's *(planned)* markers flip to built;
|
||||
`fs_unmount` ownership, ranges, presence, identity, and the volume manager
|
||||
lose their future-tense; memory updated; Ryzen bench note: pull the stick,
|
||||
watch the log gap get marked, plug it back anywhere.
|
||||
|
||||
**Not in this track** (recorded so absence is deliberate): GPT parsing beyond
|
||||
the identity read (row exists in the prober's ladder; full GPT when a GPT
|
||||
medium matters), AHCI/NVMe drivers (NVMe gated on the shm-ring data plane),
|
||||
formatting/entropy, per-process namespaces.
|
||||
@@ -63,6 +63,15 @@ pub const Device = struct {
|
||||
return self.call(.flush, {}, null, &reply) != null;
|
||||
}
|
||||
|
||||
/// Confine the process `badge` to blocks `[base_lba, base_lba + block_count)`
|
||||
/// on this device — the volume manager's per-volume grant to a filesystem.
|
||||
/// The caller must itself be unconfined (whole-device); a confined caller is
|
||||
/// refused, so a filesystem cannot widen its own range. See `DefineRange`.
|
||||
pub fn defineRange(self: Device, badge: u32, base_lba: u64, block_count: u64) bool {
|
||||
var reply: [block_protocol.message_maximum]u8 = undefined;
|
||||
return self.call(.define_range, .{ .badge = badge, .base_lba = base_lba, .block_count = block_count }, null, &reply) != null;
|
||||
}
|
||||
|
||||
/// One request at the driver. `target` is always 0: one endpoint per device, so
|
||||
/// there is no object within the peer to address.
|
||||
fn call(
|
||||
|
||||
@@ -90,7 +90,7 @@ pub fn build(b: *std.Build) void {
|
||||
// a provider that beat init to the mount needs). It also owns the subscriber
|
||||
// table and the fan-out, which are expressed in the envelope's vocabulary
|
||||
// (the reserved subscribe verb, the push floor) — hence envelope.
|
||||
_ = b.addModule("service", .{
|
||||
const service = b.addModule("service", .{
|
||||
.root_source_file = b.path("service.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "channel", .module = channel },
|
||||
@@ -99,6 +99,23 @@ pub fn build(b: *std.Build) void {
|
||||
.{ .name = "process", .module = process },
|
||||
},
|
||||
});
|
||||
// The filesystem serving harness (docs/file-system-development/storage-architecture.md):
|
||||
// the engine-agnostic half of a filesystem service. It specializes `service`
|
||||
// for the vfs protocol and is block-free (durability rides a caller closure),
|
||||
// so it needs nothing from the device domain — it sits beside its sibling.
|
||||
_ = b.addModule("file-system-harness", .{
|
||||
.root_source_file = b.path("file-system-harness.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "ipc", .module = ipc },
|
||||
.{ .name = "process", .module = process },
|
||||
.{ .name = "service", .module = service },
|
||||
.{ .name = "time", .module = time },
|
||||
.{ .name = "file-system", .module = file_system },
|
||||
.{ .name = "envelope", .module = protocol.module("envelope") },
|
||||
.{ .name = "vfs-protocol", .module = protocol.module("vfs-protocol") },
|
||||
.{ .name = "logging", .module = logging },
|
||||
},
|
||||
});
|
||||
_ = b.addModule("start", .{
|
||||
.root_source_file = b.path("start.zig"),
|
||||
.imports = &.{ .{ .name = "process", .module = process }, .{ .name = "logging", .module = logging } },
|
||||
|
||||
@@ -0,0 +1,343 @@
|
||||
//! The filesystem serving harness: the block-client-and-engine-agnostic half of
|
||||
//! a filesystem service (docs/file-system-development/storage-architecture.md).
|
||||
//! Everything a filesystem process does that is NOT its on-disk format lives
|
||||
//! here — establishment, the badge-scoped open-node table, the nine vfs-protocol
|
||||
//! handlers, mount registration, the not-mounted-yet politeness, the exit sweep,
|
||||
//! and the durable-on-close flush. A filesystem is then an ENGINE (the pure,
|
||||
//! host-testable format code behind a small method set) plus a `main` that wires
|
||||
//! it in, so a second filesystem reuses this wholesale — the reason it is a
|
||||
//! shared library rather than per-filesystem code.
|
||||
//!
|
||||
//! Placement: `library/kernel`, beside its sibling `service` (the generic
|
||||
//! serving harness this specializes for the vfs protocol). It is block-free —
|
||||
//! the caller's `Volume.flush` closure owns durability — so it needs nothing
|
||||
//! from the device domain and introduces no backwards dependency.
|
||||
//!
|
||||
//! `Server(Engine)` is generic over the engine TYPE, checked at compile time by
|
||||
//! the calls below. An engine must expose:
|
||||
//! - `pub const Node` with fields `is_directory: bool`, `size`, `mtime`;
|
||||
//! - `pub const Listing` with `name_buffer`, `name_len`, `is_directory`, `size`;
|
||||
//! - `current_time_epoch` a settable field (the harness stamps it per turn);
|
||||
//! - resolve, createFile, createDirectory, removeFile, rename, truncate,
|
||||
//! readFile, writeFile, listEntry — the signatures fat's engine.zig already has.
|
||||
|
||||
const std = @import("std");
|
||||
const ipc = @import("ipc");
|
||||
const process = @import("process");
|
||||
const service = @import("service");
|
||||
const time = @import("time");
|
||||
const file_system = @import("file-system");
|
||||
const envelope = @import("envelope");
|
||||
const vfs_protocol = @import("vfs-protocol");
|
||||
const logging = @import("logging");
|
||||
|
||||
/// One prefix this filesystem mounts into the kernel mount table. `rewrite` is
|
||||
/// the backend-relative prefix a path is rewritten to before it reaches the
|
||||
/// engine (empty = mount the volume root at `prefix`, the common case).
|
||||
pub const MountSpec = struct { prefix: []const u8, rewrite: []const u8 = "" };
|
||||
|
||||
/// The engine's node type, inferred from `resolve`'s return (`?Node`) so the
|
||||
/// engine need not re-export it as a member — engine.zig keeps `Node` at module
|
||||
/// scope, and this harness stays purely additive on the engine side.
|
||||
fn NodeType(comptime Engine: type) type {
|
||||
return @typeInfo(@typeInfo(@TypeOf(Engine.resolve)).@"fn".return_type.?).optional.child;
|
||||
}
|
||||
|
||||
pub fn Server(comptime Engine: type) type {
|
||||
return struct {
|
||||
const Node = NodeType(Engine);
|
||||
|
||||
/// What a bring-up produces: the mounted engine (a stable pointer the
|
||||
/// caller owns), the prefixes to install, and a durability closure the
|
||||
/// harness calls on every close (the caller checks its own dirty state).
|
||||
pub const Volume = struct {
|
||||
engine: *Engine,
|
||||
mounts: []const MountSpec,
|
||||
flush: *const fn () void,
|
||||
};
|
||||
|
||||
pub const Callbacks = struct {
|
||||
/// Acquire and mount the volume, or null to retry on the timer. The
|
||||
/// caller does the filesystem-specific bring-up (find the block
|
||||
/// device, set up DMA, mount the engine) and returns a `Volume`.
|
||||
bringUp: *const fn (endpoint: ipc.Handle) ?Volume,
|
||||
/// The vfs contract name to bind. A filesystem serving one volume
|
||||
/// binds "vfs" today; the volume-manager era hands each per-volume
|
||||
/// process its own establishment and this fades.
|
||||
service_name: ?[]const u8 = "vfs",
|
||||
};
|
||||
|
||||
// --- the harness's own state, one set per instantiation ---------------
|
||||
// A filesystem binary instantiates Server once, so these globals are the
|
||||
// one server's state, exactly where fat's file-scoped globals were.
|
||||
|
||||
const Serve = vfs_protocol.Protocol.Provider(void);
|
||||
const Invocation = envelope.Invocation;
|
||||
const Answer = envelope.Answer;
|
||||
|
||||
/// What a handler returns when the thing asked for is not there — a bad
|
||||
/// node id, someone else's node, an unresolved path, a refused mutation.
|
||||
/// One errno for all: a filesystem's failures are all "no such thing" to
|
||||
/// the file API, and *someone else's* must be indistinguishable from
|
||||
/// *nobody's*, or the refusal would leak which ids are live.
|
||||
const refused: isize = -envelope.ENOENT;
|
||||
|
||||
/// How often to retry bring-up while unmounted. Storage arriving is
|
||||
/// event-shaped (the usb chain registering, maybe after a restart), but
|
||||
/// there is no subscription; a slow poll keeps the service responsive
|
||||
/// (ping, terminate) while it waits and alive to catch late storage.
|
||||
const mount_retry_ms = 500;
|
||||
|
||||
const OpenNode = struct { used: bool = false, node: Node = undefined, owner: u32 = 0 };
|
||||
var open_nodes = [_]OpenNode{.{}} ** 32;
|
||||
|
||||
var callbacks: Callbacks = undefined;
|
||||
var service_endpoint: ipc.Handle = 0;
|
||||
var engine_ptr: ?*Engine = null;
|
||||
var volume_flush: *const fn () void = undefined;
|
||||
var mounted: bool = false;
|
||||
|
||||
fn allocOpen() ?usize {
|
||||
for (&open_nodes, 0..) |*o, i| {
|
||||
if (!o.used) return i;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// The open node `id` names **for `owner`** — null unless in range, in
|
||||
/// use, and this client's own. Ids are small integers from a table of 32,
|
||||
/// trivially guessable, so this badge check is the scope
|
||||
/// (docs/os-development/protocol-namespace.md). The owner is a TASK, not a
|
||||
/// process, because the badge is: a threaded client reads a node from the
|
||||
/// thread that opened it, and the exit sweep releases a worker's handles.
|
||||
fn openFor(id: u64, owner: u32) ?*OpenNode {
|
||||
if (id >= open_nodes.len) return null;
|
||||
const o = &open_nodes[@intCast(id)];
|
||||
if (!o.used or o.owner != owner) return null;
|
||||
return o;
|
||||
}
|
||||
|
||||
const ParentLeaf = struct { parent: []const u8, leaf: []const u8 };
|
||||
|
||||
// Split a path: "/a/b" -> ("/a", "b"); "/b" -> ("/", "b"); "b" -> ("/", "b").
|
||||
fn splitParent(path: []const u8) ParentLeaf {
|
||||
const slash = std.mem.lastIndexOfScalar(u8, path, '/');
|
||||
return .{
|
||||
.parent = if (slash) |s| (if (s == 0) "/" else path[0..s]) else "/",
|
||||
.leaf = if (slash) |s| path[s + 1 ..] else path,
|
||||
};
|
||||
}
|
||||
|
||||
fn onOpen(_: void, invocation: Invocation(vfs_protocol.Open), answer: Answer(vfs_protocol.Opened)) isize {
|
||||
const fs = engine_ptr orelse return refused;
|
||||
const path = invocation.tail;
|
||||
const flags = invocation.request.flags;
|
||||
var node = fs.resolve(path);
|
||||
if (node == null and flags & vfs_protocol.create != 0) {
|
||||
const split = splitParent(path);
|
||||
const parent = fs.resolve(split.parent) orelse return refused;
|
||||
node = fs.createFile(parent, split.leaf);
|
||||
}
|
||||
var resolved = node orelse return refused;
|
||||
// O_TRUNC: replace contents rather than overwrite in place (frees the
|
||||
// old chain, so a shorter rewrite leaves no stale tail).
|
||||
if (flags & vfs_protocol.truncate != 0 and !resolved.is_directory) {
|
||||
fs.truncate(&resolved);
|
||||
}
|
||||
const index = allocOpen() orelse return refused;
|
||||
open_nodes[index] = .{ .used = true, .node = resolved, .owner = invocation.sender };
|
||||
answer.set(.{ .node = index });
|
||||
return 0;
|
||||
}
|
||||
|
||||
fn onRead(_: void, invocation: Invocation(vfs_protocol.Read), answer: Answer(void)) isize {
|
||||
const fs = engine_ptr orelse return refused;
|
||||
const o = openFor(invocation.target, invocation.sender) orelse return refused;
|
||||
const into = answer.tail();
|
||||
const want = @min(@as(usize, invocation.request.len), into.len);
|
||||
return @intCast(fs.readFile(o.node, @intCast(invocation.request.offset), into[0..want]));
|
||||
}
|
||||
|
||||
fn onWrite(_: void, invocation: Invocation(vfs_protocol.Write), answer: Answer(vfs_protocol.Written)) isize {
|
||||
const fs = engine_ptr orelse return refused;
|
||||
const o = openFor(invocation.target, invocation.sender) orelse return refused;
|
||||
const data = invocation.tail[0..@min(invocation.tail.len, invocation.request.len)];
|
||||
const n = fs.writeFile(&o.node, @intCast(invocation.request.offset), data);
|
||||
answer.set(.{ .count = @intCast(n) });
|
||||
return 0;
|
||||
}
|
||||
|
||||
fn onStatus(_: void, invocation: Invocation(void), answer: Answer(vfs_protocol.FileStatus)) isize {
|
||||
const o = openFor(invocation.target, invocation.sender) orelse return refused;
|
||||
const kind: vfs_protocol.NodeKind = if (o.node.is_directory) .directory else .regular;
|
||||
answer.set(.{ .size = o.node.size, .kind = @intFromEnum(kind), .mtime = o.node.mtime });
|
||||
return 0;
|
||||
}
|
||||
|
||||
/// One entry per call. End of directory — not a directory, or a cursor
|
||||
/// past the last child — is an entry with no name.
|
||||
fn onReaddir(_: void, invocation: Invocation(vfs_protocol.Readdir), answer: Answer(vfs_protocol.DirectoryEntry)) isize {
|
||||
const fs = engine_ptr orelse return refused;
|
||||
const o = openFor(invocation.target, invocation.sender) orelse return refused;
|
||||
if (!o.node.is_directory) {
|
||||
answer.set(.{});
|
||||
return 0;
|
||||
}
|
||||
const listing = fs.listEntry(o.node, @intCast(invocation.request.cursor)) orelse {
|
||||
answer.set(.{});
|
||||
return 0;
|
||||
};
|
||||
const kind: vfs_protocol.NodeKind = if (listing.is_directory) .directory else .regular;
|
||||
const into = answer.tail();
|
||||
const name_len = @min(listing.name_len, into.len);
|
||||
@memcpy(into[0..name_len], listing.name_buffer[0..name_len]);
|
||||
answer.set(.{ .kind = @intFromEnum(kind), .name_len = @intCast(name_len), .size = listing.size });
|
||||
return @intCast(name_len);
|
||||
}
|
||||
|
||||
/// Closing is scoped like any other node operation: a client releases its
|
||||
/// own handles and nobody else's, and a foreign/free/out-of-range id is
|
||||
/// refused identically so a close cannot probe which ids are live.
|
||||
fn onClose(_: void, invocation: Invocation(void), _: Answer(void)) isize {
|
||||
const o = openFor(invocation.target, invocation.sender) orelse return refused;
|
||||
o.used = false;
|
||||
// Durable-on-close: the caller's flush commits any device write cache
|
||||
// to stable media now. This is what makes init's shutdown log flush
|
||||
// survive a real power-off, and the right default for removable media.
|
||||
volume_flush();
|
||||
return 0;
|
||||
}
|
||||
|
||||
fn onMakeDirectory(_: void, invocation: Invocation(void), _: Answer(void)) isize {
|
||||
const fs = engine_ptr orelse return refused;
|
||||
const path = invocation.tail;
|
||||
if (fs.resolve(path) != null) return refused; // already exists
|
||||
const split = splitParent(path);
|
||||
const parent = fs.resolve(split.parent) orelse return refused;
|
||||
if (fs.createDirectory(parent, split.leaf) == null) return refused;
|
||||
return 0;
|
||||
}
|
||||
|
||||
fn onUnlink(_: void, invocation: Invocation(void), _: Answer(void)) isize {
|
||||
const fs = engine_ptr orelse return refused;
|
||||
const split = splitParent(invocation.tail);
|
||||
const parent = fs.resolve(split.parent) orelse return refused;
|
||||
if (!fs.removeFile(parent, split.leaf)) return refused;
|
||||
return 0;
|
||||
}
|
||||
|
||||
fn onRename(_: void, invocation: Invocation(void), _: Answer(void)) isize {
|
||||
const fs = engine_ptr orelse return refused;
|
||||
const both = invocation.tail;
|
||||
const separator = std.mem.indexOfScalar(u8, both, 0) orelse return refused;
|
||||
const old_split = splitParent(both[0..separator]);
|
||||
const new_split = splitParent(both[separator + 1 ..]);
|
||||
if (!std.mem.eql(u8, old_split.parent, new_split.parent)) return refused; // same-directory only
|
||||
const parent = fs.resolve(old_split.parent) orelse return refused;
|
||||
if (!fs.rename(parent, old_split.leaf, new_split.leaf)) return refused;
|
||||
return 0;
|
||||
}
|
||||
|
||||
/// The verbs this backend implements. `mount`/`unmount`/`bind` are absent
|
||||
/// on purpose — path routing is the kernel's, and only init serves `bind`.
|
||||
const handlers = Serve.Handlers{
|
||||
.open = onOpen,
|
||||
.close = onClose,
|
||||
.read = onRead,
|
||||
.write = onWrite,
|
||||
.status = onStatus,
|
||||
.readdir = onReaddir,
|
||||
.mkdir = onMakeDirectory,
|
||||
.unlink = onUnlink,
|
||||
.rename = onRename,
|
||||
};
|
||||
|
||||
/// The vfs protocol carries no capability, so `arrived` is never claimed
|
||||
/// — the harness's ownership rule then closes whatever a caller attached,
|
||||
/// so a request carrying one cannot spend a slot of this server's table.
|
||||
fn onMessage(message: []const u8, out: []u8, sender: u32, arrived: *ipc.Arrival) usize {
|
||||
_ = arrived;
|
||||
const fs = engine_ptr;
|
||||
// Storage not up yet: fail politely, whatever was asked — clients retry.
|
||||
if (!mounted or fs == null) {
|
||||
const status = envelope.Status{ .status = refused, .len = 0 };
|
||||
@memcpy(out[0..envelope.prefix_size], std.mem.asBytes(&status));
|
||||
return envelope.prefix_size;
|
||||
}
|
||||
// Stamp create/write with the current wall-clock time (mtime): cheap,
|
||||
// and it keeps the engine pure (it takes the time as data, not a call).
|
||||
fs.?.current_time_epoch = time.wallClock();
|
||||
return Serve.dispatch({}, handlers, message, sender, null, out);
|
||||
}
|
||||
|
||||
/// A process-exit event releases every open handle the dead client held,
|
||||
/// so a crashed reader cannot pin table slots.
|
||||
fn onNotification(badge: u64) void {
|
||||
const got = ipc.Received{ .len = 0, .badge = badge, .cap = null };
|
||||
if (got.isTimer()) {
|
||||
tryBringUp();
|
||||
if (!mounted) _ = time.timerOnce(service_endpoint, mount_retry_ms);
|
||||
return;
|
||||
}
|
||||
if (!got.isChildExit()) return;
|
||||
const dead = got.childProcessId();
|
||||
var released: u32 = 0;
|
||||
for (&open_nodes) |*o| {
|
||||
if (o.used and o.owner == dead) {
|
||||
o.* = .{};
|
||||
released += 1;
|
||||
}
|
||||
}
|
||||
if (released != 0) std.log.info("released {d} handle(s) for dead client {d}", .{ released, dead });
|
||||
}
|
||||
|
||||
/// One bring-up attempt: ask the caller for a mounted volume, and on
|
||||
/// success install its mounts and go live. A failure leaves everything
|
||||
/// untouched for the next tick.
|
||||
fn tryBringUp() void {
|
||||
if (mounted) return;
|
||||
const volume = callbacks.bringUp(service_endpoint) orelse return;
|
||||
engine_ptr = volume.engine;
|
||||
volume_flush = volume.flush;
|
||||
for (volume.mounts) |m| {
|
||||
const ok = if (m.rewrite.len == 0)
|
||||
file_system.mount(m.prefix, service_endpoint)
|
||||
else
|
||||
file_system.mountRewritten(m.prefix, service_endpoint, m.rewrite);
|
||||
if (ok) {
|
||||
std.log.info("mounted {s}", .{m.prefix});
|
||||
} else {
|
||||
// The failure diagnostic goes to the kernel ring directly, not
|
||||
// through std.log — the mount that failed may be /system/logs
|
||||
// itself, and a routed record would then have nowhere to land.
|
||||
// Three appends rather than a formatted line so there is no
|
||||
// scratch buffer (and so no fixed length to justify).
|
||||
_ = logging.write("file-system: could not mount ");
|
||||
_ = logging.write(m.prefix);
|
||||
_ = logging.write("\n");
|
||||
}
|
||||
}
|
||||
mounted = true;
|
||||
}
|
||||
|
||||
fn initialise(endpoint: ipc.Handle) bool {
|
||||
service_endpoint = endpoint;
|
||||
// Sweep a dead client's open handles via the published exit events —
|
||||
// clients hold OUR node ids directly, so a crash must not pin slots.
|
||||
_ = process.subscribeExits(endpoint);
|
||||
tryBringUp();
|
||||
if (!mounted) _ = time.timerOnce(endpoint, mount_retry_ms);
|
||||
return true; // serve regardless: requests fail politely until storage mounts
|
||||
}
|
||||
|
||||
pub fn run(cb: Callbacks) void {
|
||||
callbacks = cb;
|
||||
service.run(vfs_protocol.message_maximum, .{
|
||||
.service = cb.service_name,
|
||||
.init = initialise,
|
||||
.on_message = onMessage,
|
||||
.on_notification = onNotification,
|
||||
});
|
||||
}
|
||||
};
|
||||
}
|
||||
@@ -37,9 +37,47 @@ pub const Transfer = extern struct {
|
||||
/// How many blocks a transfer actually moved.
|
||||
pub const Transferred = extern struct { count: u32 };
|
||||
|
||||
/// `define_range(badge, base_lba, block_count)`: confine the sender identified by
|
||||
/// `badge` to blocks `[base_lba, base_lba + block_count)`. The volume manager
|
||||
/// calls this for each filesystem process it hands a channel to — the badge is
|
||||
/// the filesystem's kernel-stamped task id, and the range is the partition it
|
||||
/// mounts. A confined sender's read/write LBAs are then volume-relative (the
|
||||
/// driver adds `base_lba`) and a transfer past `block_count` is refused. A
|
||||
/// sender with no range is unconfined (the whole device), the default until the
|
||||
/// volume manager defines one. The clamp lives at the provider because a channel
|
||||
/// must carry exactly the authority it grants (storage-architecture.md): handing
|
||||
/// a filesystem the whole disk plus a base offset would let it reach the
|
||||
/// neighbouring partition. **A confined caller may not call this** — a filesystem
|
||||
/// cannot redefine its own range and escape; only an unconfined party (the
|
||||
/// volume manager) confines others.
|
||||
pub const DefineRange = extern struct {
|
||||
badge: u32,
|
||||
_padding: u32 = 0,
|
||||
base_lba: u64,
|
||||
block_count: u64,
|
||||
};
|
||||
|
||||
/// The `medium_changed` event payload: whether a medium is now present, and a
|
||||
/// monotonic counter so a subscriber that missed an edge still sees that
|
||||
/// SOMETHING changed. Pushed by a driver whose transport can tell medium from
|
||||
/// device (a card reader, an ATAPI tray): the device stays, the medium comes and
|
||||
/// goes. The volume manager consumes it into the same unmount/remount path it
|
||||
/// runs on device death — one lifecycle, two triggers
|
||||
/// (docs/file-system-development/storage-architecture.md). Presence only, never
|
||||
/// content: the driver reports that the medium changed, not what is on it.
|
||||
pub const MediumChanged = extern struct {
|
||||
present: u8, // 1 present, 0 absent
|
||||
_padding: u8 = 0,
|
||||
_padding2: u16 = 0,
|
||||
change_count: u32,
|
||||
};
|
||||
|
||||
pub const Protocol = envelope.Define(.{
|
||||
.name = "block",
|
||||
.version = 1,
|
||||
.events = &.{
|
||||
.{ .name = "medium_changed", .payload = MediumChanged },
|
||||
},
|
||||
.operations = &.{
|
||||
.{ .name = "geometry", .reply = Geometry },
|
||||
.{ .name = "read", .request = Transfer, .reply = Transferred },
|
||||
@@ -60,8 +98,13 @@ pub const Protocol = envelope.Define(.{
|
||||
// makes is revocable by the granter while alive; death remains the
|
||||
// mechanical backstop (storage-architecture.md, the lifecycle rule).
|
||||
.{ .name = "detach" },
|
||||
// define_range(): confine a sender to a block sub-range — the partition
|
||||
// it mounts. See `DefineRange`. Appended, so every verb above keeps its
|
||||
// number.
|
||||
.{ .name = "define_range", .request = DefineRange },
|
||||
},
|
||||
});
|
||||
|
||||
pub const Operation = Protocol.Operation;
|
||||
pub const Event = Protocol.Event;
|
||||
pub const message_maximum: usize = Protocol.message_maximum;
|
||||
|
||||
@@ -31,6 +31,7 @@ pub fn build(b: *std.Build) void {
|
||||
.{ .name = "block-protocol", .root = "block/block-protocol.zig" },
|
||||
.{ .name = "usb-transfer-protocol", .root = "usb-transfer/usb-transfer-protocol.zig" },
|
||||
.{ .name = "device-manager-protocol", .root = "device-manager/device-manager-protocol.zig" },
|
||||
.{ .name = "volume-manager-protocol", .root = "volume-manager/volume-manager-protocol.zig" },
|
||||
.{ .name = "display-protocol", .root = "display/display-protocol.zig" },
|
||||
.{ .name = "scanout-protocol", .root = "scanout/scanout-protocol.zig" },
|
||||
.{ .name = "power-protocol", .root = "power/power-protocol.zig" },
|
||||
|
||||
@@ -0,0 +1,36 @@
|
||||
//! The volume-manager protocol (docs/file-system-development/storage-architecture.md):
|
||||
//! what a filesystem service says to the volume manager over
|
||||
//! `/protocol/volume-manager`. Defined through the envelope, so every packet
|
||||
//! begins with the folded `Header`.
|
||||
//!
|
||||
//! One verb. A filesystem the volume manager spawned announces itself with the
|
||||
//! volume id it was given as argv[1] (folded into `Header.target`); the reply
|
||||
//! carries that volume's block channel — already range-confined to the
|
||||
//! filesystem's badge — as the call's returned capability. The filesystem never
|
||||
//! finds its storage by name and never sees the whole device; establishment is
|
||||
//! by lineage, exactly as a driver reaches its controller (communication.md
|
||||
//! "Establishment: two planes"). No channel in the reply means the volume is not
|
||||
//! ready yet — retryable, never a verdict.
|
||||
|
||||
const envelope = @import("envelope");
|
||||
|
||||
pub const version: u16 = 1;
|
||||
|
||||
/// The filesystem's handshake. Carries only its protocol version; the volume it
|
||||
/// serves is `Header.target`, and the block channel it needs comes back as the
|
||||
/// reply's capability.
|
||||
pub const Hello = extern struct {
|
||||
version: u16 = version,
|
||||
_padding: u16 = 0,
|
||||
};
|
||||
|
||||
pub const Protocol = envelope.Define(.{
|
||||
.name = "volume-manager",
|
||||
.version = 1,
|
||||
.operations = &.{
|
||||
.{ .name = "hello", .request = Hello },
|
||||
},
|
||||
});
|
||||
|
||||
pub const Operation = Protocol.Operation;
|
||||
pub const message_maximum: usize = Protocol.message_maximum;
|
||||
@@ -14,7 +14,9 @@
|
||||
# service args...
|
||||
/system/services/input
|
||||
/system/services/device-manager
|
||||
/system/services/fat
|
||||
# fat is not here: the volume manager spawns one filesystem per volume it finds,
|
||||
# confined to that volume's partition (docs/file-system-development/storage-architecture.md).
|
||||
/system/services/volume-manager
|
||||
/system/services/display
|
||||
/system/services/display-demo
|
||||
/system/services/logger
|
||||
|
||||
|
@@ -55,7 +55,10 @@
|
||||
# --- the services init spawns from init.csv ---------------------------------
|
||||
/system/services/input, /system/services/init, bind, input
|
||||
/system/services/device-manager, /system/services/init, bind, device-manager
|
||||
/system/services/fat, /system/services/init, bind, vfs
|
||||
/system/services/volume-manager, /system/services/init, bind, volume-manager
|
||||
# fat is spawned and supervised by the volume manager now, not init — the volume
|
||||
# manager confines it to its partition and hands it the block channel.
|
||||
/system/services/fat, /system/services/volume-manager, bind, vfs
|
||||
/system/services/display, /system/services/init, bind, display
|
||||
|
||||
# The discovery service ships under one neutral name per firmware (docs/discovery.md);
|
||||
@@ -94,12 +97,14 @@
|
||||
# ============================================================================
|
||||
|
||||
# --- init's own services ----------------------------------------------------
|
||||
# fat reaches the device manager to be routed to its volume's block provider
|
||||
# (block is not a name — see the bind section); the compositor reaches the
|
||||
# scanout its driver announced, its own endpoint (the mouse-listener thread
|
||||
# opens /protocol/display like any other client — threads share no handles),
|
||||
# and the input stream that moves the cursor.
|
||||
/system/services/fat, /system/services/init, open, device-manager
|
||||
# fat reaches the volume manager to be handed its volume's block channel
|
||||
# (range-confined); the compositor reaches the scanout its driver announced, its
|
||||
# own endpoint (the mouse-listener thread opens /protocol/display like any other
|
||||
# client — threads share no handles), and the input stream that moves the cursor.
|
||||
/system/services/fat, /system/services/volume-manager, open, volume-manager
|
||||
# The volume manager reaches the device manager to be routed to each storage
|
||||
# provider's block channel, then confines a filesystem to each volume.
|
||||
/system/services/volume-manager, /system/services/init, open, device-manager
|
||||
/system/services/display, /system/services/init, open, scanout
|
||||
/system/services/display, /system/services/init, open, display
|
||||
/system/services/display, /system/services/init, open, input
|
||||
|
||||
|
Can't render this file because it contains an unexpected character in line 12 and column 15.
|
@@ -25,9 +25,11 @@ const bot = @import("bulk-only-transport.zig");
|
||||
const envelope = @import("envelope");
|
||||
const block_protocol = @import("block-protocol");
|
||||
|
||||
/// The generated block dispatch. One device per process, so the handler context
|
||||
/// is empty and the geometry stays in this file's globals.
|
||||
const Serve = block_protocol.Protocol.Provider(void);
|
||||
/// The generated block dispatch plus the subscriber machinery the harness owns
|
||||
/// (subscribe/unsubscribe, the exit sweep, the fan-out) — usb-storage publishes
|
||||
/// `medium_changed`, so it is a Subscribers provider, not a bare Provider. One
|
||||
/// device per process, so the handler context is empty.
|
||||
const Serve = service.Subscribers(block_protocol.Protocol, void);
|
||||
|
||||
const Invocation = envelope.Invocation;
|
||||
const Answer = envelope.Answer;
|
||||
@@ -47,6 +49,21 @@ var next_tag: u32 = 1;
|
||||
var block_size: u32 = 512;
|
||||
var block_count: u64 = 0;
|
||||
|
||||
// --- medium presence --------------------------------------------------------
|
||||
//
|
||||
// A slow TEST UNIT READY poll tracks whether the medium is present; on a
|
||||
// transition the driver publishes `medium_changed` to its subscribers (the
|
||||
// volume manager). This is the second removal trigger — the DEVICE stays while
|
||||
// the MEDIUM leaves (a card reader, an ATAPI tray) — which channel death cannot
|
||||
// see (docs/file-system-development/storage-architecture.md). Presence only,
|
||||
// never content. A device that is genuinely unplugged is reaped by the device
|
||||
// manager instead; a poll failure just before that death publishes absent
|
||||
// harmlessly.
|
||||
var service_endpoint: ipc.Handle = 0;
|
||||
var medium_present: bool = true; // a successful bring-up means the medium is here
|
||||
var medium_change_count: u32 = 0;
|
||||
const presence_poll_ms = 1000;
|
||||
|
||||
/// One Bulk-Only-Transport command: send the CBW, run the data stage (to/from
|
||||
/// `data_physical`), read and validate the CSW. Returns true on a passed status.
|
||||
fn transact(cdb: []const u8, direction_in: bool, data_physical: u64, data_length: u32) bool {
|
||||
@@ -82,6 +99,7 @@ fn transact(cdb: []const u8, direction_in: bool, data_physical: u64, data_length
|
||||
var bring_up_failed = false;
|
||||
|
||||
fn initialise(endpoint: ipc.Handle) bool {
|
||||
service_endpoint = endpoint;
|
||||
// One hello, both directions: the block-serving endpoint goes UP (the
|
||||
// manager routes fat's consumer hello here — this driver serves one
|
||||
// volume, one process per stick, so `block` is never a registry name),
|
||||
@@ -154,6 +172,8 @@ fn initialise(endpoint: ipc.Handle) bool {
|
||||
const sector: [*]const u8 = @ptrFromInt(command_data.virtual);
|
||||
std.log.info("block 0 signature 0x{x:0>2}{x:0>2}", .{ sector[510], sector[511] });
|
||||
}
|
||||
// Bring-up succeeded, so the medium is present; start the presence poll.
|
||||
_ = time.timerOnce(endpoint, presence_poll_ms);
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -167,14 +187,73 @@ fn initialise(endpoint: ipc.Handle) bool {
|
||||
/// command did not complete — so there is one errno for all of them.
|
||||
const refused: isize = -envelope.ENOENT;
|
||||
|
||||
fn onGeometry(_: void, _: Invocation(void), answer: Answer(block_protocol.Geometry)) isize {
|
||||
// --- per-sender range confinement -------------------------------------------
|
||||
//
|
||||
// The volume manager confines each filesystem to the partition it mounts
|
||||
// (define_range); a confined sender addresses volume-relative LBAs from 0 and
|
||||
// the driver translates and bounds-checks against its range. A sender with no
|
||||
// range is unconfined — the whole device — which is the default until a range
|
||||
// is defined (behaviour-neutral for a single-volume boot), and is what the
|
||||
// volume manager itself uses to probe partitions before it confines anyone.
|
||||
|
||||
/// bound: filesystem processes confined to sub-ranges of this device at once
|
||||
/// decided-by: ours
|
||||
/// protects: the per-badge range table below
|
||||
/// at-limit: refuse - define_range past it returns -ENOSPC; a runaway detector for
|
||||
/// a compromised volume manager, not a real-partition limit (real disks carry a
|
||||
/// handful of volumes, far under this)
|
||||
/// observed-by: the -ENOSPC a define_range caller gets when the table is full
|
||||
const maximum_ranges = 64;
|
||||
|
||||
const Range = struct { used: bool = false, badge: u32 = 0, base: u64 = 0, count: u64 = 0 };
|
||||
var ranges = [_]Range{.{}} ** maximum_ranges;
|
||||
|
||||
/// The one party allowed to confine others — the first unconfined caller to
|
||||
/// define a range, which is the volume manager (it probes and confines every
|
||||
/// filesystem before handing it a channel). Without this, any unconfined
|
||||
/// opener could install a range for another live client's badge and silently
|
||||
/// redirect its I/O. Released on the controller's death so a restarted volume
|
||||
/// manager re-takes it.
|
||||
var range_controller: ?u32 = null;
|
||||
|
||||
fn rangeFor(badge: u32) ?*Range {
|
||||
for (&ranges) |*r| {
|
||||
if (r.used and r.badge == badge) return r;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Resolve a caller's transfer to an absolute LBA, or null if it falls outside
|
||||
/// the caller's confinement. Unconfined callers (no range) pass through against
|
||||
/// the whole device.
|
||||
///
|
||||
/// The bound is written to survive a hostile confined caller: `lba + count`
|
||||
/// would WRAP for an `lba` near u64 max, sail under a naive `> r.count` check,
|
||||
/// and translate to a wild absolute block — so the check is phrased as two
|
||||
/// subtractions that cannot overflow (`lba` within the range, and `count`
|
||||
/// within what remains). `r.base + lba` cannot overflow once `lba <= r.count`,
|
||||
/// because the volume manager sets `base + count` inside the device.
|
||||
fn resolveTransfer(sender: u32, lba: u64, count: u32) ?u64 {
|
||||
const r = rangeFor(sender) orelse return lba; // unconfined: whole device
|
||||
if (lba > r.count or r.count - lba < count) return null; // past the volume's end
|
||||
return r.base + lba;
|
||||
}
|
||||
|
||||
fn onGeometry(_: void, invocation: Invocation(void), answer: Answer(block_protocol.Geometry)) isize {
|
||||
// A confined caller sees ITS volume's size, not the device's — so a
|
||||
// filesystem mounts against the geometry it is actually allowed to touch.
|
||||
if (rangeFor(invocation.sender)) |r| {
|
||||
answer.set(.{ .block_size = block_size, .block_count = r.count });
|
||||
return 0;
|
||||
}
|
||||
answer.set(.{ .block_size = block_size, .block_count = block_count });
|
||||
return 0;
|
||||
}
|
||||
|
||||
fn onRead(_: void, invocation: Invocation(block_protocol.Transfer), answer: Answer(block_protocol.Transferred)) isize {
|
||||
const request = invocation.request;
|
||||
const cdb = scsi.read10(@intCast(request.lba), @intCast(request.count));
|
||||
const abs = resolveTransfer(invocation.sender, request.lba, request.count) orelse return refused;
|
||||
const cdb = scsi.read10(@intCast(abs), @intCast(request.count));
|
||||
if (!transact(&cdb, true, request.physical, request.count * block_size)) return refused;
|
||||
answer.set(.{ .count = request.count });
|
||||
return 0;
|
||||
@@ -182,12 +261,39 @@ fn onRead(_: void, invocation: Invocation(block_protocol.Transfer), answer: Answ
|
||||
|
||||
fn onWrite(_: void, invocation: Invocation(block_protocol.Transfer), answer: Answer(block_protocol.Transferred)) isize {
|
||||
const request = invocation.request;
|
||||
const cdb = scsi.write10(@intCast(request.lba), @intCast(request.count));
|
||||
const abs = resolveTransfer(invocation.sender, request.lba, request.count) orelse return refused;
|
||||
const cdb = scsi.write10(@intCast(abs), @intCast(request.count));
|
||||
if (!transact(&cdb, false, request.physical, request.count * block_size)) return refused;
|
||||
answer.set(.{ .count = request.count });
|
||||
return 0;
|
||||
}
|
||||
|
||||
/// Confine a sender to a block sub-range (the volume manager's per-volume grant).
|
||||
/// Refused if the CALLER is itself confined — a filesystem cannot widen its own
|
||||
/// range or confine anyone; only an unconfined party (the volume manager) may.
|
||||
fn onDefineRange(_: void, invocation: Invocation(block_protocol.DefineRange), _: Answer(void)) isize {
|
||||
const sender = invocation.sender;
|
||||
// A confined caller may never confine — no self-widening, no escape.
|
||||
if (rangeFor(sender) != null) return -envelope.EPERM;
|
||||
// Confinement has a single controller (the volume manager). Whoever defines
|
||||
// the first range takes it; only they may thereafter, so a second unconfined
|
||||
// opener cannot install a range for a badge it does not own.
|
||||
if (range_controller) |c| {
|
||||
if (sender != c) return -envelope.EPERM;
|
||||
} else {
|
||||
range_controller = sender;
|
||||
}
|
||||
const request = invocation.request;
|
||||
const slot = rangeFor(request.badge) orelse free: {
|
||||
for (&ranges) |*r| {
|
||||
if (!r.used) break :free r;
|
||||
}
|
||||
break :free null;
|
||||
} orelse return -envelope.ENOSPC;
|
||||
slot.* = .{ .used = true, .badge = request.badge, .base = request.base_lba, .count = request.block_count };
|
||||
return 0;
|
||||
}
|
||||
|
||||
/// SYNCHRONIZE CACHE: commit the device's write cache to flash. No data stage.
|
||||
/// Makes prior writes durable before a caller (init at shutdown) cuts power. A
|
||||
/// device without a volatile cache reports success anyway.
|
||||
@@ -218,12 +324,47 @@ const handlers = Serve.Handlers{
|
||||
.flush = onFlush,
|
||||
.attach = onAttach,
|
||||
.detach = onDetach,
|
||||
.define_range = onDefineRange,
|
||||
};
|
||||
|
||||
fn onMessage(message: []const u8, reply: []u8, sender: u32, arrived: *ipc.Arrival) usize {
|
||||
// Peeked, never taken: `attach` forwards the capability and the controller's
|
||||
// binding takes its own reference, so this copy stays the turn's to close.
|
||||
return Serve.dispatch({}, handlers, message, sender, arrived.peek(), reply);
|
||||
// The harness peeks the arrival and takes it only if a handler (subscribe)
|
||||
// claimed it; attach/detach forward the capability without claiming, so the
|
||||
// turn still closes their copy after the controller took its own reference.
|
||||
return Serve.dispatch({}, handlers, message, sender, arrived, reply);
|
||||
}
|
||||
|
||||
/// A slow TEST UNIT READY poll: success means the medium is present, failure
|
||||
/// means it is not. On a transition, bump the counter and publish. (Sense-key
|
||||
/// inspection to tell "medium absent" from other transport errors is a
|
||||
/// refinement; a clean eject — what QEMU and a card reader produce — makes
|
||||
/// TEST UNIT READY report not-ready, which this reads correctly.)
|
||||
fn pollPresence() void {
|
||||
const ready = scsi.testUnitReady();
|
||||
const now = transact(&ready, false, 0, 0);
|
||||
if (now == medium_present) return;
|
||||
medium_present = now;
|
||||
medium_change_count +%= 1;
|
||||
std.log.info("medium {s}", .{if (now) "present" else "absent"});
|
||||
Serve.publish(.medium_changed, 0, .{ .present = @intFromBool(now), .change_count = medium_change_count });
|
||||
}
|
||||
|
||||
fn onNotification(badge: u64) void {
|
||||
const got = ipc.Received{ .len = 0, .badge = badge, .cap = null };
|
||||
if (got.isTimer()) {
|
||||
pollPresence();
|
||||
_ = time.timerOnce(service_endpoint, presence_poll_ms);
|
||||
return;
|
||||
}
|
||||
// A client died. The harness sweep (Serve.hooks) covers only the SUBSCRIBER
|
||||
// table; the per-badge range table is ours to reclaim, or confine/die cycles
|
||||
// (the medium-removal lifecycle) would exhaust it. A dead controller also
|
||||
// releases confinement authority to its successor.
|
||||
if (got.isChildExit()) {
|
||||
const dead = got.childProcessId();
|
||||
if (rangeFor(dead)) |r| r.* = .{};
|
||||
if (range_controller == dead) range_controller = null;
|
||||
}
|
||||
}
|
||||
|
||||
pub fn main(init: process.Init) void {
|
||||
@@ -241,6 +382,8 @@ pub fn main(init: process.Init) void {
|
||||
service.run(block_protocol.message_maximum, .{
|
||||
.init = initialise,
|
||||
.on_message = onMessage,
|
||||
.on_notification = onNotification,
|
||||
.subscribers = Serve.hooks,
|
||||
});
|
||||
// A failure exit (nonzero -> .aborted) tells the device manager to restart
|
||||
// us with backoff; a clean return means there was nothing to serve.
|
||||
|
||||
@@ -2146,9 +2146,18 @@ fn systemFsMount(state: *architecture.CpuState) void {
|
||||
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
// The ownership gate. A LIVE owner's mount is its own: nobody else may
|
||||
// replace it — displacement-by-remount would be worse than unmounting.
|
||||
// A DEAD owner's mount is replaceable by anyone with a backend: that is
|
||||
// the restart story (a respawned filesystem is a new task retaking its
|
||||
// prefix). Kernel-installed mounts (owner 0) are never displaceable.
|
||||
if (vfs.mountOwner(prefix)) |owner| {
|
||||
const displaceable = owner != 0 and (owner == t.id or scheduler.taskByIdLocked(owner) == null);
|
||||
if (!displaceable) return failErr(state, ipc.EPERM);
|
||||
}
|
||||
const endpoint = ipc.resolveHandle(t, backend_handle) orelse return failErr(state, ipc.EBADF);
|
||||
endpoint.refcount += 1; // the mount table's reference
|
||||
if (!vfs.mountBackend(prefix, endpoint, rewrite)) {
|
||||
if (!vfs.mountBackend(prefix, endpoint, rewrite, t.id)) {
|
||||
ipc.dropRef(endpoint);
|
||||
return fail(state);
|
||||
}
|
||||
@@ -2166,6 +2175,12 @@ fn systemFsUnmount(state: *architecture.CpuState) void {
|
||||
if (!user_memory.copyFromUser(t.address_space, prefix_ptr, prefix)) return failErr(state, ipc.EFAULT);
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
// Only the mounting task unmounts. No dead-owner exception here: a dead
|
||||
// owner's mount is already being swept lazily by resolution, and a
|
||||
// stranger gains nothing legitimate by racing that — the restart story
|
||||
// goes through remount-replace, never through unmount.
|
||||
const owner = vfs.mountOwner(prefix) orelse return fail(state);
|
||||
if (owner != t.id) return failErr(state, ipc.EPERM);
|
||||
if (!vfs.unmount(prefix)) return fail(state);
|
||||
architecture.setSystemCallResult(state, 0);
|
||||
}
|
||||
|
||||
@@ -265,6 +265,8 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void {
|
||||
deviceTransferTest(boot_information);
|
||||
} else if (eql(case, "device-authority")) {
|
||||
deviceAuthorityTest(boot_information);
|
||||
} else if (eql(case, "block-range")) {
|
||||
blockRangeTest(boot_information);
|
||||
} else if (eql(case, "device-manager")) {
|
||||
deviceManagerTest(boot_information);
|
||||
} else if (eql(case, "protocol-registry")) {
|
||||
@@ -3034,6 +3036,33 @@ fn fatMountTest(boot_information: *const BootInformation) void {
|
||||
result();
|
||||
}
|
||||
|
||||
/// Per-sender range confinement (V2a, docs/volume-manager-plan.md): the fixture
|
||||
/// acquires the block channel, confines ITSELF to a sub-range, and asserts it
|
||||
/// cannot read past that range or widen it. Boots init in REGISTRY-ONLY mode
|
||||
/// plus the device manager (which brings up the USB storage chain) — deliberately
|
||||
/// NOT the full tree, because the volume manager would take the confinement
|
||||
/// controller first and refuse the fixture's define_range. Without it the fixture
|
||||
/// is the sole definer, exactly as the volume manager is in a real boot.
|
||||
fn blockRangeTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: block-range\n", .{});
|
||||
if (boot_information.initial_ramdisk_len == 0) {
|
||||
check("bootloader handed over the initial_ramdisk", false);
|
||||
result();
|
||||
return;
|
||||
}
|
||||
const ramdisk = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||
const rd = initial_ramdisk.Reader.init(ramdisk) orelse {
|
||||
check("initial_ramdisk image is valid", false);
|
||||
result();
|
||||
return;
|
||||
};
|
||||
process.setInitialRamdisk(ramdisk);
|
||||
check("registry (init) spawned", spawnRegistry(rd));
|
||||
check("device-manager spawned (boots the USB storage chain)", spawnNamed(rd, "device-manager"));
|
||||
check("block-range-test spawned", spawnNamedWithArg(rd, "block-range-test", "run"));
|
||||
result();
|
||||
}
|
||||
|
||||
fn bootServiceTreeTest(boot_information: *const BootInformation, comptime label: []const u8) void {
|
||||
log("DANOS-TEST-BEGIN: " ++ label ++ "\n", .{});
|
||||
if (boot_information.initial_ramdisk_len == 0) {
|
||||
|
||||
+28
-6
@@ -69,6 +69,10 @@ const Mount = struct {
|
||||
backend: ?*ipc.Endpoint = null, // referenced while mounted
|
||||
rewrite: [maximum_rewrite]u8 = undefined,
|
||||
rewrite_len: usize = 0,
|
||||
// The task that mounted this prefix — the ownership `fs_unmount` and
|
||||
// remount-replace are gated on (storage-architecture.md, the lifecycle
|
||||
// rule). Zero for kernel-installed mounts, which no task may displace.
|
||||
owner: u32 = 0,
|
||||
|
||||
fn prefixSlice(self: *const Mount) []const u8 {
|
||||
return self.prefix[0..self.prefix_len];
|
||||
@@ -375,11 +379,25 @@ fn refusesProtocolMount(prefix: []const u8) bool {
|
||||
return protocolBound(); // /protocol itself: first mount wins
|
||||
}
|
||||
|
||||
/// Mount `backend` at `prefix` with an optional backend-side `rewrite` prefix.
|
||||
/// The endpoint reference is taken by the caller (process.zig bumps it); refuses
|
||||
/// shadowing or replacing the initrd trees (/system, /test) — except the two
|
||||
/// carve-outs in `initrd_carve_outs`, the writable configuration/log subtrees.
|
||||
pub fn mountBackend(prefix: []const u8, backend: *ipc.Endpoint, rewrite: []const u8) bool {
|
||||
/// The task a backend mount at exactly `prefix` is recorded against, or null
|
||||
/// when nothing backend-shaped is mounted there. The syscall layer consults
|
||||
/// this before allowing a replace or an unmount — the ownership gate lives
|
||||
/// there, where the task table is; this table only remembers the fact.
|
||||
pub fn mountOwner(prefix: []const u8) ?u32 {
|
||||
for (&mounts) |*m| {
|
||||
if (m.used and m.kind == .backend and std.mem.eql(u8, m.prefixSlice(), prefix)) return m.owner;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Mount `backend` at `prefix` with an optional backend-side `rewrite` prefix,
|
||||
/// recorded against `owner`. The endpoint reference is taken by the caller
|
||||
/// (process.zig bumps it); refuses shadowing or replacing the initrd trees
|
||||
/// (/system, /test) — except the two carve-outs in `initrd_carve_outs`, the
|
||||
/// writable configuration/log subtrees. The replace-vs-refuse decision for an
|
||||
/// already-mounted prefix is the CALLER's (it can see task liveness); by the
|
||||
/// time this runs, replacing is decided.
|
||||
pub fn mountBackend(prefix: []const u8, backend: *ipc.Endpoint, rewrite: []const u8, owner: u32) bool {
|
||||
if (!isAbsolute(prefix) or prefix.len < 2 or prefix.len > maximum_prefix) return false;
|
||||
if (rewrite.len > maximum_rewrite) return false;
|
||||
if (refusesProtocolMount(prefix)) return false; // the registry's prefix is claimed once
|
||||
@@ -389,12 +407,16 @@ pub fn mountBackend(prefix: []const u8, backend: *ipc.Endpoint, rewrite: []const
|
||||
}
|
||||
}
|
||||
installMount(prefix, .backend, backend, rewrite);
|
||||
for (&mounts) |*m| {
|
||||
if (m.used and std.mem.eql(u8, m.prefixSlice(), prefix)) m.owner = owner;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
pub fn unmount(prefix: []const u8) bool {
|
||||
// Unmounting /protocol would delete the naming layer for everyone; nobody
|
||||
// may, init included. The mount lasts the boot.
|
||||
// may, init included. The mount lasts the boot. Ownership is checked by
|
||||
// the syscall layer (mountOwner) before this runs.
|
||||
if (std.mem.eql(u8, prefix, protocol_root)) return false;
|
||||
for (&mounts) |*m| {
|
||||
if (m.used and m.kind == .backend and std.mem.eql(u8, m.prefixSlice(), prefix)) {
|
||||
|
||||
@@ -10,10 +10,9 @@ pub fn build(b: *std.Build) void {
|
||||
.name = "fat",
|
||||
.root_source_file = b.path("fat.zig"),
|
||||
.imports = &.{
|
||||
"block", "channel", "device-manager-protocol", "driver",
|
||||
"envelope", "file-system", "ipc", "logging",
|
||||
"memory", "process", "service", "time",
|
||||
"vfs-protocol",
|
||||
"block", "channel", "envelope", "file-system-harness",
|
||||
"ipc", "logging", "memory", "process",
|
||||
"time", "volume-manager-protocol",
|
||||
},
|
||||
});
|
||||
b.installArtifact(exe);
|
||||
|
||||
+89
-358
@@ -1,40 +1,31 @@
|
||||
//! system/services/fat — the FAT filesystem server. Spawned as a boot service, it
|
||||
//! opens the block device (a USB stick via usb-storage) under `.block`, mounts the
|
||||
//! FAT filesystem on it (the pure engine in engine.zig), and mounts itself into
|
||||
//! the VFS at /volumes/usb. From then on the VFS forwards every open/read/write/
|
||||
//! status/readdir/close under /volumes/usb to this server, which serves the same
|
||||
//! vfs-protocol as a backend — turning block reads into file reads.
|
||||
//! system/services/fat — the FAT filesystem service. This is FAT's FAT-specific
|
||||
//! half: it finds its block device, sets up the DMA bounce buffer, mounts the
|
||||
//! FAT engine on it, and hands the mounted volume to the shared filesystem
|
||||
//! harness (library/kernel/file-system-harness), which owns everything else —
|
||||
//! the vfs-protocol serving, the open-node table, mount registration, the exit
|
||||
//! sweep, durable-on-close. The engine (engine.zig) is the pure, host-testable
|
||||
//! format code; on-disk.zig its byte layout. A second filesystem reuses the
|
||||
//! harness and supplies its own engine
|
||||
//! (docs/file-system-development/storage-architecture.md).
|
||||
//!
|
||||
//! The block data path never crosses IPC: a DMA bounce buffer is handed to the
|
||||
//! block driver by physical address, and the engine copies sectors in and out of
|
||||
//! it.
|
||||
//! block driver by physical address, and the engine copies sectors in and out.
|
||||
|
||||
const std = @import("std");
|
||||
const channel = @import("channel");
|
||||
const device_manager_protocol = @import("device-manager-protocol");
|
||||
const driver = @import("driver");
|
||||
const volume_manager_protocol = @import("volume-manager-protocol");
|
||||
const ipc = @import("ipc");
|
||||
const process = @import("process");
|
||||
const service = @import("service");
|
||||
const time = @import("time");
|
||||
const block = @import("block");
|
||||
const file_system = @import("file-system");
|
||||
const memory = @import("memory");
|
||||
const logging = @import("logging");
|
||||
const time = @import("time");
|
||||
const engine = @import("engine.zig");
|
||||
const on_disk = @import("on-disk.zig");
|
||||
const envelope = @import("envelope");
|
||||
const vfs_protocol = @import("vfs-protocol");
|
||||
const harness = @import("file-system-harness");
|
||||
|
||||
/// The generated vfs dispatch, bound to this server. There is one FAT volume per
|
||||
/// process, so the handler context is empty and the state stays where it was: in
|
||||
/// this file's globals.
|
||||
const Serve = vfs_protocol.Protocol.Provider(void);
|
||||
|
||||
const Invocation = envelope.Invocation;
|
||||
const Answer = envelope.Answer;
|
||||
|
||||
const mount_point = "/volumes/usb";
|
||||
/// The serving harness, specialized for the FAT engine. One volume per process.
|
||||
const Harness = harness.Server(engine.FileSystem);
|
||||
|
||||
// The engine's BlockDevice, backed by the `.block` driver plus a DMA bounce
|
||||
// buffer the driver reads/writes by physical address.
|
||||
@@ -68,149 +59,93 @@ var ipc_block: IpcBlock = undefined;
|
||||
// file close — so writes are committed to stable media before a power-off.
|
||||
var device_dirty: bool = false;
|
||||
var filesystem: engine.FileSystem = undefined;
|
||||
/// The volume this FAT process serves, its id given as argv[1] by the volume
|
||||
/// manager that spawned it. The startup hello names it so the manager returns
|
||||
/// the right volume's channel.
|
||||
var my_volume_id: u64 = 0;
|
||||
|
||||
// Open handles clients hold against this backend: each maps a node id to a
|
||||
// resolved engine node, and to the client that opened it. `owner` is the
|
||||
// kernel-stamped badge of the opening task — the only source identity there is.
|
||||
const OpenNode = struct { used: bool = false, node: engine.Node = undefined, owner: u32 = 0 };
|
||||
var open_nodes = [_]OpenNode{.{}} ** 32;
|
||||
/// The prefixes this volume installs: /volumes/usb from the volume root, plus
|
||||
/// the two hierarchy subtrees the boot volume carries (rewrite == prefix), so
|
||||
/// hierarchy paths (the logger's /system/logs) stay decoupled from which volume
|
||||
/// backs them.
|
||||
const fat_mounts = [_]harness.MountSpec{
|
||||
.{ .prefix = "/volumes/usb" },
|
||||
.{ .prefix = "/system/configuration", .rewrite = "/system/configuration" },
|
||||
.{ .prefix = "/system/logs", .rewrite = "/system/logs" },
|
||||
};
|
||||
|
||||
fn allocOpen() ?usize {
|
||||
for (&open_nodes, 0..) |*o, i| {
|
||||
if (!o.used) return i;
|
||||
/// Get this volume's block channel from the volume manager (establishment by
|
||||
/// lineage, communication.md "Establishment: two planes" — `block` is not a
|
||||
/// registry name). The manager spawned this process, confined it to its
|
||||
/// partition, and answers the hello with the channel; the channel is
|
||||
/// range-confined to this process's badge, so reads and writes are
|
||||
/// volume-relative and cannot reach the neighbouring partition. Null until the
|
||||
/// manager has the volume ready — this retries.
|
||||
fn acquireVolume() ?block.Device {
|
||||
var attempts: u32 = 0;
|
||||
const vm = while (attempts < 500) : (attempts += 1) {
|
||||
if (channel.openEndpoint("volume-manager")) |handle| break handle;
|
||||
time.sleepMillis(20);
|
||||
} else return null;
|
||||
|
||||
attempts = 0;
|
||||
while (attempts < 500) : (attempts += 1) {
|
||||
var packet: [volume_manager_protocol.message_maximum]u8 = undefined;
|
||||
const framed = volume_manager_protocol.Protocol.encodeRequest(.hello, my_volume_id, .{}, &.{}, &packet) orelse return null;
|
||||
var reply: [volume_manager_protocol.message_maximum]u8 = undefined;
|
||||
const answered = ipc.callCap(vm, framed, &reply, null) catch return null;
|
||||
const status = envelope.statusOf(reply[0..answered.len]) orelse return null;
|
||||
if (status.status != 0) {
|
||||
if (answered.cap) |stray| _ = ipc.close(stray);
|
||||
_ = logging.write("/system/services/fat: volume manager refused the hello\n");
|
||||
return null;
|
||||
}
|
||||
if (answered.cap) |bus| return .{ .endpoint = bus };
|
||||
// Acked with no channel: the volume is not ready yet — retry.
|
||||
time.sleepMillis(20);
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// The open node `id` names **for `owner`** — null unless the id is in range, in
|
||||
/// use, and this client's own. Node ids are small integers drawn from a table of
|
||||
/// thirty-two, so they are trivially guessable; before this check every client
|
||||
/// honoured every other client's ids, which is the hole
|
||||
/// docs/os-development/protocol-namespace.md names ("handles must be scoped per
|
||||
/// client — validated against the badge"). Nothing else about them changed: they
|
||||
/// are still per-session, still swept when their owner dies.
|
||||
///
|
||||
/// The owner is a *task*, not a process, because the badge is: a threaded client
|
||||
/// reads and writes a node from the thread that opened it, exactly as the exit
|
||||
/// sweep already released a worker thread's handles when that thread died.
|
||||
fn openFor(id: u64, owner: u32) ?*OpenNode {
|
||||
if (id >= open_nodes.len) return null;
|
||||
const o = &open_nodes[@intCast(id)];
|
||||
if (!o.used or o.owner != owner) return null;
|
||||
return o;
|
||||
}
|
||||
|
||||
/// What a handler returns when the thing asked for is not there — a bad node id,
|
||||
/// a node that is someone else's, a path that does not resolve, a mutation the
|
||||
/// volume refused. One errno for all of them, because a filesystem's failures are
|
||||
/// all "no such thing" as far as the file API can act on them — and because
|
||||
/// *someone else's* must be indistinguishable from *nobody's*, or the refusal
|
||||
/// would itself tell a prober which ids are live (the same discipline the
|
||||
/// protocol namespace's refused open follows).
|
||||
const refused: isize = -envelope.ENOENT;
|
||||
|
||||
/// How often to look for a block device while none is mounted. Storage arriving
|
||||
/// is EVENT-shaped (the usb chain registering, possibly after a driver restart),
|
||||
/// but the registry has no subscription — a slow poll from our own harness loop
|
||||
/// keeps the service responsive (ping, terminate) while it waits, and keeps it
|
||||
/// alive to catch storage that appears LATE (a restarted usb-storage after a
|
||||
/// transient failure — the resilience half of docs/logging.md's storage story).
|
||||
const mount_retry_ms = 500;
|
||||
|
||||
var mounted = false;
|
||||
var service_endpoint: ipc.Handle = 0;
|
||||
/// The one channel to the device manager, opened on first need and kept — the
|
||||
/// poll retries on it, never spending a handle-table slot per attempt.
|
||||
var manager_handle: ?ipc.Handle = null;
|
||||
|
||||
/// Find the volume's provider through the device manager (establishment by
|
||||
/// lineage, communication.md "Establishment: two planes" — `block` is not a
|
||||
/// registry name; one storage process serves each stick): enumerate the
|
||||
/// manager's tree, take the FIRST usb mass-storage child by enumeration order
|
||||
/// (deterministic within a boot; single-volume by construction, and choosing
|
||||
/// the BOOT volume by content when two sticks are present is the M21 remount
|
||||
/// track), and consumer-hello for the channel of the driver bound to it.
|
||||
/// Null until the chain is up — the caller's poll retries.
|
||||
fn acquireVolume() ?block.Device {
|
||||
const manager = manager_handle orelse opened: {
|
||||
const handle = channel.openEndpoint("device-manager") orelse return null;
|
||||
manager_handle = handle;
|
||||
break :opened handle;
|
||||
};
|
||||
|
||||
// The envelope's reserved `enumerate` verb, PAGED: one reply carries only
|
||||
// a handful of entries and a real tree (a dozen ACPI nodes before the
|
||||
// first USB child) is bigger, so `Header.target` is the start cursor and
|
||||
// a short page is the end. Identity is the bus's native triple, for USB
|
||||
// (base << 16) | (class << 8) | protocol — mass storage is base 0x08,
|
||||
// subclass 0x06 (SCSI transparent), the same key devices.csv matches on.
|
||||
const Entry = device_manager_protocol.ChildEntry;
|
||||
var start: u64 = 0;
|
||||
while (true) {
|
||||
const enumerate = envelope.Header{ .operation = envelope.operation_enumerate, .target = start };
|
||||
var reply: [device_manager_protocol.message_maximum]u8 = undefined;
|
||||
const length = ipc.call(manager, std.mem.asBytes(&enumerate), &reply) catch return null;
|
||||
const status = envelope.statusOf(reply[0..length]) orelse return null;
|
||||
if (status.status != 0) return null;
|
||||
|
||||
const carried = @min(@as(usize, status.len), length -| envelope.prefix_size);
|
||||
const tail = reply[envelope.prefix_size..][0..carried];
|
||||
const count = tail.len / @sizeOf(Entry);
|
||||
if (count == 0) return null; // the tree is exhausted; no volume yet
|
||||
var index: usize = 0;
|
||||
while (index < count) : (index += 1) {
|
||||
const entry = std.mem.bytesToValue(Entry, tail[index * @sizeOf(Entry) ..][0..@sizeOf(Entry)]);
|
||||
if (entry.device_id == device_manager_protocol.no_device) continue;
|
||||
if ((entry.identity >> 16) & 0xff != 0x08 or (entry.identity >> 8) & 0xff != 0x06) continue;
|
||||
const exchanged = driver.helloOn(manager, .consumer, entry.device_id, null, true) orelse return null;
|
||||
const provider = exchanged.channel orelse continue; // its driver not up yet — next tick
|
||||
return .{ .endpoint = provider };
|
||||
}
|
||||
start += count;
|
||||
/// Durable-on-close: commit the device write cache if any block reached it since
|
||||
/// the last flush. The harness calls this on every close; the dirty check keeps
|
||||
/// it cheap. `device_dirty` lives here because `IpcBlock.writeBlocks` sets it.
|
||||
fn flushIfDirty() void {
|
||||
if (device_dirty) {
|
||||
_ = ipc_block.device.flush();
|
||||
device_dirty = false;
|
||||
}
|
||||
}
|
||||
|
||||
fn initialise(endpoint: ipc.Handle) bool {
|
||||
service_endpoint = endpoint;
|
||||
_ = logging.write("/system/services/fat: starting, waiting for a block device\n");
|
||||
// With the router in the kernel, clients hold OUR node ids directly; sweep
|
||||
// a dead client's open handles via the published exit events (the pattern
|
||||
// the old userspace router used for its own table).
|
||||
_ = process.subscribeExits(endpoint);
|
||||
tryBringUp();
|
||||
if (!mounted) _ = time.timerOnce(endpoint, mount_retry_ms);
|
||||
return true; // serve regardless: requests fail politely until storage mounts
|
||||
}
|
||||
|
||||
/// One storage bring-up attempt: block device -> FAT mount -> VFS mounts. Sets
|
||||
/// `mounted` on success; a failure leaves everything untouched for the next tick.
|
||||
fn tryBringUp() void {
|
||||
if (mounted) return;
|
||||
const device = acquireVolume() orelse return;
|
||||
/// FAT bring-up: find the block device, set up DMA, mount the engine, and hand
|
||||
/// the volume to the harness — or null to retry on the harness's timer.
|
||||
fn fatBringUp(endpoint: ipc.Handle) ?Harness.Volume {
|
||||
_ = endpoint;
|
||||
const device = acquireVolume() orelse return null;
|
||||
const geometry = device.geometry() orelse {
|
||||
_ = logging.write("/system/services/fat: block geometry unavailable\n");
|
||||
return;
|
||||
return null;
|
||||
};
|
||||
// Shareable so the buffer's capability can be attached down the chain (block server
|
||||
// -> controller), making its physical addresses reachable by the device under an
|
||||
// enforcing IOMMU. No-op binding otherwise.
|
||||
const bounce = memory.dmaAlloc(engine.max_transfer_sectors * 512, memory.dma_coherent | memory.dma_shareable) orelse return;
|
||||
// Shareable so the buffer's capability can be attached down the chain (block
|
||||
// server -> controller), making its physical addresses reachable by the
|
||||
// device under an enforcing IOMMU. No-op binding otherwise.
|
||||
const bounce = memory.dmaAlloc(engine.max_transfer_sectors * 512, memory.dma_coherent | memory.dma_shareable) orelse return null;
|
||||
if (bounce.handle) |handle| {
|
||||
// Attach, detach, and attach again: the round trip exercises BOTH verbs
|
||||
// of the DMA-window lifecycle through the whole chain (fat → storage →
|
||||
// bus → kernel) on every boot, so a broken detach fails every fat case
|
||||
// of the DMA-window lifecycle through the whole chain (fat -> storage ->
|
||||
// bus -> kernel) on every boot, so a broken detach fails every fat case
|
||||
// rather than lying dormant until the first buffer replacement.
|
||||
if (!device.attach(handle)) {
|
||||
_ = logging.write("/system/services/fat: could not attach the DMA bounce buffer\n");
|
||||
return;
|
||||
return null;
|
||||
}
|
||||
if (!device.detach(handle)) {
|
||||
_ = logging.write("/system/services/fat: could not detach the DMA bounce buffer\n");
|
||||
return;
|
||||
return null;
|
||||
}
|
||||
if (!device.attach(handle)) {
|
||||
_ = logging.write("/system/services/fat: could not re-attach the DMA bounce buffer\n");
|
||||
return;
|
||||
return null;
|
||||
}
|
||||
_ = ipc.close(handle); // the binding holds its own reference now
|
||||
}
|
||||
@@ -225,222 +160,18 @@ fn tryBringUp() void {
|
||||
};
|
||||
filesystem = engine.FileSystem.mount(block_device) orelse {
|
||||
_ = logging.write("/system/services/fat: not a FAT filesystem\n");
|
||||
return;
|
||||
return null;
|
||||
};
|
||||
std.log.info("mounted FAT ({s}, {d} clusters, partition lba {d})", .{ @tagName(filesystem.geometry.fat_type), filesystem.geometry.cluster_count, filesystem.base_lba });
|
||||
|
||||
// Mount ourselves into the kernel VFS at /volumes/usb — and serve
|
||||
// /system/configuration and /system/logs from the volume's identically-named
|
||||
// subtrees (the boot volume is hierarchy-shaped, so rewrite == prefix), so
|
||||
// hierarchy paths (the logger's /system/logs) stay decoupled from which
|
||||
// volume carries them.
|
||||
if (file_system.mount(mount_point, endpointForMount())) {
|
||||
std.log.info("mounted {s}", .{mount_point});
|
||||
} else {
|
||||
_ = logging.write("/system/services/fat: could not mount /volumes/usb\n");
|
||||
return .{ .engine = &filesystem, .mounts = &fat_mounts, .flush = flushIfDirty };
|
||||
}
|
||||
|
||||
pub fn main(init: process.Init) void {
|
||||
// The volume manager spawns this process with its volume id as argv[1].
|
||||
if (init.arguments.get(1)) |id| {
|
||||
my_volume_id = std.fmt.parseInt(u64, id, 10) catch 0;
|
||||
}
|
||||
if (file_system.mountRewritten("/system/configuration", endpointForMount(), "/system/configuration")) {
|
||||
std.log.info("mounted /system/configuration", .{});
|
||||
} else {
|
||||
_ = logging.write("/system/services/fat: could not mount /system/configuration\n");
|
||||
}
|
||||
if (file_system.mountRewritten("/system/logs", endpointForMount(), "/system/logs")) {
|
||||
std.log.info("mounted /system/logs", .{});
|
||||
} else {
|
||||
_ = logging.write("/system/services/fat: could not mount /system/logs\n");
|
||||
}
|
||||
mounted = true;
|
||||
}
|
||||
|
||||
fn endpointForMount() ipc.Handle {
|
||||
return service_endpoint;
|
||||
}
|
||||
|
||||
/// A subscribed process-exit event: release every open handle the dead client
|
||||
/// held, so a crashed reader can't pin table slots (or, later, locks).
|
||||
fn onNotification(badge: u64) void {
|
||||
const got = ipc.Received{ .len = 0, .badge = badge, .cap = null };
|
||||
if (got.isTimer()) {
|
||||
tryBringUp();
|
||||
if (!mounted) _ = time.timerOnce(service_endpoint, mount_retry_ms);
|
||||
return;
|
||||
}
|
||||
if (!got.isChildExit()) return;
|
||||
const dead = got.childProcessId();
|
||||
var released: u32 = 0;
|
||||
for (&open_nodes) |*o| {
|
||||
if (o.used and o.owner == dead) {
|
||||
o.* = .{};
|
||||
released += 1;
|
||||
}
|
||||
}
|
||||
if (released != 0) std.log.info("released {d} handle(s) for dead client {d}", .{ released, dead });
|
||||
}
|
||||
|
||||
const ParentLeaf = struct { parent: []const u8, leaf: []const u8 };
|
||||
|
||||
// Split a path into its parent directory and final component: "/a/b" -> ("/a",
|
||||
// "b"); "/b" -> ("/", "b"); "b" -> ("/", "b").
|
||||
fn splitParent(path: []const u8) ParentLeaf {
|
||||
const slash = std.mem.lastIndexOfScalar(u8, path, '/');
|
||||
return .{
|
||||
.parent = if (slash) |s| (if (s == 0) "/" else path[0..s]) else "/",
|
||||
.leaf = if (slash) |s| path[s + 1 ..] else path,
|
||||
};
|
||||
}
|
||||
|
||||
fn onOpen(_: void, invocation: Invocation(vfs_protocol.Open), answer: Answer(vfs_protocol.Opened)) isize {
|
||||
const path = invocation.tail;
|
||||
const flags = invocation.request.flags;
|
||||
var node = filesystem.resolve(path);
|
||||
if (node == null and flags & vfs_protocol.create != 0) {
|
||||
const split = splitParent(path);
|
||||
const parent = filesystem.resolve(split.parent) orelse return refused;
|
||||
node = filesystem.createFile(parent, split.leaf);
|
||||
}
|
||||
var resolved = node orelse return refused;
|
||||
// O_TRUNC: replace an existing file's contents rather than overwriting in place
|
||||
// (frees the old chain, so a shorter rewrite leaves no stale tail).
|
||||
if (flags & vfs_protocol.truncate != 0 and !resolved.is_directory) {
|
||||
filesystem.truncate(&resolved);
|
||||
}
|
||||
const index = allocOpen() orelse return refused;
|
||||
open_nodes[index] = .{ .used = true, .node = resolved, .owner = invocation.sender };
|
||||
answer.set(.{ .node = index });
|
||||
return 0;
|
||||
}
|
||||
|
||||
fn onRead(_: void, invocation: Invocation(vfs_protocol.Read), answer: Answer(void)) isize {
|
||||
const o = openFor(invocation.target, invocation.sender) orelse return refused;
|
||||
const into = answer.tail();
|
||||
const want = @min(@as(usize, invocation.request.len), into.len);
|
||||
return @intCast(filesystem.readFile(o.node, @intCast(invocation.request.offset), into[0..want]));
|
||||
}
|
||||
|
||||
fn onWrite(_: void, invocation: Invocation(vfs_protocol.Write), answer: Answer(vfs_protocol.Written)) isize {
|
||||
const o = openFor(invocation.target, invocation.sender) orelse return refused;
|
||||
const data = invocation.tail[0..@min(invocation.tail.len, invocation.request.len)];
|
||||
const n = filesystem.writeFile(&o.node, @intCast(invocation.request.offset), data);
|
||||
answer.set(.{ .count = @intCast(n) });
|
||||
return 0;
|
||||
}
|
||||
|
||||
fn onStatus(_: void, invocation: Invocation(void), answer: Answer(vfs_protocol.FileStatus)) isize {
|
||||
const o = openFor(invocation.target, invocation.sender) orelse return refused;
|
||||
const kind: vfs_protocol.NodeKind = if (o.node.is_directory) .directory else .regular;
|
||||
answer.set(.{ .size = o.node.size, .kind = @intFromEnum(kind), .mtime = o.node.mtime });
|
||||
return 0;
|
||||
}
|
||||
|
||||
/// One entry per call. End of directory — a node that is not a directory, or a
|
||||
/// cursor past the last child — is an entry with no name, which is how the
|
||||
/// protocol spells it now that the reply's length always counts the fixed part.
|
||||
fn onReaddir(_: void, invocation: Invocation(vfs_protocol.Readdir), answer: Answer(vfs_protocol.DirectoryEntry)) isize {
|
||||
const o = openFor(invocation.target, invocation.sender) orelse return refused;
|
||||
if (!o.node.is_directory) {
|
||||
answer.set(.{});
|
||||
return 0;
|
||||
}
|
||||
const listing = filesystem.listEntry(o.node, @intCast(invocation.request.cursor)) orelse {
|
||||
answer.set(.{});
|
||||
return 0;
|
||||
};
|
||||
const kind: vfs_protocol.NodeKind = if (listing.is_directory) .directory else .regular;
|
||||
const into = answer.tail();
|
||||
const name_len = @min(listing.name_len, into.len);
|
||||
@memcpy(into[0..name_len], listing.name_buffer[0..name_len]);
|
||||
answer.set(.{ .kind = @intFromEnum(kind), .name_len = @intCast(name_len), .size = listing.size });
|
||||
return @intCast(name_len);
|
||||
}
|
||||
|
||||
/// Closing is an operation on a node like any other, so it is scoped like any
|
||||
/// other: a client may release its own handles and nobody else's. An id that is
|
||||
/// not the caller's — free, out of range, or another client's — is refused
|
||||
/// identically, so a close cannot be used to ask which ids are live either.
|
||||
fn onClose(_: void, invocation: Invocation(void), _: Answer(void)) isize {
|
||||
const o = openFor(invocation.target, invocation.sender) orelse return refused;
|
||||
o.used = false;
|
||||
// Durable-on-close: if any block reached the device since the last flush,
|
||||
// commit its cache to stable media now (best-effort). This is what makes
|
||||
// init's shutdown log flush survive a real power-off, and is the right
|
||||
// default for removable media the user may unplug.
|
||||
if (device_dirty) {
|
||||
_ = ipc_block.device.flush();
|
||||
device_dirty = false;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
fn onMakeDirectory(_: void, invocation: Invocation(void), _: Answer(void)) isize {
|
||||
const path = invocation.tail;
|
||||
if (filesystem.resolve(path) != null) return refused; // already exists — no duplicate entries
|
||||
const split = splitParent(path);
|
||||
const parent = filesystem.resolve(split.parent) orelse return refused;
|
||||
if (filesystem.createDirectory(parent, split.leaf) == null) return refused;
|
||||
return 0;
|
||||
}
|
||||
|
||||
fn onUnlink(_: void, invocation: Invocation(void), _: Answer(void)) isize {
|
||||
const split = splitParent(invocation.tail);
|
||||
const parent = filesystem.resolve(split.parent) orelse return refused;
|
||||
if (!filesystem.removeFile(parent, split.leaf)) return refused;
|
||||
return 0;
|
||||
}
|
||||
|
||||
fn onRename(_: void, invocation: Invocation(void), _: Answer(void)) isize {
|
||||
const both = invocation.tail;
|
||||
const separator = std.mem.indexOfScalar(u8, both, 0) orelse return refused;
|
||||
const old_split = splitParent(both[0..separator]);
|
||||
const new_split = splitParent(both[separator + 1 ..]);
|
||||
// Same-directory rename only.
|
||||
if (!std.mem.eql(u8, old_split.parent, new_split.parent)) return refused;
|
||||
const parent = filesystem.resolve(old_split.parent) orelse return refused;
|
||||
if (!filesystem.rename(parent, old_split.leaf, new_split.leaf)) return refused;
|
||||
return 0;
|
||||
}
|
||||
|
||||
/// The verbs this backend implements. The three it leaves out — `mount`,
|
||||
/// `unmount`, `bind` — answer `-ENOSYS` from the generated dispatch, which is
|
||||
/// exactly right: path routing is the kernel's now, and only init implements
|
||||
/// `bind` (docs/os-development/protocol-namespace.md). `describe` is the
|
||||
/// envelope's own.
|
||||
const handlers = Serve.Handlers{
|
||||
.open = onOpen,
|
||||
.close = onClose,
|
||||
.read = onRead,
|
||||
.write = onWrite,
|
||||
.status = onStatus,
|
||||
.readdir = onReaddir,
|
||||
.mkdir = onMakeDirectory,
|
||||
.unlink = onUnlink,
|
||||
.rename = onRename,
|
||||
};
|
||||
|
||||
/// The vfs protocol has no operation that takes a capability, so `arrived` is
|
||||
/// never claimed here — which, under the harness's ownership rule, means the
|
||||
/// loop closes whatever a caller attached. That is the point of the rule: this
|
||||
/// callback used to discard a `?ipc.Handle` and every request carrying one — a
|
||||
/// legal thing for any client to do — spent a slot of the VFS server's
|
||||
/// thirty-two until it could accept no capability at all.
|
||||
fn onMessage(message: []const u8, out: []u8, sender: u32, arrived: *ipc.Arrival) usize {
|
||||
_ = arrived;
|
||||
// Storage not up (yet): fail politely, whatever was asked — clients retry.
|
||||
if (!mounted) {
|
||||
const status = envelope.Status{ .status = refused, .len = 0 };
|
||||
@memcpy(out[0..envelope.prefix_size], std.mem.asBytes(&status));
|
||||
return envelope.prefix_size;
|
||||
}
|
||||
// Stamp create/write with the current wall-clock time (mtime). Cheap, and it
|
||||
// keeps the engine pure (it takes the time as data, not a syscall).
|
||||
filesystem.current_time_epoch = time.wallClock();
|
||||
return Serve.dispatch({}, handlers, message, sender, null, out);
|
||||
}
|
||||
|
||||
pub fn main() void {
|
||||
service.run(vfs_protocol.message_maximum, .{
|
||||
.service = "vfs",
|
||||
.init = initialise,
|
||||
.on_message = onMessage,
|
||||
.on_notification = onNotification,
|
||||
});
|
||||
_ = logging.write("/system/services/fat: starting, waiting for a block device\n");
|
||||
Harness.run(.{ .bringUp = fatBringUp });
|
||||
}
|
||||
|
||||
@@ -0,0 +1,30 @@
|
||||
//! The volume-manager service as a binary package (docs/build-packages-plan.md):
|
||||
//! this file names the binary and EXACTLY the modules its source imports —
|
||||
//! build-support resolves each name from the domains this zon declares.
|
||||
|
||||
const std = @import("std");
|
||||
const build_support = @import("build-support");
|
||||
|
||||
pub fn build(b: *std.Build) void {
|
||||
const exe = build_support.userBinary(b, .{
|
||||
.name = "volume-manager",
|
||||
.root_source_file = b.path("volume-manager.zig"),
|
||||
.imports = &.{
|
||||
"block", "channel", "device-manager-protocol", "driver",
|
||||
"envelope", "ipc", "logging", "memory",
|
||||
"process", "service", "time", "volume-manager-protocol",
|
||||
},
|
||||
});
|
||||
b.installArtifact(exe);
|
||||
|
||||
// Standalone `zig build test` for the partition parser; the root build keeps
|
||||
// its aggregate test step.
|
||||
const test_step = b.step("test", "Run the partition-parser unit tests");
|
||||
const tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("partition.zig"),
|
||||
.target = b.resolveTargetQuery(.{}),
|
||||
}),
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(tests).step);
|
||||
}
|
||||
@@ -0,0 +1,16 @@
|
||||
.{
|
||||
.name = .volume_manager,
|
||||
.version = "0.0.0",
|
||||
.fingerprint = 0x911b01e1eaeabcf5, // Changing this has security and trust implications.
|
||||
.minimum_zig_version = "0.16.0",
|
||||
.dependencies = .{
|
||||
// build-support supplies the shared recipe; kernel is implicit in every
|
||||
// binary. device (block, driver) and protocol (device-manager-protocol,
|
||||
// envelope) are the homes of this service's remaining imports.
|
||||
.@"build-support" = .{ .path = "../../../build-support" },
|
||||
.kernel = .{ .path = "../../../library/kernel" },
|
||||
.device = .{ .path = "../../../library/device" },
|
||||
.protocol = .{ .path = "../../../library/protocol" },
|
||||
},
|
||||
.paths = .{""},
|
||||
}
|
||||
@@ -0,0 +1,116 @@
|
||||
//! Partition-table parsing, the policy the storage architecture places above the
|
||||
//! block driver and below the filesystem (docs/file-system-development/
|
||||
//! storage-architecture.md): read block 0, decide what block sub-ranges are
|
||||
//! volumes, and read each volume's content identity. The block DRIVER never does
|
||||
//! this — it clamps ranges it is told about; this is what tells it the numbers.
|
||||
//!
|
||||
//! Today: MBR (the four-entry table at offset 446) plus the bare-FAT case (a boot
|
||||
//! sector right at LBA 0). GPT is the next entry in the identity ladder and slots
|
||||
//! in here without touching anything above or below.
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
/// One volume the parser found on the device: the block sub-range it occupies
|
||||
/// and a content identity stable for the volume's life (the mount map keys on
|
||||
/// it; the boot volume is recorded by it). `identity` is derived from the medium,
|
||||
/// never from a port — a moved drive keeps it.
|
||||
pub const Volume = struct {
|
||||
base_lba: u64,
|
||||
block_count: u64,
|
||||
identity: u64,
|
||||
};
|
||||
|
||||
/// The MBR disk signature (offset 440, 4 bytes LE) — a 32-bit id written at
|
||||
/// partition time. Weak (dd-cloned disks share it) but on the medium, and the
|
||||
/// simplest rung of the identity ladder; the fuller rungs (GPT partition GUID,
|
||||
/// FAT volume serial) refine `identityOf` without changing the shape.
|
||||
fn diskSignature(block0: []const u8) u32 {
|
||||
if (block0.len < 444) return 0;
|
||||
return std.mem.readInt(u32, block0[440..444], .little);
|
||||
}
|
||||
|
||||
/// The identity of the volume at partition index `index`: the disk signature
|
||||
/// paired with the index, so two partitions of one disk stay distinct. For a
|
||||
/// bare FAT (no table) the index is 0.
|
||||
fn identityOf(block0: []const u8, index: u8) u64 {
|
||||
return (@as(u64, diskSignature(block0)) << 8) | index;
|
||||
}
|
||||
|
||||
/// Whether block 0 looks like a partition table (the 0x55AA boot signature). A
|
||||
/// bare FAT also carries it, so the caller distinguishes by whether any partition
|
||||
/// entry is non-empty.
|
||||
fn hasBootSignature(block0: []const u8) bool {
|
||||
return block0.len >= 512 and block0[510] == 0x55 and block0[511] == 0xAA;
|
||||
}
|
||||
|
||||
/// The first volume on a device whose block 0 is `block0` and whose whole-device
|
||||
/// size is `device_blocks`, or null if none is found. An MBR with a non-empty
|
||||
/// entry yields that partition's [start, size); otherwise a boot signature with
|
||||
/// no partitions is treated as a bare FAT spanning the whole device.
|
||||
pub fn firstVolume(block0: []const u8, device_blocks: u64) ?Volume {
|
||||
if (!hasBootSignature(block0)) return null;
|
||||
var index: u8 = 0;
|
||||
while (index < 4) : (index += 1) {
|
||||
const entry = block0[446 + @as(usize, index) * 16 ..][0..16];
|
||||
const kind = entry[4];
|
||||
const start = std.mem.readInt(u32, entry[8..12], .little);
|
||||
const size = std.mem.readInt(u32, entry[12..16], .little);
|
||||
if (kind == 0 or start == 0 or size == 0) continue;
|
||||
// These bytes come off an untrusted removable medium. A partition that
|
||||
// does not fit inside the device is not a partition — skip it. This is
|
||||
// where the driver's confinement-safety invariant is established: the
|
||||
// clamp's overflow-safety rests on base + count staying inside the
|
||||
// device (usb-storage.zig resolveTransfer), which only holds because the
|
||||
// range handed down is validated here. The subtraction cannot overflow.
|
||||
if (start > device_blocks or device_blocks - start < size) continue;
|
||||
return .{ .base_lba = start, .block_count = size, .identity = identityOf(block0, index) };
|
||||
}
|
||||
// No partition entries: a bare FAT spanning the device.
|
||||
return .{ .base_lba = 0, .block_count = device_blocks, .identity = identityOf(block0, 0) };
|
||||
}
|
||||
|
||||
test "an MBR with one partition yields its range and a distinct identity" {
|
||||
var block0 = [_]u8{0} ** 512;
|
||||
block0[510] = 0x55;
|
||||
block0[511] = 0xAA;
|
||||
std.mem.writeInt(u32, block0[440..444], 0xDEADBEEF, .little);
|
||||
// partition 0: type 0x0c (FAT32 LBA), start 2048, size 100000
|
||||
block0[446 + 4] = 0x0c;
|
||||
std.mem.writeInt(u32, block0[446 + 8 ..][0..4], 2048, .little);
|
||||
std.mem.writeInt(u32, block0[446 + 12 ..][0..4], 100000, .little);
|
||||
const v = firstVolume(&block0, 200000).?;
|
||||
try std.testing.expectEqual(@as(u64, 2048), v.base_lba);
|
||||
try std.testing.expectEqual(@as(u64, 100000), v.block_count);
|
||||
try std.testing.expectEqual((@as(u64, 0xDEADBEEF) << 8) | 0, v.identity);
|
||||
}
|
||||
|
||||
test "a boot signature with no partitions is a bare FAT over the whole device" {
|
||||
var block0 = [_]u8{0} ** 512;
|
||||
block0[510] = 0x55;
|
||||
block0[511] = 0xAA;
|
||||
const v = firstVolume(&block0, 65536).?;
|
||||
try std.testing.expectEqual(@as(u64, 0), v.base_lba);
|
||||
try std.testing.expectEqual(@as(u64, 65536), v.block_count);
|
||||
}
|
||||
|
||||
test "no boot signature is no volume" {
|
||||
const block0 = [_]u8{0} ** 512;
|
||||
try std.testing.expect(firstVolume(&block0, 65536) == null);
|
||||
}
|
||||
|
||||
test "a partition that runs past the device is skipped, not trusted" {
|
||||
var block0 = [_]u8{0} ** 512;
|
||||
block0[510] = 0x55;
|
||||
block0[511] = 0xAA;
|
||||
// partition 0: start 0xFFFFFF00, size 0x400 — far past a 200000-block device.
|
||||
block0[446 + 4] = 0x0c;
|
||||
std.mem.writeInt(u32, block0[446 + 8 ..][0..4], 0xFFFFFF00, .little);
|
||||
std.mem.writeInt(u32, block0[446 + 12 ..][0..4], 0x400, .little);
|
||||
// partition 1: start 2048, size 1000 — fits.
|
||||
block0[462 + 4] = 0x0c;
|
||||
std.mem.writeInt(u32, block0[462 + 8 ..][0..4], 2048, .little);
|
||||
std.mem.writeInt(u32, block0[462 + 12 ..][0..4], 1000, .little);
|
||||
const v = firstVolume(&block0, 200000).?;
|
||||
try std.testing.expectEqual(@as(u64, 2048), v.base_lba); // the fitting one, not the overflowing one
|
||||
try std.testing.expectEqual(@as(u64, 1000), v.block_count);
|
||||
}
|
||||
@@ -0,0 +1,342 @@
|
||||
//! system/services/volume-manager — the storage layer's policy home
|
||||
//! (docs/file-system-development/storage-architecture.md). Beside the device
|
||||
//! manager: that owns the DEVICE tree, this owns the VOLUME layer. It probes a
|
||||
//! storage provider's partition table, confines each filesystem to its
|
||||
//! partition, spawns one filesystem per volume, and answers that filesystem's
|
||||
//! startup hello with the range-confined block channel — so the filesystem
|
||||
//! never finds its storage by name and never sees the whole device. It
|
||||
//! supervises the filesystems it spawns, exactly as the device manager
|
||||
//! supervises drivers.
|
||||
//!
|
||||
//! This increment (V3b) is the flip: the FAT service stops acquiring its own
|
||||
//! volume and is spawned here instead, confined to its partition, and handed
|
||||
//! its channel over the volume-manager protocol. Single volume for now; the
|
||||
//! mount map (volumes.csv) and multi-volume land next.
|
||||
|
||||
const std = @import("std");
|
||||
const channel = @import("channel");
|
||||
const device_manager_protocol = @import("device-manager-protocol");
|
||||
const volume_manager_protocol = @import("volume-manager-protocol");
|
||||
const driver = @import("driver");
|
||||
const ipc = @import("ipc");
|
||||
const block = @import("block");
|
||||
const memory = @import("memory");
|
||||
const logging = @import("logging");
|
||||
const process = @import("process");
|
||||
const service = @import("service");
|
||||
const time = @import("time");
|
||||
const envelope = @import("envelope");
|
||||
const partition = @import("partition.zig");
|
||||
|
||||
const Serve = volume_manager_protocol.Protocol.Provider(void);
|
||||
const Invocation = envelope.Invocation;
|
||||
const Answer = envelope.Answer;
|
||||
|
||||
/// The single volume this increment handles: its provider channel, its block
|
||||
/// sub-range, its identity, the id it is addressed by, and the filesystem
|
||||
/// process serving it (0 until spawned; reset on death for respawn).
|
||||
const Volume = struct {
|
||||
storage: block.Device,
|
||||
storage_device_id: u64, // the device-manager id this volume's provider serves
|
||||
base_lba: u64,
|
||||
block_count: u64,
|
||||
identity: u64,
|
||||
id: u64,
|
||||
filesystem_pid: u32 = 0,
|
||||
};
|
||||
|
||||
/// The filesystem binary a probed volume is served by. The signature->binary
|
||||
/// map (filesystems.csv) lands with the identity ladder; for now every FAT-shaped
|
||||
/// volume gets the FAT service.
|
||||
const filesystem_binary = "/system/services/fat";
|
||||
const volume_id: u64 = 1;
|
||||
|
||||
var service_endpoint: ipc.Handle = 0;
|
||||
var manager_handle: ?ipc.Handle = null;
|
||||
var bounce: memory.DmaRegion = undefined;
|
||||
var bounce_ready = false;
|
||||
/// The currently-mounted volume, or null while no storage is present. The whole
|
||||
/// removal lifecycle is this field going null and back: the poll sees the
|
||||
/// storage provider leave the device tree (a pulled stick), kills the filesystem
|
||||
/// and clears this; when it returns, the poll re-acquires and re-mounts.
|
||||
var volume: ?Volume = null;
|
||||
var logged_no_volume = false;
|
||||
/// How often the poll checks whether the storage provider is present. Fast
|
||||
/// enough that an unplug unmounts promptly; the poll is a bare device-manager
|
||||
/// enumerate, no channel work, so it is cheap to run continuously.
|
||||
const poll_interval_ms = 500;
|
||||
|
||||
// Filesystem supervision, mirroring the device manager's (device-manager.zig):
|
||||
// a clean exit is not restarted, a fault restarts with backoff, and a fast
|
||||
// crash loop gives up rather than spinning. Without this a faulting filesystem
|
||||
// respawns in a zero-delay loop.
|
||||
const fast_death_ns: u64 = 2_000_000_000;
|
||||
const crash_loop_cap: u32 = 3;
|
||||
const backoff_base_ms: u64 = 300;
|
||||
var fs_restarts: u32 = 0;
|
||||
var fs_spawn_ns: u64 = 0;
|
||||
var fs_failed = false;
|
||||
/// A fat restart is due at `restart_due_ns`; the poll loop performs it once the
|
||||
/// backoff has elapsed (one timer, folded into the poll — no second timer).
|
||||
var restart_pending = false;
|
||||
var restart_due_ns: u64 = 0;
|
||||
|
||||
fn deviceManager() ?ipc.Handle {
|
||||
if (manager_handle) |h| return h;
|
||||
const handle = channel.openEndpoint("device-manager") orelse return null;
|
||||
manager_handle = handle;
|
||||
return handle;
|
||||
}
|
||||
|
||||
const OpenedStorage = struct { device_id: u64, device: block.Device };
|
||||
|
||||
/// The first mass-storage provider whose block channel actually opens, with its
|
||||
/// device id. A device-manager tree can carry more than one entry of the
|
||||
/// mass-storage identity — a phantom that no driver is bound to answers a
|
||||
/// consumer hello with NO channel — so this tries each and takes the first that
|
||||
/// yields a channel, exactly as a filesystem's own acquisition loop does.
|
||||
/// Called only when there is no volume (an insertion), so the hellos it makes
|
||||
/// are not per-poll churn.
|
||||
fn openAnyStorage() ?OpenedStorage {
|
||||
const manager = deviceManager() orelse return null;
|
||||
const Entry = device_manager_protocol.ChildEntry;
|
||||
var start: u64 = 0;
|
||||
while (true) {
|
||||
const enumerate = envelope.Header{ .operation = envelope.operation_enumerate, .target = start };
|
||||
var reply: [device_manager_protocol.message_maximum]u8 = undefined;
|
||||
const length = ipc.call(manager, std.mem.asBytes(&enumerate), &reply) catch return null;
|
||||
const status = envelope.statusOf(reply[0..length]) orelse return null;
|
||||
if (status.status != 0) return null;
|
||||
const carried = @min(@as(usize, status.len), length -| envelope.prefix_size);
|
||||
const tail = reply[envelope.prefix_size..][0..carried];
|
||||
const count = tail.len / @sizeOf(Entry);
|
||||
if (count == 0) return null;
|
||||
var index: usize = 0;
|
||||
while (index < count) : (index += 1) {
|
||||
const entry = std.mem.bytesToValue(Entry, tail[index * @sizeOf(Entry) ..][0..@sizeOf(Entry)]);
|
||||
if (entry.device_id == device_manager_protocol.no_device) continue;
|
||||
if ((entry.identity >> 16) & 0xff != 0x08 or (entry.identity >> 8) & 0xff != 0x06) continue;
|
||||
const exchanged = driver.helloOn(manager, .consumer, entry.device_id, null, true) orelse continue;
|
||||
const provider = exchanged.channel orelse continue; // a phantom / not-yet-bound entry
|
||||
return .{ .device_id = entry.device_id, .device = .{ .endpoint = provider } };
|
||||
}
|
||||
start += count;
|
||||
}
|
||||
}
|
||||
|
||||
/// Whether `device_id` is still in the device-manager tree — a bare enumerate,
|
||||
/// no consumer-hello, so it is cheap to call every poll. This is how removal is
|
||||
/// detected: the specific device the mounted volume sits on disappears.
|
||||
fn isDevicePresent(device_id: u64) bool {
|
||||
const manager = deviceManager() orelse return false;
|
||||
const Entry = device_manager_protocol.ChildEntry;
|
||||
var start: u64 = 0;
|
||||
while (true) {
|
||||
const enumerate = envelope.Header{ .operation = envelope.operation_enumerate, .target = start };
|
||||
var reply: [device_manager_protocol.message_maximum]u8 = undefined;
|
||||
const length = ipc.call(manager, std.mem.asBytes(&enumerate), &reply) catch return false;
|
||||
const status = envelope.statusOf(reply[0..length]) orelse return false;
|
||||
if (status.status != 0) return false;
|
||||
const carried = @min(@as(usize, status.len), length -| envelope.prefix_size);
|
||||
const tail = reply[envelope.prefix_size..][0..carried];
|
||||
const count = tail.len / @sizeOf(Entry);
|
||||
if (count == 0) return false;
|
||||
var index: usize = 0;
|
||||
while (index < count) : (index += 1) {
|
||||
const entry = std.mem.bytesToValue(Entry, tail[index * @sizeOf(Entry) ..][0..@sizeOf(Entry)]);
|
||||
if (entry.device_id == device_id) return true;
|
||||
}
|
||||
start += count;
|
||||
}
|
||||
}
|
||||
|
||||
/// Spawn the filesystem for `v`, confine it to the volume's range, and record
|
||||
/// its pid. The confinement is defined for the fresh pid BEFORE the filesystem
|
||||
/// runs, so its first read is already bounded; the volume manager is the
|
||||
/// confinement controller (it defines the first range on the device).
|
||||
fn spawnFilesystem(v: *Volume) void {
|
||||
if (fs_failed) return;
|
||||
const pid = process.spawnSupervised(filesystem_binary, &.{"1"}, service_endpoint) orelse {
|
||||
_ = logging.write("volume-manager: could not spawn the filesystem; retrying\n");
|
||||
armRestart();
|
||||
return;
|
||||
};
|
||||
if (!v.storage.defineRange(pid, v.base_lba, v.block_count)) {
|
||||
_ = logging.write("volume-manager: could not confine the filesystem to its volume; retrying\n");
|
||||
_ = process.kill(pid);
|
||||
armRestart();
|
||||
return;
|
||||
}
|
||||
v.filesystem_pid = pid;
|
||||
fs_spawn_ns = time.clock();
|
||||
std.log.info("volume 0x{x} -> {s} (pid {d}), lba {d}, {d} blocks", .{ v.identity, filesystem_binary, pid, v.base_lba, v.block_count });
|
||||
}
|
||||
|
||||
/// Schedule a fat restart after backoff; the poll loop performs it once due.
|
||||
fn armRestart() void {
|
||||
const delay = if (fs_restarts == 0) backoff_base_ms else backoff_base_ms << @intCast(@min(fs_restarts - 1, 5));
|
||||
restart_due_ns = time.clock() + delay * 1_000_000;
|
||||
restart_pending = true;
|
||||
}
|
||||
|
||||
/// A storage provider just appeared: open its channel, read block 0, parse the
|
||||
/// volume, and spawn its filesystem. On any failure the channel is closed (so a
|
||||
/// present-but-unreadable device does not leak a handle every poll) and `volume`
|
||||
/// stays null — the next poll retries. A fresh medium gets a fresh supervision
|
||||
/// budget.
|
||||
fn bringUpVolume() void {
|
||||
if (!bounce_ready) {
|
||||
bounce = memory.dmaAlloc(512, memory.dma_coherent | memory.dma_shareable) orelse return;
|
||||
bounce_ready = true;
|
||||
}
|
||||
const opened = openAnyStorage() orelse return;
|
||||
const device = opened.device;
|
||||
// Attach the read buffer to THIS device (a no-op without an enforcing IOMMU).
|
||||
// The handle is kept, not closed, so it can be re-attached to the next
|
||||
// device after a replug.
|
||||
if (bounce.handle) |handle| {
|
||||
if (!device.attach(handle)) {
|
||||
_ = ipc.close(device.endpoint);
|
||||
return;
|
||||
}
|
||||
}
|
||||
const geometry = device.geometry() orelse {
|
||||
_ = ipc.close(device.endpoint);
|
||||
return;
|
||||
};
|
||||
if (!device.read(0, 1, bounce.physical)) {
|
||||
_ = ipc.close(device.endpoint);
|
||||
return;
|
||||
}
|
||||
const sector: [*]const u8 = @ptrFromInt(bounce.virtual);
|
||||
const found = partition.firstVolume(sector[0..512], geometry.block_count) orelse {
|
||||
if (!logged_no_volume) {
|
||||
_ = logging.write("volume-manager: storage present but no recognizable volume\n");
|
||||
logged_no_volume = true;
|
||||
}
|
||||
_ = ipc.close(device.endpoint);
|
||||
return;
|
||||
};
|
||||
logged_no_volume = false;
|
||||
fs_restarts = 0;
|
||||
fs_failed = false;
|
||||
restart_pending = false;
|
||||
volume = .{ .storage = device, .storage_device_id = opened.device_id, .base_lba = found.base_lba, .block_count = found.block_count, .identity = found.identity, .id = volume_id };
|
||||
spawnFilesystem(&volume.?);
|
||||
}
|
||||
|
||||
/// The storage provider left the device tree (a pulled stick): kill the
|
||||
/// filesystem so its mounts are retired. Retirement is lazy, not an eager
|
||||
/// death-time sweep — killing the process marks the filesystem's backend
|
||||
/// endpoint dead, and the VFS router drops each mount that endpoint backed on
|
||||
/// the next path resolution under it (that resolve frees the slot and returns
|
||||
/// not_found). Then drop the now-dead channel and clear the volume; the next
|
||||
/// poll that sees storage return re-mounts.
|
||||
fn removeVolume() void {
|
||||
const v = volume orelse return;
|
||||
std.log.info("storage for volume {d} removed; unmounting", .{v.id});
|
||||
if (v.filesystem_pid != 0) _ = process.kill(v.filesystem_pid);
|
||||
_ = ipc.close(v.storage.endpoint);
|
||||
volume = null;
|
||||
restart_pending = false;
|
||||
fs_restarts = 0;
|
||||
fs_failed = false;
|
||||
}
|
||||
|
||||
/// One poll tick. Removal is checked FIRST and supersedes a pending restart: if
|
||||
/// the device is gone there is nothing to restart fat onto, and respawning it
|
||||
/// against the dead channel would just churn until the crash cap. Only once the
|
||||
/// device is confirmed present does a due restart fire.
|
||||
fn pollTick() void {
|
||||
if (volume) |v| {
|
||||
// Serving: watch for the specific device leaving (a pulled stick).
|
||||
if (!isDevicePresent(v.storage_device_id)) {
|
||||
removeVolume();
|
||||
return;
|
||||
}
|
||||
if (restart_pending and time.clock() >= restart_due_ns) {
|
||||
restart_pending = false;
|
||||
spawnFilesystem(&volume.?);
|
||||
}
|
||||
} else {
|
||||
// Idle: try to bring a present storage device up.
|
||||
bringUpVolume();
|
||||
}
|
||||
}
|
||||
|
||||
/// A filesystem announces itself for the volume it was spawned to serve. Reply
|
||||
/// with that volume's block channel (already range-confined to this filesystem's
|
||||
/// badge) as the call's returned capability. No channel means the volume is not
|
||||
/// ready — the filesystem retries.
|
||||
fn onHello(_: void, invocation: Invocation(volume_manager_protocol.Hello), _: Answer(void)) isize {
|
||||
const v = volume orelse return 0; // not probed yet — retryable, no cap
|
||||
if (invocation.target != v.id) return 0; // unknown volume — retryable
|
||||
if (invocation.sender != v.filesystem_pid) {
|
||||
// Not the filesystem we spawned for this volume. Refuse: only the
|
||||
// confined filesystem gets the channel.
|
||||
std.log.info("refused hello for volume {d} from process {d}", .{ invocation.target, invocation.sender });
|
||||
return -envelope.EPERM;
|
||||
}
|
||||
service.replyWithCapability(v.storage.endpoint);
|
||||
std.log.info("handed volume {d} to pid {d}", .{ v.id, invocation.sender });
|
||||
return 0;
|
||||
}
|
||||
|
||||
const handlers = Serve.Handlers{ .hello = onHello };
|
||||
|
||||
fn onMessage(message: []const u8, out: []u8, sender: u32, arrived: *ipc.Arrival) usize {
|
||||
// No verb takes a capability up, so the turn closes whatever arrives.
|
||||
return Serve.dispatch({}, handlers, message, sender, arrived.peek(), out);
|
||||
}
|
||||
|
||||
fn initialise(endpoint: ipc.Handle) bool {
|
||||
service_endpoint = endpoint;
|
||||
_ = logging.write("volume-manager: starting, waiting for a storage device\n");
|
||||
_ = process.subscribeExits(endpoint);
|
||||
pollTick();
|
||||
_ = time.timerOnce(endpoint, poll_interval_ms); // the poll runs for the life of the boot
|
||||
return true;
|
||||
}
|
||||
|
||||
fn onNotification(badge: u64) void {
|
||||
const got = ipc.Received{ .len = 0, .badge = badge, .cap = null };
|
||||
if (got.isTimer()) {
|
||||
pollTick();
|
||||
_ = time.timerOnce(service_endpoint, poll_interval_ms); // always re-arm: presence is watched continuously
|
||||
return;
|
||||
}
|
||||
// A filesystem died. The exit reason drives the decision, exactly as the
|
||||
// device manager supervises drivers: a clean exit meant to stop; a fault
|
||||
// restarts with backoff until a fast crash loop gives up. The old range is
|
||||
// reclaimed by the driver on the same death; the respawn confines afresh.
|
||||
if (got.isChildExit()) {
|
||||
const dead = got.childProcessId();
|
||||
const v = &(volume orelse return);
|
||||
if (v.filesystem_pid != dead) return;
|
||||
v.filesystem_pid = 0;
|
||||
const reason = process.exitReason(dead) orelse .fault;
|
||||
if (reason == .exited) {
|
||||
std.log.info("filesystem for volume {d} exited cleanly; not restarting", .{v.id});
|
||||
return;
|
||||
}
|
||||
const alive = time.clock() -| fs_spawn_ns;
|
||||
fs_restarts = if (alive < fast_death_ns) fs_restarts + 1 else 1;
|
||||
if (fs_restarts >= crash_loop_cap) {
|
||||
fs_failed = true;
|
||||
std.log.info("filesystem for volume {d} is failing repeatedly; giving up", .{v.id});
|
||||
return;
|
||||
}
|
||||
std.log.info("filesystem for volume {d} died ({s}); restarting", .{ v.id, @tagName(reason) });
|
||||
armRestart();
|
||||
}
|
||||
}
|
||||
|
||||
pub fn main(init: process.Init) void {
|
||||
_ = init;
|
||||
service.run(volume_manager_protocol.message_maximum, .{
|
||||
.service = "volume-manager",
|
||||
.init = initialise,
|
||||
.on_message = onMessage,
|
||||
.on_notification = onNotification,
|
||||
});
|
||||
}
|
||||
+63
-1
@@ -79,7 +79,9 @@ ARCHES = {
|
||||
"-device", "usb-kbd,bus=xhci.0",
|
||||
"-device", "usb-mouse,bus=xhci.0",
|
||||
"-drive", f"if=none,id=bootusb,format=raw,file={boot_volume}",
|
||||
"-device", "usb-storage,bus=xhci.0,drive=bootusb,removable=on,bootindex=0",
|
||||
# id=bootstorage + an explicit port so the volume-replug drill can
|
||||
# device_del/device_add it back onto the same freed root port.
|
||||
"-device", "usb-storage,bus=xhci.0,port=3,drive=bootusb,removable=on,bootindex=0,id=bootstorage",
|
||||
"-net", "none",
|
||||
"-vga", "none", "-device", "VGA,edid=on,xres=1280,yres=720",
|
||||
"-display", "none",
|
||||
@@ -755,6 +757,49 @@ CASES = [
|
||||
"timeout": 150,
|
||||
"expect": r"fat: mounted /volumes/usb[\s\S]*fat-test: ok",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# The removal lifecycle (V4, docs/volume-manager-plan.md): pull the boot stick
|
||||
# mid-run. device_del the usb-storage device -> the bus reports the port empty
|
||||
# -> the device manager reaps usb-storage -> the mass-storage child leaves the
|
||||
# tree -> the volume manager's poll sees it gone and kills the FAT service, so
|
||||
# its mounts retire (an honest unmount). The tail (mounted -> removed) can only
|
||||
# be the removal, since the mount precedes the unplug. Discrimination: before
|
||||
# V4 the volume manager stopped polling after the first probe, so it never
|
||||
# noticed the removal — this line is absent.
|
||||
#
|
||||
# The RE-mount on replug is not asserted here: QEMU's device_add of usb-storage
|
||||
# to the boot xHCI controller is not re-presented to the guest (no port-connect
|
||||
# on any port), so it cannot drive the reappearance in this harness. On real
|
||||
# hardware the bus's per-tick port poll catches a reconnect's PORTSC change
|
||||
# (H1/usb-root-replug proves reconnect works on a second controller), and the
|
||||
# volume manager's bringUpVolume remounts when the device returns — bench-
|
||||
# verified, not QEMU-verified. So this case proves the unmount half.
|
||||
{"name": "volume-removal",
|
||||
"build_case": "fat-mount",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"qmp_sequence": [
|
||||
{"delay": 8, "command": "device_del", "arguments": {"id": "bootstorage"}},
|
||||
],
|
||||
"expect": r"(?s)fat: mounted /volumes/usb"
|
||||
r"[\s\S]*volume-manager: storage for volume \d+ removed; unmounting",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# Volume-manager discovery + probe (V3a, docs/volume-manager-plan.md). Reuses
|
||||
# the fat-mount kernel build (the default boot now spawns the volume manager
|
||||
# from init.csv). It acquires the mass-storage block channel through the
|
||||
# device manager, reads block 0, and parses the first volume out of it — the
|
||||
# partition-table walk that used to live in the FAT engine, now above the
|
||||
# driver where it belongs. Before V3a the service did not exist, so this line
|
||||
# is absent.
|
||||
{"name": "volume-probe",
|
||||
"build_case": "fat-mount",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
# The volume manager probes the partition table, then confines a filesystem
|
||||
# to the volume and hands it over — one log line naming the volume's range,
|
||||
# its identity, and the filesystem it spawned for it.
|
||||
"expect": r"volume-manager: volume 0x[0-9a-f]+ -> \S+ \(pid \d+\), lba \d+, \d+ blocks"
|
||||
r"[\s\S]*volume-manager: handed volume \d+ to pid \d+",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# Phase 2b: mkdir/unlink through the mount. Reuses the fat-mount build — the
|
||||
# fat-test client, after listing, makes a directory, writes+reads a file inside
|
||||
# it, then removes the file, exercising the whole VFS -> fat mutation path.
|
||||
@@ -1234,6 +1279,23 @@ CASES = [
|
||||
{"name": "device-authority",
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# Per-sender range confinement (V2a, docs/volume-manager-plan.md): a process
|
||||
# confines ITSELF to a block sub-range (as the volume manager confines a
|
||||
# filesystem), then proves it cannot read past the range nor widen it. The
|
||||
# security assertions are named explicitly so the case cannot pass without
|
||||
# them; a confined read crossing the range must be REFUSED and a confined
|
||||
# define_range must be REFUSED. Against pre-clamp usb-storage the define_range
|
||||
# verb does not exist, so the fixture fails to arm confinement at all.
|
||||
{"name": "block-range",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"expect": r"(?s)(?=.*block-range: ok in-range-read)"
|
||||
r"(?=.*block-range: ok out-of-range-refused)"
|
||||
r"(?=.*block-range: ok wrap-refused)"
|
||||
r"(?=.*block-range: ok geometry-is-confined)"
|
||||
r"(?=.*block-range: ok confined-cannot-redefine)"
|
||||
r"(?=.*block-range: VERDICT done)",
|
||||
"fail": r"block-range: FAILED|DANOS-TEST-RESULT: FAIL"},
|
||||
# IRQ teardown: an exiting driver's line is masked and its slot cleared (so no
|
||||
# ISR notifies a freed endpoint), and a sibling owner sharing that endpoint
|
||||
# keeps its own binding. A long-running driver never reaches this teardown path.
|
||||
|
||||
@@ -0,0 +1,154 @@
|
||||
//! block-range-test — the discrimination fixture for per-sender range
|
||||
//! confinement (V2a, docs/volume-manager-plan.md). It gets a block channel the
|
||||
//! way a filesystem does (consumer-hello the device manager for the mass-storage
|
||||
//! provider), then proves the two properties the clamp exists for:
|
||||
//!
|
||||
//! 1. an UNCONFINED caller may define a range on its own badge (the volume
|
||||
//! manager is unconfined — this stands in for it);
|
||||
//! 2. once confined, a transfer PAST the range is refused, and the volume
|
||||
//! relative LBA 0 maps inside the range (the clamp translates + bounds);
|
||||
//! 3. a CONFINED caller may NOT call define_range again (the gate — a
|
||||
//! filesystem cannot widen its own range or escape).
|
||||
//!
|
||||
//! Against pre-clamp usb-storage the verb does not exist, so (1) already fails —
|
||||
//! which is exactly the discrimination: the fixture cannot even arm confinement,
|
||||
//! let alone see a transfer refused for crossing it.
|
||||
//!
|
||||
//! It coexists with the FAT service in the same boot: ranges are per-badge, so
|
||||
//! confining THIS process touches nothing fat does on its own channel.
|
||||
|
||||
const std = @import("std");
|
||||
const channel = @import("channel");
|
||||
const device_manager_protocol = @import("device-manager-protocol");
|
||||
const driver = @import("driver");
|
||||
const ipc = @import("ipc");
|
||||
const block = @import("block");
|
||||
const memory = @import("memory");
|
||||
const logging = @import("logging");
|
||||
const process = @import("process");
|
||||
const time = @import("time");
|
||||
const envelope = @import("envelope");
|
||||
|
||||
fn verdict(ok: bool, name: []const u8) void {
|
||||
_ = logging.write("block-range: ");
|
||||
_ = logging.write(if (ok) "ok " else "FAILED ");
|
||||
_ = logging.write(name);
|
||||
_ = logging.write("\n");
|
||||
}
|
||||
|
||||
/// The mass-storage provider's block channel, via the device manager's tree —
|
||||
/// the same lineage acquisition the FAT service uses (block is not a name).
|
||||
fn acquireBlock() ?block.Device {
|
||||
var tries: u32 = 0;
|
||||
const manager = while (tries < 200) : (tries += 1) {
|
||||
if (channel.openEndpoint("device-manager")) |h| break h;
|
||||
time.sleepMillis(20);
|
||||
} else return null;
|
||||
|
||||
// The whole USB storage chain (enumeration, bring-up) takes a few seconds to
|
||||
// appear in the manager's tree, so retry the enumerate-and-hello with a pause
|
||||
// between rounds — 500 x 20 ms ~ 10 s, well within the case timeout.
|
||||
const Entry = device_manager_protocol.ChildEntry;
|
||||
var attempt: u32 = 0;
|
||||
while (attempt < 500) : (attempt += 1) {
|
||||
var start: u64 = 0;
|
||||
while (true) {
|
||||
const enumerate = envelope.Header{ .operation = envelope.operation_enumerate, .target = start };
|
||||
var reply: [device_manager_protocol.message_maximum]u8 = undefined;
|
||||
const length = ipc.call(manager, std.mem.asBytes(&enumerate), &reply) catch break;
|
||||
const status = envelope.statusOf(reply[0..length]) orelse break;
|
||||
if (status.status != 0) break;
|
||||
const carried = @min(@as(usize, status.len), length -| envelope.prefix_size);
|
||||
const tail = reply[envelope.prefix_size..][0..carried];
|
||||
const count = tail.len / @sizeOf(Entry);
|
||||
if (count == 0) break;
|
||||
var index: usize = 0;
|
||||
while (index < count) : (index += 1) {
|
||||
const entry = std.mem.bytesToValue(Entry, tail[index * @sizeOf(Entry) ..][0..@sizeOf(Entry)]);
|
||||
if (entry.device_id == device_manager_protocol.no_device) continue;
|
||||
if ((entry.identity >> 16) & 0xff != 0x08 or (entry.identity >> 8) & 0xff != 0x06) continue;
|
||||
const exchanged = driver.helloOn(manager, .consumer, entry.device_id, null, true) orelse break;
|
||||
const provider = exchanged.channel orelse continue;
|
||||
return .{ .endpoint = provider };
|
||||
}
|
||||
start += count;
|
||||
}
|
||||
time.sleepMillis(20);
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
pub fn main(init: process.Init) void {
|
||||
// Bundled fixtures are swept up and spawned bare on every boot; stay silent
|
||||
// unless the kernel test explicitly runs us, or we would contend for the
|
||||
// block channel and print markers into unrelated cases.
|
||||
const arg = init.arguments.get(1) orelse return;
|
||||
if (!std.mem.eql(u8, arg, "run")) return;
|
||||
|
||||
const device = acquireBlock() orelse {
|
||||
verdict(false, "acquire-block");
|
||||
return;
|
||||
};
|
||||
const geometry = device.geometry() orelse {
|
||||
verdict(false, "geometry");
|
||||
return;
|
||||
};
|
||||
// Need at least a few blocks to carve a range out of; every FAT image is far
|
||||
// larger, so this only guards a nonsense device.
|
||||
if (geometry.block_count < 4) {
|
||||
verdict(false, "device-too-small");
|
||||
return;
|
||||
}
|
||||
|
||||
// A one-block DMA buffer for the positive-control read. Shareable so it can be
|
||||
// attached under an enforcing IOMMU (a no-op success otherwise).
|
||||
const bounce = memory.dmaAlloc(512, memory.dma_coherent | memory.dma_shareable) orelse {
|
||||
verdict(false, "dma-alloc");
|
||||
return;
|
||||
};
|
||||
if (bounce.handle) |handle| {
|
||||
if (!device.attach(handle)) {
|
||||
verdict(false, "attach");
|
||||
return;
|
||||
}
|
||||
_ = ipc.close(handle);
|
||||
}
|
||||
|
||||
// Baseline: an unconfined read of block 0 succeeds — so a later refusal is
|
||||
// the clamp, not a broken read path.
|
||||
verdict(device.read(0, 1, bounce.physical), "unconfined-read");
|
||||
|
||||
const me = process.taskId();
|
||||
|
||||
// (1) An unconfined caller confines itself to blocks [1, 3). Against pre-clamp
|
||||
// usb-storage this verb does not exist and the call fails here.
|
||||
if (!device.defineRange(me, 1, 2)) {
|
||||
verdict(false, "define-range");
|
||||
return;
|
||||
}
|
||||
verdict(true, "define-range");
|
||||
|
||||
// (2) Confined now: volume-relative LBA 0 maps to device block 1 (inside the
|
||||
// range) and succeeds; LBA 2 would reach device block 3, past the 2-block
|
||||
// range, and must be refused.
|
||||
verdict(device.read(0, 1, bounce.physical), "in-range-read");
|
||||
verdict(!device.read(2, 1, bounce.physical), "out-of-range-refused");
|
||||
// The wrap attack, calibrated to be exploitable against a naive bound: this
|
||||
// process is confined with base 1, so a volume-relative LBA of maxInt(u64)
|
||||
// makes base + lba wrap to absolute block 0 — a real, readable block OUTSIDE
|
||||
// the range (the boot sector). A naive `lba + count > count` check also
|
||||
// wraps to 0 and waves it through; the overflow-safe bound refuses it.
|
||||
verdict(!device.read(std.math.maxInt(u64), 1, bounce.physical), "wrap-refused");
|
||||
|
||||
// Geometry now reports the CONFINED size, not the device's.
|
||||
const confined = device.geometry() orelse {
|
||||
verdict(false, "confined-geometry");
|
||||
return;
|
||||
};
|
||||
verdict(confined.block_count == 2, "geometry-is-confined");
|
||||
|
||||
// (3) The gate: a confined caller cannot define_range — no widening, no escape.
|
||||
verdict(!device.defineRange(me, 0, geometry.block_count), "confined-cannot-redefine");
|
||||
|
||||
_ = logging.write("block-range: VERDICT done\n");
|
||||
}
|
||||
@@ -0,0 +1,19 @@
|
||||
//! The block-range-test fixture as a binary package (docs/build-packages-plan.md):
|
||||
//! this file names the binary and EXACTLY the modules its source imports —
|
||||
//! build-support resolves each name from the domains this zon declares.
|
||||
|
||||
const std = @import("std");
|
||||
const build_support = @import("build-support");
|
||||
|
||||
pub fn build(b: *std.Build) void {
|
||||
const exe = build_support.userBinary(b, .{
|
||||
.name = "block-range-test",
|
||||
.root_source_file = b.path("block-range-test.zig"),
|
||||
.imports = &.{
|
||||
"block", "channel", "device-manager-protocol", "driver",
|
||||
"envelope", "ipc", "logging", "memory",
|
||||
"process", "time",
|
||||
},
|
||||
});
|
||||
b.installArtifact(exe);
|
||||
}
|
||||
@@ -0,0 +1,16 @@
|
||||
.{
|
||||
.name = .block_range_test,
|
||||
.version = "0.0.0",
|
||||
.fingerprint = 0xa7f2045ed72783c3, // Changing this has security and trust implications.
|
||||
.minimum_zig_version = "0.16.0",
|
||||
.dependencies = .{
|
||||
// build-support supplies the shared recipe; kernel is implicit in every
|
||||
// binary. device (block, driver) and protocol (device-manager-protocol,
|
||||
// envelope) are the homes of this fixture's remaining imports.
|
||||
.@"build-support" = .{ .path = "../../../../build-support" },
|
||||
.kernel = .{ .path = "../../../../library/kernel" },
|
||||
.device = .{ .path = "../../../../library/device" },
|
||||
.protocol = .{ .path = "../../../../library/protocol" },
|
||||
},
|
||||
.paths = .{""},
|
||||
}
|
||||
@@ -84,6 +84,28 @@ fn park() void {
|
||||
_ = logging.write("vfstest: park open failed\n");
|
||||
return;
|
||||
}
|
||||
|
||||
// Mount ownership (V0, docs/volume-manager-plan.md): the volume is
|
||||
// provably mounted (the parked file just opened on it), it is FAT's mount,
|
||||
// and this process is not fat — unmounting it must be REFUSED and the
|
||||
// subtree must still resolve afterwards. Bailing here withholds the
|
||||
// "parked" marker, which fails the vfs-client-death case: before the
|
||||
// ownership gate existed, any process could unmount any prefix, and this
|
||||
// fixture would have deleted the volume out from under the whole boot.
|
||||
if (fs.fsUnmount("/volumes/usb")) {
|
||||
_ = logging.write("vfstest: foreign unmount was ALLOWED\n");
|
||||
return;
|
||||
}
|
||||
if (fs.open("/volumes/usb/parked", .{})) |resolved| {
|
||||
var verification = resolved;
|
||||
verification.close(); // the park below must be the client's ONLY open
|
||||
// handle — the kernel test string-matches "released 1 handle(s)".
|
||||
} else {
|
||||
_ = logging.write("vfstest: /volumes/usb gone after refused unmount\n");
|
||||
return;
|
||||
}
|
||||
_ = logging.write("vfstest: foreign unmount refused\n");
|
||||
|
||||
while (true) {
|
||||
_ = logging.write("vfstest: parked\n");
|
||||
time.sleepMillis(500);
|
||||
|
||||
Reference in New Issue
Block a user