diff --git a/library/protocol/build.zig b/library/protocol/build.zig index 2a81968..371e13e 100644 --- a/library/protocol/build.zig +++ b/library/protocol/build.zig @@ -31,6 +31,7 @@ pub fn build(b: *std.Build) void { .{ .name = "block-protocol", .root = "block/block-protocol.zig" }, .{ .name = "usb-transfer-protocol", .root = "usb-transfer/usb-transfer-protocol.zig" }, .{ .name = "device-manager-protocol", .root = "device-manager/device-manager-protocol.zig" }, + .{ .name = "volume-manager-protocol", .root = "volume-manager/volume-manager-protocol.zig" }, .{ .name = "display-protocol", .root = "display/display-protocol.zig" }, .{ .name = "scanout-protocol", .root = "scanout/scanout-protocol.zig" }, .{ .name = "power-protocol", .root = "power/power-protocol.zig" }, diff --git a/library/protocol/volume-manager/volume-manager-protocol.zig b/library/protocol/volume-manager/volume-manager-protocol.zig new file mode 100644 index 0000000..9fa9f21 --- /dev/null +++ b/library/protocol/volume-manager/volume-manager-protocol.zig @@ -0,0 +1,36 @@ +//! The volume-manager protocol (docs/file-system-development/storage-architecture.md): +//! what a filesystem service says to the volume manager over +//! `/protocol/volume-manager`. Defined through the envelope, so every packet +//! begins with the folded `Header`. +//! +//! One verb. A filesystem the volume manager spawned announces itself with the +//! volume id it was given as argv[1] (folded into `Header.target`); the reply +//! carries that volume's block channel — already range-confined to the +//! filesystem's badge — as the call's returned capability. The filesystem never +//! finds its storage by name and never sees the whole device; establishment is +//! by lineage, exactly as a driver reaches its controller (communication.md +//! "Establishment: two planes"). No channel in the reply means the volume is not +//! ready yet — retryable, never a verdict. + +const envelope = @import("envelope"); + +pub const version: u16 = 1; + +/// The filesystem's handshake. Carries only its protocol version; the volume it +/// serves is `Header.target`, and the block channel it needs comes back as the +/// reply's capability. +pub const Hello = extern struct { + version: u16 = version, + _padding: u16 = 0, +}; + +pub const Protocol = envelope.Define(.{ + .name = "volume-manager", + .version = 1, + .operations = &.{ + .{ .name = "hello", .request = Hello }, + }, +}); + +pub const Operation = Protocol.Operation; +pub const message_maximum: usize = Protocol.message_maximum; diff --git a/system/configuration/init.csv b/system/configuration/init.csv index 847cbcb..7b31032 100644 --- a/system/configuration/init.csv +++ b/system/configuration/init.csv @@ -14,8 +14,9 @@ # service args... /system/services/input /system/services/device-manager +# fat is not here: the volume manager spawns one filesystem per volume it finds, +# confined to that volume's partition (docs/file-system-development/storage-architecture.md). /system/services/volume-manager -/system/services/fat /system/services/display /system/services/display-demo /system/services/logger diff --git a/system/configuration/protocol.csv b/system/configuration/protocol.csv index 92a8327..f749122 100644 --- a/system/configuration/protocol.csv +++ b/system/configuration/protocol.csv @@ -56,7 +56,9 @@ /system/services/input, /system/services/init, bind, input /system/services/device-manager, /system/services/init, bind, device-manager /system/services/volume-manager, /system/services/init, bind, volume-manager -/system/services/fat, /system/services/init, bind, vfs +# fat is spawned and supervised by the volume manager now, not init — the volume +# manager confines it to its partition and hands it the block channel. +/system/services/fat, /system/services/volume-manager, bind, vfs /system/services/display, /system/services/init, bind, display # The discovery service ships under one neutral name per firmware (docs/discovery.md); @@ -95,14 +97,13 @@ # ============================================================================ # --- init's own services ---------------------------------------------------- -# fat reaches the device manager to be routed to its volume's block provider -# (block is not a name — see the bind section); the compositor reaches the -# scanout its driver announced, its own endpoint (the mouse-listener thread -# opens /protocol/display like any other client — threads share no handles), -# and the input stream that moves the cursor. -/system/services/fat, /system/services/init, open, device-manager +# fat reaches the volume manager to be handed its volume's block channel +# (range-confined); the compositor reaches the scanout its driver announced, its +# own endpoint (the mouse-listener thread opens /protocol/display like any other +# client — threads share no handles), and the input stream that moves the cursor. +/system/services/fat, /system/services/volume-manager, open, volume-manager # The volume manager reaches the device manager to be routed to each storage -# provider's block channel, the same lineage acquisition fat makes today. +# provider's block channel, then confines a filesystem to each volume. /system/services/volume-manager, /system/services/init, open, device-manager /system/services/display, /system/services/init, open, scanout /system/services/display, /system/services/init, open, display diff --git a/system/kernel/tests.zig b/system/kernel/tests.zig index 2c323b4..c707998 100644 --- a/system/kernel/tests.zig +++ b/system/kernel/tests.zig @@ -3036,11 +3036,13 @@ fn fatMountTest(boot_information: *const BootInformation) void { result(); } -/// Per-sender range confinement (V2a, docs/volume-manager-plan.md): boot the -/// full tree so the USB storage chain is up, then spawn block-range-test, which +/// Per-sender range confinement (V2a, docs/volume-manager-plan.md): the fixture /// acquires the block channel, confines ITSELF to a sub-range, and asserts it -/// cannot read past that range or widen it. The fixture's markers are the -/// assertion (the QEMU expect regex matches them); this only boots and spawns. +/// cannot read past that range or widen it. Boots init in REGISTRY-ONLY mode +/// plus the device manager (which brings up the USB storage chain) — deliberately +/// NOT the full tree, because the volume manager would take the confinement +/// controller first and refuse the fixture's define_range. Without it the fixture +/// is the sole definer, exactly as the volume manager is in a real boot. fn blockRangeTest(boot_information: *const BootInformation) void { log("DANOS-TEST-BEGIN: block-range\n", .{}); if (boot_information.initial_ramdisk_len == 0) { @@ -3055,8 +3057,8 @@ fn blockRangeTest(boot_information: *const BootInformation) void { return; }; process.setInitialRamdisk(ramdisk); - const spawned = if (process.spawnBundled("/system/services/init")) true else |_| false; - check("init spawned (boots the USB storage chain)", spawned); + check("registry (init) spawned", spawnRegistry(rd)); + check("device-manager spawned (boots the USB storage chain)", spawnNamed(rd, "device-manager")); check("block-range-test spawned", spawnNamedWithArg(rd, "block-range-test", "run")); result(); } diff --git a/system/services/fat/build.zig b/system/services/fat/build.zig index b398bfc..b9141f0 100644 --- a/system/services/fat/build.zig +++ b/system/services/fat/build.zig @@ -10,9 +10,9 @@ pub fn build(b: *std.Build) void { .name = "fat", .root_source_file = b.path("fat.zig"), .imports = &.{ - "block", "channel", "device-manager-protocol", - "driver", "envelope", "file-system-harness", - "ipc", "logging", "memory", + "block", "channel", "envelope", "file-system-harness", + "ipc", "logging", "memory", "process", + "time", "volume-manager-protocol", }, }); b.installArtifact(exe); diff --git a/system/services/fat/fat.zig b/system/services/fat/fat.zig index 700ab91..2146ea5 100644 --- a/system/services/fat/fat.zig +++ b/system/services/fat/fat.zig @@ -13,12 +13,13 @@ const std = @import("std"); const channel = @import("channel"); -const device_manager_protocol = @import("device-manager-protocol"); -const driver = @import("driver"); +const volume_manager_protocol = @import("volume-manager-protocol"); const ipc = @import("ipc"); +const process = @import("process"); const block = @import("block"); const memory = @import("memory"); const logging = @import("logging"); +const time = @import("time"); const engine = @import("engine.zig"); const envelope = @import("envelope"); const harness = @import("file-system-harness"); @@ -58,9 +59,10 @@ var ipc_block: IpcBlock = undefined; // file close — so writes are committed to stable media before a power-off. var device_dirty: bool = false; var filesystem: engine.FileSystem = undefined; -/// The one channel to the device manager, opened on first need and kept — the -/// poll retries on it, never spending a handle-table slot per attempt. -var manager_handle: ?ipc.Handle = null; +/// The volume this FAT process serves, its id given as argv[1] by the volume +/// manager that spawned it. The startup hello names it so the manager returns +/// the right volume's channel. +var my_volume_id: u64 = 0; /// The prefixes this volume installs: /volumes/usb from the volume root, plus /// the two hierarchy subtrees the boot volume carries (rewrite == prefix), so @@ -72,51 +74,37 @@ const fat_mounts = [_]harness.MountSpec{ .{ .prefix = "/system/logs", .rewrite = "/system/logs" }, }; -/// Find the volume's provider through the device manager (establishment by +/// Get this volume's block channel from the volume manager (establishment by /// lineage, communication.md "Establishment: two planes" — `block` is not a -/// registry name; one storage process serves each stick): enumerate the -/// manager's tree, take the FIRST usb mass-storage child by enumeration order -/// (deterministic within a boot; single-volume by construction, and choosing -/// the BOOT volume by content when two sticks are present is the volume-manager -/// track), and consumer-hello for the channel of the driver bound to it. -/// Null until the chain is up — the harness's poll retries. +/// registry name). The manager spawned this process, confined it to its +/// partition, and answers the hello with the channel; the channel is +/// range-confined to this process's badge, so reads and writes are +/// volume-relative and cannot reach the neighbouring partition. Null until the +/// manager has the volume ready — this retries. fn acquireVolume() ?block.Device { - const manager = manager_handle orelse opened: { - const handle = channel.openEndpoint("device-manager") orelse return null; - manager_handle = handle; - break :opened handle; - }; + var attempts: u32 = 0; + const vm = while (attempts < 500) : (attempts += 1) { + if (channel.openEndpoint("volume-manager")) |handle| break handle; + time.sleepMillis(20); + } else return null; - // The envelope's reserved `enumerate` verb, PAGED: one reply carries only - // a handful of entries and a real tree (a dozen ACPI nodes before the - // first USB child) is bigger, so `Header.target` is the start cursor and - // a short page is the end. Identity is the bus's native triple, for USB - // (base << 16) | (class << 8) | protocol — mass storage is base 0x08, - // subclass 0x06 (SCSI transparent), the same key devices.csv matches on. - const Entry = device_manager_protocol.ChildEntry; - var start: u64 = 0; - while (true) { - const enumerate = envelope.Header{ .operation = envelope.operation_enumerate, .target = start }; - var reply: [device_manager_protocol.message_maximum]u8 = undefined; - const length = ipc.call(manager, std.mem.asBytes(&enumerate), &reply) catch return null; - const status = envelope.statusOf(reply[0..length]) orelse return null; - if (status.status != 0) return null; - - const carried = @min(@as(usize, status.len), length -| envelope.prefix_size); - const tail = reply[envelope.prefix_size..][0..carried]; - const count = tail.len / @sizeOf(Entry); - if (count == 0) return null; // the tree is exhausted; no volume yet - var index: usize = 0; - while (index < count) : (index += 1) { - const entry = std.mem.bytesToValue(Entry, tail[index * @sizeOf(Entry) ..][0..@sizeOf(Entry)]); - if (entry.device_id == device_manager_protocol.no_device) continue; - if ((entry.identity >> 16) & 0xff != 0x08 or (entry.identity >> 8) & 0xff != 0x06) continue; - const exchanged = driver.helloOn(manager, .consumer, entry.device_id, null, true) orelse return null; - const provider = exchanged.channel orelse continue; // its driver not up yet — next tick - return .{ .endpoint = provider }; + attempts = 0; + while (attempts < 500) : (attempts += 1) { + var packet: [volume_manager_protocol.message_maximum]u8 = undefined; + const framed = volume_manager_protocol.Protocol.encodeRequest(.hello, my_volume_id, .{}, &.{}, &packet) orelse return null; + var reply: [volume_manager_protocol.message_maximum]u8 = undefined; + const answered = ipc.callCap(vm, framed, &reply, null) catch return null; + const status = envelope.statusOf(reply[0..answered.len]) orelse return null; + if (status.status != 0) { + if (answered.cap) |stray| _ = ipc.close(stray); + _ = logging.write("/system/services/fat: volume manager refused the hello\n"); + return null; } - start += count; + if (answered.cap) |bus| return .{ .endpoint = bus }; + // Acked with no channel: the volume is not ready yet — retry. + time.sleepMillis(20); } + return null; } /// Durable-on-close: commit the device write cache if any block reached it since @@ -179,7 +167,11 @@ fn fatBringUp(endpoint: ipc.Handle) ?Harness.Volume { return .{ .engine = &filesystem, .mounts = &fat_mounts, .flush = flushIfDirty }; } -pub fn main() void { +pub fn main(init: process.Init) void { + // The volume manager spawns this process with its volume id as argv[1]. + if (init.arguments.get(1)) |id| { + my_volume_id = std.fmt.parseInt(u64, id, 10) catch 0; + } _ = logging.write("/system/services/fat: starting, waiting for a block device\n"); Harness.run(.{ .bringUp = fatBringUp }); } diff --git a/system/services/volume-manager/build.zig b/system/services/volume-manager/build.zig index 5b353c1..aa567e0 100644 --- a/system/services/volume-manager/build.zig +++ b/system/services/volume-manager/build.zig @@ -10,9 +10,9 @@ pub fn build(b: *std.Build) void { .name = "volume-manager", .root_source_file = b.path("volume-manager.zig"), .imports = &.{ - "block", "channel", "device-manager-protocol", "driver", - "envelope", "ipc", "logging", "memory", - "process", "service", "time", + "block", "channel", "device-manager-protocol", "driver", + "envelope", "ipc", "logging", "memory", + "process", "service", "time", "volume-manager-protocol", }, }); b.installArtifact(exe); diff --git a/system/services/volume-manager/volume-manager.zig b/system/services/volume-manager/volume-manager.zig index fb756be..bc10b28 100644 --- a/system/services/volume-manager/volume-manager.zig +++ b/system/services/volume-manager/volume-manager.zig @@ -1,20 +1,22 @@ //! system/services/volume-manager — the storage layer's policy home -//! (docs/file-system-development/storage-architecture.md). It sits beside the -//! device manager: the device manager owns the DEVICE tree; this owns the VOLUME -//! layer. It hears about storage providers, probes their partition tables and -//! content identity, and — in later increments — confines each filesystem to its -//! partition and spawns one per volume, answering that filesystem's startup -//! hello with the (range-confined) block channel. +//! (docs/file-system-development/storage-architecture.md). Beside the device +//! manager: that owns the DEVICE tree, this owns the VOLUME layer. It probes a +//! storage provider's partition table, confines each filesystem to its +//! partition, spawns one filesystem per volume, and answers that filesystem's +//! startup hello with the range-confined block channel — so the filesystem +//! never finds its storage by name and never sees the whole device. It +//! supervises the filesystems it spawns, exactly as the device manager +//! supervises drivers. //! -//! This increment (V3a) is discovery and probe only: find the mass-storage -//! provider, read block 0, parse the first volume out of it, and log what it -//! found — additive, with the FAT service still acquiring its own volume. The -//! delegation (confine + spawn + hand over the channel) and the mount map land -//! next, keeping the FAT service working throughout. +//! This increment (V3b) is the flip: the FAT service stops acquiring its own +//! volume and is spawned here instead, confined to its partition, and handed +//! its channel over the volume-manager protocol. Single volume for now; the +//! mount map (volumes.csv) and multi-volume land next. const std = @import("std"); const channel = @import("channel"); const device_manager_protocol = @import("device-manager-protocol"); +const volume_manager_protocol = @import("volume-manager-protocol"); const driver = @import("driver"); const ipc = @import("ipc"); const block = @import("block"); @@ -26,16 +28,36 @@ const time = @import("time"); const envelope = @import("envelope"); const partition = @import("partition.zig"); +const Serve = volume_manager_protocol.Protocol.Provider(void); +const Invocation = envelope.Invocation; +const Answer = envelope.Answer; + +/// The single volume this increment handles: its provider channel, its block +/// sub-range, its identity, the id it is addressed by, and the filesystem +/// process serving it (0 until spawned; reset on death for respawn). +const Volume = struct { + storage: block.Device, + base_lba: u64, + block_count: u64, + identity: u64, + id: u64, + filesystem_pid: u32 = 0, +}; + +/// The filesystem binary a probed volume is served by. The signature->binary +/// map (filesystems.csv) lands with the identity ladder; for now every FAT-shaped +/// volume gets the FAT service. +const filesystem_binary = "/system/services/fat"; +const volume_id: u64 = 1; + var service_endpoint: ipc.Handle = 0; var manager_handle: ?ipc.Handle = null; var bounce: memory.DmaRegion = undefined; var bounce_ready = false; var probed = false; +var volume: ?Volume = null; const probe_retry_ms = 500; -/// The first mass-storage provider's block channel, via the device manager's -/// tree — the same lineage acquisition a filesystem makes (block is not a -/// registry name). One enumerate sweep; null until the chain is up. fn acquireStorage() ?block.Device { const manager = manager_handle orelse opened: { const handle = channel.openEndpoint("device-manager") orelse return null; @@ -67,9 +89,24 @@ fn acquireStorage() ?block.Device { } } -/// One probe attempt: acquire the storage channel, read block 0, and parse the -/// first volume. Sets `probed` and logs on success; a failure leaves everything -/// for the next tick. +/// Spawn the filesystem for `v`, confine it to the volume's range, and record +/// its pid. The confinement is defined for the fresh pid BEFORE the filesystem +/// runs, so its first read is already bounded; the volume manager is the +/// confinement controller (it defines the first range on the device). +fn spawnFilesystem(v: *Volume) void { + const pid = process.spawnSupervised(filesystem_binary, &.{"1"}, service_endpoint) orelse { + _ = logging.write("volume-manager: could not spawn the filesystem\n"); + return; + }; + if (!v.storage.defineRange(pid, v.base_lba, v.block_count)) { + _ = logging.write("volume-manager: could not confine the filesystem to its volume\n"); + _ = process.kill(pid); + return; + } + v.filesystem_pid = pid; + std.log.info("volume 0x{x} -> {s} (pid {d}), lba {d}, {d} blocks", .{ v.identity, filesystem_binary, pid, v.base_lba, v.block_count }); +} + fn tryProbe() void { if (probed) return; if (!bounce_ready) { @@ -77,29 +114,53 @@ fn tryProbe() void { bounce_ready = true; } const device = acquireStorage() orelse return; - // Attach the read buffer to the controller (a no-op success without an - // enforcing IOMMU). The volume manager is unconfined — it reads the whole - // device to probe — so no range is defined here. if (bounce.handle) |handle| { if (!device.attach(handle)) return; _ = ipc.close(handle); - bounce.handle = null; // attached once; do not re-forward on a retry + bounce.handle = null; } const geometry = device.geometry() orelse return; if (!device.read(0, 1, bounce.physical)) return; const sector: [*]const u8 = @ptrFromInt(bounce.virtual); - const volume = partition.firstVolume(sector[0..512], geometry.block_count) orelse { + const found = partition.firstVolume(sector[0..512], geometry.block_count) orelse { _ = logging.write("volume-manager: no volume found on the storage device\n"); - probed = true; // a device with no recognizable volume is not retried + probed = true; return; }; - std.log.info("volume 0x{x} at lba {d}, {d} blocks", .{ volume.identity, volume.base_lba, volume.block_count }); + volume = .{ .storage = device, .base_lba = found.base_lba, .block_count = found.block_count, .identity = found.identity, .id = volume_id }; probed = true; + spawnFilesystem(&volume.?); +} + +/// A filesystem announces itself for the volume it was spawned to serve. Reply +/// with that volume's block channel (already range-confined to this filesystem's +/// badge) as the call's returned capability. No channel means the volume is not +/// ready — the filesystem retries. +fn onHello(_: void, invocation: Invocation(volume_manager_protocol.Hello), _: Answer(void)) isize { + const v = volume orelse return 0; // not probed yet — retryable, no cap + if (invocation.target != v.id) return 0; // unknown volume — retryable + if (invocation.sender != v.filesystem_pid) { + // Not the filesystem we spawned for this volume. Refuse: only the + // confined filesystem gets the channel. + std.log.info("refused hello for volume {d} from process {d}", .{ invocation.target, invocation.sender }); + return -envelope.EPERM; + } + service.replyWithCapability(v.storage.endpoint); + std.log.info("handed volume {d} to pid {d}", .{ v.id, invocation.sender }); + return 0; +} + +const handlers = Serve.Handlers{ .hello = onHello }; + +fn onMessage(message: []const u8, out: []u8, sender: u32, arrived: *ipc.Arrival) usize { + // No verb takes a capability up, so the turn closes whatever arrives. + return Serve.dispatch({}, handlers, message, sender, arrived.peek(), out); } fn initialise(endpoint: ipc.Handle) bool { service_endpoint = endpoint; _ = logging.write("volume-manager: starting, waiting for a storage device\n"); + _ = process.subscribeExits(endpoint); tryProbe(); if (!probed) _ = time.timerOnce(endpoint, probe_retry_ms); return true; @@ -110,23 +171,26 @@ fn onNotification(badge: u64) void { if (got.isTimer()) { tryProbe(); if (!probed) _ = time.timerOnce(service_endpoint, probe_retry_ms); + return; + } + // A filesystem died. Its old range is reclaimed by the driver on the same + // death; respawn it, confined afresh to the same volume (a fresh pid, a + // fresh range). The reap-and-rebuild the device manager proved, one layer up. + if (got.isChildExit()) { + const dead = got.childProcessId(); + if (volume) |*v| { + if (v.filesystem_pid == dead) { + v.filesystem_pid = 0; + std.log.info("filesystem for volume {d} died; respawning", .{v.id}); + spawnFilesystem(v); + } + } } -} - -/// No clients yet: a filesystem hello lands here in the next increment. Until -/// then, refuse politely. -fn onMessage(message: []const u8, out: []u8, sender: u32, arrived: *ipc.Arrival) usize { - _ = message; - _ = sender; - _ = arrived; - const status = envelope.Status{ .status = -envelope.ENOSYS, .len = 0 }; - @memcpy(out[0..envelope.prefix_size], std.mem.asBytes(&status)); - return envelope.prefix_size; } pub fn main(init: process.Init) void { _ = init; - service.run(device_manager_protocol.message_maximum, .{ + service.run(volume_manager_protocol.message_maximum, .{ .service = "volume-manager", .init = initialise, .on_message = onMessage, diff --git a/test/qemu_test.py b/test/qemu_test.py index d380eb8..e3b39ab 100644 --- a/test/qemu_test.py +++ b/test/qemu_test.py @@ -766,7 +766,11 @@ CASES = [ "build_case": "fat-mount", "smp": 4, "timeout": 150, - "expect": r"volume-manager: volume 0x[0-9a-f]+ at lba \d+, \d+ blocks", + # The volume manager probes the partition table, then confines a filesystem + # to the volume and hands it over — one log line naming the volume's range, + # its identity, and the filesystem it spawned for it. + "expect": r"volume-manager: volume 0x[0-9a-f]+ -> \S+ \(pid \d+\), lba \d+, \d+ blocks" + r"[\s\S]*volume-manager: handed volume \d+ to pid \d+", "fail": r"DANOS-TEST-RESULT: FAIL"}, # Phase 2b: mkdir/unlink through the mount. Reuses the fat-mount build — the # fat-test client, after listing, makes a directory, writes+reads a file inside