From d56b1b81c0851a237ea865a1ac6ea3eab8c2947b Mon Sep 17 00:00:00 2001 From: Daniel Samson <12231216+daniel-samson@users.noreply.github.com> Date: Sun, 9 Aug 2026 18:10:43 +0100 Subject: [PATCH] =?UTF-8?q?volume-manager:=20discovery=20and=20probe=20?= =?UTF-8?q?=E2=80=94=20V3a?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The storage layer gains its policy home (storage-architecture.md): a new system/services/volume-manager, spawned by init, that acquires the mass- storage block channel through the device manager (the same lineage a filesystem uses), reads block 0, and parses the first volume out of it. The partition-table walk that lived in the FAT engine moves here, above the driver where it belongs (partition.zig, host-tested: MBR entry, bare-FAT, no-signature). Identity is the MBR disk signature + partition index — the weak rung of the ladder; GPT GUID and FAT serial refine identityOf without changing shape. This increment is discovery + probe + log only, additive: the FAT service still acquires its own volume, so nothing changes for it. Confining each filesystem to its partition and spawning one per volume (the flip) lands next, keeping fat working throughout. Grants + wiring: init.csv spawns it after the device manager; protocol.csv grants bind volume-manager + open device-manager. Verified: volume-probe asserts the parse (bare-FAT volume at lba 0), neutral 10/10 across storage, restart, display, logging, confinement — the volume manager now runs in every boot and disturbs nothing. --- build.zig | 1 + build.zig.zon | 1 + system/configuration/init.csv | 1 + system/configuration/protocol.csv | 4 + system/services/volume-manager/build.zig | 30 ++++ system/services/volume-manager/build.zig.zon | 16 +++ system/services/volume-manager/partition.zig | 92 ++++++++++++ .../volume-manager/volume-manager.zig | 135 ++++++++++++++++++ test/qemu_test.py | 13 ++ 9 files changed, 293 insertions(+) create mode 100644 system/services/volume-manager/build.zig create mode 100644 system/services/volume-manager/build.zig.zon create mode 100644 system/services/volume-manager/partition.zig create mode 100644 system/services/volume-manager/volume-manager.zig diff --git a/build.zig b/build.zig index f8e6d6c..28c7786 100644 --- a/build.zig +++ b/build.zig @@ -125,6 +125,7 @@ const production_ship = [_]ShipRow{ service("display"), service("display-demo"), service("device-manager"), + service("volume-manager"), service("input"), service("logger"), driver("pci-bus"), diff --git a/build.zig.zon b/build.zig.zon index 0198bf7..2b2a5c9 100644 --- a/build.zig.zon +++ b/build.zig.zon @@ -49,6 +49,7 @@ .display = .{ .path = "system/services/display" }, .@"display-demo" = .{ .path = "system/services/display-demo" }, .@"device-manager" = .{ .path = "system/services/device-manager" }, + .@"volume-manager" = .{ .path = "system/services/volume-manager" }, .input = .{ .path = "system/services/input" }, .logger = .{ .path = "system/services/logger" }, // The discovery pair and the /test fixtures are lazy: only what a diff --git a/system/configuration/init.csv b/system/configuration/init.csv index 2609284..847cbcb 100644 --- a/system/configuration/init.csv +++ b/system/configuration/init.csv @@ -14,6 +14,7 @@ # service args... /system/services/input /system/services/device-manager +/system/services/volume-manager /system/services/fat /system/services/display /system/services/display-demo diff --git a/system/configuration/protocol.csv b/system/configuration/protocol.csv index c2cbd0b..92a8327 100644 --- a/system/configuration/protocol.csv +++ b/system/configuration/protocol.csv @@ -55,6 +55,7 @@ # --- the services init spawns from init.csv --------------------------------- /system/services/input, /system/services/init, bind, input /system/services/device-manager, /system/services/init, bind, device-manager +/system/services/volume-manager, /system/services/init, bind, volume-manager /system/services/fat, /system/services/init, bind, vfs /system/services/display, /system/services/init, bind, display @@ -100,6 +101,9 @@ # opens /protocol/display like any other client — threads share no handles), # and the input stream that moves the cursor. /system/services/fat, /system/services/init, open, device-manager +# The volume manager reaches the device manager to be routed to each storage +# provider's block channel, the same lineage acquisition fat makes today. +/system/services/volume-manager, /system/services/init, open, device-manager /system/services/display, /system/services/init, open, scanout /system/services/display, /system/services/init, open, display /system/services/display, /system/services/init, open, input diff --git a/system/services/volume-manager/build.zig b/system/services/volume-manager/build.zig new file mode 100644 index 0000000..5b353c1 --- /dev/null +++ b/system/services/volume-manager/build.zig @@ -0,0 +1,30 @@ +//! The volume-manager service as a binary package (docs/build-packages-plan.md): +//! this file names the binary and EXACTLY the modules its source imports — +//! build-support resolves each name from the domains this zon declares. + +const std = @import("std"); +const build_support = @import("build-support"); + +pub fn build(b: *std.Build) void { + const exe = build_support.userBinary(b, .{ + .name = "volume-manager", + .root_source_file = b.path("volume-manager.zig"), + .imports = &.{ + "block", "channel", "device-manager-protocol", "driver", + "envelope", "ipc", "logging", "memory", + "process", "service", "time", + }, + }); + b.installArtifact(exe); + + // Standalone `zig build test` for the partition parser; the root build keeps + // its aggregate test step. + const test_step = b.step("test", "Run the partition-parser unit tests"); + const tests = b.addTest(.{ + .root_module = b.createModule(.{ + .root_source_file = b.path("partition.zig"), + .target = b.resolveTargetQuery(.{}), + }), + }); + test_step.dependOn(&b.addRunArtifact(tests).step); +} diff --git a/system/services/volume-manager/build.zig.zon b/system/services/volume-manager/build.zig.zon new file mode 100644 index 0000000..990d7d3 --- /dev/null +++ b/system/services/volume-manager/build.zig.zon @@ -0,0 +1,16 @@ +.{ + .name = .volume_manager, + .version = "0.0.0", + .fingerprint = 0x911b01e1eaeabcf5, // Changing this has security and trust implications. + .minimum_zig_version = "0.16.0", + .dependencies = .{ + // build-support supplies the shared recipe; kernel is implicit in every + // binary. device (block, driver) and protocol (device-manager-protocol, + // envelope) are the homes of this service's remaining imports. + .@"build-support" = .{ .path = "../../../build-support" }, + .kernel = .{ .path = "../../../library/kernel" }, + .device = .{ .path = "../../../library/device" }, + .protocol = .{ .path = "../../../library/protocol" }, + }, + .paths = .{""}, +} diff --git a/system/services/volume-manager/partition.zig b/system/services/volume-manager/partition.zig new file mode 100644 index 0000000..00ea139 --- /dev/null +++ b/system/services/volume-manager/partition.zig @@ -0,0 +1,92 @@ +//! Partition-table parsing, the policy the storage architecture places above the +//! block driver and below the filesystem (docs/file-system-development/ +//! storage-architecture.md): read block 0, decide what block sub-ranges are +//! volumes, and read each volume's content identity. The block DRIVER never does +//! this — it clamps ranges it is told about; this is what tells it the numbers. +//! +//! Today: MBR (the four-entry table at offset 446) plus the bare-FAT case (a boot +//! sector right at LBA 0). GPT is the next entry in the identity ladder and slots +//! in here without touching anything above or below. + +const std = @import("std"); + +/// One volume the parser found on the device: the block sub-range it occupies +/// and a content identity stable for the volume's life (the mount map keys on +/// it; the boot volume is recorded by it). `identity` is derived from the medium, +/// never from a port — a moved drive keeps it. +pub const Volume = struct { + base_lba: u64, + block_count: u64, + identity: u64, +}; + +/// The MBR disk signature (offset 440, 4 bytes LE) — a 32-bit id written at +/// partition time. Weak (dd-cloned disks share it) but on the medium, and the +/// simplest rung of the identity ladder; the fuller rungs (GPT partition GUID, +/// FAT volume serial) refine `identityOf` without changing the shape. +fn diskSignature(block0: []const u8) u32 { + if (block0.len < 444) return 0; + return std.mem.readInt(u32, block0[440..444], .little); +} + +/// The identity of the volume at partition index `index`: the disk signature +/// paired with the index, so two partitions of one disk stay distinct. For a +/// bare FAT (no table) the index is 0. +fn identityOf(block0: []const u8, index: u8) u64 { + return (@as(u64, diskSignature(block0)) << 8) | index; +} + +/// Whether block 0 looks like a partition table (the 0x55AA boot signature). A +/// bare FAT also carries it, so the caller distinguishes by whether any partition +/// entry is non-empty. +fn hasBootSignature(block0: []const u8) bool { + return block0.len >= 512 and block0[510] == 0x55 and block0[511] == 0xAA; +} + +/// The first volume on a device whose block 0 is `block0` and whose whole-device +/// size is `device_blocks`, or null if none is found. An MBR with a non-empty +/// entry yields that partition's [start, size); otherwise a boot signature with +/// no partitions is treated as a bare FAT spanning the whole device. +pub fn firstVolume(block0: []const u8, device_blocks: u64) ?Volume { + if (!hasBootSignature(block0)) return null; + var index: u8 = 0; + while (index < 4) : (index += 1) { + const entry = block0[446 + @as(usize, index) * 16 ..][0..16]; + const kind = entry[4]; + const start = std.mem.readInt(u32, entry[8..12], .little); + const size = std.mem.readInt(u32, entry[12..16], .little); + if (kind == 0 or start == 0 or size == 0) continue; + return .{ .base_lba = start, .block_count = size, .identity = identityOf(block0, index) }; + } + // No partition entries: a bare FAT spanning the device. + return .{ .base_lba = 0, .block_count = device_blocks, .identity = identityOf(block0, 0) }; +} + +test "an MBR with one partition yields its range and a distinct identity" { + var block0 = [_]u8{0} ** 512; + block0[510] = 0x55; + block0[511] = 0xAA; + std.mem.writeInt(u32, block0[440..444], 0xDEADBEEF, .little); + // partition 0: type 0x0c (FAT32 LBA), start 2048, size 100000 + block0[446 + 4] = 0x0c; + std.mem.writeInt(u32, block0[446 + 8 ..][0..4], 2048, .little); + std.mem.writeInt(u32, block0[446 + 12 ..][0..4], 100000, .little); + const v = firstVolume(&block0, 200000).?; + try std.testing.expectEqual(@as(u64, 2048), v.base_lba); + try std.testing.expectEqual(@as(u64, 100000), v.block_count); + try std.testing.expectEqual((@as(u64, 0xDEADBEEF) << 8) | 0, v.identity); +} + +test "a boot signature with no partitions is a bare FAT over the whole device" { + var block0 = [_]u8{0} ** 512; + block0[510] = 0x55; + block0[511] = 0xAA; + const v = firstVolume(&block0, 65536).?; + try std.testing.expectEqual(@as(u64, 0), v.base_lba); + try std.testing.expectEqual(@as(u64, 65536), v.block_count); +} + +test "no boot signature is no volume" { + const block0 = [_]u8{0} ** 512; + try std.testing.expect(firstVolume(&block0, 65536) == null); +} diff --git a/system/services/volume-manager/volume-manager.zig b/system/services/volume-manager/volume-manager.zig new file mode 100644 index 0000000..fb756be --- /dev/null +++ b/system/services/volume-manager/volume-manager.zig @@ -0,0 +1,135 @@ +//! system/services/volume-manager — the storage layer's policy home +//! (docs/file-system-development/storage-architecture.md). It sits beside the +//! device manager: the device manager owns the DEVICE tree; this owns the VOLUME +//! layer. It hears about storage providers, probes their partition tables and +//! content identity, and — in later increments — confines each filesystem to its +//! partition and spawns one per volume, answering that filesystem's startup +//! hello with the (range-confined) block channel. +//! +//! This increment (V3a) is discovery and probe only: find the mass-storage +//! provider, read block 0, parse the first volume out of it, and log what it +//! found — additive, with the FAT service still acquiring its own volume. The +//! delegation (confine + spawn + hand over the channel) and the mount map land +//! next, keeping the FAT service working throughout. + +const std = @import("std"); +const channel = @import("channel"); +const device_manager_protocol = @import("device-manager-protocol"); +const driver = @import("driver"); +const ipc = @import("ipc"); +const block = @import("block"); +const memory = @import("memory"); +const logging = @import("logging"); +const process = @import("process"); +const service = @import("service"); +const time = @import("time"); +const envelope = @import("envelope"); +const partition = @import("partition.zig"); + +var service_endpoint: ipc.Handle = 0; +var manager_handle: ?ipc.Handle = null; +var bounce: memory.DmaRegion = undefined; +var bounce_ready = false; +var probed = false; +const probe_retry_ms = 500; + +/// The first mass-storage provider's block channel, via the device manager's +/// tree — the same lineage acquisition a filesystem makes (block is not a +/// registry name). One enumerate sweep; null until the chain is up. +fn acquireStorage() ?block.Device { + const manager = manager_handle orelse opened: { + const handle = channel.openEndpoint("device-manager") orelse return null; + manager_handle = handle; + break :opened handle; + }; + const Entry = device_manager_protocol.ChildEntry; + var start: u64 = 0; + while (true) { + const enumerate = envelope.Header{ .operation = envelope.operation_enumerate, .target = start }; + var reply: [device_manager_protocol.message_maximum]u8 = undefined; + const length = ipc.call(manager, std.mem.asBytes(&enumerate), &reply) catch return null; + const status = envelope.statusOf(reply[0..length]) orelse return null; + if (status.status != 0) return null; + const carried = @min(@as(usize, status.len), length -| envelope.prefix_size); + const tail = reply[envelope.prefix_size..][0..carried]; + const count = tail.len / @sizeOf(Entry); + if (count == 0) return null; + var index: usize = 0; + while (index < count) : (index += 1) { + const entry = std.mem.bytesToValue(Entry, tail[index * @sizeOf(Entry) ..][0..@sizeOf(Entry)]); + if (entry.device_id == device_manager_protocol.no_device) continue; + if ((entry.identity >> 16) & 0xff != 0x08 or (entry.identity >> 8) & 0xff != 0x06) continue; + const exchanged = driver.helloOn(manager, .consumer, entry.device_id, null, true) orelse return null; + const provider = exchanged.channel orelse continue; + return .{ .endpoint = provider }; + } + start += count; + } +} + +/// One probe attempt: acquire the storage channel, read block 0, and parse the +/// first volume. Sets `probed` and logs on success; a failure leaves everything +/// for the next tick. +fn tryProbe() void { + if (probed) return; + if (!bounce_ready) { + bounce = memory.dmaAlloc(512, memory.dma_coherent | memory.dma_shareable) orelse return; + bounce_ready = true; + } + const device = acquireStorage() orelse return; + // Attach the read buffer to the controller (a no-op success without an + // enforcing IOMMU). The volume manager is unconfined — it reads the whole + // device to probe — so no range is defined here. + if (bounce.handle) |handle| { + if (!device.attach(handle)) return; + _ = ipc.close(handle); + bounce.handle = null; // attached once; do not re-forward on a retry + } + const geometry = device.geometry() orelse return; + if (!device.read(0, 1, bounce.physical)) return; + const sector: [*]const u8 = @ptrFromInt(bounce.virtual); + const volume = partition.firstVolume(sector[0..512], geometry.block_count) orelse { + _ = logging.write("volume-manager: no volume found on the storage device\n"); + probed = true; // a device with no recognizable volume is not retried + return; + }; + std.log.info("volume 0x{x} at lba {d}, {d} blocks", .{ volume.identity, volume.base_lba, volume.block_count }); + probed = true; +} + +fn initialise(endpoint: ipc.Handle) bool { + service_endpoint = endpoint; + _ = logging.write("volume-manager: starting, waiting for a storage device\n"); + tryProbe(); + if (!probed) _ = time.timerOnce(endpoint, probe_retry_ms); + return true; +} + +fn onNotification(badge: u64) void { + const got = ipc.Received{ .len = 0, .badge = badge, .cap = null }; + if (got.isTimer()) { + tryProbe(); + if (!probed) _ = time.timerOnce(service_endpoint, probe_retry_ms); + } +} + +/// No clients yet: a filesystem hello lands here in the next increment. Until +/// then, refuse politely. +fn onMessage(message: []const u8, out: []u8, sender: u32, arrived: *ipc.Arrival) usize { + _ = message; + _ = sender; + _ = arrived; + const status = envelope.Status{ .status = -envelope.ENOSYS, .len = 0 }; + @memcpy(out[0..envelope.prefix_size], std.mem.asBytes(&status)); + return envelope.prefix_size; +} + +pub fn main(init: process.Init) void { + _ = init; + service.run(device_manager_protocol.message_maximum, .{ + .service = "volume-manager", + .init = initialise, + .on_message = onMessage, + .on_notification = onNotification, + }); +} diff --git a/test/qemu_test.py b/test/qemu_test.py index 6e91485..d380eb8 100644 --- a/test/qemu_test.py +++ b/test/qemu_test.py @@ -755,6 +755,19 @@ CASES = [ "timeout": 150, "expect": r"fat: mounted /volumes/usb[\s\S]*fat-test: ok", "fail": r"DANOS-TEST-RESULT: FAIL"}, + # Volume-manager discovery + probe (V3a, docs/volume-manager-plan.md). Reuses + # the fat-mount kernel build (the default boot now spawns the volume manager + # from init.csv). It acquires the mass-storage block channel through the + # device manager, reads block 0, and parses the first volume out of it — the + # partition-table walk that used to live in the FAT engine, now above the + # driver where it belongs. Before V3a the service did not exist, so this line + # is absent. + {"name": "volume-probe", + "build_case": "fat-mount", + "smp": 4, + "timeout": 150, + "expect": r"volume-manager: volume 0x[0-9a-f]+ at lba \d+, \d+ blocks", + "fail": r"DANOS-TEST-RESULT: FAIL"}, # Phase 2b: mkdir/unlink through the mount. Reuses the fat-mount build — the # fat-test client, after listing, makes a directory, writes+reads a file inside # it, then removes the file, exercising the whole VFS -> fat mutation path.