diff --git a/build.zig b/build.zig index f8e6d6c..28c7786 100644 --- a/build.zig +++ b/build.zig @@ -125,6 +125,7 @@ const production_ship = [_]ShipRow{ service("display"), service("display-demo"), service("device-manager"), + service("volume-manager"), service("input"), service("logger"), driver("pci-bus"), diff --git a/build.zig.zon b/build.zig.zon index 0198bf7..2b2a5c9 100644 --- a/build.zig.zon +++ b/build.zig.zon @@ -49,6 +49,7 @@ .display = .{ .path = "system/services/display" }, .@"display-demo" = .{ .path = "system/services/display-demo" }, .@"device-manager" = .{ .path = "system/services/device-manager" }, + .@"volume-manager" = .{ .path = "system/services/volume-manager" }, .input = .{ .path = "system/services/input" }, .logger = .{ .path = "system/services/logger" }, // The discovery pair and the /test fixtures are lazy: only what a diff --git a/system/configuration/init.csv b/system/configuration/init.csv index 2609284..847cbcb 100644 --- a/system/configuration/init.csv +++ b/system/configuration/init.csv @@ -14,6 +14,7 @@ # service args... /system/services/input /system/services/device-manager +/system/services/volume-manager /system/services/fat /system/services/display /system/services/display-demo diff --git a/system/configuration/protocol.csv b/system/configuration/protocol.csv index c2cbd0b..92a8327 100644 --- a/system/configuration/protocol.csv +++ b/system/configuration/protocol.csv @@ -55,6 +55,7 @@ # --- the services init spawns from init.csv --------------------------------- /system/services/input, /system/services/init, bind, input /system/services/device-manager, /system/services/init, bind, device-manager +/system/services/volume-manager, /system/services/init, bind, volume-manager /system/services/fat, /system/services/init, bind, vfs /system/services/display, /system/services/init, bind, display @@ -100,6 +101,9 @@ # opens /protocol/display like any other client — threads share no handles), # and the input stream that moves the cursor. /system/services/fat, /system/services/init, open, device-manager +# The volume manager reaches the device manager to be routed to each storage +# provider's block channel, the same lineage acquisition fat makes today. +/system/services/volume-manager, /system/services/init, open, device-manager /system/services/display, /system/services/init, open, scanout /system/services/display, /system/services/init, open, display /system/services/display, /system/services/init, open, input diff --git a/system/services/volume-manager/build.zig b/system/services/volume-manager/build.zig new file mode 100644 index 0000000..5b353c1 --- /dev/null +++ b/system/services/volume-manager/build.zig @@ -0,0 +1,30 @@ +//! The volume-manager service as a binary package (docs/build-packages-plan.md): +//! this file names the binary and EXACTLY the modules its source imports — +//! build-support resolves each name from the domains this zon declares. + +const std = @import("std"); +const build_support = @import("build-support"); + +pub fn build(b: *std.Build) void { + const exe = build_support.userBinary(b, .{ + .name = "volume-manager", + .root_source_file = b.path("volume-manager.zig"), + .imports = &.{ + "block", "channel", "device-manager-protocol", "driver", + "envelope", "ipc", "logging", "memory", + "process", "service", "time", + }, + }); + b.installArtifact(exe); + + // Standalone `zig build test` for the partition parser; the root build keeps + // its aggregate test step. + const test_step = b.step("test", "Run the partition-parser unit tests"); + const tests = b.addTest(.{ + .root_module = b.createModule(.{ + .root_source_file = b.path("partition.zig"), + .target = b.resolveTargetQuery(.{}), + }), + }); + test_step.dependOn(&b.addRunArtifact(tests).step); +} diff --git a/system/services/volume-manager/build.zig.zon b/system/services/volume-manager/build.zig.zon new file mode 100644 index 0000000..990d7d3 --- /dev/null +++ b/system/services/volume-manager/build.zig.zon @@ -0,0 +1,16 @@ +.{ + .name = .volume_manager, + .version = "0.0.0", + .fingerprint = 0x911b01e1eaeabcf5, // Changing this has security and trust implications. + .minimum_zig_version = "0.16.0", + .dependencies = .{ + // build-support supplies the shared recipe; kernel is implicit in every + // binary. device (block, driver) and protocol (device-manager-protocol, + // envelope) are the homes of this service's remaining imports. + .@"build-support" = .{ .path = "../../../build-support" }, + .kernel = .{ .path = "../../../library/kernel" }, + .device = .{ .path = "../../../library/device" }, + .protocol = .{ .path = "../../../library/protocol" }, + }, + .paths = .{""}, +} diff --git a/system/services/volume-manager/partition.zig b/system/services/volume-manager/partition.zig new file mode 100644 index 0000000..00ea139 --- /dev/null +++ b/system/services/volume-manager/partition.zig @@ -0,0 +1,92 @@ +//! Partition-table parsing, the policy the storage architecture places above the +//! block driver and below the filesystem (docs/file-system-development/ +//! storage-architecture.md): read block 0, decide what block sub-ranges are +//! volumes, and read each volume's content identity. The block DRIVER never does +//! this — it clamps ranges it is told about; this is what tells it the numbers. +//! +//! Today: MBR (the four-entry table at offset 446) plus the bare-FAT case (a boot +//! sector right at LBA 0). GPT is the next entry in the identity ladder and slots +//! in here without touching anything above or below. + +const std = @import("std"); + +/// One volume the parser found on the device: the block sub-range it occupies +/// and a content identity stable for the volume's life (the mount map keys on +/// it; the boot volume is recorded by it). `identity` is derived from the medium, +/// never from a port — a moved drive keeps it. +pub const Volume = struct { + base_lba: u64, + block_count: u64, + identity: u64, +}; + +/// The MBR disk signature (offset 440, 4 bytes LE) — a 32-bit id written at +/// partition time. Weak (dd-cloned disks share it) but on the medium, and the +/// simplest rung of the identity ladder; the fuller rungs (GPT partition GUID, +/// FAT volume serial) refine `identityOf` without changing the shape. +fn diskSignature(block0: []const u8) u32 { + if (block0.len < 444) return 0; + return std.mem.readInt(u32, block0[440..444], .little); +} + +/// The identity of the volume at partition index `index`: the disk signature +/// paired with the index, so two partitions of one disk stay distinct. For a +/// bare FAT (no table) the index is 0. +fn identityOf(block0: []const u8, index: u8) u64 { + return (@as(u64, diskSignature(block0)) << 8) | index; +} + +/// Whether block 0 looks like a partition table (the 0x55AA boot signature). A +/// bare FAT also carries it, so the caller distinguishes by whether any partition +/// entry is non-empty. +fn hasBootSignature(block0: []const u8) bool { + return block0.len >= 512 and block0[510] == 0x55 and block0[511] == 0xAA; +} + +/// The first volume on a device whose block 0 is `block0` and whose whole-device +/// size is `device_blocks`, or null if none is found. An MBR with a non-empty +/// entry yields that partition's [start, size); otherwise a boot signature with +/// no partitions is treated as a bare FAT spanning the whole device. +pub fn firstVolume(block0: []const u8, device_blocks: u64) ?Volume { + if (!hasBootSignature(block0)) return null; + var index: u8 = 0; + while (index < 4) : (index += 1) { + const entry = block0[446 + @as(usize, index) * 16 ..][0..16]; + const kind = entry[4]; + const start = std.mem.readInt(u32, entry[8..12], .little); + const size = std.mem.readInt(u32, entry[12..16], .little); + if (kind == 0 or start == 0 or size == 0) continue; + return .{ .base_lba = start, .block_count = size, .identity = identityOf(block0, index) }; + } + // No partition entries: a bare FAT spanning the device. + return .{ .base_lba = 0, .block_count = device_blocks, .identity = identityOf(block0, 0) }; +} + +test "an MBR with one partition yields its range and a distinct identity" { + var block0 = [_]u8{0} ** 512; + block0[510] = 0x55; + block0[511] = 0xAA; + std.mem.writeInt(u32, block0[440..444], 0xDEADBEEF, .little); + // partition 0: type 0x0c (FAT32 LBA), start 2048, size 100000 + block0[446 + 4] = 0x0c; + std.mem.writeInt(u32, block0[446 + 8 ..][0..4], 2048, .little); + std.mem.writeInt(u32, block0[446 + 12 ..][0..4], 100000, .little); + const v = firstVolume(&block0, 200000).?; + try std.testing.expectEqual(@as(u64, 2048), v.base_lba); + try std.testing.expectEqual(@as(u64, 100000), v.block_count); + try std.testing.expectEqual((@as(u64, 0xDEADBEEF) << 8) | 0, v.identity); +} + +test "a boot signature with no partitions is a bare FAT over the whole device" { + var block0 = [_]u8{0} ** 512; + block0[510] = 0x55; + block0[511] = 0xAA; + const v = firstVolume(&block0, 65536).?; + try std.testing.expectEqual(@as(u64, 0), v.base_lba); + try std.testing.expectEqual(@as(u64, 65536), v.block_count); +} + +test "no boot signature is no volume" { + const block0 = [_]u8{0} ** 512; + try std.testing.expect(firstVolume(&block0, 65536) == null); +} diff --git a/system/services/volume-manager/volume-manager.zig b/system/services/volume-manager/volume-manager.zig new file mode 100644 index 0000000..fb756be --- /dev/null +++ b/system/services/volume-manager/volume-manager.zig @@ -0,0 +1,135 @@ +//! system/services/volume-manager — the storage layer's policy home +//! (docs/file-system-development/storage-architecture.md). It sits beside the +//! device manager: the device manager owns the DEVICE tree; this owns the VOLUME +//! layer. It hears about storage providers, probes their partition tables and +//! content identity, and — in later increments — confines each filesystem to its +//! partition and spawns one per volume, answering that filesystem's startup +//! hello with the (range-confined) block channel. +//! +//! This increment (V3a) is discovery and probe only: find the mass-storage +//! provider, read block 0, parse the first volume out of it, and log what it +//! found — additive, with the FAT service still acquiring its own volume. The +//! delegation (confine + spawn + hand over the channel) and the mount map land +//! next, keeping the FAT service working throughout. + +const std = @import("std"); +const channel = @import("channel"); +const device_manager_protocol = @import("device-manager-protocol"); +const driver = @import("driver"); +const ipc = @import("ipc"); +const block = @import("block"); +const memory = @import("memory"); +const logging = @import("logging"); +const process = @import("process"); +const service = @import("service"); +const time = @import("time"); +const envelope = @import("envelope"); +const partition = @import("partition.zig"); + +var service_endpoint: ipc.Handle = 0; +var manager_handle: ?ipc.Handle = null; +var bounce: memory.DmaRegion = undefined; +var bounce_ready = false; +var probed = false; +const probe_retry_ms = 500; + +/// The first mass-storage provider's block channel, via the device manager's +/// tree — the same lineage acquisition a filesystem makes (block is not a +/// registry name). One enumerate sweep; null until the chain is up. +fn acquireStorage() ?block.Device { + const manager = manager_handle orelse opened: { + const handle = channel.openEndpoint("device-manager") orelse return null; + manager_handle = handle; + break :opened handle; + }; + const Entry = device_manager_protocol.ChildEntry; + var start: u64 = 0; + while (true) { + const enumerate = envelope.Header{ .operation = envelope.operation_enumerate, .target = start }; + var reply: [device_manager_protocol.message_maximum]u8 = undefined; + const length = ipc.call(manager, std.mem.asBytes(&enumerate), &reply) catch return null; + const status = envelope.statusOf(reply[0..length]) orelse return null; + if (status.status != 0) return null; + const carried = @min(@as(usize, status.len), length -| envelope.prefix_size); + const tail = reply[envelope.prefix_size..][0..carried]; + const count = tail.len / @sizeOf(Entry); + if (count == 0) return null; + var index: usize = 0; + while (index < count) : (index += 1) { + const entry = std.mem.bytesToValue(Entry, tail[index * @sizeOf(Entry) ..][0..@sizeOf(Entry)]); + if (entry.device_id == device_manager_protocol.no_device) continue; + if ((entry.identity >> 16) & 0xff != 0x08 or (entry.identity >> 8) & 0xff != 0x06) continue; + const exchanged = driver.helloOn(manager, .consumer, entry.device_id, null, true) orelse return null; + const provider = exchanged.channel orelse continue; + return .{ .endpoint = provider }; + } + start += count; + } +} + +/// One probe attempt: acquire the storage channel, read block 0, and parse the +/// first volume. Sets `probed` and logs on success; a failure leaves everything +/// for the next tick. +fn tryProbe() void { + if (probed) return; + if (!bounce_ready) { + bounce = memory.dmaAlloc(512, memory.dma_coherent | memory.dma_shareable) orelse return; + bounce_ready = true; + } + const device = acquireStorage() orelse return; + // Attach the read buffer to the controller (a no-op success without an + // enforcing IOMMU). The volume manager is unconfined — it reads the whole + // device to probe — so no range is defined here. + if (bounce.handle) |handle| { + if (!device.attach(handle)) return; + _ = ipc.close(handle); + bounce.handle = null; // attached once; do not re-forward on a retry + } + const geometry = device.geometry() orelse return; + if (!device.read(0, 1, bounce.physical)) return; + const sector: [*]const u8 = @ptrFromInt(bounce.virtual); + const volume = partition.firstVolume(sector[0..512], geometry.block_count) orelse { + _ = logging.write("volume-manager: no volume found on the storage device\n"); + probed = true; // a device with no recognizable volume is not retried + return; + }; + std.log.info("volume 0x{x} at lba {d}, {d} blocks", .{ volume.identity, volume.base_lba, volume.block_count }); + probed = true; +} + +fn initialise(endpoint: ipc.Handle) bool { + service_endpoint = endpoint; + _ = logging.write("volume-manager: starting, waiting for a storage device\n"); + tryProbe(); + if (!probed) _ = time.timerOnce(endpoint, probe_retry_ms); + return true; +} + +fn onNotification(badge: u64) void { + const got = ipc.Received{ .len = 0, .badge = badge, .cap = null }; + if (got.isTimer()) { + tryProbe(); + if (!probed) _ = time.timerOnce(service_endpoint, probe_retry_ms); + } +} + +/// No clients yet: a filesystem hello lands here in the next increment. Until +/// then, refuse politely. +fn onMessage(message: []const u8, out: []u8, sender: u32, arrived: *ipc.Arrival) usize { + _ = message; + _ = sender; + _ = arrived; + const status = envelope.Status{ .status = -envelope.ENOSYS, .len = 0 }; + @memcpy(out[0..envelope.prefix_size], std.mem.asBytes(&status)); + return envelope.prefix_size; +} + +pub fn main(init: process.Init) void { + _ = init; + service.run(device_manager_protocol.message_maximum, .{ + .service = "volume-manager", + .init = initialise, + .on_message = onMessage, + .on_notification = onNotification, + }); +} diff --git a/test/qemu_test.py b/test/qemu_test.py index 6e91485..d380eb8 100644 --- a/test/qemu_test.py +++ b/test/qemu_test.py @@ -755,6 +755,19 @@ CASES = [ "timeout": 150, "expect": r"fat: mounted /volumes/usb[\s\S]*fat-test: ok", "fail": r"DANOS-TEST-RESULT: FAIL"}, + # Volume-manager discovery + probe (V3a, docs/volume-manager-plan.md). Reuses + # the fat-mount kernel build (the default boot now spawns the volume manager + # from init.csv). It acquires the mass-storage block channel through the + # device manager, reads block 0, and parses the first volume out of it — the + # partition-table walk that used to live in the FAT engine, now above the + # driver where it belongs. Before V3a the service did not exist, so this line + # is absent. + {"name": "volume-probe", + "build_case": "fat-mount", + "smp": 4, + "timeout": 150, + "expect": r"volume-manager: volume 0x[0-9a-f]+ at lba \d+, \d+ blocks", + "fail": r"DANOS-TEST-RESULT: FAIL"}, # Phase 2b: mkdir/unlink through the mount. Reuses the fat-mount build — the # fat-test client, after listing, makes a directory, writes+reads a file inside # it, then removes the file, exercising the whole VFS -> fat mutation path.