volume-manager: the flip — fat is spawned, confined, and handed its channel (V3b)
The load-bearing step. The FAT service stops acquiring its own volume: the volume manager spawns it (per volume), defines its partition range on the storage driver BEFORE it runs, and answers its startup hello with the range-confined block channel over a new volume-manager protocol. fat never finds its storage by name and never sees the whole device — establishment by lineage, one layer up from the driver tree. - New library/protocol/volume-manager: one verb, hello(volume-id) -> the block channel as the reply capability (the P0 reply-cap path). - The volume manager becomes the confinement CONTROLLER: it defines the first range on usb-storage, so no other party can confine a filesystem. It supervises the filesystems it spawns and respawns one on death (the reap- and-rebuild the device manager proved, one layer up). - fat: drops acquireVolume(device-manager); hellos the volume manager for its channel; reads its volume id from argv[1]. main takes process.Init now. - init.csv no longer spawns fat (the volume manager does); protocol.csv rewires fat to be supervised by the volume manager (bind vfs, open volume-manager) and drops fat open device-manager. - The block-range fixture boots registry + device-manager only (not the full tree), so the volume manager is absent and the fixture stays the sole confinement definer — otherwise the volume manager would take the controller first and refuse it. Verified end to end (VM probes -> spawns fat -> confines it -> hands over the channel -> fat mounts) and neutral: 18/18 across the fat family, logging, shutdown, both IOMMU variants, usb restart, vfs, conformance, confinement.
This commit is contained in:
@@ -10,9 +10,9 @@ pub fn build(b: *std.Build) void {
|
||||
.name = "volume-manager",
|
||||
.root_source_file = b.path("volume-manager.zig"),
|
||||
.imports = &.{
|
||||
"block", "channel", "device-manager-protocol", "driver",
|
||||
"envelope", "ipc", "logging", "memory",
|
||||
"process", "service", "time",
|
||||
"block", "channel", "device-manager-protocol", "driver",
|
||||
"envelope", "ipc", "logging", "memory",
|
||||
"process", "service", "time", "volume-manager-protocol",
|
||||
},
|
||||
});
|
||||
b.installArtifact(exe);
|
||||
|
||||
@@ -1,20 +1,22 @@
|
||||
//! system/services/volume-manager — the storage layer's policy home
|
||||
//! (docs/file-system-development/storage-architecture.md). It sits beside the
|
||||
//! device manager: the device manager owns the DEVICE tree; this owns the VOLUME
|
||||
//! layer. It hears about storage providers, probes their partition tables and
|
||||
//! content identity, and — in later increments — confines each filesystem to its
|
||||
//! partition and spawns one per volume, answering that filesystem's startup
|
||||
//! hello with the (range-confined) block channel.
|
||||
//! (docs/file-system-development/storage-architecture.md). Beside the device
|
||||
//! manager: that owns the DEVICE tree, this owns the VOLUME layer. It probes a
|
||||
//! storage provider's partition table, confines each filesystem to its
|
||||
//! partition, spawns one filesystem per volume, and answers that filesystem's
|
||||
//! startup hello with the range-confined block channel — so the filesystem
|
||||
//! never finds its storage by name and never sees the whole device. It
|
||||
//! supervises the filesystems it spawns, exactly as the device manager
|
||||
//! supervises drivers.
|
||||
//!
|
||||
//! This increment (V3a) is discovery and probe only: find the mass-storage
|
||||
//! provider, read block 0, parse the first volume out of it, and log what it
|
||||
//! found — additive, with the FAT service still acquiring its own volume. The
|
||||
//! delegation (confine + spawn + hand over the channel) and the mount map land
|
||||
//! next, keeping the FAT service working throughout.
|
||||
//! This increment (V3b) is the flip: the FAT service stops acquiring its own
|
||||
//! volume and is spawned here instead, confined to its partition, and handed
|
||||
//! its channel over the volume-manager protocol. Single volume for now; the
|
||||
//! mount map (volumes.csv) and multi-volume land next.
|
||||
|
||||
const std = @import("std");
|
||||
const channel = @import("channel");
|
||||
const device_manager_protocol = @import("device-manager-protocol");
|
||||
const volume_manager_protocol = @import("volume-manager-protocol");
|
||||
const driver = @import("driver");
|
||||
const ipc = @import("ipc");
|
||||
const block = @import("block");
|
||||
@@ -26,16 +28,36 @@ const time = @import("time");
|
||||
const envelope = @import("envelope");
|
||||
const partition = @import("partition.zig");
|
||||
|
||||
const Serve = volume_manager_protocol.Protocol.Provider(void);
|
||||
const Invocation = envelope.Invocation;
|
||||
const Answer = envelope.Answer;
|
||||
|
||||
/// The single volume this increment handles: its provider channel, its block
|
||||
/// sub-range, its identity, the id it is addressed by, and the filesystem
|
||||
/// process serving it (0 until spawned; reset on death for respawn).
|
||||
const Volume = struct {
|
||||
storage: block.Device,
|
||||
base_lba: u64,
|
||||
block_count: u64,
|
||||
identity: u64,
|
||||
id: u64,
|
||||
filesystem_pid: u32 = 0,
|
||||
};
|
||||
|
||||
/// The filesystem binary a probed volume is served by. The signature->binary
|
||||
/// map (filesystems.csv) lands with the identity ladder; for now every FAT-shaped
|
||||
/// volume gets the FAT service.
|
||||
const filesystem_binary = "/system/services/fat";
|
||||
const volume_id: u64 = 1;
|
||||
|
||||
var service_endpoint: ipc.Handle = 0;
|
||||
var manager_handle: ?ipc.Handle = null;
|
||||
var bounce: memory.DmaRegion = undefined;
|
||||
var bounce_ready = false;
|
||||
var probed = false;
|
||||
var volume: ?Volume = null;
|
||||
const probe_retry_ms = 500;
|
||||
|
||||
/// The first mass-storage provider's block channel, via the device manager's
|
||||
/// tree — the same lineage acquisition a filesystem makes (block is not a
|
||||
/// registry name). One enumerate sweep; null until the chain is up.
|
||||
fn acquireStorage() ?block.Device {
|
||||
const manager = manager_handle orelse opened: {
|
||||
const handle = channel.openEndpoint("device-manager") orelse return null;
|
||||
@@ -67,9 +89,24 @@ fn acquireStorage() ?block.Device {
|
||||
}
|
||||
}
|
||||
|
||||
/// One probe attempt: acquire the storage channel, read block 0, and parse the
|
||||
/// first volume. Sets `probed` and logs on success; a failure leaves everything
|
||||
/// for the next tick.
|
||||
/// Spawn the filesystem for `v`, confine it to the volume's range, and record
|
||||
/// its pid. The confinement is defined for the fresh pid BEFORE the filesystem
|
||||
/// runs, so its first read is already bounded; the volume manager is the
|
||||
/// confinement controller (it defines the first range on the device).
|
||||
fn spawnFilesystem(v: *Volume) void {
|
||||
const pid = process.spawnSupervised(filesystem_binary, &.{"1"}, service_endpoint) orelse {
|
||||
_ = logging.write("volume-manager: could not spawn the filesystem\n");
|
||||
return;
|
||||
};
|
||||
if (!v.storage.defineRange(pid, v.base_lba, v.block_count)) {
|
||||
_ = logging.write("volume-manager: could not confine the filesystem to its volume\n");
|
||||
_ = process.kill(pid);
|
||||
return;
|
||||
}
|
||||
v.filesystem_pid = pid;
|
||||
std.log.info("volume 0x{x} -> {s} (pid {d}), lba {d}, {d} blocks", .{ v.identity, filesystem_binary, pid, v.base_lba, v.block_count });
|
||||
}
|
||||
|
||||
fn tryProbe() void {
|
||||
if (probed) return;
|
||||
if (!bounce_ready) {
|
||||
@@ -77,29 +114,53 @@ fn tryProbe() void {
|
||||
bounce_ready = true;
|
||||
}
|
||||
const device = acquireStorage() orelse return;
|
||||
// Attach the read buffer to the controller (a no-op success without an
|
||||
// enforcing IOMMU). The volume manager is unconfined — it reads the whole
|
||||
// device to probe — so no range is defined here.
|
||||
if (bounce.handle) |handle| {
|
||||
if (!device.attach(handle)) return;
|
||||
_ = ipc.close(handle);
|
||||
bounce.handle = null; // attached once; do not re-forward on a retry
|
||||
bounce.handle = null;
|
||||
}
|
||||
const geometry = device.geometry() orelse return;
|
||||
if (!device.read(0, 1, bounce.physical)) return;
|
||||
const sector: [*]const u8 = @ptrFromInt(bounce.virtual);
|
||||
const volume = partition.firstVolume(sector[0..512], geometry.block_count) orelse {
|
||||
const found = partition.firstVolume(sector[0..512], geometry.block_count) orelse {
|
||||
_ = logging.write("volume-manager: no volume found on the storage device\n");
|
||||
probed = true; // a device with no recognizable volume is not retried
|
||||
probed = true;
|
||||
return;
|
||||
};
|
||||
std.log.info("volume 0x{x} at lba {d}, {d} blocks", .{ volume.identity, volume.base_lba, volume.block_count });
|
||||
volume = .{ .storage = device, .base_lba = found.base_lba, .block_count = found.block_count, .identity = found.identity, .id = volume_id };
|
||||
probed = true;
|
||||
spawnFilesystem(&volume.?);
|
||||
}
|
||||
|
||||
/// A filesystem announces itself for the volume it was spawned to serve. Reply
|
||||
/// with that volume's block channel (already range-confined to this filesystem's
|
||||
/// badge) as the call's returned capability. No channel means the volume is not
|
||||
/// ready — the filesystem retries.
|
||||
fn onHello(_: void, invocation: Invocation(volume_manager_protocol.Hello), _: Answer(void)) isize {
|
||||
const v = volume orelse return 0; // not probed yet — retryable, no cap
|
||||
if (invocation.target != v.id) return 0; // unknown volume — retryable
|
||||
if (invocation.sender != v.filesystem_pid) {
|
||||
// Not the filesystem we spawned for this volume. Refuse: only the
|
||||
// confined filesystem gets the channel.
|
||||
std.log.info("refused hello for volume {d} from process {d}", .{ invocation.target, invocation.sender });
|
||||
return -envelope.EPERM;
|
||||
}
|
||||
service.replyWithCapability(v.storage.endpoint);
|
||||
std.log.info("handed volume {d} to pid {d}", .{ v.id, invocation.sender });
|
||||
return 0;
|
||||
}
|
||||
|
||||
const handlers = Serve.Handlers{ .hello = onHello };
|
||||
|
||||
fn onMessage(message: []const u8, out: []u8, sender: u32, arrived: *ipc.Arrival) usize {
|
||||
// No verb takes a capability up, so the turn closes whatever arrives.
|
||||
return Serve.dispatch({}, handlers, message, sender, arrived.peek(), out);
|
||||
}
|
||||
|
||||
fn initialise(endpoint: ipc.Handle) bool {
|
||||
service_endpoint = endpoint;
|
||||
_ = logging.write("volume-manager: starting, waiting for a storage device\n");
|
||||
_ = process.subscribeExits(endpoint);
|
||||
tryProbe();
|
||||
if (!probed) _ = time.timerOnce(endpoint, probe_retry_ms);
|
||||
return true;
|
||||
@@ -110,23 +171,26 @@ fn onNotification(badge: u64) void {
|
||||
if (got.isTimer()) {
|
||||
tryProbe();
|
||||
if (!probed) _ = time.timerOnce(service_endpoint, probe_retry_ms);
|
||||
return;
|
||||
}
|
||||
// A filesystem died. Its old range is reclaimed by the driver on the same
|
||||
// death; respawn it, confined afresh to the same volume (a fresh pid, a
|
||||
// fresh range). The reap-and-rebuild the device manager proved, one layer up.
|
||||
if (got.isChildExit()) {
|
||||
const dead = got.childProcessId();
|
||||
if (volume) |*v| {
|
||||
if (v.filesystem_pid == dead) {
|
||||
v.filesystem_pid = 0;
|
||||
std.log.info("filesystem for volume {d} died; respawning", .{v.id});
|
||||
spawnFilesystem(v);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// No clients yet: a filesystem hello lands here in the next increment. Until
|
||||
/// then, refuse politely.
|
||||
fn onMessage(message: []const u8, out: []u8, sender: u32, arrived: *ipc.Arrival) usize {
|
||||
_ = message;
|
||||
_ = sender;
|
||||
_ = arrived;
|
||||
const status = envelope.Status{ .status = -envelope.ENOSYS, .len = 0 };
|
||||
@memcpy(out[0..envelope.prefix_size], std.mem.asBytes(&status));
|
||||
return envelope.prefix_size;
|
||||
}
|
||||
|
||||
pub fn main(init: process.Init) void {
|
||||
_ = init;
|
||||
service.run(device_manager_protocol.message_maximum, .{
|
||||
service.run(volume_manager_protocol.message_maximum, .{
|
||||
.service = "volume-manager",
|
||||
.init = initialise,
|
||||
.on_message = onMessage,
|
||||
|
||||
Reference in New Issue
Block a user