library: five protocols speak the envelope

The folded header stops being a rule in a document and becomes the layout
on the wire. Verbs number from sixteen, leaving describe, enumerate,
subscribe and unsubscribe reserved and answered the same way by every
provider — none of them writes a line to do it. What each protocol used to
carry in a field of its own now travels in the header: a vfs node and a
display layer are the packet's target, and a reply opens with a status the
envelope stamps rather than one each protocol spelled for itself.

Display gains the most. One forty-byte request had served eleven verbs, so
attach_scanout smuggled stride through x, refresh through y and format
through colour, and every coordinate crossed as a bitcast. Per-operation
structs end all three: the fields have their own names and their own signs,
and the tile payload grows to 224 bytes because the prefix shrank. Scanout
loses a message maximum of 64 it had no business declaring — it answers
calls, and the floor for a call is 256 — and virtio-gpu stops hard-coding
that number at its harness.

Two changes are semantic rather than notational. A directory now ends at an
entry with no name, because the fixed part of a reply always travels and a
zero-length reply no longer exists to mean anything. And input joins the
service harness, the last loop in the tree that answered no ping and heard
no terminate; its subscriber table, its pruning and its fan-out are the
same code, and a shutdown now asks it to stop instead of killing it.

A new conformance case reads the registry's own listing and asks every
protocol it finds for its name, its version and its verb count, then offers
a verb nobody defines and requires -ENOSYS — the envelope's promise,
checked against providers rather than against itself. What it cannot reach
in that boot it names on the serial line instead of passing quietly.

Suite 110/110.
This commit is contained in:
Daniel Samson
2026-08-01 06:15:25 +01:00
parent b004b9c3eb
commit d2dfbcabf8
40 changed files with 1882 additions and 1008 deletions
+33 -24
View File
@@ -13,6 +13,7 @@ const memory = @import("memory");
const logging = @import("logging");
const compositor = @import("compositor.zig");
const envelope = @import("envelope");
const scanout_protocol = @import("scanout-protocol");
const Rect = compositor.Rect;
const Surface = compositor.Surface;
@@ -188,45 +189,53 @@ pub const VirtioGpu = struct {
/// accumulated; the driver transfers + fenced-flushes the whole frame.
pub fn present(self: *const VirtioGpu, damage: []const Rect) void {
_ = damage;
var request = scanout_protocol.Request{
.operation = @intFromEnum(scanout_protocol.Operation.present),
.width = self.width,
.height = self.height,
};
var reply: [scanout_protocol.reply_size]u8 = undefined;
_ = ipc.call(self.scanout, std.mem.asBytes(&request), &reply) catch {};
_ = call(self.scanout, .present, .{ .width = self.width, .height = self.height });
}
/// Fill `out` with the driver's offered modes; returns how many were written.
pub fn modes(self: *const VirtioGpu, out: []Mode) usize {
var request = scanout_protocol.Request{ .operation = @intFromEnum(scanout_protocol.Operation.get_modes) };
var reply: [scanout_protocol.modes_reply_size]u8 = undefined;
const n = ipc.call(self.scanout, std.mem.asBytes(&request), &reply) catch return 0;
if (n < scanout_protocol.modes_reply_size) return 0;
const answer = std.mem.bytesToValue(scanout_protocol.ModesReply, reply[0..scanout_protocol.modes_reply_size]);
if (answer.status != 0) return 0;
const count = @min(@min(answer.count, scanout_protocol.max_modes), out.len);
for (0..count) |i| out[i] = answer.modes[i];
const answered = call(self.scanout, .get_modes, {}) orelse return 0;
const offered = Scanout.decodeReply(.get_modes, answered.packet[0..answered.len]) orelse return 0;
const count = @min(@min(offered.count, scanout_protocol.max_modes), out.len);
for (0..count) |i| out[i] = offered.modes[i];
return count;
}
/// Change the scanout resolution. On success the active `width`/`height` update (the shared
/// surface — sized to the max mode — is unchanged, so `stride` stays put).
pub fn setMode(self: *VirtioGpu, w: u32, h: u32) bool {
if (w == 0 or h == 0 or w > self.stride) return false;
var request = scanout_protocol.Request{
.operation = @intFromEnum(scanout_protocol.Operation.set_mode),
.width = w,
.height = h,
};
var reply: [scanout_protocol.reply_size]u8 = undefined;
const n = ipc.call(self.scanout, std.mem.asBytes(&request), &reply) catch return false;
if (n < scanout_protocol.reply_size) return false;
if (std.mem.bytesToValue(scanout_protocol.Reply, reply[0..scanout_protocol.reply_size]).status != 0) return false;
_ = call(self.scanout, .set_mode, .{ .width = w, .height = h }) orelse return false;
self.width = w;
self.height = h;
return true;
}
};
const Scanout = scanout_protocol.Protocol;
/// A reply the driver answered with, kept whole so the caller can decode the
/// verb's own fixed part out of it.
const Answered = struct {
packet: [scanout_protocol.message_maximum]u8,
len: usize,
};
/// One request at the scanout driver. Null covers both a transport failure and a
/// driver that refused — a present that did not happen is a present that did not
/// happen, and this backend has nothing to do about either but skip the frame.
fn call(
scanout: ipc.Handle,
comptime operation: Scanout.Operation,
request: Scanout.RequestOf(operation),
) ?Answered {
var packet: [scanout_protocol.message_maximum]u8 = undefined;
const framed = Scanout.encodeRequest(operation, 0, request, &.{}, &packet) orelse return null;
var answered: Answered = .{ .packet = undefined, .len = 0 };
answered.len = ipc.call(scanout, framed, &answered.packet) catch return null;
const status = envelope.statusOf(answered.packet[0..answered.len]) orelse return null;
if (status.status != 0) return null;
return answered;
}
/// The pluggable scanout backend. A tagged union so the compositor holds one value and
/// dispatches without caring which is active; the `virtio` native backend joins `gop` at V4.
pub const Backend = union(enum) {
+3 -2
View File
@@ -10,8 +10,9 @@ pub fn build(b: *std.Build) void {
.name = "display",
.root_source_file = b.path("display.zig"),
.imports = &.{
"channel", "display-client", "display-protocol", "driver", "input-client", "ipc",
"logging", "memory", "scanout-protocol", "service", "thread", "time",
"channel", "display-client", "display-protocol", "driver", "envelope", "input-client",
"ipc", "logging", "memory", "scanout-protocol", "service",
"thread", "time",
},
.threaded = true, // real atomics/TLS (docs/threading.md)
});
+121 -81
View File
@@ -28,10 +28,22 @@ const logging = @import("logging");
const compositor = @import("compositor.zig");
const backend_mod = @import("backend.zig");
const envelope = @import("envelope");
const display_protocol = @import("display-protocol");
const Rect = compositor.Rect;
const Surface = compositor.Surface;
/// The generated display dispatch. One compositor per process, so the handler
/// context is empty and the layer stack stays in this file's globals.
const Serve = display_protocol.Protocol.Provider(void);
const Invocation = envelope.Invocation;
const Answer = envelope.Answer;
/// What a handler returns when the layer named in `Header.target` is not one of
/// ours, or a mode was refused.
const refused: isize = -envelope.ENOENT;
/// The active scanout backend — the GOP framebuffer at boot, upgraded to a native driver
/// (virtio-gpu) when one announces itself (V4).
var backend: backend_mod.Backend = undefined;
@@ -314,11 +326,18 @@ fn verifyNativePresent() void {
/// `systemSharedMemoryMap`) — the pixels stay ours after the handle naming them
/// goes, and a driver that dies and re-announces no longer costs a handle slot
/// per restart.
fn attachScanout(stride: u32, width: u32, height: u32, format: u32, refresh_hz: u32, arrived: *ipc.Arrival, reply: []u8) usize {
const cap = arrived.peek() orelse return fail(reply);
if (width == 0 or height == 0 or stride < width) return fail(reply);
const mapped = memory.sharedMap(cap) orelse return fail(reply);
const scanout = channel.openEndpoint("scanout") orelse return fail(reply);
fn onAttachScanout(_: void, invocation: Invocation(display_protocol.AttachScanout), _: Answer(void)) isize {
const announce = invocation.request;
const stride = announce.stride;
const width = announce.width;
const height = announce.height;
const format = announce.format;
const refresh_hz = announce.refresh_hz;
const cap = invocation.capability orelse return refused;
if (width == 0 or height == 0 or stride < width) return refused;
const mapped = memory.sharedMap(cap) orelse return refused;
const scanout = channel.openEndpoint("scanout") orelse return refused;
// A second announce means the driver died and was restarted (V6): re-attach to its fresh
// scanout. (The previous shared mapping leaks — there is no shared_memory_unmap syscall yet — but the
// frames are the dead driver's, reclaimed on its exit; a handful across a crash is benign.)
@@ -346,7 +365,7 @@ fn attachScanout(stride: u32, width: u32, height: u32, format: u32, refresh_hz:
"display: scanout re-attached\n"
else
"display: scanout upgraded to virtio-gpu\n");
return ok(reply);
return 0;
}
/// After the native upgrade is verified, prove the runtime-resolution-change and fenced-present
@@ -609,88 +628,109 @@ fn initialise(endpoint: ipc.Handle) bool {
return true;
}
fn writeReply(reply: []u8, value: display_protocol.Reply) usize {
const bytes = std.mem.asBytes(&value);
@memcpy(reply[0..bytes.len], bytes);
return bytes.len;
// --- the protocol handlers --------------------------------------------------
//
// A layer id is `Header.target` on every verb that names one, so no handler
// reads a layer out of its own request any more. `target` is a u64 and a layer
// id a u32: a value that does not fit is not a layer of ours, and `layerAt`
// refuses it the same way an out-of-range one is refused.
fn targetLayer(target: u64) ?u32 {
if (target > std.math.maxInt(u32)) return null;
return @intCast(target);
}
fn ok(reply: []u8) usize {
return writeReply(reply, .{ .status = 0 });
fn onInfo(_: void, _: Invocation(void), answer: Answer(display_protocol.Info)) isize {
const mode = backend.info();
answer.set(.{ .width = mode.width, .height = mode.height, .pitch = mode.pitch, .format = mode.format });
return 0;
}
fn fail(reply: []u8) usize {
return writeReply(reply, .{ .status = -1 });
fn onCreateLayer(_: void, invocation: Invocation(display_protocol.CreateLayer), answer: Answer(display_protocol.Created)) isize {
const request = invocation.request;
const slot = createLayer(request.x, request.y, request.width, request.height, request.z, request.visible != 0) orelse return refused;
answer.set(.{ .layer = slot });
return 0;
}
fn onConfigureLayer(_: void, invocation: Invocation(display_protocol.ConfigureLayer), _: Answer(void)) isize {
const id = targetLayer(invocation.target) orelse return refused;
const request = invocation.request;
return if (configureLayer(id, request.x, request.y, request.z, request.visible != 0)) 0 else refused;
}
fn onDestroyLayer(_: void, invocation: Invocation(void), _: Answer(void)) isize {
const id = targetLayer(invocation.target) orelse return refused;
return if (destroyLayer(id)) 0 else refused;
}
fn onFillRect(_: void, invocation: Invocation(display_protocol.FillRect), _: Answer(void)) isize {
const id = targetLayer(invocation.target) orelse return refused;
const request = invocation.request;
const local = Rect.init(request.x, request.y, @intCast(request.width), @intCast(request.height));
return if (fillLayer(id, local, request.colour)) 0 else refused;
}
fn onBlitTile(_: void, invocation: Invocation(display_protocol.BlitTile), _: Answer(void)) isize {
const id = targetLayer(invocation.target) orelse return refused;
const request = invocation.request;
return if (blitLayer(id, request.x, request.y, request.width, request.height, invocation.tail)) 0 else refused;
}
fn onDamage(_: void, invocation: Invocation(display_protocol.Damage), _: Answer(void)) isize {
const id = targetLayer(invocation.target) orelse return refused;
const l = layerAt(id) orelse return refused;
const request = invocation.request;
const screen = Rect{ .x = l.x + request.x, .y = l.y + request.y, .w = @intCast(request.width), .h = @intCast(request.height) };
addDamage(screen.intersect(layerScreenRect(l)));
return 0;
}
fn onPresent(_: void, _: Invocation(void), _: Answer(void)) isize {
// Scheduled, not immediate: the frame clock composites the accumulated damage
// at the next tick, so back-to-back client presents coalesce into one frame.
schedulePresent();
return 0;
}
fn onSetMode(_: void, invocation: Invocation(display_protocol.SetMode), _: Answer(void)) isize {
if (!backend.setMode(invocation.request.width, invocation.request.height)) return refused;
addDamage(screenRect()); // repaint the whole screen at the new resolution
present();
return 0;
}
fn onGetModes(_: void, _: Invocation(void), answer: Answer(display_protocol.Modes)) isize {
var list: [4]backend_mod.Mode = undefined;
const count = backend.modes(&list);
var modes = display_protocol.Modes{ .count = @intCast(count) };
for (0..@min(count, display_protocol.max_modes)) |i| {
modes.modes[i] = .{ .width = list[i].width, .height = list[i].height };
}
answer.set(modes);
return 0;
}
const handlers = Serve.Handlers{
.info = onInfo,
.create_layer = onCreateLayer,
.configure_layer = onConfigureLayer,
.destroy_layer = onDestroyLayer,
.fill_rect = onFillRect,
.blit_tile = onBlitTile,
.damage = onDamage,
.present = onPresent,
.attach_scanout = onAttachScanout,
.set_mode = onSetMode,
.get_modes = onGetModes,
};
fn onMessage(message: []const u8, reply: []u8, sender: u32, arrived: *ipc.Arrival) usize {
_ = sender;
if (message.len < display_protocol.request_size) return fail(reply);
const request = std.mem.bytesToValue(display_protocol.Request, message[0..display_protocol.request_size]);
const payload = message[display_protocol.request_size..];
// Switch on the raw operation value — an out-of-range one must fail cleanly, not
// panic an `@enumFromInt`.
switch (request.operation) {
@intFromEnum(display_protocol.Operation.info) => {
const m = backend.info();
return writeReply(reply, .{ .status = 0, .width = m.width, .height = m.height, .pitch = m.pitch, .format = m.format });
},
@intFromEnum(display_protocol.Operation.create_layer) => {
// x/y are signed coordinates carried in the u32 wire fields — reinterpret the
// bits (@bitCast), don't range-check (@intCast) which a negative would fail.
const slot = createLayer(@bitCast(request.x), @bitCast(request.y), request.width, request.height, request.z, request.visible != 0) orelse return fail(reply);
return writeReply(reply, .{ .status = 0, .layer = slot });
},
@intFromEnum(display_protocol.Operation.configure_layer) => {
return if (configureLayer(request.layer, @bitCast(request.x), @bitCast(request.y), request.z, request.visible != 0)) ok(reply) else fail(reply);
},
@intFromEnum(display_protocol.Operation.destroy_layer) => {
return if (destroyLayer(request.layer)) ok(reply) else fail(reply);
},
@intFromEnum(display_protocol.Operation.fill_rect) => {
const local = Rect.init(@bitCast(request.x), @bitCast(request.y), @intCast(request.width), @intCast(request.height));
return if (fillLayer(request.layer, local, request.colour)) ok(reply) else fail(reply);
},
@intFromEnum(display_protocol.Operation.blit_tile) => {
return if (blitLayer(request.layer, @bitCast(request.x), @bitCast(request.y), request.width, request.height, payload)) ok(reply) else fail(reply);
},
@intFromEnum(display_protocol.Operation.damage) => {
const l = layerAt(request.layer) orelse return fail(reply);
const screen = Rect{ .x = l.x + @as(i32, @bitCast(request.x)), .y = l.y + @as(i32, @bitCast(request.y)), .w = @intCast(request.width), .h = @intCast(request.height) };
addDamage(screen.intersect(layerScreenRect(l)));
return ok(reply);
},
@intFromEnum(display_protocol.Operation.present) => {
// Scheduled, not immediate: the frame clock composites the accumulated damage
// at the next tick, so back-to-back client presents coalesce into one frame.
schedulePresent();
return ok(reply);
},
@intFromEnum(display_protocol.Operation.attach_scanout) => {
return attachScanout(request.x, request.width, request.height, request.colour, request.y, arrived, reply);
},
@intFromEnum(display_protocol.Operation.set_mode) => {
if (!backend.setMode(request.width, request.height)) return fail(reply);
addDamage(screenRect()); // repaint the whole screen at the new resolution
present();
return ok(reply);
},
@intFromEnum(display_protocol.Operation.get_modes) => {
var list: [4]backend_mod.Mode = undefined;
const count = backend.modes(&list);
var response = display_protocol.ModesReply{ .status = 0, .count = @intCast(count), .modes = undefined };
for (0..display_protocol.max_modes) |i| {
response.modes[i] = if (i < count)
.{ .width = list[i].width, .height = list[i].height }
else
.{ .width = 0, .height = 0 };
}
const bytes = std.mem.asBytes(&response);
@memcpy(reply[0..bytes.len], bytes);
return bytes.len;
},
else => return fail(reply),
}
// The one capability this service is ever handed is the scanout driver's
// shared surface, and `attachScanout` deliberately does not claim it (the
// mapping holds its own reference) — so the capability is peeked, never
// taken, and the turn closes it.
return Serve.dispatch({}, handlers, message, sender, arrived.peek(), reply);
}
/// Two notification sources reach the compositor, and one coalesced badge can carry
+2 -2
View File
@@ -10,8 +10,8 @@ pub fn build(b: *std.Build) void {
.name = "fat",
.root_source_file = b.path("fat.zig"),
.imports = &.{
"block", "file-system", "ipc", "logging", "memory", "process", "service", "time",
"vfs-protocol",
"block", "envelope", "file-system", "ipc", "logging", "memory", "process",
"service", "time", "vfs-protocol",
},
});
b.installArtifact(exe);
+131 -99
View File
@@ -20,8 +20,17 @@ const memory = @import("memory");
const logging = @import("logging");
const engine = @import("engine.zig");
const on_disk = @import("on-disk.zig");
const envelope = @import("envelope");
const vfs_protocol = @import("vfs-protocol");
/// The generated vfs dispatch, bound to this server. There is one FAT volume per
/// process, so the handler context is empty and the state stays where it was: in
/// this file's globals.
const Serve = vfs_protocol.Protocol.Provider(void);
const Invocation = envelope.Invocation;
const Answer = envelope.Answer;
const mount_point = "/volumes/usb";
// The engine's BlockDevice, backed by the `.block` driver plus a DMA bounce
@@ -75,16 +84,11 @@ fn openAt(id: u64) ?*OpenNode {
return if (o.used) o else null;
}
fn writeReply(out: []u8, reply: vfs_protocol.Reply, payload: []const u8) usize {
@memcpy(out[0..vfs_protocol.reply_size], std.mem.asBytes(&reply));
const n = @min(payload.len, out.len - vfs_protocol.reply_size);
@memcpy(out[vfs_protocol.reply_size..][0..n], payload[0..n]);
return vfs_protocol.reply_size + n;
}
fn fail(out: []u8) usize {
return writeReply(out, .{ .status = -1 }, &.{});
}
/// What a handler returns when the thing asked for is not there — a bad node id,
/// a path that does not resolve, a mutation the volume refused. One errno for all
/// of them, because a filesystem's failures are all "no such thing" as far as the
/// file API can act on them.
const refused: isize = -envelope.ENOENT;
/// How often to look for a block device while none is mounted. Storage arriving
/// is EVENT-shaped (the usb chain registering, possibly after a driver restart),
@@ -204,24 +208,128 @@ fn splitParent(path: []const u8) ParentLeaf {
};
}
fn handleOpen(out: []u8, path: []const u8, flags: u32, sender: u32) usize {
fn onOpen(_: void, invocation: Invocation(vfs_protocol.Open), answer: Answer(vfs_protocol.Opened)) isize {
const path = invocation.tail;
const flags = invocation.request.flags;
var node = filesystem.resolve(path);
if (node == null and flags & vfs_protocol.create != 0) {
const split = splitParent(path);
const parent = filesystem.resolve(split.parent) orelse return fail(out);
const parent = filesystem.resolve(split.parent) orelse return refused;
node = filesystem.createFile(parent, split.leaf);
}
var resolved = node orelse return fail(out);
var resolved = node orelse return refused;
// O_TRUNC: replace an existing file's contents rather than overwriting in place
// (frees the old chain, so a shorter rewrite leaves no stale tail).
if (flags & vfs_protocol.truncate != 0 and !resolved.is_directory) {
filesystem.truncate(&resolved);
}
const index = allocOpen() orelse return fail(out);
open_nodes[index] = .{ .used = true, .node = resolved, .owner = sender };
return writeReply(out, .{ .status = 0, .node = index }, &.{});
const index = allocOpen() orelse return refused;
open_nodes[index] = .{ .used = true, .node = resolved, .owner = invocation.sender };
answer.set(.{ .node = index });
return 0;
}
fn onRead(_: void, invocation: Invocation(vfs_protocol.Read), answer: Answer(void)) isize {
const o = openAt(invocation.target) orelse return refused;
const into = answer.tail();
const want = @min(@as(usize, invocation.request.len), into.len);
return @intCast(filesystem.readFile(o.node, @intCast(invocation.request.offset), into[0..want]));
}
fn onWrite(_: void, invocation: Invocation(vfs_protocol.Write), answer: Answer(vfs_protocol.Written)) isize {
const o = openAt(invocation.target) orelse return refused;
const data = invocation.tail[0..@min(invocation.tail.len, invocation.request.len)];
const n = filesystem.writeFile(&o.node, @intCast(invocation.request.offset), data);
answer.set(.{ .count = @intCast(n) });
return 0;
}
fn onStatus(_: void, invocation: Invocation(void), answer: Answer(vfs_protocol.FileStatus)) isize {
const o = openAt(invocation.target) orelse return refused;
const kind: vfs_protocol.NodeKind = if (o.node.is_directory) .directory else .regular;
answer.set(.{ .size = o.node.size, .kind = @intFromEnum(kind), .mtime = o.node.mtime });
return 0;
}
/// One entry per call. End of directory — a node that is not a directory, or a
/// cursor past the last child — is an entry with no name, which is how the
/// protocol spells it now that the reply's length always counts the fixed part.
fn onReaddir(_: void, invocation: Invocation(vfs_protocol.Readdir), answer: Answer(vfs_protocol.DirectoryEntry)) isize {
const o = openAt(invocation.target) orelse return refused;
if (!o.node.is_directory) {
answer.set(.{});
return 0;
}
const listing = filesystem.listEntry(o.node, @intCast(invocation.request.cursor)) orelse {
answer.set(.{});
return 0;
};
const kind: vfs_protocol.NodeKind = if (listing.is_directory) .directory else .regular;
const into = answer.tail();
const name_len = @min(listing.name_len, into.len);
@memcpy(into[0..name_len], listing.name_buffer[0..name_len]);
answer.set(.{ .kind = @intFromEnum(kind), .name_len = @intCast(name_len), .size = listing.size });
return @intCast(name_len);
}
fn onClose(_: void, invocation: Invocation(void), _: Answer(void)) isize {
if (openAt(invocation.target)) |o| o.used = false;
// Durable-on-close: if any block reached the device since the last flush,
// commit its cache to stable media now (best-effort). This is what makes
// init's shutdown log flush survive a real power-off, and is the right
// default for removable media the user may unplug.
if (device_dirty) {
_ = ipc_block.device.flush();
device_dirty = false;
}
return 0;
}
fn onMakeDirectory(_: void, invocation: Invocation(void), _: Answer(void)) isize {
const path = invocation.tail;
if (filesystem.resolve(path) != null) return refused; // already exists — no duplicate entries
const split = splitParent(path);
const parent = filesystem.resolve(split.parent) orelse return refused;
if (filesystem.createDirectory(parent, split.leaf) == null) return refused;
return 0;
}
fn onUnlink(_: void, invocation: Invocation(void), _: Answer(void)) isize {
const split = splitParent(invocation.tail);
const parent = filesystem.resolve(split.parent) orelse return refused;
if (!filesystem.removeFile(parent, split.leaf)) return refused;
return 0;
}
fn onRename(_: void, invocation: Invocation(void), _: Answer(void)) isize {
const both = invocation.tail;
const separator = std.mem.indexOfScalar(u8, both, 0) orelse return refused;
const old_split = splitParent(both[0..separator]);
const new_split = splitParent(both[separator + 1 ..]);
// Same-directory rename only.
if (!std.mem.eql(u8, old_split.parent, new_split.parent)) return refused;
const parent = filesystem.resolve(old_split.parent) orelse return refused;
if (!filesystem.rename(parent, old_split.leaf, new_split.leaf)) return refused;
return 0;
}
/// The verbs this backend implements. The three it leaves out — `mount`,
/// `unmount`, `bind` — answer `-ENOSYS` from the generated dispatch, which is
/// exactly right: path routing is the kernel's now, and only init implements
/// `bind` (docs/os-development/protocol-namespace.md). `describe` is the
/// envelope's own.
const handlers = Serve.Handlers{
.open = onOpen,
.close = onClose,
.read = onRead,
.write = onWrite,
.status = onStatus,
.readdir = onReaddir,
.mkdir = onMakeDirectory,
.unlink = onUnlink,
.rename = onRename,
};
/// The vfs protocol has no operation that takes a capability, so `arrived` is
/// never claimed here — which, under the harness's ownership rule, means the
/// loop closes whatever a caller attached. That is the point of the rule: this
@@ -230,92 +338,16 @@ fn handleOpen(out: []u8, path: []const u8, flags: u32, sender: u32) usize {
/// thirty-two until it could accept no capability at all.
fn onMessage(message: []const u8, out: []u8, sender: u32, arrived: *ipc.Arrival) usize {
_ = arrived;
if (!mounted) return fail(out); // storage not up (yet): fail politely, clients retry
if (message.len < vfs_protocol.request_size) return fail(out);
const request = std.mem.bytesToValue(vfs_protocol.Request, message[0..vfs_protocol.request_size]);
const payload = message[vfs_protocol.request_size..];
// Storage not up (yet): fail politely, whatever was asked — clients retry.
if (!mounted) {
const status = envelope.Status{ .status = refused, .len = 0 };
@memcpy(out[0..envelope.prefix_size], std.mem.asBytes(&status));
return envelope.prefix_size;
}
// Stamp create/write with the current wall-clock time (mtime). Cheap, and it
// keeps the engine pure (it takes the time as data, not a syscall).
filesystem.current_time_epoch = time.wallClock();
switch (request.operation) {
.open => return handleOpen(out, payload[0..@min(payload.len, request.len)], request.flags, sender),
.read => {
const o = openAt(request.node) orelse return fail(out);
var buffer: [vfs_protocol.maximum_payload]u8 = undefined;
const want = @min(@as(usize, request.len), buffer.len);
const n = filesystem.readFile(o.node, @intCast(request.offset), buffer[0..want]);
return writeReply(out, .{ .status = 0, .len = @intCast(n) }, buffer[0..n]);
},
.write => {
const o = openAt(request.node) orelse return fail(out);
const data = payload[0..@min(payload.len, request.len)];
const n = filesystem.writeFile(&o.node, @intCast(request.offset), data);
return writeReply(out, .{ .status = 0, .len = @intCast(n) }, &.{});
},
.status => {
const o = openAt(request.node) orelse return fail(out);
const kind: vfs_protocol.NodeKind = if (o.node.is_directory) .directory else .regular;
const status = vfs_protocol.FileStatus{ .size = o.node.size, .kind = @intFromEnum(kind), .mtime = o.node.mtime };
return writeReply(out, .{ .status = 0, .len = @sizeOf(vfs_protocol.FileStatus) }, std.mem.asBytes(&status));
},
.readdir => {
const o = openAt(request.node) orelse return fail(out);
if (!o.node.is_directory) return writeReply(out, .{ .status = 0, .len = 0 }, &.{});
const listing = filesystem.listEntry(o.node, @intCast(request.offset)) orelse return writeReply(out, .{ .status = 0, .len = 0 }, &.{});
const kind: vfs_protocol.NodeKind = if (listing.is_directory) .directory else .regular;
const header = vfs_protocol.DirectoryEntry{ .kind = @intFromEnum(kind), .name_len = @intCast(listing.name_len), .size = listing.size };
var buffer: [vfs_protocol.maximum_payload]u8 = undefined;
@memcpy(buffer[0..vfs_protocol.directory_entry_size], std.mem.asBytes(&header));
const nlen = @min(listing.name_len, buffer.len - vfs_protocol.directory_entry_size);
@memcpy(buffer[vfs_protocol.directory_entry_size..][0..nlen], listing.name_buffer[0..nlen]);
const total = vfs_protocol.directory_entry_size + nlen;
return writeReply(out, .{ .status = 0, .len = @intCast(total) }, buffer[0..total]);
},
.close => {
if (openAt(request.node)) |o| o.used = false;
// Durable-on-close: if any block reached the device since the last
// flush, commit its cache to stable media now (best-effort). This is
// what makes init's shutdown log flush survive a real power-off, and is
// the right default for removable media the user may unplug.
if (device_dirty) {
_ = ipc_block.device.flush();
device_dirty = false;
}
return writeReply(out, .{ .status = 0 }, &.{});
},
.mkdir => {
const path = payload[0..@min(payload.len, request.len)];
if (filesystem.resolve(path) != null) return fail(out); // already exists — no duplicate entries
const split = splitParent(path);
const parent = filesystem.resolve(split.parent) orelse return fail(out);
if (filesystem.createDirectory(parent, split.leaf) == null) return fail(out);
return writeReply(out, .{ .status = 0 }, &.{});
},
.unlink => {
const split = splitParent(payload[0..@min(payload.len, request.len)]);
const parent = filesystem.resolve(split.parent) orelse return fail(out);
if (!filesystem.removeFile(parent, split.leaf)) return fail(out);
return writeReply(out, .{ .status = 0 }, &.{});
},
.rename => {
const both = payload[0..@min(payload.len, request.len)];
const sep = std.mem.indexOfScalar(u8, both, 0) orelse return fail(out);
const old_split = splitParent(both[0..sep]);
const new_split = splitParent(both[sep + 1 ..]);
// Same-directory rename only.
if (!std.mem.eql(u8, old_split.parent, new_split.parent)) return fail(out);
const parent = filesystem.resolve(old_split.parent) orelse return fail(out);
if (!filesystem.rename(parent, old_split.leaf, new_split.leaf)) return fail(out);
return writeReply(out, .{ .status = 0 }, &.{});
},
// A backend is never itself a mount target.
// Router verbs, and the registry's claim verb: a file backend answers
// none of them (docs/os-development/protocol-namespace.md — only init
// implements `bind`).
.mount, .unmount, .bind => return fail(out),
}
return Serve.dispatch({}, handlers, message, sender, null, out);
}
pub fn main() void {
+50 -30
View File
@@ -548,34 +548,43 @@ var heartbeat_running = false;
/// `pending_capability`. `arrived` is the capability the *request* carried, owned
/// by the turn — nothing here has to close it, only `bind` has to claim it.
fn serveRegistry(request_bytes: []const u8, reply: []u8, sender: u32, arrived: *Arrival) usize {
if (request_bytes.len < vfs_protocol.request_size)
return answer(reply, -envelope.EPROTO, 0, 0);
// The header is read field by field rather than reinterpreted whole: the
// operation is an enum on the wire and the bytes come from anyone at all, so
// a value outside it must be a refusal, never a decoded enum.
if (request_bytes.len < envelope.prefix_size)
return answer(reply, -envelope.EPROTO, 0);
// The header is read field by field rather than reinterpreted whole, and the
// verb is compared as a number rather than decoded into the generated
// `Operation`: the bytes come from anyone at all, so a value outside the enum
// must be a refusal, never an `@enumFromInt`. This is deliberately NOT
// `Protocol.Provider.dispatch` for the same reason — PID 1 reads a stranger's
// packet, and it reads it by hand.
const operation = std.mem.readInt(u32, request_bytes[0..4], .little);
const cursor = std.mem.readInt(u64, request_bytes[16..24], .little);
const declared = std.mem.readInt(u32, request_bytes[24..28], .little);
const payload_len = @min(@as(usize, declared), request_bytes.len - vfs_protocol.request_size);
const payload = request_bytes[vfs_protocol.request_size..][0..payload_len];
const body = request_bytes[envelope.prefix_size..];
if (operation == @intFromEnum(vfs_protocol.Operation.bind))
return answer(reply, onBind(sender, payload, arrived), 0, 0);
return answer(reply, onBind(sender, body, arrived), 0);
// Only `bind` claims a capability; one attached to anything else is closed by
// the turn's `defer` in the loop, along with the ones sent to a request that
// was too short to name a verb at all.
if (operation == @intFromEnum(vfs_protocol.Operation.open)) return onOpen(reply, sender, payload);
if (operation == @intFromEnum(vfs_protocol.Operation.readdir)) return onReaddir(reply, cursor);
if (operation == @intFromEnum(vfs_protocol.Operation.open)) {
// `open`'s fixed part is the flags word, which means nothing to a
// namespace; the name follows it as the packet's tail.
if (body.len < @sizeOf(vfs_protocol.Open)) return answer(reply, -envelope.EPROTO, 0);
return onOpen(reply, sender, body[@sizeOf(vfs_protocol.Open)..]);
}
if (operation == @intFromEnum(vfs_protocol.Operation.readdir)) {
if (body.len < @sizeOf(vfs_protocol.Readdir)) return answer(reply, -envelope.EPROTO, 0);
return onReaddir(reply, std.mem.readInt(u64, body[0..8], .little));
}
// Everything else a filesystem answers is meaningless here: `/protocol` holds
// contracts, not bytes.
return answer(reply, -envelope.ENOSYS, 0, 0);
return answer(reply, -envelope.ENOSYS, 0);
}
/// Lay down a vfs reply header (and say how many payload bytes follow it).
fn answer(reply: []u8, status: i32, node: u64, payload_len: usize) usize {
const header = vfs_protocol.Reply{ .status = status, .node = node, .len = @intCast(payload_len) };
@memcpy(reply[0..vfs_protocol.reply_size], std.mem.asBytes(&header));
return vfs_protocol.reply_size + payload_len;
/// Lay down the envelope's reply prefix (and say how many payload bytes the
/// caller has already written after it).
fn answer(reply: []u8, status: i32, payload_len: usize) usize {
const header = envelope.Status{ .status = status, .len = @intCast(payload_len) };
@memcpy(reply[0..envelope.prefix_size], std.mem.asBytes(&header));
return envelope.prefix_size + payload_len;
}
/// `bind(name, capability = the provider's endpoint)`. The capability is the
@@ -667,15 +676,19 @@ fn onBind(sender: u32, raw_name: []const u8, arrived: *Arrival) i32 {
/// other, a line in a world-readable log ring, or a serial write costing
/// milliseconds.)
fn onOpen(reply: []u8, sender: u32, raw_name: []const u8) usize {
const name = contractName(raw_name) orelse return answer(reply, -envelope.ENOENT, 0, 0);
const name = contractName(raw_name) orelse return answer(reply, -envelope.ENOENT, 0);
refreshProcessTable();
const identity = identify(sender);
const permitted = if (identity) |who| mayOpen(who, name) else false;
const binding = findBinding(name);
if (!permitted) return answer(reply, -envelope.ENOENT, 0, 0);
const found = binding orelse return answer(reply, -envelope.ENOENT, 0, 0);
if (!permitted) return answer(reply, -envelope.ENOENT, 0);
const found = binding orelse return answer(reply, -envelope.ENOENT, 0);
pending_capability = found.endpoint;
return answer(reply, 0, 0, 0);
// A contract node has no node id — the capability is the whole answer — but
// the protocol says an `open` reply carries one, so it carries a zero.
const opened = vfs_protocol.Opened{ .node = 0 };
@memcpy(reply[envelope.prefix_size..][0..@sizeOf(vfs_protocol.Opened)], std.mem.asBytes(&opened));
return answer(reply, 0, @sizeOf(vfs_protocol.Opened));
}
/// `readdir(cursor)` — the namespace, browsable. One entry per turn, as the vfs
@@ -690,18 +703,25 @@ fn onReaddir(reply: []u8, cursor: u64) usize {
continue;
}
const name = binding.nameSlice();
const entry = vfs_protocol.DirectoryEntry{
return writeEntry(reply, .{
.kind = @intFromEnum(vfs_protocol.NodeKind.protocol),
.name_len = @intCast(name.len),
.size = binding.task,
};
const total = vfs_protocol.directory_entry_size + name.len;
if (vfs_protocol.reply_size + total > reply.len) return answer(reply, -envelope.EPROTO, 0, 0);
@memcpy(reply[vfs_protocol.reply_size..][0..vfs_protocol.directory_entry_size], std.mem.asBytes(&entry));
@memcpy(reply[vfs_protocol.reply_size + vfs_protocol.directory_entry_size ..][0..name.len], name);
return answer(reply, 0, 0, total);
}, name);
}
return answer(reply, 0, 0, 0); // end of directory
// End of directory, which the envelope spells as an entry with no name: the
// reply's own length cannot say it any more, because the fixed reply part
// always travels.
return writeEntry(reply, .{}, &.{});
}
/// One `readdir` reply: the entry, then its name inline.
fn writeEntry(reply: []u8, entry: vfs_protocol.DirectoryEntry, name: []const u8) usize {
const total = vfs_protocol.directory_entry_size + name.len;
if (envelope.prefix_size + total > reply.len) return answer(reply, -envelope.EPROTO, 0);
@memcpy(reply[envelope.prefix_size..][0..vfs_protocol.directory_entry_size], std.mem.asBytes(&entry));
@memcpy(reply[envelope.prefix_size + vfs_protocol.directory_entry_size ..][0..name.len], name);
return answer(reply, 0, total);
}
pub fn main(startup: process.Init) void {
+1 -1
View File
@@ -9,7 +9,7 @@ pub fn build(b: *std.Build) void {
const exe = build_support.userBinary(b, .{
.name = "input",
.root_source_file = b.path("input.zig"),
.imports = &.{ "channel", "input-client", "input-protocol", "ipc", "logging", "process", "service" },
.imports = &.{ "envelope", "input-protocol", "ipc", "logging", "process", "service" },
});
b.installArtifact(exe);
}
+73 -69
View File
@@ -17,17 +17,29 @@
//!
//! A subscriber registers by handing the service its own endpoint as a capability (M13
//! capability passing — this service is its first real consumer). The service keeps that
//! handle and `ipc.send`s each event to it.
//! handle and `ipc.send`s each event to it. That is the envelope's reserved `subscribe`
//! verb, which this protocol adopts rather than defining its own.
//!
//! P4a moved this service onto the shared harness (library/kernel/service.zig). It was the
//! last hand-rolled receive loop in the tree, and the one service that answered neither the
//! universal ping nor a `terminate` signal — so a shutdown had to kill it. The subscriber
//! table, the fan-out, and the prune-on-subscribe below are unchanged; lifting *those* into
//! the harness is a later milestone, and doing it here would have hidden this one.
const std = @import("std");
const channel = @import("channel");
const envelope = @import("envelope");
const ipc = @import("ipc");
const process = @import("process");
const service = @import("service");
const input = @import("input-client");
const logging = @import("logging");
const input_protocol = @import("input-protocol");
/// The generated input dispatch. One fan-out point per process, so the handler
/// context is empty and the subscriber table stays in this file's globals.
const Serve = input_protocol.Protocol.Provider(void);
const Invocation = envelope.Invocation;
const Answer = envelope.Answer;
/// One registered subscriber: the endpoint we push events to (a capability it handed us at
/// subscribe time) and the task id that owns it (the subscribe call's badge), so a slot
/// left behind by a subscriber that exited can be reclaimed.
@@ -83,80 +95,72 @@ fn addSubscriber(endpoint: ipc.Handle, task_id: u32, device_mask: u32) bool {
/// Push `event` to every subscriber whose interest mask includes its device class.
/// `ipc.send` never blocks, so a slow or dead subscriber cannot stall delivery to others.
///
/// The class is the packet's operation, so the fan-out picks the event by device and the
/// packet is framed once, outside the loop — every subscriber of a class gets identical
/// bytes, which is what "one fan-out point per event domain" means on the wire.
fn broadcast(event: input_protocol.InputEvent) void {
const bytes = std.mem.asBytes(&event);
const class = input_protocol.eventOfDevice(event.device) orelse return; // no class wants it
var packet: [envelope.post_maximum]u8 = undefined;
const framed = switch (class) {
.keyboard => input_protocol.Protocol.encodeEvent(.keyboard, 0, event.asKeyboard() orelse return, &packet),
.mouse => input_protocol.Protocol.encodeEvent(.mouse, 0, event.asMouse() orelse return, &packet),
.joystick => input_protocol.Protocol.encodeEvent(.joystick, 0, event.asJoystick() orelse return, &packet),
} orelse return;
const bit = input_protocol.deviceBit(event.device);
for (&subscribers) |*sub| {
if (sub.used and sub.device_mask & bit != 0) _ = ipc.send(sub.endpoint, bytes);
if (sub.used and sub.device_mask & bit != 0) _ = ipc.send(sub.endpoint, framed);
}
}
/// Handle one request. `got` carries the sender badge (a task id); `arrived` carries the
/// capability the request came with, under the same ownership rule the service harness
/// states (`ipc.Arrival`): **it belongs to the turn, and only a handler that means to keep
/// it says `take`.** Everything else here — a short message, a `publish`, a subscribe that
/// finds the table full — simply returns, and the loop closes what arrived. Writes a
/// `Reply` into `out` and returns its length.
fn handle(message: []const u8, got: ipc.Received, out: []u8, arrived: *ipc.Arrival) usize {
const reply = struct {
fn write(buffer: []u8, status: i32) usize {
const header = input_protocol.Reply{ .status = status };
@memcpy(buffer[0..input_protocol.reply_size], std.mem.asBytes(&header));
return input_protocol.reply_size;
}
};
/// Set by `onSubscribe` when the subscriber table has taken ownership of the capability the
/// call carried, and read by `onMessage`, which is where the turn's `Arrival` lives. The
/// generated dispatch hands a handler the raw handle rather than the `Arrival` — deliberately,
/// since a handler has no business closing the turn's property — so the *claim* has to travel
/// back out this way. One turn, one handler, one thread: there is nothing here to race.
var capability_claimed = false;
if (message.len < input_protocol.request_size) return reply.write(out, -1);
const request = std.mem.bytesToValue(input_protocol.Request, message[0..input_protocol.request_size]);
/// The reserved `subscribe` verb: register the caller's endpoint (the call's capability) for
/// the classes in the packet's tail. Refusals simply return, and the turn closes what arrived
/// — the ownership rule the harness states (`ipc.Arrival`), unchanged by the move onto it.
fn onSubscribe(_: void, invocation: Invocation(void), _: Answer(void)) isize {
const endpoint = invocation.capability orelse return -envelope.EPROTO; // no endpoint passed
// A zero mask means "everything" (a subscriber that named no class still wants input).
const requested = input_protocol.decodeSubscribe(invocation.tail).device_mask;
const mask = if (requested == 0) input_protocol.device_all else requested;
pruneDeadSubscribers();
if (!addSubscriber(endpoint, invocation.sender, mask)) return -envelope.ENOSPC; // table full
capability_claimed = true; // the subscriber table holds it until that task dies
return 0;
}
switch (@as(input_protocol.Operation, @enumFromInt(request.operation))) {
.subscribe => {
const endpoint = arrived.peek() orelse return reply.write(out, -1); // no endpoint passed
// A zero mask means "everything" (a subscriber that named no class still wants input).
const mask = if (request.device_mask == 0) input_protocol.device_all else request.device_mask;
pruneDeadSubscribers();
if (!addSubscriber(endpoint, @intCast(got.badge), mask)) return reply.write(out, -1); // table full
_ = arrived.take(); // claimed: the subscriber table holds it until that task dies
return reply.write(out, 0);
},
.publish => {
broadcast(request.event);
return reply.write(out, 0);
},
}
fn onPublish(_: void, invocation: Invocation(input_protocol.InputEvent), _: Answer(void)) isize {
broadcast(invocation.request);
return 0;
}
const handlers = Serve.Handlers{ .publish = onPublish, .subscribe = onSubscribe };
fn onMessage(message: []const u8, out: []u8, sender: u32, arrived: *ipc.Arrival) usize {
capability_claimed = false;
const written = Serve.dispatch({}, handlers, message, sender, arrived.peek(), out);
if (capability_claimed) _ = arrived.take();
return written;
}
fn initialise(_: ipc.Handle) bool {
// The harness has already bound `/protocol/input` by the time this runs, so
// "ready" still means what it always meant: the name is claimed and the loop
// is about to serve it.
_ = logging.write("/system/services/input: ready\n");
return true;
}
pub fn main() void {
const endpoint = ipc.createIpcEndpoint() orelse {
_ = logging.write("/system/services/input: no endpoint\n");
return;
};
if (!channel.bindPatiently("input", endpoint)) {
_ = logging.write("/system/services/input: could not bind /protocol/input\n");
return;
}
_ = logging.write("/system/services/input: ready\n");
var reply_buffer: [input_protocol.reply_size]u8 = undefined;
var reply_len: usize = 0;
var receive: [input_protocol.request_size]u8 = undefined;
while (true) {
const got = ipc.replyWait(endpoint, reply_buffer[0..reply_len], &receive, null);
// Whatever capability came with this turn is the turn's, and the turn closes it
// unless `handle` claims it (`ipc.Arrival`). The kernel installs a sent capability
// whatever the message's length or kind, so this covers the notification
// `continue` and every refusal inside `handle` — otherwise about thirty-two
// capability-carrying calls, which need no authorization at all, exhaust this
// service's handle table and no further subscribe can ever land.
var arrived: ipc.Arrival = .{ .handle = got.cap };
defer arrived.release();
// Only synchronous client requests (subscribe/publish) arrive here; nothing sends
// this service asynchronous messages, so a notification wake would be spurious.
if (got.isNotification()) {
reply_len = 0;
continue;
}
reply_len = handle(receive[0..got.len], got, &reply_buffer, &arrived);
}
service.run(input_protocol.message_maximum, .{
.service = "input",
.init = initialise,
.on_message = onMessage,
});
}