init: /protocol replaces the ServiceId registry

A protocol is reached by name now, not by a compile-time integer. Init is
PID 1 and already knows which binary it started, so init serves /protocol
as a vfs backend: bind claims a contract with the provider's endpoint
attached, open answers with that endpoint as the reply's capability, and
readdir lists what is bound with the task and binary behind it. The kernel
reserves the prefix — nothing may mount over it, under it, or unmount it —
and ServiceId, ipc_register and ipc_lookup are gone, their syscall numbers
left vacant.

A bind is authorized by who the caller *is*: the kernel-stamped binary
together with the supervising task's identity, matched against
/system/configuration/protocol.csv. Identity, not spelling — spawn is
ungated, so an attacker can run any bundled binary, and a name-only rule
would have let it launder grants through an init of its own making. A name
a live process holds is refused to everyone else; a dead one's is released.

Three review rounds against a hostile ring-3 process found what 108 green
tests could not, because the suite contains no attacker. Publishing init's
supervision endpoint as the registry put PID 1's mailbox in every process's
hands, where two forged bytes reached the shutdown path: privileged traffic
is now believed only from the task that holds the contract it speaks for.
A capability arriving on a request outlived every path that ignored it,
one handle per call until the table was full — in init, and in the harness
ten services share — so the arriving capability is owned by the turn and
released unless a handler says otherwise. And the kernel let anyone holding
an endpoint handle aim signals, timers, exit notices and interrupts at it:
binding now requires having created it.

Suite 108/108. The new protocol-registry case asserts eleven properties,
each one an attack that must fail.
This commit is contained in:
Daniel Samson
2026-08-01 02:39:07 +01:00
parent 1ff0991452
commit 1379b699f3
66 changed files with 2191 additions and 392 deletions
+13 -20
View File
@@ -1,6 +1,6 @@
//! The **private kernel ↔ runtime** ABI: the raw system_call contract — the call
//! numbers, `mmap` protection flags, the page size those calls work in, and the IPC
//! name-registry ids and notification bit. Shared by the kernel dispatcher
//! notification bits. Shared by the kernel dispatcher
//! (system/kernel/process.zig) and the user-space runtime library (library/runtime/),
//! so the two can never drift.
//!
@@ -33,8 +33,13 @@ pub const SystemCall = enum(u64) {
mmap = 4, // mmap(len, prot) -> base: grant zeroed, page-aligned user pages
munmap = 5, // munmap(base, len): release pages from a prior mmap
create_ipc_endpoint = 6, // create_ipc_endpoint() -> handle: a new IPC endpoint
ipc_register = 7, // ipc_register(service_id, handle): publish an endpoint by well-known id
ipc_lookup = 8, // ipc_lookup(service_id) -> handle: find a published endpoint
// 7 and 8 were ipc_register/ipc_lookup — the flat ServiceId name registry,
// retired with the protocol namespace (docs/os-development/protocol-namespace.md).
// A service now binds its name at the registry (init, over /protocol) and a
// client resolves and opens that path; neither is a system call any more. The
// numbers stay vacant rather than being reused: every other entry is
// position-fixed by an explicit value, so a hole costs nothing and a reused
// number would silently mean two things across a rebuild boundary.
ipc_call = 9, // ipc_call(h, message, len, reply, cap) -> reply_len: send + block for reply
ipc_reply_wait = 10, // ipc_reply_wait(h, reply, len, receive, cap) -> receive_len (+badge in rdx)
device_enumerate = 11, // device_enumerate(buffer, maximum) -> count: snapshot the device table
@@ -284,23 +289,11 @@ pub const KlogStatus = extern struct {
boot_unix_seconds: u64, // wall-clock time of boot (RTC anchor)
};
/// Well-known IPC service ids for the bootstrap name registry (create_ipc_endpoint +
/// ipc_register/ipc_lookup). Small integers, so no string interning is needed
/// during bring-up. The VFS server registers under `vfs`; clients look it up.
pub const ServiceId = enum(u32) {
vfs = 1, // RETIRED: the router moved into the kernel (fs_resolve); the slot stays reserved
input = 2,
ps2_bus = 3, // the 8042 owner; child device drivers attach here for raw bytes
device_manager = 4, // the tree, the matcher, the supervisor (docs/device-manager.md)
power = 5, // system power: events (button, lid, battery) + shutdown (docs/power.md; domain-named per docs/discovery.md — the acpi service registers it on x86, a PSCI service will on ARM)
usb_bus = 6, // the xHCI host-controller driver's transfer endpoint; USB class drivers look it up and `callCap`-open their device to get a private per-device transfer channel (docs/driver-model.md)
block = 7, // a block-device driver (USB mass storage today): read/write of fixed-size blocks, the storage a filesystem sits on
fat = 8, // the FAT filesystem server; the VFS mounts it and forwards paths under its mount point (/volumes/usb) to it
display = 9, // the display service: owns the framebuffer, composites a layer stack, presents frames (docs/display.md)
shared_memory_test = 10, // the shared-memory test server (V2): a client passes it a shared-memory capability, it maps + verifies (docs/display-v2.md)
scanout = 11, // a native scanout driver (virtio-gpu): the compositor finds it here to upgrade off the GOP framebuffer (docs/display-v2.md)
_,
};
// The `ServiceId` enum lived here: a flat, compile-time list of well-known
// service ids backed by a 16-slot kernel table. It is gone with the protocol
// namespace — names are strings resolved under `/protocol` at run time, so a
// third-party program can introduce a contract the ABI never heard of, and the
// registrar (init) decides who may claim one.
/// Protection flags for `mmap` (matching the usual C bit values).
pub const prot_read: u64 = 1;
+70
View File
@@ -0,0 +1,70 @@
# /system/configuration/protocol.csv — who may claim, and who may reach, a name
# under /protocol (docs/os-development/protocol-namespace.md).
#
# init is the registrar: it serves /protocol, and every bind is checked against
# this file. It is AUTHORITATIVE — a name no row grants cannot be bound, and a
# missing file means nothing may be bound at all.
#
# '#' starts a comment (whole-line or trailing); blank lines are ignored.
# Whitespace around a field is trimmed, so columns may be padded. Four
# comma-separated fields per row:
#
# binary the claimant's binary path, exactly as the kernel stamped it at
# spawn (argv[0]) — unforgeable, read from the process records
# supervisor the authorized supervising TASK, written as the binary it runs —
# the path init was started as for its own services, the device
# manager's path for the drivers it starts. The one word that is not
# a path is 'kernel', because a kernel task has no binary; that is
# what the test harness's direct spawns look like.
# Matched by IDENTITY, not by spelling. Name alone is not identity —
# spawn is ungated, so a hostile process can start a granted binary
# itself and inherit its grants; and it can equally start its own
# instance of the *supervisor's* binary and have that spawn the
# granted one, at which point both names read correctly (the
# laundering deputy). So init also asks which task the supervisor
# is: 'kernel' means supervisor id 0, which only the kernel can
# confer; init's own path means this init; any other path means a
# task init spawned itself or one the kernel spawned. Task ids are
# monotonic and never reused, so an id cannot be borrowed.
# permission bind (provide this contract) | open (speak to it)
# name the contract, relative to /protocol
#
# A trailing '*' on any field matches any tail — how a subtree is granted whole.
#
# NOTE: 'open' rows are parsed but not yet enforced; every open resolves today.
# The milestone that turns them into refusals is P3 (docs/security-track-plan.md).
#
# binary supervisor permission name
# --- the services init spawns from init.csv ---------------------------------
/system/services/input, /system/services/init, bind, input
/system/services/device-manager, /system/services/init, bind, device-manager
/system/services/fat, /system/services/init, bind, vfs
/system/services/display, /system/services/init, bind, display
# The discovery service ships under one neutral name per firmware (docs/discovery.md);
# on x86 it is the acpi service, and what it provides is the power contract.
/system/services/discovery, /system/services/device-manager, bind, power
# --- the drivers, which the device manager spawns ---------------------------
/system/drivers/ps2-bus, /system/services/device-manager, bind, ps2-bus
/system/drivers/usb-xhci-bus, /system/services/device-manager, bind, usb-transfer
/system/drivers/usb-storage, /system/services/device-manager, bind, block
/system/drivers/virtio-gpu, /system/services/device-manager, bind, scanout
# --- the same providers when the kernel test harness starts them directly ---
# A scenario boot spawns its own providers instead of letting init do it
# (docs/security-track-plan.md, decision 9), so the same binaries appear with
# 'kernel' as the supervisor. Nothing else changes: the binary must still match.
/system/services/input, kernel, bind, input
/system/services/device-manager, kernel, bind, device-manager
/system/services/fat, kernel, bind, vfs
/system/services/display, kernel, bind, display
/system/services/discovery, kernel, bind, power
# --- test fixtures ----------------------------------------------------------
# The subtree rule, dogfooded: anything installed under /test may claim anything
# under /protocol/test, and nothing above it — whether the harness spawned it or
# another fixture did.
/test/*, kernel, bind, test/*
/test/*, /test/*, bind, test/*
1 # /system/configuration/protocol.csv — who may claim, and who may reach, a name
2 # under /protocol (docs/os-development/protocol-namespace.md).
3 #
4 # init is the registrar: it serves /protocol, and every bind is checked against
5 # this file. It is AUTHORITATIVE — a name no row grants cannot be bound, and a
6 # missing file means nothing may be bound at all.
7 #
8 # '#' starts a comment (whole-line or trailing); blank lines are ignored.
9 # Whitespace around a field is trimmed, so columns may be padded. Four
10 # comma-separated fields per row:
11 #
12 # binary the claimant's binary path, exactly as the kernel stamped it at
13 # spawn (argv[0]) — unforgeable, read from the process records
14 # supervisor the authorized supervising TASK, written as the binary it runs —
15 # the path init was started as for its own services, the device
16 # manager's path for the drivers it starts. The one word that is not
17 # a path is 'kernel', because a kernel task has no binary; that is
18 # what the test harness's direct spawns look like.
19 # Matched by IDENTITY, not by spelling. Name alone is not identity —
20 # spawn is ungated, so a hostile process can start a granted binary
21 # itself and inherit its grants; and it can equally start its own
22 # instance of the *supervisor's* binary and have that spawn the
23 # granted one, at which point both names read correctly (the
24 # laundering deputy). So init also asks which task the supervisor
25 # is: 'kernel' means supervisor id 0, which only the kernel can
26 # confer; init's own path means this init; any other path means a
27 # task init spawned itself or one the kernel spawned. Task ids are
28 # monotonic and never reused, so an id cannot be borrowed.
29 # permission bind (provide this contract) | open (speak to it)
30 # name the contract, relative to /protocol
31 #
32 # A trailing '*' on any field matches any tail — how a subtree is granted whole.
33 #
34 # NOTE: 'open' rows are parsed but not yet enforced; every open resolves today.
35 # The milestone that turns them into refusals is P3 (docs/security-track-plan.md).
36 #
37 # binary supervisor permission name
38 # --- the services init spawns from init.csv ---------------------------------
39 /system/services/input, /system/services/init, bind, input
40 /system/services/device-manager, /system/services/init, bind, device-manager
41 /system/services/fat, /system/services/init, bind, vfs
42 /system/services/display, /system/services/init, bind, display
43 # The discovery service ships under one neutral name per firmware (docs/discovery.md);
44 # on x86 it is the acpi service, and what it provides is the power contract.
45 /system/services/discovery, /system/services/device-manager, bind, power
46 # --- the drivers, which the device manager spawns ---------------------------
47 /system/drivers/ps2-bus, /system/services/device-manager, bind, ps2-bus
48 /system/drivers/usb-xhci-bus, /system/services/device-manager, bind, usb-transfer
49 /system/drivers/usb-storage, /system/services/device-manager, bind, block
50 /system/drivers/virtio-gpu, /system/services/device-manager, bind, scanout
51 # --- the same providers when the kernel test harness starts them directly ---
52 # A scenario boot spawns its own providers instead of letting init do it
53 # (docs/security-track-plan.md, decision 9), so the same binaries appear with
54 # 'kernel' as the supervisor. Nothing else changes: the binary must still match.
55 /system/services/input, kernel, bind, input
56 /system/services/device-manager, kernel, bind, device-manager
57 /system/services/fat, kernel, bind, vfs
58 /system/services/display, kernel, bind, display
59 /system/services/discovery, kernel, bind, power
60 # --- test fixtures ----------------------------------------------------------
61 # The subtree rule, dogfooded: anything installed under /test may claim anything
62 # under /protocol/test, and nothing above it — whether the harness spawned it or
63 # another fixture did.
64 /test/*, kernel, bind, test/*
65 /test/*, /test/*, bind, test/*
+2 -2
View File
@@ -245,11 +245,11 @@ fn registerAndReport(bus: u64, dev: u64, function: u64, class_triple: u32) void
};
}
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Handle) usize {
fn onMessage(message: []const u8, reply: []u8, sender: u32, arrived: *ipc.Arrival) usize {
_ = message;
_ = reply;
_ = sender;
_ = capability;
_ = arrived; // nothing here takes a capability: the harness closes what arrives
return 0;
}
+5 -5
View File
@@ -9,7 +9,7 @@ pub fn build(b: *std.Build) void {
const ps2_bus_exe = build_support.userBinary(b, .{
.name = "ps2-bus",
.root_source_file = b.path("ps2-bus.zig"),
.imports = &.{ "acpi-ids", "driver", "ipc", "logging", "memory", "process", "service", "time" },
.imports = &.{ "acpi-ids", "channel", "driver", "ipc", "logging", "memory", "process", "service", "time" },
});
b.installArtifact(ps2_bus_exe);
@@ -17,8 +17,8 @@ pub fn build(b: *std.Build) void {
.name = "ps2-keyboard",
.root_source_file = b.path("keyboard.zig"),
.imports = &.{
"acpi-ids", "driver", "input-client", "input-protocol", "ipc", "logging", "memory",
"process", "time", "xkeyboard-config",
"acpi-ids", "channel", "driver", "input-client", "input-protocol", "ipc",
"logging", "memory", "process", "time", "xkeyboard-config",
},
});
b.installArtifact(ps2_keyboard_exe);
@@ -27,8 +27,8 @@ pub fn build(b: *std.Build) void {
.name = "ps2-mouse",
.root_source_file = b.path("mouse.zig"),
.imports = &.{
"acpi-ids", "driver", "input-client", "input-protocol", "ipc", "logging", "memory",
"process", "time",
"acpi-ids", "channel", "driver", "input-client", "input-protocol", "ipc", "logging",
"memory", "process", "time",
},
});
b.installArtifact(ps2_mouse_exe);
+4 -3
View File
@@ -16,6 +16,7 @@
const std = @import("std");
const device = @import("driver");
const channel = @import("channel");
const ipc = @import("ipc");
const process = @import("process");
const time = @import("time");
@@ -27,12 +28,12 @@ const ps2 = @import("ps2-library.zig");
const scancode = @import("scancode.zig");
const input_protocol = @import("input-protocol");
/// Look up the ps2-bus service, retrying while the bus (which spawned us before
/// registering) is still coming up.
/// Open `/protocol/ps2-bus`, retrying while the bus (which spawned us before
/// binding) is still coming up.
fn lookupBus() ?ipc.Handle {
var attempts: usize = 0;
while (attempts < 100) : (attempts += 1) {
if (ipc.lookup(.ps2_bus)) |handle| return handle;
if (channel.openEndpoint("ps2-bus")) |handle| return handle;
time.sleepMillis(50);
}
return null;
+4 -3
View File
@@ -16,6 +16,7 @@
const std = @import("std");
const device = @import("driver");
const channel = @import("channel");
const ipc = @import("ipc");
const process = @import("process");
const time = @import("time");
@@ -26,12 +27,12 @@ const ps2 = @import("ps2-library.zig");
const mouse_packet = @import("mouse-packet.zig");
const input_protocol = @import("input-protocol");
/// Look up the ps2-bus service, retrying while the bus (which spawned us before
/// registering) is still coming up.
/// Open `/protocol/ps2-bus`, retrying while the bus (which spawned us before
/// binding) is still coming up.
fn lookupBus() ?ipc.Handle {
var attempts: usize = 0;
while (attempts < 100) : (attempts += 1) {
if (ipc.lookup(.ps2_bus)) |handle| return handle;
if (channel.openEndpoint("ps2-bus")) |handle| return handle;
time.sleepMillis(50);
}
return null;
+29 -9
View File
@@ -11,6 +11,7 @@
//! - irq 0xc len 0x1
const std = @import("std");
const device = @import("driver");
const channel = @import("channel");
const ipc = @import("ipc");
const process = @import("process");
const service = @import("service");
@@ -62,7 +63,13 @@ var port_device_types = [_]?ps2.DeviceType{ null, null };
/// Handle a child driver's `AttachRequest`: record the endpoint capability it
/// passed as the forwarding target for the port whose device matches its type.
/// Writes an `AttachReply` into `out` and returns its length.
fn handleAttach(message: []const u8, got: ipc.Received, out: []u8) usize {
/// A child driver's AttachRequest. The endpoint it hands over arrives under the
/// same ownership rule the service harness states (`ipc.Arrival`): the turn owns
/// it, and only the path that records it in `port_endpoints` says `take`. Every
/// refusal here simply returns, and the loop closes what arrived — otherwise a
/// stranger (this is a named contract, reachable by anyone) spends one of this
/// driver's thirty-two handle slots per malformed attach.
fn handleAttach(message: []const u8, out: []u8, arrived: *ipc.Arrival) usize {
const reply = struct {
fn write(buffer: []u8, status: ps2.AttachStatus) usize {
const header = ps2.AttachReply{ .status = @intFromEnum(status) };
@@ -73,12 +80,17 @@ fn handleAttach(message: []const u8, got: ipc.Received, out: []u8) usize {
if (message.len < @sizeOf(ps2.AttachRequest)) return reply.write(out, .invalid_request);
const request = std.mem.bytesToValue(ps2.AttachRequest, message[0..@sizeOf(ps2.AttachRequest)]);
const endpoint = got.cap orelse return reply.write(out, .missing_endpoint);
const endpoint = arrived.peek() orelse return reply.write(out, .missing_endpoint);
for (&port_device_types, 0..) |maybe_type, port_index| {
const device_type = maybe_type orelse continue;
if (@intFromEnum(device_type) != request.device_type) continue;
port_endpoints[port_index] = endpoint;
// Claimed. A re-attach supersedes the previous driver's endpoint, and the
// one it displaces is closed: the slot holds exactly one reference.
if (port_endpoints[port_index]) |previous| {
if (previous != endpoint) _ = ipc.close(previous);
}
port_endpoints[port_index] = arrived.take();
std.log.info("{s} driver attached", .{@tagName(device_type)});
return reply.write(out, .ok);
}
@@ -213,15 +225,16 @@ pub fn main() void {
return;
};
// The endpoint the child drivers attach to and IRQ1 wakes. Registered under a
// well-known id so the children can find it, the way input subscribers find
// the input service.
// The endpoint the child drivers attach to and IRQ1 wakes. Bound as the
// `ps2-bus` contract so the children can find it by name, the way input
// subscribers find the input service. This driver runs its own loop rather
// than the service harness, so it binds by hand — same call the harness makes.
const endpoint = ipc.createIpcEndpoint() orelse {
_ = logging.write("/system/drivers/ps2-bus: no endpoint\n");
return;
};
if (!ipc.register(.ps2_bus, endpoint)) {
_ = logging.write("/system/drivers/ps2-bus: register failed\n");
if (!channel.bindPatiently("ps2-bus", endpoint)) {
_ = logging.write("/system/drivers/ps2-bus: could not bind /protocol/ps2-bus\n");
return;
}
@@ -273,6 +286,13 @@ pub fn main() void {
var receive: [@sizeOf(ps2.AttachRequest)]u8 = undefined;
while (true) {
const got = ipc.replyWait(endpoint, reply_buffer[0..reply_len], &receive, null);
// The turn owns whatever capability arrived and closes it unless
// `handleAttach` claims it (`ipc.Arrival`) — the kernel installs one
// whatever the message's length or kind, so this covers the notification
// path and every refusal below it.
var arrived: ipc.Arrival = .{ .handle = got.cap };
defer arrived.release();
if (got.isNotification()) {
reply_len = 0;
if (got.isMessage() or got.isChildExit()) continue; // nothing sends us these
@@ -301,6 +321,6 @@ pub fn main() void {
}
continue;
}
reply_len = handleAttach(receive[0..got.len], got, &reply_buffer);
reply_len = handleAttach(receive[0..got.len], &reply_buffer, &arrived);
}
}
+7 -6
View File
@@ -146,17 +146,18 @@ fn initialise(endpoint: ipc.Handle) bool {
/// Serve the block protocol: geometry, and whole-block read/write to/from the
/// caller's DMA buffer (named by physical address).
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Handle) usize {
fn onMessage(message: []const u8, reply: []u8, sender: u32, arrived: *ipc.Arrival) usize {
_ = sender;
if (message.len < block_protocol.request_size) return 0;
const request = std.mem.bytesToValue(block_protocol.Request, message[0..block_protocol.request_size]);
switch (request.operation) {
@intFromEnum(block_protocol.Operation.attach) => {
// The filesystem's DMA buffer: forward its capability to the controller so
// the device can reach it, then release our copy (the binding holds a ref).
const handle = capability orelse return writeReply(reply, .{ .status = -1, .block_size = 0, .block_count = 0 });
// The filesystem's DMA buffer: forward its capability to the controller
// so the device can reach it. Never claimed — the binding holds its own
// reference, so our copy is the turn's to close, on this path and on the
// refusal above it alike.
const handle = arrived.peek() orelse return writeReply(reply, .{ .status = -1, .block_size = 0, .block_count = 0 });
const ok = device.attachDma(handle);
_ = ipc.close(handle);
return writeReply(reply, .{ .status = if (ok) 0 else -1, .block_size = 0, .block_count = 0 });
},
@intFromEnum(block_protocol.Operation.geometry) => {
@@ -202,7 +203,7 @@ pub fn main(init: process.Init) void {
return;
};
service.run(block_protocol.message_maximum, .{
.service = .block,
.service = "block",
.init = initialise,
.on_message = onMessage,
});
+4 -3
View File
@@ -10,9 +10,10 @@ pub fn build(b: *std.Build) void {
.name = "usb-xhci-bus",
.root_source_file = b.path("usb-xhci-bus.zig"),
.imports = &.{
"device-manager-protocol", "driver", "input-client", "ipc", "logging", "memory",
"mmio", "pci", "process", "service", "time", "usb-abi", "usb-ids",
"usb-transfer-protocol",
"channel", "device-manager-protocol", "driver", "input-client",
"ipc", "logging", "memory", "mmio",
"pci", "process", "service", "time",
"usb-abi", "usb-ids", "usb-transfer-protocol",
},
});
b.installArtifact(exe);
+42 -13
View File
@@ -15,6 +15,7 @@
const std = @import("std");
const device = @import("driver");
const channel = @import("channel");
const ipc = @import("ipc");
const process = @import("process");
const service = @import("service");
@@ -68,19 +69,25 @@ const Open = struct {
};
var opens = [_]Open{.{}} ** 16;
fn recordOpen(device_token: u64, report_endpoint: usize) void {
/// Remember (or replace) the endpoint that reports for `device_token`. Returns
/// whether the table kept the handle — false means the caller still owns it and
/// must dispose of it. A re-open supersedes the previous endpoint, and the one
/// it displaced is closed here: the table holds exactly one reference per slot.
fn recordOpen(device_token: u64, report_endpoint: usize) bool {
for (&opens) |*open| {
if (open.used and open.device_token == device_token) {
if (open.report_endpoint != report_endpoint) _ = ipc.close(open.report_endpoint);
open.report_endpoint = report_endpoint;
return;
return true;
}
}
for (&opens) |*open| {
if (!open.used) {
open.* = .{ .used = true, .device_token = device_token, .report_endpoint = report_endpoint };
return;
return true;
}
}
return false; // table full: not kept
}
fn reportEndpointFor(device_token: u64) ?usize {
@@ -97,6 +104,22 @@ var controller_id: u64 = device_manager_protocol.no_device;
/// manager reads as "meant to stop" — a missing assignment is not a crash loop.
fn initialise(endpoint: ipc.Handle) bool {
service_endpoint = endpoint;
// The transfer contract, bound by hand rather than through the harness's
// `.service`, because **losing it is not fatal here**. One machine can carry
// several xHCI controllers and the driver model spawns one process per
// controller, so several processes provide the same contract for different
// hardware — and `/protocol` holds exactly one name, deliberately (addressing
// lives inside the protocol, never in the path). Whoever binds first is the
// one clients reach by name; a later instance still owns its controller,
// enumerates its bus, and reports its children to the device manager, so it
// keeps running. **Known gap:** a class driver behind a second controller
// cannot reach it — the transfer protocol has no controller field for
// `target`, and the fix is either one process multiplexing every controller
// or the spawner wiring the child's channel (P5), not a second name.
if (!channel.bindPatiently("usb-transfer", endpoint))
_ = logging.write("/system/drivers/usb-xhci-bus: /protocol/usb-transfer is another controller's; serving mine unnamed\n");
if (!device.claim(controller_id)) {
std.log.info("unable to claim controller device {d}", .{controller_id});
return false;
@@ -475,28 +498,29 @@ fn reportInterface(manager: ipc.Handle, port: u32, interface: library.InterfaceI
/// Serve the USB transfer protocol: a class driver opens its device, then issues
/// control / interrupt-subscribe / bulk requests against it.
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Handle) usize {
fn onMessage(message: []const u8, reply: []u8, sender: u32, arrived: *ipc.Arrival) usize {
_ = sender;
if (message.len < 4) return 0;
const operation = std.mem.readInt(u32, message[0..4], .little);
return switch (operation) {
@intFromEnum(usb_transfer_protocol.Operation.open) => handleOpen(message, reply, capability),
@intFromEnum(usb_transfer_protocol.Operation.open) => handleOpen(message, reply, arrived),
@intFromEnum(usb_transfer_protocol.Operation.control) => handleControl(message, reply),
@intFromEnum(usb_transfer_protocol.Operation.interrupt_subscribe) => handleSubscribe(message, reply),
@intFromEnum(usb_transfer_protocol.Operation.bulk) => handleBulk(message, reply),
@intFromEnum(usb_transfer_protocol.Operation.dma_attach) => handleDmaAttach(message, reply, capability),
@intFromEnum(usb_transfer_protocol.Operation.dma_attach) => handleDmaAttach(message, reply, arrived),
else => 0,
};
}
/// dma_attach: bind the class driver's DMA-region capability into the controller's IOMMU
/// domain, so the controller may DMA to the physical addresses inside that buffer. The
/// binding holds its own kernel reference, so the forwarded capability is closed here.
fn handleDmaAttach(message: []const u8, reply: []u8, capability: ?ipc.Handle) usize {
/// binding holds its own kernel reference, so this never claims the arriving handle —
/// the turn's `defer` in the harness is the close, on the failure paths as well as this
/// one.
fn handleDmaAttach(message: []const u8, reply: []u8, arrived: *ipc.Arrival) usize {
if (message.len < @sizeOf(usb_transfer_protocol.DmaAttachRequest)) return writeReply(reply, usb_transfer_protocol.DmaAttachReply{ .status = -1 });
const handle = capability orelse return writeReply(reply, usb_transfer_protocol.DmaAttachReply{ .status = -1 });
const handle = arrived.peek() orelse return writeReply(reply, usb_transfer_protocol.DmaAttachReply{ .status = -1 });
const ok = device.dmaBind(controller_id, handle);
_ = ipc.close(handle);
return writeReply(reply, usb_transfer_protocol.DmaAttachReply{ .status = if (ok) 0 else -1 });
}
@@ -509,13 +533,17 @@ fn writeReply(reply: []u8, value: anytype) usize {
/// open: resolve the assigned device id to an interface, remember the caller's
/// endpoint (for interrupt reports), and answer with a device token + the
/// interface's endpoints so the class driver need not re-read the config.
fn handleOpen(message: []const u8, reply: []u8, capability: ?ipc.Handle) usize {
fn handleOpen(message: []const u8, reply: []u8, arrived: *ipc.Arrival) usize {
if (message.len < @sizeOf(usb_transfer_protocol.OpenRequest)) return writeReply(reply, usb_transfer_protocol.OpenReply{ .status = -1, .endpoint_count = 0, .device_token = 0, .interface_class = 0, .interface_subclass = 0, .interface_protocol = 0, .interface_number = 0 });
const request = std.mem.bytesToValue(usb_transfer_protocol.OpenRequest, message[0..@sizeOf(usb_transfer_protocol.OpenRequest)]);
const engine = if (controller) |*c| c else return writeReply(reply, usb_transfer_protocol.OpenReply{ .status = -1, .endpoint_count = 0, .device_token = 0, .interface_class = 0, .interface_subclass = 0, .interface_protocol = 0, .interface_number = 0 });
const found = engine.findInterface(request.device_id) orelse return writeReply(reply, usb_transfer_protocol.OpenReply{ .status = -1, .endpoint_count = 0, .device_token = 0, .interface_class = 0, .interface_subclass = 0, .interface_protocol = 0, .interface_number = 0 });
if (capability) |endpoint| recordOpen(request.device_id, endpoint);
// The report endpoint is claimed only if the open table actually keeps it;
// a full table leaves it to the turn to close.
if (arrived.peek()) |endpoint| {
if (recordOpen(request.device_id, endpoint)) _ = arrived.take();
}
var open_reply = usb_transfer_protocol.OpenReply{
.status = 0,
@@ -673,7 +701,8 @@ pub fn main(init: process.Init) void {
return;
};
service.run(usb_transfer_protocol.message_maximum, .{
.service = .usb_bus,
// No `.service`: the contract is bound inside `initialise`, where losing
// it to another controller's driver is survivable rather than fatal.
.init = initialise,
.on_message = onMessage,
.on_notification = onNotification,
+2 -2
View File
@@ -10,8 +10,8 @@ pub fn build(b: *std.Build) void {
.name = "virtio-gpu",
.root_source_file = b.path("virtio-gpu.zig"),
.imports = &.{
"display-protocol", "driver", "ipc", "logging", "memory", "mmio", "pci", "process",
"scanout-protocol", "service", "time",
"channel", "display-protocol", "driver", "ipc", "logging", "memory", "mmio", "pci",
"process", "scanout-protocol", "service", "time",
},
});
b.installArtifact(exe);
+5 -4
View File
@@ -15,6 +15,7 @@
const std = @import("std");
const device = @import("driver");
const channel = @import("channel");
const ipc = @import("ipc");
const process = @import("process");
const service = @import("service");
@@ -475,7 +476,7 @@ fn presentFull() bool {
fn announce() void {
var tries: u32 = 0;
const display = while (tries < 50) : (tries += 1) {
if (ipc.lookup(.display)) |h| break h;
if (channel.openEndpoint("display")) |h| break h;
time.sleepMillis(20);
} else {
std.log.info("no display service to announce to (scanout-only)", .{});
@@ -507,9 +508,9 @@ fn scanoutStatus(reply: []u8, ok: bool) usize {
/// The `.scanout` service: the compositor drives present / mode queries here. The pixels are
/// already in the shared surface, so a present is a transfer-to-host + fenced flush; a mode
/// change just re-points the scanout rectangle (the surface is sized to the largest mode).
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Handle) usize {
fn onMessage(message: []const u8, reply: []u8, sender: u32, arrived: *ipc.Arrival) usize {
_ = sender;
_ = capability;
_ = arrived; // nothing here takes a capability: the harness closes what arrives
if (message.len < scanout_protocol.request_size) return 0;
const request = std.mem.bytesToValue(scanout_protocol.Request, message[0..scanout_protocol.request_size]);
switch (request.operation) {
@@ -547,7 +548,7 @@ pub fn main(init: process.Init) void {
return;
};
service.run(256, .{
.service = .scanout,
.service = "scanout",
.init = initialise,
.on_message = onMessage,
});
+79 -36
View File
@@ -40,16 +40,13 @@ const Task = scheduler.Task;
pub const MESSAGE_MAXIMUM: usize = 256;
pub const maximum_handles = scheduler.ipc_maximum_handles;
// The name registry is indexed directly by ServiceId, so this must exceed the
// largest id (currently fat = 8). Sized with headroom for new services.
pub const maximum_services = 16;
/// Errno-style failures, returned as `-value` in the system_call result register.
pub const EBADF: i64 = 1; // bad handle
pub const E2BIG: i64 = 2; // message exceeds MESSAGE_MAXIMUM
pub const EFAULT: i64 = 3; // buffer unmapped / out of the user half
pub const ENOENT: i64 = 4; // no such registered service
pub const ENOSPC: i64 = 5; // handle table or registry full
pub const ENOENT: i64 = 4; // no such name
pub const ENOSPC: i64 = 5; // handle table full
pub const ENOMEM: i64 = 6; // out of memory
pub const EPEER: i64 = 7; // peer died before replying (its process exited or was killed)
pub const ESRCH: i64 = 8; // no such process (process_kill of an unknown/dead id)
@@ -90,12 +87,21 @@ const PostSlot = struct {
const user_half_end: u64 = user_memory.user_half_end;
/// A rendezvous endpoint. Allocated from the kernel heap; referenced by handle
/// (per process) and/or by a registry slot, counted by `refcount`.
/// (per process) and by whoever a capability was passed to, counted by `refcount`.
pub const Endpoint = struct {
refcount: u32 = 1,
/// Next in the list of every live endpoint. Endpoints are otherwise reachable
/// only through the handle tables that name them, and the death path has to
/// find a dying task's endpoints without one — see `live_endpoints`.
next_live: ?*Endpoint = null,
// The task that created it. When that task dies, the endpoint is marked `dead` so a caller
// gets -EPEER instead of blocking forever on a service that will never reply again (V6).
owner: u32 = 0,
// The *process* that created it — `owner`'s leader, snapshotted at creation so the
// answer survives the creating thread. `owner` alone cannot answer "is this mine?"
// for a threaded service, and the question has to be answerable after that thread is
// gone; see `ownedBy`.
owner_leader: u32 = 0,
dead: bool = false,
// Callers blocked in `call`, awaiting receive, in FIFO order (threaded via
// Task.next; each such task is .blocked and in no scheduler queue).
@@ -115,29 +121,78 @@ pub const Endpoint = struct {
post_tail: u16 = 0,
};
/// Every live endpoint, singly linked through `next_live`. The list exists for
/// exactly one purpose: the death path must mark a dying task's endpoints dead,
/// and a handle table only answers the other question (which endpoints does this
/// task *hold*). Mutated under the big kernel lock, like every other IPC global.
var live_endpoints: ?*Endpoint = null;
pub fn createIpcEndpoint() ?*Endpoint {
const creator = scheduler.current();
const endpoint = heap.allocator().create(Endpoint) catch return null;
endpoint.* = .{ .owner = scheduler.currentId() };
endpoint.* = .{ .owner = creator.id, .owner_leader = creator.leader, .next_live = live_endpoints };
live_endpoints = endpoint;
return endpoint;
}
/// A task is dying: kill the endpoints it registered as services. Mark each `dead` (so a later
/// `call` returns -EPEER rather than blocking on a reply that will never come), wake anyone
/// already parked sending to it with that error, and vacate its registry slot. Only *registered*
/// endpoints are reachable from here; unregistered ones drop with the task's handle table. The
/// caller holds the big kernel lock (this runs on the death path). See docs/display-v2.md (V6).
/// Whether `t` may have the kernel post **notifications** — signals, timer
/// landings, exit notices, interrupts — into `endpoint`: whether the endpoint is
/// its process's own.
///
/// Holding a *handle* to an endpoint is not ownership of it. `fs_resolve`
/// installs a mounted backend's capability in any caller's table
/// (`installHandleDeduped`), and any capability may be passed along a call, so a
/// sendable handle means only "you may talk to this". A kernel notification is
/// different in kind: it makes the kernel speak *into* someone else's mailbox
/// with a badge that receiver cannot distinguish from one it asked for — a
/// genuine signal badge, a genuine timer landing. That is how a forged
/// `terminate` reached PID 1's shutdown path: the attacker aimed **its own**
/// signal delivery at init's endpoint with `signal_bind` and then signalled
/// itself, and every bit the kernel stamped was authentic. Refusing the *bind*
/// is the only place the distinction still exists.
///
/// Threads: ownership is the **process's**, not the task's, so any thread may
/// bind an endpoint a sibling created — the same normalization `process_signal`
/// and `process_kill` perform when they resolve a member to its leader. The
/// creating task's own id is honoured too, which is what keeps kernel tasks
/// (leader 0) from being treated as one process.
pub fn ownedBy(endpoint: *const Endpoint, t: *const Task) bool {
if (endpoint.owner == t.id) return true;
return t.leader != 0 and endpoint.owner_leader == t.leader;
}
/// Unlink a freed endpoint from the live list. O(n) in the number of live
/// endpoints, which is tens.
fn forgetEndpoint(endpoint: *Endpoint) void {
var link = &live_endpoints;
while (link.*) |current| {
if (current == endpoint) {
link.* = current.next_live;
return;
}
link = &current.next_live;
}
}
/// A task is dying: kill every endpoint it created. Mark each `dead` (so a later
/// `call` returns -EPEER rather than blocking on a reply that will never come) and wake
/// anyone already parked sending to it with that error. This is what makes a provider's
/// death visible to the clients holding its capability — the naming layer's restart
/// story (a client re-resolves on -EPEER) rests on it, as does the VFS router's lazy
/// unmount of a backend that died. The endpoint object itself lives until the last
/// handle naming it drops. The caller holds the big kernel lock (this runs on the death
/// path). See docs/display-v2.md (V6).
pub fn killOwnedEndpointsLocked(task_id: u32) void {
for (&registry) |*slot| {
const endpoint = slot.* orelse continue;
if (endpoint.owner != task_id) continue;
var current = live_endpoints;
while (current) |endpoint| {
current = endpoint.next_live;
if (endpoint.owner != task_id or endpoint.dead) continue;
endpoint.dead = true;
while (dequeueSender(endpoint)) |sender| {
sender.ipc_status = -EPEER;
sender.ipc_received_cap = abi.no_cap;
scheduler.readyLocked(sender);
}
slot.* = null;
dropRef(endpoint);
}
}
@@ -147,6 +202,7 @@ pub fn dropRef(endpoint: *Endpoint) void {
if (endpoint.refcount > 1) {
endpoint.refcount -= 1;
} else {
forgetEndpoint(endpoint);
heap.allocator().destroy(endpoint);
}
}
@@ -633,22 +689,9 @@ fn dropEntry(entry: scheduler.HandleObject) void {
}
}
var registry: [maximum_services]?*Endpoint = .{null} ** maximum_services;
/// Publish `endpoint` under well-known `id` (takes a reference). Returns 0 or -errno.
pub fn register(id: u32, endpoint: *Endpoint) i64 {
if (id >= maximum_services) return -ENOENT;
if (registry[id]) |old| dropRef(old);
endpoint.refcount += 1;
registry[id] = endpoint;
return 0;
}
/// Find the endpoint published under `id`, taking a reference for the caller to
/// install in its handle table. Null if nothing is registered there.
pub fn lookup(id: u32) ?*Endpoint {
if (id >= maximum_services) return null;
const endpoint = registry[id] orelse return null;
endpoint.refcount += 1;
return endpoint;
}
// The flat `ServiceId` registry lived here — a 16-slot table any process could
// write, indexed by a compile-time enum. Naming is user-space's job now: init
// serves `/protocol` and decides who may claim a name
// (docs/os-development/protocol-namespace.md). The kernel keeps only what is
// genuinely kernel work — moving capabilities and telling clients their provider
// died (`killOwnedEndpointsLocked`).
+2 -2
View File
@@ -47,8 +47,8 @@ pub const maximum_gsi = 24;
var bound: [maximum_gsi]?*ipc_sync.Endpoint = .{null} ** maximum_gsi;
/// Task that owns each binding. Teardown is keyed on *this*, not on the endpoint
/// pointer: an endpoint can be shared between processes (ipc_register/ipc_lookup hand
/// out extra references), so "every GSI pointing at this endpoint" is not the same
/// pointer: an endpoint can be shared between processes (a capability passed in a message
/// hands out extra references), so "every GSI pointing at this endpoint" is not the same
/// set as "every GSI this process bound", and releasing the former on exit would mask
/// a live sibling's device line.
var bound_owner: [maximum_gsi]u32 = .{0} ** maximum_gsi;
+50 -41
View File
@@ -229,8 +229,6 @@ fn system_call(state: *architecture.CpuState) void {
.mmap => systemMmap(state),
.munmap => systemMunmap(state),
.create_ipc_endpoint => systemCreateIpcEndpoint(state),
.ipc_register => systemIpcRegister(state),
.ipc_lookup => systemIpcLookup(state),
.ipc_call => systemIpcCall(state),
.ipc_reply_wait => systemIpcReplyWait(state),
.ipc_send => systemIpcSend(state),
@@ -320,36 +318,6 @@ fn systemCreateIpcEndpoint(state: *architecture.CpuState) void {
architecture.setSystemCallResult(state, @intCast(h));
}
/// ipc_register(service_id, handle): publish the caller's endpoint under a
/// well-known id so other processes can find it.
fn systemIpcRegister(state: *architecture.CpuState) void {
// Under the big kernel lock: mutates the global service registry and endpoint
// refcounts, which threads of the same (or another) process can race.
const flags = sync.enter();
defer sync.leave(flags);
const id: u32 = @truncate(architecture.systemCallArg(state, 0));
const endpoint = ipc.resolveHandle(scheduler.current(), architecture.systemCallArg(state, 1)) orelse return failErr(state, ipc.EBADF);
architecture.setSystemCallResult(state, @bitCast(ipc.register(id, endpoint)));
}
/// ipc_lookup(service_id) -> handle: find a published endpoint and install a
/// handle to it in the caller.
fn systemIpcLookup(state: *architecture.CpuState) void {
// Under the big kernel lock: reads the global registry, takes an endpoint reference,
// and installs a handle — all racy against concurrent threads (this is the path the
// display's mouse-listener thread takes to reach the compositor endpoint).
const flags = sync.enter();
defer sync.leave(flags);
const id: u32 = @truncate(architecture.systemCallArg(state, 0));
const endpoint = ipc.lookup(id) orelse return failErr(state, ipc.ENOENT);
const h = ipc.installHandle(scheduler.current(), endpoint);
if (h < 0) {
ipc.dropRef(endpoint);
return failErr(state, ipc.ENOSPC);
}
architecture.setSystemCallResult(state, @intCast(h));
}
/// ipc_call(handle, message_ptr, message_len, reply_ptr, reply_cap) -> reply_len.
/// Blocks until the server replies; the trap frame lives on this task's kernel
/// stack, so it survives the block and receives the result on resume.
@@ -989,10 +957,17 @@ fn systemSpawn(state: *architecture.CpuState) void {
if (len == 0 or len > scheduler.maximum_task_name or ptr >= user_half_end or ptr + len > user_half_end) return fail(state);
if (arguments_len > maximum_argument_bytes) return fail(state);
if (arguments_len != 0 and (arguments_ptr >= user_half_end or arguments_ptr + arguments_len > user_half_end)) return fail(state);
// The exit endpoint is a notification binding like signal_bind's and
// timer_bind's, so it obeys the same rule: the caller's own mailbox, never a
// stranger's. Otherwise any process could have the kernel post child-exit
// badges into PID 1 by spawning throwaway children against init's endpoint.
const exit_endpoint: ?*ipc.Endpoint = if (exit_handle == abi.no_cap)
null
else
ipc.resolveHandle(t, exit_handle) orelse return failErr(state, ipc.EBADF);
else block: {
const endpoint = ipc.resolveHandle(t, exit_handle) orelse return failErr(state, ipc.EBADF);
if (!ipc.ownedBy(endpoint, t)) return failErr(state, ipc.EPERM);
break :block endpoint;
};
const image = ramdisk_image orelse return fail(state);
const rd = initial_ramdisk.Reader.init(image) orelse return fail(state);
@@ -1040,11 +1015,15 @@ fn systemThreadSpawn(state: *architecture.CpuState) void {
if (t.address_space == 0) return fail(state); // kernel tasks own no address space to share
if (entry == 0 or entry >= user_half_end) return fail(state);
if (stack_top == 0 or stack_top > user_half_end) return fail(state);
// The endpoint the thread notifies on exit (how join waits), or none.
// The endpoint the thread notifies on exit (how join waits), or none — the
// caller's own, like every other notification binding.
const exit_endpoint: ?*ipc.Endpoint = if (exit_handle == abi.no_cap)
null
else
ipc.resolveHandle(t, exit_handle) orelse return failErr(state, ipc.EBADF);
else block: {
const endpoint = ipc.resolveHandle(t, exit_handle) orelse return failErr(state, ipc.EBADF);
if (!ipc.ownedBy(endpoint, t)) return failErr(state, ipc.EPERM);
break :block endpoint;
};
const tid = spawnThreadSupervised(t.address_space, entry, stack_top, arg, t.priority, t.id, exit_endpoint, t.leader);
if (tid == -ipc.ESRCH) return failErr(state, ipc.ESRCH); // dying group admits no member
if (tid < 0) return fail(state);
@@ -1538,13 +1517,20 @@ const exit_subscriber_capacity = 8;
const ExitSubscriber = struct { endpoint: *ipc.Endpoint, owner: u32 };
var exit_subscribers: [exit_subscriber_capacity]?ExitSubscriber = .{null} ** exit_subscriber_capacity;
/// process_subscribe(endpoint): subscribe the caller's endpoint to published exit
/// events. Ungated, like process_enumerate — what is running (and dying) is not a
/// secret between cooperating processes. -ENOSPC when the table is full.
/// process_subscribe(endpoint): subscribe the **caller's own** endpoint to
/// published exit events. *Which* deaths one may hear of is ungated, like
/// process_enumerate — what is running (and dying) is not a secret between
/// cooperating processes. *Whose mailbox* they land in is not: the endpoint must
/// be the caller's (`ipc.ownedBy`), or any process could aim the firehose at a
/// stranger — filling PID 1's mailbox with exit notices it reads as its own
/// children's, and spending the eight-slot table so the services that need
/// deaths (the VFS's handle sweep) cannot subscribe at all. -EPERM otherwise,
/// -ENOSPC when the table is full.
fn systemProcessSubscribe(state: *architecture.CpuState) void {
const t = scheduler.current();
if (t.address_space == 0) return fail(state);
const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 0)) orelse return failErr(state, ipc.EBADF);
if (!ipc.ownedBy(endpoint, t)) return failErr(state, ipc.EPERM);
const flags = sync.enter();
defer sync.leave(flags);
for (&exit_subscribers) |*slot| {
@@ -1561,10 +1547,19 @@ fn systemProcessSubscribe(state: *architecture.CpuState) void {
/// IRQ-as-IPC pattern a fourth time (docs/process-lifecycle.md). Replacing a
/// binding drops the old reference; signals that pended while unbound are
/// delivered immediately on bind, coalesced into one notification.
///
/// The endpoint must be the caller's own (`ipc.ownedBy`), or `signal_bind`
/// becomes a signal *forgery* primitive: `process_signal` is deliberately loose
/// about the target (a task may always signal itself) because the delivery point
/// was assumed to be the target's own mailbox. Aim it elsewhere and a stranger
/// signalling itself makes the kernel stamp a genuine `terminate` badge into
/// somebody else's queue — which is a shutdown request PID 1 has no way to
/// disbelieve.
fn systemSignalBind(state: *architecture.CpuState) void {
const t = scheduler.current();
if (t.address_space == 0) return fail(state);
const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 0)) orelse return failErr(state, ipc.EBADF);
if (!ipc.ownedBy(endpoint, t)) return failErr(state, ipc.EPERM);
const flags = sync.enter();
defer sync.leave(flags);
if (t.signal_endpoint) |raw| ipc.dropRef(@ptrCast(@alignCast(raw)));
@@ -1633,11 +1628,19 @@ fn timerSweepLocked() void {
}
}
/// timer_bind(endpoint, ms): arm a one-shot timer. -ENOSPC when the table is full.
/// timer_bind(endpoint, ms): arm a one-shot timer on an endpoint of the caller's
/// own (`ipc.ownedBy`; -EPERM otherwise). A timer landing carries no identity —
/// that is the whole reason a service may keep exactly one in flight — so a
/// timer armed on someone else's endpoint is indistinguishable from one they
/// armed themselves, and a loop that re-arms on every landing (init's heartbeat)
/// multiplies: N forged timers leave N+1 self-perpetuating beats. The
/// sixteen-slot table is a shared resource on top of that. -ENOSPC when it is
/// full.
fn systemTimerBind(state: *architecture.CpuState) void {
const t = scheduler.current();
if (t.address_space == 0) return fail(state);
const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 0)) orelse return failErr(state, ipc.EBADF);
if (!ipc.ownedBy(endpoint, t)) return failErr(state, ipc.EPERM);
const ms = architecture.systemCallArg(state, 1);
const flags = sync.enter();
defer sync.leave(flags);
@@ -1677,12 +1680,17 @@ fn ownedGsi(t: *scheduler.Task, device_id: u64, resource_index: u64) ?u32 {
/// irq_bind(device_id, resource_index, endpoint) -> 0/-1: deliver that device's IRQ to the
/// endpoint as an asynchronous IPC notification. The driver then blocks in
/// IPC_ReplyWait and is woken by the ISR; see system/kernel/irq.zig for the cycle.
/// Two gates, both necessary: the device must be *claimed* by the caller
/// (`ownedGsi`), and the endpoint must be the caller's own (`ipc.ownedBy`) — a
/// claim entitles a driver to its own interrupts, not to post them into a
/// stranger's mailbox.
fn systemIrqBind(state: *architecture.CpuState) void {
const t = scheduler.current();
if (t.address_space == 0) return fail(state);
const gsi = ownedGsi(t, architecture.systemCallArg(state, 0), architecture.systemCallArg(state, 1)) orelse
return fail(state);
const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 2)) orelse return fail(state);
if (!ipc.ownedBy(endpoint, t)) return failErr(state, ipc.EPERM);
const flags = sync.enter();
defer sync.leave(flags);
@@ -1703,6 +1711,7 @@ fn systemMsiBind(state: *architecture.CpuState) void {
const owner = devices_broker.ownerOf(device_id) orelse return fail(state);
if (owner != t.id) return fail(state); // not claimed by this process
const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 1)) orelse return failErr(state, ipc.EBADF);
if (!ipc.ownedBy(endpoint, t)) return failErr(state, ipc.EPERM); // interrupts land in your own mailbox
const flags = sync.enter();
defer sync.leave(flags);
+116 -17
View File
@@ -246,6 +246,8 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void {
containmentTest();
} else if (eql(case, "device-manager")) {
deviceManagerTest(boot_information);
} else if (eql(case, "protocol-registry")) {
protocolRegistryTest(boot_information);
} else if (eql(case, "reboot")) {
rebootTest();
} else {
@@ -2209,7 +2211,7 @@ fn processKillTest(boot_information: *const BootInformation) void {
while (i < rd.count) : (i += 1) {
const item = rd.entry(i) orelse continue;
if (!eql(initial_ramdisk.basename(item.name), "process-test")) continue;
spinner = process.spawnProcessSupervised(item.blob, 4, &.{ "process-test", "spinner" }, me, endpoint) catch 0;
spinner = process.spawnProcessSupervised(item.blob, 4, &.{ item.name, "spinner" }, me, endpoint) catch 0;
break;
}
check("process-test spawned as the supervised spinner victim", spinner != 0);
@@ -2395,13 +2397,14 @@ fn signalsTest(boot_information: *const BootInformation) void {
};
process.setInitialRamdisk(image); // the parent system_spawns its children by name
_ = spawnRegistry(rd); // the service child binds /protocol/test/process
process.write_count = 0;
var runner: u32 = 0;
var i: u32 = 0;
while (i < rd.count) : (i += 1) {
const item = rd.entry(i) orelse continue;
if (!eql(initial_ramdisk.basename(item.name), "process-test")) continue;
runner = process.spawnProcessSupervised(item.blob, 4, &.{ "process-test", "signal-run" }, scheduler.currentId(), null) catch 0;
runner = process.spawnProcessSupervised(item.blob, 4, &.{ item.name, "signal-run" }, scheduler.currentId(), null) catch 0;
break;
}
check("signal-run parent spawned", runner != 0);
@@ -2443,13 +2446,14 @@ fn driverRestartTest(boot_information: *const BootInformation) void {
};
process.setInitialRamdisk(image); // the manager system_spawns drivers by name
_ = spawnRegistry(rd); // the drivers bind their contracts
process.write_count = 0;
var manager: u32 = 0;
var i: u32 = 0;
while (i < rd.count) : (i += 1) {
const item = rd.entry(i) orelse continue;
if (!eql(initial_ramdisk.basename(item.name), "device-manager")) continue;
manager = process.spawnProcessSupervised(item.blob, 4, &.{ "device-manager", "test-restart" }, scheduler.currentId(), null) catch 0;
manager = process.spawnProcessSupervised(item.blob, 4, &.{ item.name, "test-restart" }, scheduler.currentId(), null) catch 0;
break;
}
check("device-manager spawned in test-restart mode", manager != 0);
@@ -2482,12 +2486,13 @@ fn usbReportTest(boot_information: *const BootInformation) void {
};
process.setInitialRamdisk(image);
_ = spawnRegistry(rd); // the xhci driver binds /protocol/usb-transfer
var manager: u32 = 0;
var i: u32 = 0;
while (i < rd.count) : (i += 1) {
const item = rd.entry(i) orelse continue;
if (!eql(initial_ramdisk.basename(item.name), "device-manager")) continue;
manager = process.spawnProcessSupervised(item.blob, 4, &.{ "device-manager", "test-usb-restart" }, scheduler.currentId(), null) catch 0;
manager = process.spawnProcessSupervised(item.blob, 4, &.{ item.name, "test-usb-restart" }, scheduler.currentId(), null) catch 0;
break;
}
check("device-manager spawned in test-usb-restart mode", manager != 0);
@@ -2514,12 +2519,13 @@ fn deviceListTest(boot_information: *const BootInformation) void {
};
process.setInitialRamdisk(image);
_ = spawnRegistry(rd); // the fixture opens /protocol/device-manager
var manager: u32 = 0;
var i: u32 = 0;
while (i < rd.count) : (i += 1) {
const item = rd.entry(i) orelse continue;
if (!eql(initial_ramdisk.basename(item.name), "device-manager")) continue;
manager = process.spawnProcessSupervised(item.blob, 4, &.{ "device-manager", "test-usb-restart" }, scheduler.currentId(), null) catch 0;
manager = process.spawnProcessSupervised(item.blob, 4, &.{ item.name, "test-usb-restart" }, scheduler.currentId(), null) catch 0;
break;
}
check("device-manager spawned in test-usb-restart mode", manager != 0);
@@ -2548,13 +2554,14 @@ fn pciCapsTest(boot_information: *const BootInformation) void {
};
process.setInitialRamdisk(image);
_ = spawnRegistry(rd); // the manager binds /protocol/device-manager
// Plain mode — no restart drill, whose kill would race the fixture's claim.
var manager: u32 = 0;
var i: u32 = 0;
while (i < rd.count) : (i += 1) {
const item = rd.entry(i) orelse continue;
if (!eql(initial_ramdisk.basename(item.name), "device-manager")) continue;
manager = process.spawnProcessSupervised(item.blob, 4, &.{"device-manager"}, scheduler.currentId(), null) catch 0;
manager = process.spawnProcessSupervised(item.blob, 4, &.{item.name}, scheduler.currentId(), null) catch 0;
break;
}
check("device-manager spawned", manager != 0);
@@ -2582,12 +2589,13 @@ fn iommuFaultTest(boot_information: *const BootInformation) void {
check("IOMMU enabled for the enforcement test", iommu.enabled());
process.setInitialRamdisk(image);
_ = spawnRegistry(rd); // the manager binds /protocol/device-manager
var manager: u32 = 0;
var i: u32 = 0;
while (i < rd.count) : (i += 1) {
const item = rd.entry(i) orelse continue;
if (!eql(initial_ramdisk.basename(item.name), "device-manager")) continue;
manager = process.spawnProcessSupervised(item.blob, 4, &.{"device-manager"}, scheduler.currentId(), null) catch 0;
manager = process.spawnProcessSupervised(item.blob, 4, &.{item.name}, scheduler.currentId(), null) catch 0;
break;
}
check("device-manager spawned", manager != 0);
@@ -2623,12 +2631,13 @@ fn pciScanTest(boot_information: *const BootInformation) void {
check("the kernel seeded no PCI functions (the walk retired)", brokerPciCount(&buffer) == 0);
process.setInitialRamdisk(image);
_ = spawnRegistry(rd); // the manager binds /protocol/device-manager
var manager: u32 = 0;
var i: u32 = 0;
while (i < rd.count) : (i += 1) {
const item = rd.entry(i) orelse continue;
if (!eql(initial_ramdisk.basename(item.name), "device-manager")) continue;
manager = process.spawnProcessSupervised(item.blob, 4, &.{ "device-manager", "test-pci-restart" }, scheduler.currentId(), null) catch 0;
manager = process.spawnProcessSupervised(item.blob, 4, &.{ item.name, "test-pci-restart" }, scheduler.currentId(), null) catch 0;
break;
}
check("device-manager spawned (test-pci-restart mode)", manager != 0);
@@ -2776,12 +2785,13 @@ fn acpiReportTest(boot_information: *const BootInformation) void {
return;
};
process.setInitialRamdisk(image);
_ = spawnRegistry(rd); // the manager and the acpi service bind theirs
var spawned = false;
var i: u32 = 0;
while (i < rd.count) : (i += 1) {
const item = rd.entry(i) orelse continue;
if (!eql(initial_ramdisk.basename(item.name), "device-manager")) continue;
_ = process.spawnProcessSupervised(item.blob, 4, &.{"device-manager"}, scheduler.currentId(), null) catch 0;
_ = process.spawnProcessSupervised(item.blob, 4, &.{item.name}, scheduler.currentId(), null) catch 0;
spawned = true;
break;
}
@@ -2819,7 +2829,7 @@ fn acpiParseTest(boot_information: *const BootInformation) void {
while (i < rd.count) : (i += 1) {
const item = rd.entry(i) orelse continue;
if (!eql(initial_ramdisk.basename(item.name), "discovery")) continue;
_ = process.spawnProcessSupervised(item.blob, 4, &.{ "discovery", "1" }, scheduler.currentId(), null) catch 0;
_ = process.spawnProcessSupervised(item.blob, 4, &.{ item.name, "1" }, scheduler.currentId(), null) catch 0;
spawned = true;
break;
}
@@ -2854,7 +2864,7 @@ fn supervisionTest(boot_information: *const BootInformation) void {
while (i < rd.count) : (i += 1) {
const item = rd.entry(i) orelse continue;
if (!eql(initial_ramdisk.basename(item.name), "process-test")) continue;
started = if (process.spawnProcess(item.blob, 4, &.{ "process-test", "run" })) true else |_| false;
started = if (process.spawnProcess(item.blob, 4, &.{ item.name, "run" })) true else |_| false;
break;
}
check("process-test spawned as the user-space supervisor", started);
@@ -2996,6 +3006,9 @@ fn inputTest(boot_information: *const BootInformation) void {
process.write_count = 0;
process.write_from_user = false;
// init (the registry, below) reads its manifests through the kernel VFS.
process.setInitialRamdisk(image);
_ = spawnRegistry(rd); // the input service binds /protocol/input
_ = spawnNamed(rd, "input"); // the fan-out service
_ = spawnNamed(rd, "input-source"); // a synthetic keyboard publishing events
_ = spawnNamed(rd, "input-test"); // the subscriber whose "ok" line is the marker
@@ -3037,6 +3050,10 @@ fn displayServiceTest(boot_information: *const BootInformation) void {
return;
};
// init (the registry) reads its manifests through the kernel VFS.
process.setInitialRamdisk(image);
_ = spawnRegistry(rd); // the compositor binds /protocol/display
// Spawn the compositor and hand it the core. Its own serial heartbeats — `display:
// online WxH` and `display: presented frame 0` — are what the harness matches (it
// reads serial directly, like the fault cases). We don't poll for them in-kernel: a
@@ -3073,6 +3090,9 @@ fn displayCursorTest(boot_information: *const BootInformation) void {
return;
};
// init (the registry, below) reads its manifests through the kernel VFS.
process.setInitialRamdisk(image);
_ = spawnRegistry(rd); // input and display bind theirs
if (!spawnNamed(rd, "input")) {
log("display-cursor: could not spawn the input service\n", .{});
result();
@@ -3113,6 +3133,9 @@ fn displayDemoTest(boot_information: *const BootInformation) void {
return;
};
// init (the registry, below) reads its manifests through the kernel VFS.
process.setInitialRamdisk(image);
_ = spawnRegistry(rd); // the compositor binds /protocol/display
if (!spawnNamed(rd, "display")) {
log("display-demo: could not spawn the display service\n", .{});
result();
@@ -3148,6 +3171,9 @@ fn sharedMemoryTest(boot_information: *const BootInformation) void {
return;
};
// init (the registry, below) reads its manifests through the kernel VFS.
process.setInitialRamdisk(image);
_ = spawnRegistry(rd); // the server binds /protocol/test/shared-memory
if (!spawnNamed(rd, "shared-memory-server")) {
log("shared-memory: could not spawn shared-memory-server\n", .{});
result();
@@ -3185,12 +3211,13 @@ fn virtioGpuTest(boot_information: *const BootInformation) void {
// from the kernel device tree, spawns pci-bus, and matches the virtio-gpu class triple to
// spawn our driver with the function's device id as argv[1].
process.setInitialRamdisk(image);
_ = spawnRegistry(rd); // the driver binds /protocol/scanout
var manager: u32 = 0;
var i: u32 = 0;
while (i < rd.count) : (i += 1) {
const item = rd.entry(i) orelse continue;
if (!eql(initial_ramdisk.basename(item.name), "device-manager")) continue;
manager = process.spawnProcessSupervised(item.blob, 4, &.{"device-manager"}, scheduler.currentId(), null) catch 0;
manager = process.spawnProcessSupervised(item.blob, 4, &.{item.name}, scheduler.currentId(), null) catch 0;
break;
}
if (manager == 0) {
@@ -3226,12 +3253,13 @@ fn displayNativeTest(boot_information: *const BootInformation) void {
};
process.setInitialRamdisk(image);
_ = spawnRegistry(rd); // display, the manager, and the driver bind theirs
var manager: u32 = 0;
var i: u32 = 0;
while (i < rd.count) : (i += 1) {
const item = rd.entry(i) orelse continue;
if (!eql(initial_ramdisk.basename(item.name), "device-manager")) continue;
manager = process.spawnProcessSupervised(item.blob, 4, &.{"device-manager"}, scheduler.currentId(), null) catch 0;
manager = process.spawnProcessSupervised(item.blob, 4, &.{item.name}, scheduler.currentId(), null) catch 0;
break;
}
if (manager == 0) {
@@ -3270,12 +3298,13 @@ fn displayReattachTest(boot_information: *const BootInformation) void {
};
process.setInitialRamdisk(image);
_ = spawnRegistry(rd); // display and the restarted driver bind theirs
var manager: u32 = 0;
var i: u32 = 0;
while (i < rd.count) : (i += 1) {
const item = rd.entry(i) orelse continue;
if (!eql(initial_ramdisk.basename(item.name), "device-manager")) continue;
manager = process.spawnProcessSupervised(item.blob, 4, &.{ "device-manager", "test-scanout-restart" }, scheduler.currentId(), null) catch 0;
manager = process.spawnProcessSupervised(item.blob, 4, &.{ item.name, "test-scanout-restart" }, scheduler.currentId(), null) catch 0;
break;
}
if (manager == 0) {
@@ -3625,6 +3654,21 @@ fn threadTestMarkerCase(boot_information: *const BootInformation, case_name: []c
result();
}
/// Bring up the protocol namespace for a scenario that spawns its providers
/// itself. `/protocol` is served by init, PID 1 — but a scenario case wants the
/// naming layer without init's whole service list underneath it, so init is
/// started in its `registry` role: it mounts `/protocol`, reads the grants, and
/// spawns nothing (docs/os-development/protocol-namespace.md; the plan's
/// decision 9). Providers retry their bind, so racing the mount is survivable —
/// but calling this first makes the race rare.
///
/// The caller must have published the initial ramdisk already
/// (`process.setInitialRamdisk`): init reads its manifests out of it, and every
/// `/protocol` resolve goes through the same kernel VFS.
fn spawnRegistry(rd: initial_ramdisk.Reader) bool {
return spawnNamedWithArg(rd, "init", "registry");
}
fn spawnNamed(rd: initial_ramdisk.Reader, name: []const u8) bool {
var i: u32 = 0;
while (i < rd.count) : (i += 1) {
@@ -3745,6 +3789,60 @@ fn childDescriptor(hid: []const u8, start: u64, len: u64) device_abi.DeviceDescr
/// match `pci-bus`, and spawn it (with the bridge id as its argument) — and the spawned
/// pci-bus must reach its own live marker. It uses no special privilege — the same
/// `device_enumerate` any process could call.
/// P2 — the registrar (docs/os-development/protocol-namespace.md). Bring up
/// `/protocol` (init in its registry role) and hand the fixture the core: it
/// asserts that an ungranted bind is refused, that the kernel's reserved prefix
/// holds, that a name a live provider holds cannot be taken, and that killing a
/// provider makes its channel fail while re-resolving the same name reaches the
/// restarted instance.
///
/// It doubles as the security case for PID 1's shared mailbox, since resolving
/// `/protocol` hands every process a sendable handle to it: a forged power
/// payload, a redirected terminate signal, a timer or exit subscription armed on
/// a foreign endpoint, and capability-carrying ping storms against both PID 1 and
/// a harness-run service. Those assertions kill the boot when they regress rather
/// than printing anything, which is the strongest form available here.
///
/// The fixture's `protocol-registry: ok` is the marker; each step also prints its
/// own line, which the harness's ordered regex reads.
fn protocolRegistryTest(boot_information: *const BootInformation) void {
log("DANOS-TEST-BEGIN: protocol-registry\n", .{});
if (boot_information.initial_ramdisk_len == 0) {
check("bootloader handed over an initial_ramdisk", false);
result();
return;
}
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
const rd = initial_ramdisk.Reader.init(image) orelse {
check("initial_ramdisk image is valid", false);
result();
return;
};
// The fixture spawns its own providers by name, so the ramdisk must be
// published; init then mounts /protocol over the same kernel VFS.
process.setInitialRamdisk(image);
check("registry (init) spawned", spawnRegistry(rd));
check("protocol-registry-test spawned", spawnNamedWithArg(rd, "protocol-registry-test", "run"));
const pass_marker = "protocol-registry: ok";
const fail_marker = "protocol-registry: FAIL";
scheduler.setPriority(1);
const deadline = architecture.millis() + 20000;
var saw_pass = false;
var saw_fail = false;
while (architecture.millis() < deadline and !saw_pass and !saw_fail) {
if (bufferHas(pass_marker)) saw_pass = true;
if (bufferHas(fail_marker)) saw_fail = true;
scheduler.yield();
}
scheduler.setPriority(4);
check("no step of the registry contract failed", !saw_fail);
check("the fixture completed every registry assertion", saw_pass);
result();
}
fn deviceManagerTest(boot_information: *const BootInformation) void {
log("DANOS-TEST-BEGIN: device-manager\n", .{});
if (boot_information.initial_ramdisk_len == 0) {
@@ -3764,6 +3862,7 @@ fn deviceManagerTest(boot_information: *const BootInformation) void {
// all, it's because the manager discovered the PCI host bridge, matched, and
// spawned it.
process.setInitialRamdisk(image);
_ = spawnRegistry(rd); // the manager binds /protocol/device-manager
process.write_count = 0;
process.write_from_user = false;
@@ -3845,9 +3944,9 @@ fn hpetDeviceId() ?u64 {
/// 2. After `releaseOwner` for the binding's owner, that same entry is masked again.
///
/// And one property that can only be checked from kernel state: a *different* owner's
/// binding on the same endpoint survives. Endpoints are shared (ipc_register hands out
/// references), so teardown keyed on the endpoint pointer rather than the owning task
/// would mask a live sibling driver's device line.
/// binding on the same endpoint survives. Endpoints are shared (a capability passed in a
/// message hands out extra references), so teardown keyed on the endpoint pointer rather
/// than the owning task would mask a live sibling driver's device line.
fn irqFreeTest() void {
log("DANOS-TEST-BEGIN: irqfree\n", .{});
+33
View File
@@ -168,6 +168,11 @@ fn installMount(prefix: []const u8, kind: MountKind, backend: ?*ipc.Endpoint, re
var slot: ?*Mount = null;
for (&mounts) |*m| {
if (m.used and std.mem.eql(u8, m.prefixSlice(), prefix)) {
// ...except the protocol namespace. Remount-replace is how a
// restarted FAT retakes /volumes/usb; letting it retake /protocol
// would hand the whole naming layer to whoever asked second.
// First mount wins, and init (PID 1) is always first.
if (std.mem.eql(u8, prefix, protocol_root)) return;
if (m.backend) |old| ipc.dropRef(old);
slot = m;
break;
@@ -346,6 +351,30 @@ fn isInitrdCarveOut(prefix: []const u8) bool {
return false;
}
/// The protocol namespace's root — a reserved prefix, like the initrd trees.
/// Init (PID 1) mounts the registry here once at boot and the prefix then
/// refuses everything: a second mount at it, any mount *under* it (which would
/// shadow one contract), and its unmount. That is the whole kernel-side residue
/// of the naming layer — the registrar authority itself never leaves init
/// (docs/os-development/protocol-namespace.md).
const protocol_root = "/protocol";
fn protocolBound() bool {
for (&mounts) |*m| {
if (m.used and std.mem.eql(u8, m.prefixSlice(), protocol_root)) return true;
}
return false;
}
/// Whether mounting at `prefix` would touch the protocol namespace. Exactly
/// `/protocol` is allowed once — while nothing holds it; anything under it,
/// ever, is refused.
fn refusesProtocolMount(prefix: []const u8) bool {
const relative = underMount(prefix, protocol_root) orelse return false;
if (relative.len != 1) return true; // strictly under /protocol: never
return protocolBound(); // /protocol itself: first mount wins
}
/// Mount `backend` at `prefix` with an optional backend-side `rewrite` prefix.
/// The endpoint reference is taken by the caller (process.zig bumps it); refuses
/// shadowing or replacing the initrd trees (/system, /test) — except the two
@@ -353,6 +382,7 @@ fn isInitrdCarveOut(prefix: []const u8) bool {
pub fn mountBackend(prefix: []const u8, backend: *ipc.Endpoint, rewrite: []const u8) bool {
if (!isAbsolute(prefix) or prefix.len < 2 or prefix.len > maximum_prefix) return false;
if (rewrite.len > maximum_rewrite) return false;
if (refusesProtocolMount(prefix)) return false; // the registry's prefix is claimed once
for (&mounts) |*m| { // the initrd trees are not shadowable (carve-outs aside)
if (m.used and m.kind == .kernel_initrd and underMount(prefix, m.prefixSlice()) != null) {
if (!isInitrdCarveOut(prefix)) return false;
@@ -363,6 +393,9 @@ pub fn mountBackend(prefix: []const u8, backend: *ipc.Endpoint, rewrite: []const
}
pub fn unmount(prefix: []const u8) bool {
// Unmounting /protocol would delete the naming layer for everyone; nobody
// may, init included. The mount lasts the boot.
if (std.mem.eql(u8, prefix, protocol_root)) return false;
for (&mounts) |*m| {
if (m.used and m.kind == .backend and std.mem.eql(u8, m.prefixSlice(), prefix)) {
if (m.backend) |endpoint| ipc.dropRef(endpoint);
+26 -5
View File
@@ -12,6 +12,7 @@
const std = @import("std");
const device = @import("driver");
const channel = @import("channel");
const ipc = @import("ipc");
const process = @import("process");
const service = @import("service");
@@ -189,14 +190,32 @@ pub fn main(init: process.Init) void {
readFadt(fadt);
s5_valid = readSleepS5(&persistent_namespace);
// Every name this service needs, resolved before it becomes a provider — see
// `manager_channel`. Best-effort, as it has always been: a standalone
// bring-up with no device manager still serves power.
manager_channel = channel.openEndpoint("device-manager");
service.run(power_protocol.message_maximum, .{
.service = .power,
.service = "power",
.init = onInit,
.on_message = onMessage,
.on_notification = onNotification,
});
}
/// The device manager's channel, opened **before** this service binds its own
/// contract — deliberately, and load-bearing.
///
/// init is the registrar, and init is also this service's one subscriber: the
/// moment `power` is bound, init calls us to subscribe. init has a single thread,
/// so while it is blocked in that call it cannot answer anyone — including us. If
/// we opened a name after binding, the two could cross: init blocked calling us,
/// us blocked asking init to resolve a name, neither ever replying. Resolving
/// everything we need first makes that impossible, because after the bind this
/// service only ever talks to the device manager (which never calls init) and
/// then parks in the harness loop, where init's subscribe lands.
var manager_channel: ?ipc.Handle = null;
// Static so the harness callbacks (which run after main's stack frame is gone)
// can reach the namespace and interpreter.
var persistent_namespace: aml.Namespace = undefined;
@@ -209,7 +228,7 @@ fn onInit(endpoint: ipc.Handle) bool {
registered_count = 0;
walkDevices(persistent_namespace.root, &global_interpreter);
const manager = ipc.lookup(.device_manager);
const manager = manager_channel;
var i: usize = 0;
while (i < registered_count) : (i += 1) {
const entry = registered[i];
@@ -433,15 +452,17 @@ fn onNotification(badge: u64) void {
/// The `.power` protocol: subscribe (endpoint as the call's capability),
/// shutdown (PID 1 only). Device discovery uses a different endpoint (the
/// device manager's), so nothing here handles ChildAdded.
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Handle) usize {
fn onMessage(message: []const u8, reply: []u8, sender: u32, arrived: *ipc.Arrival) usize {
if (message.len < 1) return 0;
switch (message[0]) {
@intFromEnum(power_protocol.Operation.subscribe) => {
// The subscriber's endpoint is claimed only when a slot takes it;
// a full table refuses and the turn closes what arrived.
var status: i32 = -1;
if (capability) |handle| {
if (arrived.peek() != null) {
for (&subscribers, 0..) |*slot, si| {
if (slot.* == null) {
slot.* = handle;
slot.* = arrived.take();
subscriber_tasks[si] = sender;
status = 0;
break;
+2 -2
View File
@@ -14,8 +14,8 @@ pub fn build(b: *std.Build) void {
.name = "discovery",
.root_source_file = b.path("acpi.zig"),
.imports = &.{
"acpi-ids", "aml", "device-manager-protocol", "driver", "ipc", "logging", "memory",
"power-protocol", "process", "service", "time",
"acpi-ids", "aml", "channel", "device-manager-protocol", "driver", "ipc", "logging",
"memory", "power-protocol", "process", "service", "time",
},
});
b.installArtifact(exe);
@@ -395,13 +395,13 @@ fn initialise(endpoint: ipc.Handle) bool {
return true;
}
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Handle) usize {
fn onMessage(message: []const u8, reply: []u8, sender: u32, arrived: *ipc.Arrival) usize {
if (message.len < 1) return 0;
switch (message[0]) {
@intFromEnum(device_manager_protocol.Operation.child_added) => return onChildAdded(message, reply, sender),
@intFromEnum(device_manager_protocol.Operation.child_removed) => return onChildRemoved(message, reply, sender),
@intFromEnum(device_manager_protocol.Operation.enumerate) => return onEnumerate(reply),
@intFromEnum(device_manager_protocol.Operation.subscribe) => return onSubscribe(reply, capability),
@intFromEnum(device_manager_protocol.Operation.subscribe) => return onSubscribe(reply, arrived),
@intFromEnum(device_manager_protocol.Operation.hello) => {},
else => return 0,
}
@@ -532,13 +532,15 @@ fn onEnumerate(reply: []u8) usize {
return offset;
}
/// An application subscribed: its endpoint arrived as the call's capability.
fn onSubscribe(reply: []u8, capability: ?ipc.Handle) usize {
/// An application subscribed: its endpoint arrived as the call's capability. The
/// table taking a slot is what claims it (`take`); a full table refuses and lets
/// the turn close it, so a subscribe storm cannot spend the handle table too.
fn onSubscribe(reply: []u8, arrived: *ipc.Arrival) usize {
var status: i32 = -1;
if (capability) |handle| {
if (arrived.peek() != null) {
for (&subscribers) |*slot| {
if (slot.* == null) {
slot.* = handle;
slot.* = arrived.take(); // claimed: the table holds it from here
status = 0;
break;
}
@@ -566,7 +568,7 @@ pub fn main(init: process.Init) void {
test_scanout_restart_mode = std.mem.eql(u8, mode, "test-scanout-restart");
}
service.run(device_manager_protocol.message_maximum, .{
.service = .device_manager,
.service = "device-manager",
.init = initialise,
.on_message = onMessage,
.on_notification = onNotification,
+2 -2
View File
@@ -10,8 +10,8 @@ pub fn build(b: *std.Build) void {
.name = "display",
.root_source_file = b.path("display.zig"),
.imports = &.{
"display-client", "display-protocol", "driver", "input-client", "ipc", "logging",
"memory", "scanout-protocol", "service", "thread", "time",
"channel", "display-client", "display-protocol", "driver", "input-client", "ipc",
"logging", "memory", "scanout-protocol", "service", "thread", "time",
},
.threaded = true, // real atomics/TLS (docs/threading.md)
});
+20 -10
View File
@@ -16,6 +16,7 @@
//! (docs/display-v2.md).
const std = @import("std");
const channel = @import("channel");
const ipc = @import("ipc");
const input = @import("input-client");
const Thread = @import("thread").Thread;
@@ -307,11 +308,17 @@ fn verifyNativePresent() void {
/// present channel, switch the backend to virtio-gpu, and queue a full-screen repaint. The
/// present is deferred to a timer (see `service_endpoint`) so it happens after this reply
/// unblocks the driver and it starts serving `.scanout`.
fn attachScanout(stride: u32, width: u32, height: u32, format: u32, refresh_hz: u32, capability: ?ipc.Handle, reply: []u8) usize {
const cap = capability orelse return fail(reply);
/// The surface arrives as the call's capability, and the harness's ownership rule
/// applies: nothing here claims it, so the turn closes it on every path. Safe
/// because a **mapping holds its own kernel reference** (system/kernel/process.zig
/// `systemSharedMemoryMap`) — the pixels stay ours after the handle naming them
/// goes, and a driver that dies and re-announces no longer costs a handle slot
/// per restart.
fn attachScanout(stride: u32, width: u32, height: u32, format: u32, refresh_hz: u32, arrived: *ipc.Arrival, reply: []u8) usize {
const cap = arrived.peek() orelse return fail(reply);
if (width == 0 or height == 0 or stride < width) return fail(reply);
const mapped = memory.sharedMap(cap) orelse return fail(reply);
const scanout = ipc.lookup(.scanout) orelse return fail(reply);
const scanout = channel.openEndpoint("scanout") orelse return fail(reply);
// A second announce means the driver died and was restarted (V6): re-attach to its fresh
// scanout. (The previous shared mapping leaks — there is no shared_memory_unmap syscall yet — but the
// frames are the dead driver's, reclaimed on its exit; a handful across a crash is benign.)
@@ -506,10 +513,13 @@ fn mouseListener(width: u32, height: u32) void {
return;
};
// Our own handle to the compositor's endpoint. IPC handles are per-thread, so we
// cannot reuse the main thread's service handle — we look the service up to install a
// handle in this thread's table. A poke posted here wakes the compositor loop parked
// in replyWait (docs/threading.md: handles do not cross threads).
cursor_channel.poke_endpoint = ipc.lookup(.display) orelse {
// cannot reuse the main thread's — this thread resolves and opens
// `/protocol/display` exactly like any other client would, once at startup, and
// gets its own handle. There is no special mechanism for reaching yourself: the
// registry does not know or care that the provider is this process. A poke posted
// here wakes the compositor loop parked in replyWait (docs/threading.md: handles
// do not cross threads).
cursor_channel.poke_endpoint = channel.openEndpoint("display") orelse {
_ = logging.write("display: mouse listener could not reach the compositor endpoint\n");
return;
};
@@ -613,7 +623,7 @@ fn fail(reply: []u8) usize {
return writeReply(reply, .{ .status = -1 });
}
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Handle) usize {
fn onMessage(message: []const u8, reply: []u8, sender: u32, arrived: *ipc.Arrival) usize {
_ = sender;
if (message.len < display_protocol.request_size) return fail(reply);
const request = std.mem.bytesToValue(display_protocol.Request, message[0..display_protocol.request_size]);
@@ -657,7 +667,7 @@ fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Han
return ok(reply);
},
@intFromEnum(display_protocol.Operation.attach_scanout) => {
return attachScanout(request.x, request.width, request.height, request.colour, request.y, capability, reply);
return attachScanout(request.x, request.width, request.height, request.colour, request.y, arrived, reply);
},
@intFromEnum(display_protocol.Operation.set_mode) => {
if (!backend.setMode(request.width, request.height)) return fail(reply);
@@ -696,7 +706,7 @@ fn onNotification(badge: u64) void {
pub fn main() void {
service.run(display_protocol.message_maximum, .{
.service = .display,
.service = "display",
.init = initialise,
.on_message = onMessage,
.on_notification = onNotification,
+13 -4
View File
@@ -222,8 +222,14 @@ fn handleOpen(out: []u8, path: []const u8, flags: u32, sender: u32) usize {
return writeReply(out, .{ .status = 0, .node = index }, &.{});
}
fn onMessage(message: []const u8, out: []u8, sender: u32, capability: ?ipc.Handle) usize {
_ = capability;
/// The vfs protocol has no operation that takes a capability, so `arrived` is
/// never claimed here — which, under the harness's ownership rule, means the
/// loop closes whatever a caller attached. That is the point of the rule: this
/// callback used to discard a `?ipc.Handle` and every request carrying one — a
/// legal thing for any client to do — spent a slot of the VFS server's
/// thirty-two until it could accept no capability at all.
fn onMessage(message: []const u8, out: []u8, sender: u32, arrived: *ipc.Arrival) usize {
_ = arrived;
if (!mounted) return fail(out); // storage not up (yet): fail politely, clients retry
if (message.len < vfs_protocol.request_size) return fail(out);
const request = std.mem.bytesToValue(vfs_protocol.Request, message[0..vfs_protocol.request_size]);
@@ -305,13 +311,16 @@ fn onMessage(message: []const u8, out: []u8, sender: u32, capability: ?ipc.Handl
return writeReply(out, .{ .status = 0 }, &.{});
},
// A backend is never itself a mount target.
.mount, .unmount => return fail(out),
// Router verbs, and the registry's claim verb: a file backend answers
// none of them (docs/os-development/protocol-namespace.md — only init
// implements `bind`).
.mount, .unmount, .bind => return fail(out),
}
}
pub fn main() void {
service.run(vfs_protocol.message_maximum, .{
.service = .fat,
.service = "vfs",
.init = initialise,
.on_message = onMessage,
.on_notification = onNotification,
+2 -2
View File
@@ -11,8 +11,8 @@ pub fn build(b: *std.Build) void {
.name = "init",
.root_source_file = b.path("init.zig"),
.imports = &.{
"csv", "file-system", "ipc", "logging", "memory", "power-protocol",
"process", "time",
"csv", "envelope", "file-system", "ipc", "logging", "memory", "power-protocol",
"process", "time", "vfs-protocol",
},
});
// init reads the same `serial` flag the kernel does: its liveness heartbeat
+674 -54
View File
@@ -16,6 +16,28 @@
//! power-button event runs the stop sequence over its children in reverse order
//! before asking the power service to enter S5 — lifecycle (M17) and events (M21)
//! composing into a clean poweroff.
//!
//! P2: init is also the **registrar** — it serves `/protocol`, the namespace where
//! a program finds everything it talks to
//! (docs/os-development/protocol-namespace.md). It is the natural home: it already
//! spawns the services and already holds the supervision link to each, so it is the
//! process that *knows* which binary is which. Registry traffic rides the same
//! endpoint as supervision, because one thread can only wait in one place — the
//! loop below answers vfs `open`/`readdir`/`bind` alongside signals, timers, power
//! events, and children's deaths.
//!
//! Which sets the security posture of this file. `fs_resolve` installs a mounted
//! backend's endpoint capability in *any* caller's handle table, so sharing the
//! mailbox means **every ring-3 process can send into PID 1**. Two rules follow,
//! and both are structural here rather than remembered per branch:
//!
//! - **Privileged action requires an attested sender.** What arrives is a
//! stranger's bytes; the only identity on it is the task id the kernel stamps.
//! Content never authorizes (`onPowerEvent`), and neither does a name — the
//! registrar attests a caller's supervision by task id (`supervisorSatisfies`).
//! - **A capability that arrives is closed unless it is claimed** (`Arrival`),
//! because PID 1's thirty-two handle slots are a resource an unauthenticated
//! caller would otherwise be able to spend.
const std = @import("std");
const ipc = @import("ipc");
@@ -24,6 +46,8 @@ const time = @import("time");
const memory = @import("memory");
const logging = @import("logging");
const power_protocol = @import("power-protocol");
const vfs_protocol = @import("vfs-protocol");
const envelope = @import("envelope");
const build_options = @import("build_options");
const fs = @import("file-system");
const csv = @import("csv");
@@ -73,17 +97,10 @@ var supervision_endpoint: ipc.Handle = 0;
/// uses for /system/configuration/devices.csv. A missing file means no services (the no-ramdisk
/// isolation test): loud, but not fatal.
fn loadServices() void {
var file = fs.open("/system/configuration/init.csv", .{}) orelse {
const used = readConfiguration("/system/configuration/init.csv", &init_csv) orelse {
_ = logging.write("/system/services/init: /system/configuration/init.csv missing — no services started\n");
return;
};
defer file.close();
var used: usize = 0;
while (used < init_csv.len) {
const n = file.read(init_csv[used..]) orelse break;
if (n == 0) break;
used += n;
}
var lines = std.mem.splitScalar(u8, init_csv[0..used], '\n');
while (lines.next()) |line| {
const body = csv.stripComment(line);
@@ -107,11 +124,500 @@ fn loadServices() void {
}
}
/// Read a whole configuration file into `into`, returning the byte count. Both of
/// init's manifests live in the initial ramdisk the kernel serves directly, so this
/// works before any filesystem service exists. A file that fills the buffer exactly
/// is reported: a manifest silently losing its last rows is a policy change nobody
/// asked for, and the symptom (one service refused a name) points nowhere near it.
fn readConfiguration(path: []const u8, into: []u8) ?usize {
var file = fs.open(path, .{}) orelse return null;
defer file.close();
var used: usize = 0;
while (used < into.len) {
const n = file.read(into[used..]) orelse break;
if (n == 0) break;
used += n;
}
if (used == into.len) std.log.info("{s} filled the read buffer — rows past {d} bytes are lost", .{ path, used });
return used;
}
/// Give up restarting a service after this many crashes — a crash-loop cap, so a service
/// that faults immediately on every spawn doesn't respawn forever.
const maximum_restarts = 3;
pub fn main() void {
// --- the registry: /protocol ------------------------------------------------
/// Longest contract name the namespace admits (`display`, `test/shared-memory`)
/// and the most that may be bound at once. Both static, like everything else
/// init holds.
const maximum_name = 64;
const maximum_bindings = 16;
const maximum_grants = 48;
/// One bound contract: the name, the provider's endpoint (a capability init
/// holds and hands to whoever opens the name), and the provenance a diagnostic
/// listing answers "who serves this?" with.
const Binding = struct {
used: bool = false,
name: [maximum_name]u8 = undefined,
name_len: usize = 0,
endpoint: ipc.Handle = 0,
task: u32 = 0,
binary: [64]u8 = undefined,
binary_len: usize = 0,
fn nameSlice(self: *const Binding) []const u8 {
return self.name[0..self.name_len];
}
fn binarySlice(self: *const Binding) []const u8 {
return self.binary[0..self.binary_len];
}
};
var bindings: [maximum_bindings]Binding = .{Binding{}} ** maximum_bindings;
/// What a grant row permits: claiming a name, or reaching one. `open` rows are
/// parsed and held but not yet enforced — every open resolves in P2, and P3 is
/// the milestone that turns these into refusals (docs/security-track-plan.md).
const Permission = enum { bind, open };
/// One row of `/system/configuration/protocol.csv`. Every field may end in `*`,
/// which matches any tail — the subtree scoping the design doc describes, and
/// what lets one row grant the whole `/test/` family its `test/...` names.
const Grant = struct {
binary: []const u8 = "",
supervisor: []const u8 = "",
permission: Permission = .bind,
name: []const u8 = "",
};
/// Roomier than init.csv's: this manifest carries a row per provider per spawn
/// path, its own format documentation, and grows again with the open grants.
var protocol_csv: [8192]u8 = undefined;
var grants: [maximum_grants]Grant = .{Grant{}} ** maximum_grants;
var grant_count: usize = 0;
/// Parse `/system/configuration/protocol.csv` — the grant manifest. Separate from
/// init.csv because every field there after the path is argv, and overloading that
/// would be ambiguous; separate *files* also means a grant exists for binaries init
/// never spawns (the drivers, which the device manager owns).
fn loadGrants() void {
const used = readConfiguration("/system/configuration/protocol.csv", &protocol_csv) orelse {
_ = logging.write("/system/services/init: /system/configuration/protocol.csv missing — no protocol may be bound\n");
return;
};
var lines = std.mem.splitScalar(u8, protocol_csv[0..used], '\n');
while (lines.next()) |line| {
const body = csv.stripComment(line);
if (body.len == 0) continue;
if (grant_count >= grants.len) {
_ = logging.write("/system/services/init: /system/configuration/protocol.csv has more rows than the table holds\n");
break;
}
var it = csv.fields(body);
const binary = it.next() orelse continue;
const supervisor = it.next() orelse continue;
const permission = it.next() orelse continue;
const name = it.next() orelse continue;
if (binary.len == 0 or supervisor.len == 0 or name.len == 0) continue;
const kind: Permission = if (std.mem.eql(u8, permission, "bind"))
.bind
else if (std.mem.eql(u8, permission, "open"))
.open
else
continue; // an unreadable row grants nothing rather than something wrong
grants[grant_count] = .{ .binary = binary, .supervisor = supervisor, .permission = kind, .name = name };
grant_count += 1;
}
}
/// Match a manifest field against a value: exact, or a trailing `*` matching any
/// tail. The wildcard is how a subtree is granted whole (`/test/*` for every test
/// fixture, `test/*` for every name they may claim).
fn matches(pattern: []const u8, value: []const u8) bool {
if (pattern.len != 0 and pattern[pattern.len - 1] == '*') {
const prefix = pattern[0 .. pattern.len - 1];
return value.len >= prefix.len and std.mem.eql(u8, value[0..prefix.len], prefix);
}
return std.mem.eql(u8, pattern, value);
}
/// A snapshot of the kernel's process records — the only identity in the system
/// that cannot be forged, because the kernel stamps it at spawn. Refreshed per
/// authorization; binds are rare, so the copy costs nothing that matters.
var process_table: [64]process.ProcessDescriptor = undefined;
var process_count: usize = 0;
var process_truncated = false;
fn refreshProcessTable() void {
const total = process.processes(&process_table);
process_count = @min(total, process_table.len);
process_truncated = total > process_table.len;
}
/// Whether task `id` is still alive, as the last snapshot saw it. A snapshot that
/// did not fit answers "alive" for anything it did not list: refusing a bind is
/// recoverable, stealing a live provider's name is not.
fn taskAlive(id: u32) bool {
return descriptorOf(id) != null or process_truncated;
}
fn descriptorOf(id: u32) ?*const process.ProcessDescriptor {
for (process_table[0..process_count]) |*descriptor| {
if (descriptor.id == id) return descriptor;
}
return null;
}
fn nameOf(descriptor: *const process.ProcessDescriptor) []const u8 {
const length = @min(@as(usize, descriptor.name_length), descriptor.name.len);
return descriptor.name[0..length];
}
/// The name a kernel task answers to in a grant row. Kernel tasks carry no
/// binary, so the manifest spells the harness's parentage `kernel`.
const kernel_supervisor = "kernel";
/// init's own task id, read once at startup. Ids are monotonic and never reused
/// (system/kernel/process.zig), so an id comparison is an *identity* test where a
/// name comparison is only a resemblance test — the whole basis of the
/// attestation below.
var own_task: u32 = 0;
/// Whether `id` is a process THIS init spawned: a lookup in its own child table,
/// which is the one record of "I started that one" nobody else can write.
fn spawnedByUs(id: u32) bool {
if (id == 0) return false;
for (child_ids[0..service_count]) |child| {
if (child == id) return true;
}
return false;
}
/// Who is asking, attested by the kernel: the caller's binary, and the **task**
/// that spawned it — an id, not a name.
///
/// A name alone is not identity: `spawn` is ungated, so a hostile process can
/// start a granted binary itself and would inherit its grants. Neither is the
/// supervisor's *name* enough, and this is the trap the first cut fell into —
/// init and the device manager are ordinary bundled binaries, so an attacker
/// spawns its own `/system/services/init` and lets that instance spawn
/// `/system/services/input`. Both kernel-stamped names then match the grant row
/// exactly, and walking to the root of the chain does not help either: the
/// laundered chain still roots at the real PID 1. What refuses it is asking
/// *which task* the supervisor is, and only accepting one init can vouch for.
const Identity = struct {
/// The caller's binary path, exactly as the kernel stamped it at spawn.
binary: []const u8,
/// The supervising task's id. 0 means the kernel spawned the caller, which
/// no ring-3 process can arrange: every `system_spawn` stamps the caller as
/// the child's supervisor (system/kernel/process.zig `systemSpawn`).
supervisor_task: u32,
/// The supervising task's kernel-stamped binary — the grant row's supervisor
/// column is matched against this, and the refusal log prints it. `kernel`
/// when there is no supervising task.
supervisor_binary: []const u8,
/// Whether init can vouch for how the supervising task came to exist: it is
/// this init, a process this init spawned, or a process the KERNEL spawned.
/// A supervisor init cannot vouch for satisfies no row, however well its
/// name reads — that is the laundering deputy's refusal.
supervisor_vouched: bool,
};
/// The process a task belongs to. A thread resolves to its leader: threads share
/// a binary (a thread's own record is named `thread`), and the supervision link
/// that matters is the process's.
fn leaderOf(descriptor: *const process.ProcessDescriptor) *const process.ProcessDescriptor {
if (descriptor.leader == descriptor.id) return descriptor;
return descriptorOf(descriptor.leader) orelse descriptor;
}
/// Resolve the badge on a request into an identity, one hop up the supervision
/// chain in the kernel's records — one hop is enough because the hop is attested
/// by id (see `supervisorSatisfies`), and every id in the chain init accepts is
/// one init or the kernel created.
fn identify(task: u32) ?Identity {
const caller = descriptorOf(task) orelse return null;
const leader = leaderOf(caller);
if (leader.supervisor == 0) return .{
.binary = nameOf(leader),
.supervisor_task = 0,
.supervisor_binary = kernel_supervisor,
.supervisor_vouched = true, // the kernel is the root of trust, not a claimant
};
// The supervising *task* may be a worker thread of the supervising process;
// its process is what the manifest names and what init recorded at spawn.
const supervisor = leaderOf(descriptorOf(leader.supervisor) orelse return null); // unattestable: refuse
return .{
.binary = nameOf(leader),
.supervisor_task = supervisor.id,
.supervisor_binary = nameOf(supervisor),
.supervisor_vouched = supervisor.id == own_task or
spawnedByUs(supervisor.id) or
supervisor.supervisor == 0,
};
}
/// Whether the caller's supervising task satisfies a grant row's supervisor
/// column. The column names *the authorized supervising task*, matched by
/// identity — the binary it must be, plus proof that this instance of that
/// binary is the authorized one:
///
/// - `kernel` is satisfied only by a genuinely kernel-spawned caller
/// (supervisor id 0). A ring-3 process cannot manufacture that: user
/// `system_spawn` always stamps the caller (system/kernel/process.zig).
/// - init's own binary is satisfied only when the supervising task IS this
/// init (`own_task`).
/// - any other binary — the device manager, a test fixture spawning another —
/// is satisfied only when the supervising task is one init spawned itself
/// (its own child table) or one the kernel spawned. Everything init and the
/// kernel start is therefore reachable; a chain that passes through a
/// process *neither* of them started is not.
fn supervisorSatisfies(column: []const u8, identity: Identity) bool {
if (std.mem.eql(u8, column, kernel_supervisor)) return identity.supervisor_task == 0;
if (identity.supervisor_task == 0) return false; // a kernel task answers to no binary column
if (!matches(column, identity.supervisor_binary)) return false;
return identity.supervisor_vouched;
}
/// Whether `identity` is granted `permission` on `name`.
fn granted(identity: Identity, permission: Permission, name: []const u8) bool {
for (grants[0..grant_count]) |grant| {
if (grant.permission != permission) continue;
if (!matches(grant.binary, identity.binary)) continue;
if (!supervisorSatisfies(grant.supervisor, identity)) continue;
if (!matches(grant.name, name)) continue;
return true;
}
return false;
}
fn findBinding(name: []const u8) ?*Binding {
for (&bindings) |*binding| {
if (binding.used and std.mem.eql(u8, binding.nameSlice(), name)) return binding;
}
return null;
}
/// Release a binding: the provider's endpoint capability goes back to the handle
/// table, and the name is free for the next claimant. init's own cached power
/// channel goes with it — a closed handle number is reused by the next capability
/// that arrives, and a stale copy would quietly aim the shutdown call at a
/// stranger. So does the authorized power *task*: nothing may speak for a
/// contract nobody holds.
fn releaseBinding(binding: *Binding) void {
if (std.mem.eql(u8, binding.nameSlice(), power_contract)) {
power_endpoint = null;
power_task = null;
power_pending = false;
}
_ = ipc.close(binding.endpoint);
binding.* = .{};
}
/// Drop every name a dead process held. Called when a supervised child dies (so
/// the restarted instance can bind again) and whenever a bind finds the current
/// owner gone — providers init does not supervise need the second path.
fn unbindTask(task: u32) void {
for (&bindings) |*binding| {
if (binding.used and binding.task == task) {
std.log.info("/protocol/{s} released ({s} is gone)", .{ binding.nameSlice(), binding.binarySlice() });
releaseBinding(binding);
}
}
}
/// A contract name as the namespace spells it: the mount-relative path a resolve
/// hands us ("/display") and the name a bind sends ("display") are the same thing
/// with and without a leading slash, so one normaliser serves both. Empty or
/// longer than the namespace admits is not a name.
fn contractName(raw: []const u8) ?[]const u8 {
const name = if (raw.len != 0 and raw[0] == '/') raw[1..] else raw;
if (name.len == 0 or name.len > maximum_name) return null;
return name;
}
/// The provider's endpoint that will ride the *next* reply, when the request was
/// an `open` that found its contract.
var pending_capability: ?ipc.Handle = null;
/// The ownership rule for a capability that arrives with a turn of the loop —
/// **the turn owns it until a handler takes it, and closes whatever is left** —
/// lives in `ipc.Arrival`, next to `replyWait`, because it is not PID 1's rule:
/// the service harness every other service runs (library/kernel/service.zig) had
/// the identical hole and now states the identical contract.
const Arrival = ipc.Arrival;
/// The one contract init is itself a client of. It never resolves the name — it
/// *is* the registry, so it reads its own table; the binding is what hands it the
/// channel.
const power_contract = "power";
/// Set when `power` is bound: init subscribes to it on the next turn of the loop,
/// never inside the bind — the provider is blocked on our reply until then, so
/// calling it here would deadlock the pair.
var power_pending = false;
var power_endpoint: ?ipc.Handle = null;
/// The one task authorized to deliver power events: whoever holds the `power`
/// binding. Recorded at the bind and cleared with the binding, so a provider that
/// dies and rebinds re-derives it with no further ceremony.
///
/// This is the *authentication* for the shutdown path. init's registry endpoint
/// is its supervision endpoint, and `fs_resolve("/protocol")` installs a sendable
/// handle to it in any caller's table — so after P2 every ring-3 process can post
/// into PID 1's mailbox. A power event may therefore never be believed on the
/// strength of its payload; it is believed because the kernel stamped the
/// sender's task id on it and that id is the provider's.
var power_task: ?u32 = null;
/// Whether the heartbeat's re-arming timer is running. A timer landing carries no
/// identity, so the loop cannot tell one timer from another — which means exactly
/// one may ever be in flight, or every landing re-arms and the beat doubles. (It
/// did: two beats a second is enough extra chatter to cut a driver's echoed line
/// in half on the shared serial stream.) So the deferred power subscribe borrows
/// the heartbeat's tick when there is one, and arms its own only when there is not.
var heartbeat_running = false;
/// Answer one registry request. Writes a vfs-protocol reply into `reply` and
/// returns its length; a capability the reply must carry lands in
/// `pending_capability`. `arrived` is the capability the *request* carried, owned
/// by the turn — nothing here has to close it, only `bind` has to claim it.
fn serveRegistry(request_bytes: []const u8, reply: []u8, sender: u32, arrived: *Arrival) usize {
if (request_bytes.len < vfs_protocol.request_size)
return answer(reply, -envelope.EPROTO, 0, 0);
// The header is read field by field rather than reinterpreted whole: the
// operation is an enum on the wire and the bytes come from anyone at all, so
// a value outside it must be a refusal, never a decoded enum.
const operation = std.mem.readInt(u32, request_bytes[0..4], .little);
const cursor = std.mem.readInt(u64, request_bytes[16..24], .little);
const declared = std.mem.readInt(u32, request_bytes[24..28], .little);
const payload_len = @min(@as(usize, declared), request_bytes.len - vfs_protocol.request_size);
const payload = request_bytes[vfs_protocol.request_size..][0..payload_len];
if (operation == @intFromEnum(vfs_protocol.Operation.bind))
return answer(reply, onBind(sender, payload, arrived), 0, 0);
// Only `bind` claims a capability; one attached to anything else is closed by
// the turn's `defer` in the loop, along with the ones sent to a request that
// was too short to name a verb at all.
if (operation == @intFromEnum(vfs_protocol.Operation.open)) return onOpen(reply, payload);
if (operation == @intFromEnum(vfs_protocol.Operation.readdir)) return onReaddir(reply, cursor);
// Everything else a filesystem answers is meaningless here: `/protocol` holds
// contracts, not bytes.
return answer(reply, -envelope.ENOSYS, 0, 0);
}
/// Lay down a vfs reply header (and say how many payload bytes follow it).
fn answer(reply: []u8, status: i32, node: u64, payload_len: usize) usize {
const header = vfs_protocol.Reply{ .status = status, .node = node, .len = @intCast(payload_len) };
@memcpy(reply[0..vfs_protocol.reply_size], std.mem.asBytes(&header));
return vfs_protocol.reply_size + payload_len;
}
/// `bind(name, capability = the provider's endpoint)`. The capability is the
/// point of the call, so a bind without one is malformed. Every refusal below
/// simply returns: the endpoint stays the turn's, and the turn closes it — which
/// is why there is not one `ipc.close` on the way out of any of the six of them.
/// The success path is the only one that says anything about ownership, because
/// it is the only one that keeps the capability.
fn onBind(sender: u32, raw_name: []const u8, arrived: *Arrival) i32 {
if (arrived.peek() == null) return -envelope.EPROTO;
const name = contractName(raw_name) orelse return -envelope.ENOENT;
refreshProcessTable();
const identity = identify(sender) orelse return -envelope.EPERM;
if (!granted(identity, .bind, name)) {
std.log.info("refused bind of /protocol/{s} by {s} (pid {d}, supervisor {s} pid {d})", .{
name,
identity.binary,
sender,
identity.supervisor_binary,
identity.supervisor_task,
});
return -envelope.EPERM;
}
if (findBinding(name)) |existing| {
// Collision is an error — never last-writer-wins — unless the incumbent
// is dead, which is how a restarted provider retakes its own name.
if (taskAlive(existing.task)) {
std.log.info("refused bind of /protocol/{s}: held by {s} (pid {d})", .{ name, existing.binarySlice(), existing.task });
return -envelope.EBUSY;
}
releaseBinding(existing);
}
const slot = for (&bindings) |*binding| {
if (!binding.used) break binding;
} else return -envelope.ENOSPC;
// Claimed: the binding owns the endpoint from here, and `releaseBinding` is
// what closes it.
const endpoint = arrived.take().?;
slot.* = .{ .used = true, .endpoint = endpoint, .task = sender };
@memcpy(slot.name[0..name.len], name);
slot.name_len = name.len;
const binary_len = @min(identity.binary.len, slot.binary.len);
@memcpy(slot.binary[0..binary_len], identity.binary[0..binary_len]);
slot.binary_len = binary_len;
// Provenance, at the moment it becomes true: name -> pid -> binary path.
std.log.info("/protocol/{s} -> pid {d} {s}", .{ name, sender, slot.binarySlice() });
if (std.mem.eql(u8, name, power_contract)) {
power_endpoint = endpoint;
// The bind is also the authentication: whoever holds `power` is the one
// task whose power events init will act on (see `onPowerEvent`).
power_task = sender;
power_pending = true;
// Wake ourselves once the reply has gone out; the subscribe call cannot
// happen while the power service is still blocked on it. The heartbeat's
// tick is that wake when it is running — see `heartbeat_running`.
if (!heartbeat_running) _ = time.timerOnce(supervision_endpoint, 1);
}
return 0;
}
/// `open(name)` -> the provider's endpoint, delivered as the reply's capability.
/// A name nothing has bound is `-ENOENT`; in P3 an ungranted one becomes the same
/// answer, because absence and refusal are deliberately indistinguishable.
fn onOpen(reply: []u8, raw_name: []const u8) usize {
const name = contractName(raw_name) orelse return answer(reply, -envelope.ENOENT, 0, 0);
const binding = findBinding(name) orelse return answer(reply, -envelope.ENOENT, 0, 0);
pending_capability = binding.endpoint;
return answer(reply, 0, 0, 0);
}
/// `readdir(cursor)` — the namespace, browsable. One entry per turn, as the vfs
/// protocol lists any directory: kind `protocol`, the contract's name, and the
/// provider's task id in `size`, so a plain listing answers "who serves this?".
fn onReaddir(reply: []u8, cursor: u64) usize {
var index: u64 = 0;
for (&bindings) |*binding| {
if (!binding.used) continue;
if (index != cursor) {
index += 1;
continue;
}
const name = binding.nameSlice();
const entry = vfs_protocol.DirectoryEntry{
.kind = @intFromEnum(vfs_protocol.NodeKind.protocol),
.name_len = @intCast(name.len),
.size = binding.task,
};
const total = vfs_protocol.directory_entry_size + name.len;
if (vfs_protocol.reply_size + total > reply.len) return answer(reply, -envelope.EPROTO, 0, 0);
@memcpy(reply[vfs_protocol.reply_size..][0..vfs_protocol.directory_entry_size], std.mem.asBytes(&entry));
@memcpy(reply[vfs_protocol.reply_size + vfs_protocol.directory_entry_size ..][0..name.len], name);
return answer(reply, 0, 0, total);
}
return answer(reply, 0, 0, 0); // end of directory
}
pub fn main(startup: process.Init) void {
// `registry` is the scenario mode: serve /protocol and nothing else. The
// kernel test harness spawns its own providers directly, so it wants the
// naming layer up without init's whole service list underneath it
// (docs/security-track-plan.md, decision 9).
const registry_only = if (startup.arguments.get(1)) |role| std.mem.eql(u8, role, "registry") else false;
// Prove the heap end to end: allocate through the runtime allocator (which
// mmaps pages from the kernel and carves them with the free list), write into
// that heap buffer (exercising the widened debug_write bounds check), and
@@ -128,65 +634,175 @@ pub fn main() void {
// One endpoint carries everything init waits on: children's exit
// notifications (they are spawned supervised against it), init's own
// signals, and power events it subscribes to. All arrive in the loop below.
// signals, power events it subscribes to, and — since PID 1 is the registrar
// — every /protocol request. One thread can wait in one place, so they share
// a mailbox and the loop below tells them apart.
supervision_endpoint = ipc.createIpcEndpoint() orelse {
_ = logging.write("/system/services/init: no endpoint\n");
return;
};
_ = process.bindSignals(supervision_endpoint);
// Our own id, before anything can ask us a question. It is half of the
// registrar's authority: a grant row naming init as the supervisor is
// satisfied by *this* task and no other instance of this binary
// (`supervisorSatisfies`).
own_task = process.taskId();
// The namespace goes up BEFORE anything is spawned, so a service's first
// bind lands rather than retrying. The kernel reserves the prefix: this
// mount is the only one it will ever hold.
loadGrants();
if (!fs.mount("/protocol", supervision_endpoint)) {
_ = logging.write("/system/services/init: /protocol already mounted — not the registrar\n");
}
// Load the service list, then bring each up supervised so init can stop them
// cleanly. Best-effort and silent: each service announces its own readiness,
// and with no /system/configuration/init.csv (an isolation test) the loop starts nothing.
loadServices();
for (services[0..service_count], 0..) |*service, i| {
if (process.spawnSupervised(service.path, service.arguments(), supervision_endpoint)) |id| child_ids[i] = id;
if (!registry_only) {
loadServices();
for (services[0..service_count], 0..) |*service, i| {
if (process.spawnSupervised(service.path, service.arguments(), supervision_endpoint)) |id| child_ids[i] = id;
}
}
// Subscribe to power events (retry: the power service registers well after
// init starts). Best-effort — without it, a `terminate` signal still
// triggers the same shutdown path.
subscribePower();
// A re-arming timer drives the liveness heartbeat — proof PID 1 is alive (the
// init test's marker) and a -Dserial diagnostic. It is a serial/test-build-only
// concern: a flashable (serial-off) image runs a purely event-driven PID 1 that
// wakes only for real work (signals, power events, children's exits), never for a
// periodic beat. `build_options.serial` is comptime, so the heartbeat — its timer
// and the handler below — folds away entirely when serial is off.
if (build_options.serial) _ = time.timerOnce(supervision_endpoint, 1000);
heartbeat_running = build_options.serial and !registry_only;
if (heartbeat_running) _ = time.timerOnce(supervision_endpoint, 1000);
var receive: [power_protocol.message_maximum]u8 = undefined;
var receive: [vfs_protocol.message_maximum]u8 = undefined;
var reply_buffer: [vfs_protocol.message_maximum]u8 = undefined;
var reply_len: usize = 0;
var reply_capability: ?ipc.Handle = null;
while (true) {
const got = ipc.replyWait(supervision_endpoint, &.{}, &receive, null);
if (process.signalsFrom(got.badge)) |signals| {
if (signals.has(.terminate)) shutDown();
continue;
const got = ipc.replyWait(supervision_endpoint, reply_buffer[0..reply_len], &receive, reply_capability);
reply_len = 0; // nothing owed until this turn's request says otherwise
reply_capability = null;
// Whatever capability came with this turn is the turn's, and the turn
// closes it unless a handler claims it. Structural on purpose — see
// `Arrival`; it is what keeps a zero-length call from spending a handle
// slot of PID 1's per call.
var arrived: Arrival = .{ .handle = got.cap };
defer arrived.release();
if (got.isNotification()) {
// Every badge on this branch is stamped by the KERNEL, and a stranger
// cannot stamp one: `ipc_send` — the only way a ring-3 process puts
// something in this mailbox with no reply owed — sets exactly
// `notify_badge_bit | notify_message_bit` and fills the low bits with
// the sender's own task id (system/kernel/ipc-synchronous.zig,
// `sendLocked`).
//
// That is a statement about `ipc_send`, and on its own it proved far
// too little: an attacker does not use `ipc_send` to forge a signal,
// it asks the kernel to deliver a real one *here*. `fs_resolve`
// hands any process a sendable handle to this endpoint, and
// `signal_bind`/`timer_bind`/`process_subscribe`/spawn's exit
// endpoint all used to accept any handle the caller held — so a
// stranger could point its own signal delivery at PID 1 and signal
// itself, and the terminate badge landing here was genuine in every
// bit. What makes these branches trustworthy is therefore in the
// KERNEL, not in this comment: binding a kernel notification to an
// endpoint now requires *owning* that endpoint (`ipc.ownedBy`), so a
// signal here comes only from our supervisor or our own group, a
// timer landing only from a timer we armed, and a child-exit notice
// only from a child we spawned. The buffered-message branch below is
// the one still carrying a stranger's bytes, and it is the one that
// authenticates its sender.
if (process.signalsFrom(got.badge)) |signals| {
if (signals.has(.terminate)) shutDown();
continue;
}
if (got.isTimer()) {
// The pending power subscription rides any timer landing: by the
// time one arrives, the bind's reply has left and the power
// service is serving again.
if (power_pending) {
power_pending = false;
subscribePower();
}
if (heartbeat_running) {
_ = logging.write("/system/services/init: heartbeat\n");
_ = time.timerOnce(supervision_endpoint, 1000);
}
continue;
}
if (got.isMessage()) {
// A buffered message: the only thing here an anonymous stranger
// can put in front of PID 1. Authenticated by sender, never by
// payload — see `onPowerEvent`.
onPowerEvent(got.senderTaskId(), receive[0..got.len]);
continue;
}
if (got.isChildExit()) {
restartChild(got.childProcessId());
continue;
}
continue; // anything else: keep waiting
}
if (build_options.serial and got.isTimer()) {
_ = logging.write("/system/services/init: heartbeat\n");
_ = time.timerOnce(supervision_endpoint, 1000);
continue;
}
if (got.isMessage() and got.len >= 2 and receive[0] == @intFromEnum(power_protocol.Operation.event)) {
// A power event (the only buffered messages init receives).
if (receive[1] == @intFromEnum(power_protocol.Event.power_button)) shutDown();
continue;
}
if (got.isChildExit()) {
restartChild(got.childProcessId());
continue;
}
// Anything else: keep waiting.
if (got.isNotification()) continue;
// The universal ping, answered by the empty reply. A ping may still carry
// a capability — the kernel installs one regardless of length — and this
// `continue` disposes of it through the turn's `defer`, which is exactly
// what it failed to do when the close lived in the branches.
if (got.len == 0) continue;
reply_len = serveRegistry(receive[0..got.len], &reply_buffer, got.senderTaskId(), &arrived);
reply_capability = pending_capability;
pending_capability = null;
}
}
/// A buffered message claiming to be a power event.
///
/// **Privileged control traffic is authenticated by sender, never by content.**
/// init's registry endpoint is its supervision endpoint, and `fs_resolve` installs
/// a sendable handle to any mount's backend in *any* caller's table
/// (system/kernel/process.zig), so after P2 every ring-3 process holds a handle it
/// can `ipc_send` into. Two payload bytes were once enough to reach `shutDown()`
/// from here — which stops every service and parks PID 1 in its final sleep,
/// destroying the registry for the rest of the boot, and does it for any process
/// that cares to ask.
///
/// The sender's task id is the fix, because it is not the sender's to choose: the
/// kernel stamps it into the badge's low bits as it copies the message into the
/// ring. Init is the registry, so it knows exactly which task holds `power`, and
/// that task alone is believed. A provider that dies and rebinds moves the
/// authorization with the binding; a name nothing holds authorizes nobody. Task
/// ids are never reused, so even a dead provider's id cannot be inherited.
///
/// (One task, not one process: the ACPI service is single-threaded and publishes
/// from the same task that bound the name. A threaded provider would want its
/// leader compared instead — which is a change to make when one appears, not a
/// looser rule to leave lying around for it.)
fn onPowerEvent(sender: u32, payload: []const u8) void {
const authorized = power_task orelse {
std.log.info("ignored a power event from pid {d}: nothing holds /protocol/power", .{sender});
return;
};
if (sender != authorized) {
std.log.info("ignored a power event from pid {d}: /protocol/power is pid {d}", .{ sender, authorized });
return;
}
if (payload.len < 2) return;
if (payload[0] != @intFromEnum(power_protocol.Operation.event)) return;
if (payload[1] == @intFromEnum(power_protocol.Event.power_button)) shutDown();
}
/// A supervised boot service died. Find which one and restart it — unless it exited
/// cleanly (it chose to stop, e.g. a driver with no hardware) or has hit the crash-loop
/// cap. Reclaiming the dead process is already the kernel's job (docs/process-lifecycle.md
/// iron rule 1); init only decides whether to bring it back.
fn restartChild(id: u32) void {
// Whatever it served, it serves no longer: the name goes back before the
// replacement asks for it, so the restarted instance binds rather than
// colliding with its own corpse.
unbindTask(id);
if (shutting_down) return; // deaths during the stop sequence are expected, not crashes
for (services[0..service_count], 0..) |*service, i| {
if (child_ids[i] != id) continue;
@@ -209,22 +825,26 @@ fn restartChild(id: u32) void {
// An untracked child (e.g. the log-flush one-shot): nothing to restart.
}
/// Look up the power service and subscribe our endpoint (handed over as the
/// call's capability) so events arrive as buffered messages here.
/// Subscribe our endpoint (handed over as the call's capability) to the power
/// service, so events arrive as buffered messages here. init is the registry, so
/// it never resolves `/protocol/power` — it reads its own table, which is also
/// what makes this reachable at all: the subscription is armed by the bind that
/// put the endpoint there.
///
/// **This is the only place PID 1 blocks on another process, and it is the one
/// hazard the registrar has.** One thread serves both the namespace and this
/// call, so while it is outstanding init answers nobody: if the callee were
/// itself blocked asking init to resolve a name, the pair would never move. Two
/// things keep that from happening — the call is deferred to the next turn of
/// the loop (so the provider has its bind reply and is on its way to
/// `replyWait`), and the power provider resolves every name it needs *before* it
/// binds (system/services/acpi/acpi.zig, `manager_channel`). Any future service
/// init calls owes the same discipline.
fn subscribePower() void {
var handle: ?ipc.Handle = null;
var tries: u32 = 0;
while (handle == null and tries < 200) : (tries += 1) {
handle = ipc.lookup(.power);
if (handle == null) time.sleepMillis(20);
}
// A missing power service is not fatal — init proceeds to its heartbeat and
// a `terminate` signal still drives shutdown. Silent so the no-ramdisk init
// test's heartbeat marker is the next line written.
const h = handle orelse return;
const handle = power_endpoint orelse return;
const request = power_protocol.Subscribe{};
var reply: [power_protocol.message_maximum]u8 = undefined;
_ = ipc.callCap(h, std.mem.asBytes(&request), &reply, supervision_endpoint) catch {};
_ = ipc.callCap(handle, std.mem.asBytes(&request), &reply, supervision_endpoint) catch {};
}
/// The stop sequence: persist the log while storage is still up, then terminate
@@ -242,7 +862,7 @@ fn shutDown() void {
i -= 1;
if (child_ids[i] != 0) process.stop(child_ids[i], 2000, supervision_endpoint);
}
if (ipc.lookup(.power)) |h| {
if (power_endpoint) |h| {
const request = power_protocol.Shutdown{};
var reply: [power_protocol.message_maximum]u8 = undefined;
_ = ipc.call(h, std.mem.asBytes(&request), &reply) catch {};
+1 -1
View File
@@ -9,7 +9,7 @@ pub fn build(b: *std.Build) void {
const exe = build_support.userBinary(b, .{
.name = "input",
.root_source_file = b.path("input.zig"),
.imports = &.{ "input-client", "input-protocol", "ipc", "logging", "process", "service" },
.imports = &.{ "channel", "input-client", "input-protocol", "ipc", "logging", "process", "service" },
});
b.installArtifact(exe);
}
+30 -10
View File
@@ -1,5 +1,5 @@
//! system/services/input — the user-space input service. Shipped in the initial_ramdisk,
//! spawned as a ring-3 process, and published under the well-known `input` service id. It
//! spawned as a ring-3 process, and bound at `/protocol/input`. It
//! is the fan-out point between **sources** (keyboard, mouse, and joystick/gamepad drivers)
//! and **subscribers** (any program that wants input): a source `publish`es an
//! `InputEvent`, and the service pushes it to every subscriber whose interest mask includes
@@ -20,6 +20,7 @@
//! handle and `ipc.send`s each event to it.
const std = @import("std");
const channel = @import("channel");
const ipc = @import("ipc");
const process = @import("process");
const service = @import("service");
@@ -58,7 +59,13 @@ fn pruneDeadSubscribers() void {
break;
}
}
if (!alive) sub.* = .{};
// The slot owns the endpoint capability it was handed, so reclaiming the
// slot closes it — otherwise a process that subscribes and dies costs a
// handle-table slot that never comes back.
if (!alive) {
_ = ipc.close(sub.endpoint);
sub.* = .{};
}
}
}
@@ -84,10 +91,13 @@ fn broadcast(event: input_protocol.InputEvent) void {
}
}
/// Handle one request. `got` carries the sender badge (a task id) and, for subscribe, the
/// subscriber's endpoint capability in `got.cap`. Writes a `Reply` into `out` and returns
/// its length.
fn handle(message: []const u8, got: ipc.Received, out: []u8) usize {
/// Handle one request. `got` carries the sender badge (a task id); `arrived` carries the
/// capability the request came with, under the same ownership rule the service harness
/// states (`ipc.Arrival`): **it belongs to the turn, and only a handler that means to keep
/// it says `take`.** Everything else here — a short message, a `publish`, a subscribe that
/// finds the table full — simply returns, and the loop closes what arrived. Writes a
/// `Reply` into `out` and returns its length.
fn handle(message: []const u8, got: ipc.Received, out: []u8, arrived: *ipc.Arrival) usize {
const reply = struct {
fn write(buffer: []u8, status: i32) usize {
const header = input_protocol.Reply{ .status = status };
@@ -101,11 +111,12 @@ fn handle(message: []const u8, got: ipc.Received, out: []u8) usize {
switch (@as(input_protocol.Operation, @enumFromInt(request.operation))) {
.subscribe => {
const endpoint = got.cap orelse return reply.write(out, -1); // no endpoint passed
const endpoint = arrived.peek() orelse return reply.write(out, -1); // no endpoint passed
// A zero mask means "everything" (a subscriber that named no class still wants input).
const mask = if (request.device_mask == 0) input_protocol.device_all else request.device_mask;
pruneDeadSubscribers();
if (!addSubscriber(endpoint, @intCast(got.badge), mask)) return reply.write(out, -1); // table full
_ = arrived.take(); // claimed: the subscriber table holds it until that task dies
return reply.write(out, 0);
},
.publish => {
@@ -120,8 +131,8 @@ pub fn main() void {
_ = logging.write("/system/services/input: no endpoint\n");
return;
};
if (!ipc.register(.input, endpoint)) {
_ = logging.write("/system/services/input: register failed\n");
if (!channel.bindPatiently("input", endpoint)) {
_ = logging.write("/system/services/input: could not bind /protocol/input\n");
return;
}
_ = logging.write("/system/services/input: ready\n");
@@ -131,12 +142,21 @@ pub fn main() void {
var receive: [input_protocol.request_size]u8 = undefined;
while (true) {
const got = ipc.replyWait(endpoint, reply_buffer[0..reply_len], &receive, null);
// Whatever capability came with this turn is the turn's, and the turn closes it
// unless `handle` claims it (`ipc.Arrival`). The kernel installs a sent capability
// whatever the message's length or kind, so this covers the notification
// `continue` and every refusal inside `handle` — otherwise about thirty-two
// capability-carrying calls, which need no authorization at all, exhaust this
// service's handle table and no further subscribe can ever land.
var arrived: ipc.Arrival = .{ .handle = got.cap };
defer arrived.release();
// Only synchronous client requests (subscribe/publish) arrive here; nothing sends
// this service asynchronous messages, so a notification wake would be spurious.
if (got.isNotification()) {
reply_len = 0;
continue;
}
reply_len = handle(receive[0..got.len], got, &reply_buffer);
reply_len = handle(receive[0..got.len], got, &reply_buffer, &arrived);
}
}
+2 -2
View File
@@ -99,11 +99,11 @@ fn initialise(harness_endpoint: ipc.Handle) bool {
}
/// The logger serves no protocol; the ping is answered by the harness.
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Handle) usize {
fn onMessage(message: []const u8, reply: []u8, sender: u32, arrived: *ipc.Arrival) usize {
_ = message;
_ = reply;
_ = sender;
_ = capability;
_ = arrived; // nothing here takes a capability: the harness closes what arrives
return 0;
}