init: /protocol replaces the ServiceId registry
A protocol is reached by name now, not by a compile-time integer. Init is PID 1 and already knows which binary it started, so init serves /protocol as a vfs backend: bind claims a contract with the provider's endpoint attached, open answers with that endpoint as the reply's capability, and readdir lists what is bound with the task and binary behind it. The kernel reserves the prefix — nothing may mount over it, under it, or unmount it — and ServiceId, ipc_register and ipc_lookup are gone, their syscall numbers left vacant. A bind is authorized by who the caller *is*: the kernel-stamped binary together with the supervising task's identity, matched against /system/configuration/protocol.csv. Identity, not spelling — spawn is ungated, so an attacker can run any bundled binary, and a name-only rule would have let it launder grants through an init of its own making. A name a live process holds is refused to everyone else; a dead one's is released. Three review rounds against a hostile ring-3 process found what 108 green tests could not, because the suite contains no attacker. Publishing init's supervision endpoint as the registry put PID 1's mailbox in every process's hands, where two forged bytes reached the shutdown path: privileged traffic is now believed only from the task that holds the contract it speaks for. A capability arriving on a request outlived every path that ignored it, one handle per call until the table was full — in init, and in the harness ten services share — so the arriving capability is owned by the turn and released unless a handler says otherwise. And the kernel let anyone holding an endpoint handle aim signals, timers, exit notices and interrupts at it: binding now requires having created it. Suite 108/108. The new protocol-registry case asserts eleven properties, each one an attack that must fail.
This commit is contained in:
@@ -40,16 +40,13 @@ const Task = scheduler.Task;
|
||||
pub const MESSAGE_MAXIMUM: usize = 256;
|
||||
|
||||
pub const maximum_handles = scheduler.ipc_maximum_handles;
|
||||
// The name registry is indexed directly by ServiceId, so this must exceed the
|
||||
// largest id (currently fat = 8). Sized with headroom for new services.
|
||||
pub const maximum_services = 16;
|
||||
|
||||
/// Errno-style failures, returned as `-value` in the system_call result register.
|
||||
pub const EBADF: i64 = 1; // bad handle
|
||||
pub const E2BIG: i64 = 2; // message exceeds MESSAGE_MAXIMUM
|
||||
pub const EFAULT: i64 = 3; // buffer unmapped / out of the user half
|
||||
pub const ENOENT: i64 = 4; // no such registered service
|
||||
pub const ENOSPC: i64 = 5; // handle table or registry full
|
||||
pub const ENOENT: i64 = 4; // no such name
|
||||
pub const ENOSPC: i64 = 5; // handle table full
|
||||
pub const ENOMEM: i64 = 6; // out of memory
|
||||
pub const EPEER: i64 = 7; // peer died before replying (its process exited or was killed)
|
||||
pub const ESRCH: i64 = 8; // no such process (process_kill of an unknown/dead id)
|
||||
@@ -90,12 +87,21 @@ const PostSlot = struct {
|
||||
const user_half_end: u64 = user_memory.user_half_end;
|
||||
|
||||
/// A rendezvous endpoint. Allocated from the kernel heap; referenced by handle
|
||||
/// (per process) and/or by a registry slot, counted by `refcount`.
|
||||
/// (per process) and by whoever a capability was passed to, counted by `refcount`.
|
||||
pub const Endpoint = struct {
|
||||
refcount: u32 = 1,
|
||||
/// Next in the list of every live endpoint. Endpoints are otherwise reachable
|
||||
/// only through the handle tables that name them, and the death path has to
|
||||
/// find a dying task's endpoints without one — see `live_endpoints`.
|
||||
next_live: ?*Endpoint = null,
|
||||
// The task that created it. When that task dies, the endpoint is marked `dead` so a caller
|
||||
// gets -EPEER instead of blocking forever on a service that will never reply again (V6).
|
||||
owner: u32 = 0,
|
||||
// The *process* that created it — `owner`'s leader, snapshotted at creation so the
|
||||
// answer survives the creating thread. `owner` alone cannot answer "is this mine?"
|
||||
// for a threaded service, and the question has to be answerable after that thread is
|
||||
// gone; see `ownedBy`.
|
||||
owner_leader: u32 = 0,
|
||||
dead: bool = false,
|
||||
// Callers blocked in `call`, awaiting receive, in FIFO order (threaded via
|
||||
// Task.next; each such task is .blocked and in no scheduler queue).
|
||||
@@ -115,29 +121,78 @@ pub const Endpoint = struct {
|
||||
post_tail: u16 = 0,
|
||||
};
|
||||
|
||||
/// Every live endpoint, singly linked through `next_live`. The list exists for
|
||||
/// exactly one purpose: the death path must mark a dying task's endpoints dead,
|
||||
/// and a handle table only answers the other question (which endpoints does this
|
||||
/// task *hold*). Mutated under the big kernel lock, like every other IPC global.
|
||||
var live_endpoints: ?*Endpoint = null;
|
||||
|
||||
pub fn createIpcEndpoint() ?*Endpoint {
|
||||
const creator = scheduler.current();
|
||||
const endpoint = heap.allocator().create(Endpoint) catch return null;
|
||||
endpoint.* = .{ .owner = scheduler.currentId() };
|
||||
endpoint.* = .{ .owner = creator.id, .owner_leader = creator.leader, .next_live = live_endpoints };
|
||||
live_endpoints = endpoint;
|
||||
return endpoint;
|
||||
}
|
||||
|
||||
/// A task is dying: kill the endpoints it registered as services. Mark each `dead` (so a later
|
||||
/// `call` returns -EPEER rather than blocking on a reply that will never come), wake anyone
|
||||
/// already parked sending to it with that error, and vacate its registry slot. Only *registered*
|
||||
/// endpoints are reachable from here; unregistered ones drop with the task's handle table. The
|
||||
/// caller holds the big kernel lock (this runs on the death path). See docs/display-v2.md (V6).
|
||||
/// Whether `t` may have the kernel post **notifications** — signals, timer
|
||||
/// landings, exit notices, interrupts — into `endpoint`: whether the endpoint is
|
||||
/// its process's own.
|
||||
///
|
||||
/// Holding a *handle* to an endpoint is not ownership of it. `fs_resolve`
|
||||
/// installs a mounted backend's capability in any caller's table
|
||||
/// (`installHandleDeduped`), and any capability may be passed along a call, so a
|
||||
/// sendable handle means only "you may talk to this". A kernel notification is
|
||||
/// different in kind: it makes the kernel speak *into* someone else's mailbox
|
||||
/// with a badge that receiver cannot distinguish from one it asked for — a
|
||||
/// genuine signal badge, a genuine timer landing. That is how a forged
|
||||
/// `terminate` reached PID 1's shutdown path: the attacker aimed **its own**
|
||||
/// signal delivery at init's endpoint with `signal_bind` and then signalled
|
||||
/// itself, and every bit the kernel stamped was authentic. Refusing the *bind*
|
||||
/// is the only place the distinction still exists.
|
||||
///
|
||||
/// Threads: ownership is the **process's**, not the task's, so any thread may
|
||||
/// bind an endpoint a sibling created — the same normalization `process_signal`
|
||||
/// and `process_kill` perform when they resolve a member to its leader. The
|
||||
/// creating task's own id is honoured too, which is what keeps kernel tasks
|
||||
/// (leader 0) from being treated as one process.
|
||||
pub fn ownedBy(endpoint: *const Endpoint, t: *const Task) bool {
|
||||
if (endpoint.owner == t.id) return true;
|
||||
return t.leader != 0 and endpoint.owner_leader == t.leader;
|
||||
}
|
||||
|
||||
/// Unlink a freed endpoint from the live list. O(n) in the number of live
|
||||
/// endpoints, which is tens.
|
||||
fn forgetEndpoint(endpoint: *Endpoint) void {
|
||||
var link = &live_endpoints;
|
||||
while (link.*) |current| {
|
||||
if (current == endpoint) {
|
||||
link.* = current.next_live;
|
||||
return;
|
||||
}
|
||||
link = ¤t.next_live;
|
||||
}
|
||||
}
|
||||
|
||||
/// A task is dying: kill every endpoint it created. Mark each `dead` (so a later
|
||||
/// `call` returns -EPEER rather than blocking on a reply that will never come) and wake
|
||||
/// anyone already parked sending to it with that error. This is what makes a provider's
|
||||
/// death visible to the clients holding its capability — the naming layer's restart
|
||||
/// story (a client re-resolves on -EPEER) rests on it, as does the VFS router's lazy
|
||||
/// unmount of a backend that died. The endpoint object itself lives until the last
|
||||
/// handle naming it drops. The caller holds the big kernel lock (this runs on the death
|
||||
/// path). See docs/display-v2.md (V6).
|
||||
pub fn killOwnedEndpointsLocked(task_id: u32) void {
|
||||
for (®istry) |*slot| {
|
||||
const endpoint = slot.* orelse continue;
|
||||
if (endpoint.owner != task_id) continue;
|
||||
var current = live_endpoints;
|
||||
while (current) |endpoint| {
|
||||
current = endpoint.next_live;
|
||||
if (endpoint.owner != task_id or endpoint.dead) continue;
|
||||
endpoint.dead = true;
|
||||
while (dequeueSender(endpoint)) |sender| {
|
||||
sender.ipc_status = -EPEER;
|
||||
sender.ipc_received_cap = abi.no_cap;
|
||||
scheduler.readyLocked(sender);
|
||||
}
|
||||
slot.* = null;
|
||||
dropRef(endpoint);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -147,6 +202,7 @@ pub fn dropRef(endpoint: *Endpoint) void {
|
||||
if (endpoint.refcount > 1) {
|
||||
endpoint.refcount -= 1;
|
||||
} else {
|
||||
forgetEndpoint(endpoint);
|
||||
heap.allocator().destroy(endpoint);
|
||||
}
|
||||
}
|
||||
@@ -633,22 +689,9 @@ fn dropEntry(entry: scheduler.HandleObject) void {
|
||||
}
|
||||
}
|
||||
|
||||
var registry: [maximum_services]?*Endpoint = .{null} ** maximum_services;
|
||||
|
||||
/// Publish `endpoint` under well-known `id` (takes a reference). Returns 0 or -errno.
|
||||
pub fn register(id: u32, endpoint: *Endpoint) i64 {
|
||||
if (id >= maximum_services) return -ENOENT;
|
||||
if (registry[id]) |old| dropRef(old);
|
||||
endpoint.refcount += 1;
|
||||
registry[id] = endpoint;
|
||||
return 0;
|
||||
}
|
||||
|
||||
/// Find the endpoint published under `id`, taking a reference for the caller to
|
||||
/// install in its handle table. Null if nothing is registered there.
|
||||
pub fn lookup(id: u32) ?*Endpoint {
|
||||
if (id >= maximum_services) return null;
|
||||
const endpoint = registry[id] orelse return null;
|
||||
endpoint.refcount += 1;
|
||||
return endpoint;
|
||||
}
|
||||
// The flat `ServiceId` registry lived here — a 16-slot table any process could
|
||||
// write, indexed by a compile-time enum. Naming is user-space's job now: init
|
||||
// serves `/protocol` and decides who may claim a name
|
||||
// (docs/os-development/protocol-namespace.md). The kernel keeps only what is
|
||||
// genuinely kernel work — moving capabilities and telling clients their provider
|
||||
// died (`killOwnedEndpointsLocked`).
|
||||
|
||||
Reference in New Issue
Block a user