diff --git a/build.zig b/build.zig index 493ce99..aac809e 100644 --- a/build.zig +++ b/build.zig @@ -339,6 +339,7 @@ pub fn build(b: *std.Build) void { "thread-test", // the multi-threaded fixture (its package sets .threaded) "user-memory-test", // aims deliberately bad user pointers at the checked copy layer "protocol-registry-test", // drives the registrar: ungranted bind, collision, restart + "protocol-denied-test", // restriction stage one: an ungranted open answers as absence }) |fixture| { const package = b.lazyDependency(fixture, .{}) orelse @panic("a test fixture package is missing under test/system/services"); diff --git a/build.zig.zon b/build.zig.zon index 5c01bad..3313887 100644 --- a/build.zig.zon +++ b/build.zig.zon @@ -76,6 +76,7 @@ .@"thread-test" = .{ .path = "test/system/services/thread-test", .lazy = true }, .@"user-memory-test" = .{ .path = "test/system/services/user-memory-test", .lazy = true }, .@"protocol-registry-test" = .{ .path = "test/system/services/protocol-registry-test", .lazy = true }, + .@"protocol-denied-test" = .{ .path = "test/system/services/protocol-denied-test", .lazy = true }, // See `zig fetch --save ` for a command-line interface for adding dependencies. //.example = .{ // // When updating this field to a new URL, be sure to delete the corresponding diff --git a/docs/os-development/protocol-namespace.md b/docs/os-development/protocol-namespace.md index ff3ed92..f7e0c02 100644 --- a/docs/os-development/protocol-namespace.md +++ b/docs/os-development/protocol-namespace.md @@ -1,7 +1,9 @@ # The protocol namespace -*Design, agreed 2026-07-31. Supersedes the `ServiceId` registry. Not yet implemented — -the migration plan at the end is the work list.* +*Design, agreed 2026-07-31. Supersedes the `ServiceId` registry. P1–P3 of the +migration plan at the end have landed (the envelope, the registry and the +`ServiceId` flag-day, and restriction stage one); P4 and P5 are the remaining +work list.* How a program finds, connects to, and is restricted from the things it talks to. Three ideas, kept deliberately separate: diff --git a/docs/security-track-plan.md b/docs/security-track-plan.md index 98b7c15..38fbec6 100644 --- a/docs/security-track-plan.md +++ b/docs/security-track-plan.md @@ -31,13 +31,39 @@ the logging/USB-lifecycle track). ## Status +**Live state — updated on `main` after every phase, so this file read from a +plain `main` checkout always tells the truth about where the work is.** + +| | | +|---|---| +| Working on | **P4a** — protocol rebase onto `envelope.Define` (next) | +| Branch carrying it | `feat/security-group-2` (pushed to origin) | +| On `main` | Phase 0, PM, H1, P1, P2, P3 (group 2 merged) | +| Awaiting merge | nothing — group 2 is on `main` | +| Suite | 109 cases, all passing | +| Last updated | 2026-08-01 | + +A checkbox below means the phase met its definition of green and was +committed — on the branch named above, which reaches `main` at the next +group boundary. + - [x] **Phase 0** — baseline: suite green on `main` (106/106, 2026-07-31; `zig build` + `zig build test` clean at 9a32380), plan committed - [x] **PM** — path-migration flag-day (`/etc`→`/system/configuration`, `/var/log`→`/system/logs`, `/mnt/usb`→`/volumes/usb`; vfs carve-out for the two writable `/system` subtrees, FAT's `/var` mount split in two; suite 106/106) - [x] **H1** — the `user-memory` module; nine stragglers converted; leaf U/S+W checks (plus physmap-coverage confirmation, so an `mmio_map`'d buffer cannot fault ring 0 — this also closes the same hazard on the IPC path; `fs_resolve`'s out-capacity bound made overflow-safe; suite 107/107) - [x] **merge** group 1 → main, push (f3bc23c, 2026-07-31) - [x] **P1** — envelope module + `Define`; vfs `NodeKind.protocol` + open-reply-capability; client `Channel` (mechanics only, nothing converted; suite unchanged at 107) - [x] **P2** — registry in init; `/protocol` reserved; ServiceId flag-day (11 binds, 17 lookups; `protocol.csv` grants, chain-attested identity, dead-owner rebind; the kernel's endpoint-death sweep generalized off the retired registry; suite 108/108). Three adversarial review rounds closed six defects a green suite had missed: a forged power event could shut the machine down; the ping path leaked a capability per call, first in init and then in the shared harness; supervisor attestation by name was defeated by a laundering deputy; and the kernel let any handle-holder bind signals, timers, exits and IRQs to an endpoint it did not own. -- [ ] **P3** — open grants: `protocol.csv` enforcement, denial test +- [x] **P3** — open grants: `protocol.csv` enforcement, denial test. `onOpen` + consults the manifest with the same chain-attested identity a bind uses, and a + refused caller gets the *same* answer as one naming a contract nobody bound — + `-ENOENT`, no capability, the same reply bytes, no log line, and both questions + asked on every open so there is nothing to time. Twenty-seven `open` rows cover + the whole live client set. One wrinkle the plan had not foreseen: the driver + tree is three deep (device manager → PS/2 bus → keyboard/mouse) and attestation + is one hop, so a legitimate grandchild read exactly like a laundering deputy; + the manifest gained a third permission, `supervise`, which names an authorized + supervising task per contract and is deliberately **open-only**, leaving P2's + bind attestation and every refusal it makes untouched (suite 109/109) - [ ] **merge** group 2 → main, push - [ ] **P4a** — clean protocols rebased onto `Define` (vfs, block, display, scanout, input) - [ ] **P4b** — misfit protocols rebased (device-manager, power, usb-transfer) @@ -103,6 +129,19 @@ that implements it. and both stamped names satisfy the row while the chain is entirely the attacker's. Walking to the root of the chain does not fix it either, since the laundered chain still roots at the real PID 1.)* + *(P3 amendment: a third permission, `supervise`, joins `bind|open`. One-hop + attestation cannot express the one three-deep chain in the tree — the device + manager starts the PS/2 bus, and the bus starts the keyboard and mouse + drivers — and nothing structural tells that chain apart from the laundering + deputy, since both are a granted binary spawned by a granted binary. Only + policy can: a `supervise` row names the authorized supervising task the way + every other row names a claimant (binary, its own supervisor, the contract it + concerns), and an `open` row may then name that task in its supervisor + column. The delegate is itself attested the ordinary strict way, so the chain + still anchors in init or the kernel one hop above it and the recursion stops + there. It is **open-only** on purpose — a delegate may vouch for what its + children *reach*, never for what they *claim* — so the bind path is + byte-for-byte P2's and the laundering-deputy refusal is untouched.)* 4. **Test fixtures bind under `/protocol/test/...`**, granted to any binary whose path starts `/test/` — the subtree-scoping rule from the design doc, dogfooded. `shared_memory_test` (the borrowed-ServiceId @@ -297,6 +336,34 @@ Every existing scenario doubles as conversion proof. Suite 108. the first succeeds and the second fails identically to not-found. Suite 109. +*Landed. Four things the plan did not foresee, recorded because P4 and P5 +inherit them:* + +- *`supervise` — decision 3's amendment. The PS/2 keyboard and mouse drivers + are started by the PS/2 bus driver, which the device manager started: the + tree's one three-deep chain, and one hop deeper than attestation reaches. + Nothing structural separates it from the laundering deputy, so the manifest + says which delegate is authorized, per contract. Open-only, so P2's bind + attestation is unchanged.* +- *Indistinguishability is a claim about work, not only about bytes. `onOpen` + refreshes the process table, identifies the caller, scans the grants and + scans the bindings on **every** open and forms one verdict at the end; and + it logs nothing on any branch, because `klog_read` is ungated (a line + written on one branch is a line the refused caller can read) and a serial + line is milliseconds it could time. The operator's diagnosis is the pair the + namespace publishes anyway: `readdir /protocol` for what is bound, the + manifest for who may reach it.* +- *The fixture is `protocol-denied-test`, and its scenario boots the **input + service** so the forbidden name is genuinely bound — the fixture reads the + namespace listing to prove it before asking for it. Without a live provider + the case would be comparing two boot races and asserting nothing.* +- *Two channels stay open by design, named rather than papered over: `readdir` + over `/protocol` lists every bound name to anyone (deliberate — the tree is + diagnosable), and `/system/configuration/protocol.csv` is world-readable on + the `/system` mount. Stage one hides neither the set of contracts nor the + policy; what it removes is the **oracle in the reply**, which is what stage + two's parked and faked opens depend on.* + ## P4a — clean protocols onto Define vfs, block, display, scanout, input — the modules whose shapes map diff --git a/system/configuration/protocol.csv b/system/configuration/protocol.csv index dc80ef1..57fee6f 100644 --- a/system/configuration/protocol.csv +++ b/system/configuration/protocol.csv @@ -1,9 +1,17 @@ # /system/configuration/protocol.csv — who may claim, and who may reach, a name # under /protocol (docs/os-development/protocol-namespace.md). # -# init is the registrar: it serves /protocol, and every bind is checked against -# this file. It is AUTHORITATIVE — a name no row grants cannot be bound, and a -# missing file means nothing may be bound at all. +# init is the registrar: it serves /protocol, and every bind AND every open is +# checked against this file. It is AUTHORITATIVE — a name no row grants cannot be +# bound or reached, and a missing file means nothing may be bound or reached at +# all. +# +# A refused open is answered exactly as a name nobody bound is: -ENOENT, and no +# capability. That is not politeness, it is the model — the namespace IS the +# restriction, so what a process may not open simply does not exist for it, and +# there is no "permission denied" for it to tell apart from "no such contract". +# Which is why a missing row here shows up as a client retrying forever rather +# than as an error: check this file first, and `readdir /protocol` second. # # '#' starts a comment (whole-line or trailing); blank lines are ignored. # Whitespace around a field is trimmed, so columns may be padded. Four @@ -26,13 +34,21 @@ # confer; init's own path means this init; any other path means a # task init spawned itself or one the kernel spawned. Task ids are # monotonic and never reused, so an id cannot be borrowed. -# permission bind (provide this contract) | open (speak to it) +# permission bind (provide this contract) | open (speak to it) | +# supervise (stand in someone else's chain — see below) # name the contract, relative to /protocol # # A trailing '*' on any field matches any tail — how a subtree is granted whole. # -# NOTE: 'open' rows are parsed but not yet enforced; every open resolves today. -# The milestone that turns them into refusals is P3 (docs/security-track-plan.md). +# 'supervise' exists because attestation is one hop deep and the driver tree is +# three: the device manager starts the PS/2 bus, and the bus starts the keyboard +# and mouse drivers. Init never met the bus, so it cannot vouch for it by +# acquaintance — and it must not vouch for it by name, or the laundering deputy +# walks straight in. A 'supervise' row is the manifest saying it: a task running +# this binary, under this supervisor, may be the supervising task an 'open' row +# names, for this contract and no other. It grants the delegate nothing itself, +# and it is deliberately open-only — a delegate may vouch for what its children +# REACH, never for what they CLAIM, so every bind refusal is untouched by it. # # binary supervisor permission name @@ -68,3 +84,72 @@ # another fixture did. /test/*, kernel, bind, test/* /test/*, /test/*, bind, test/* + + +# ============================================================================ +# open — who may REACH each contract. One row per client per contract; a client +# with no row here simply finds the name absent, forever. +# ============================================================================ + +# --- init's own services ---------------------------------------------------- +# fat reaches the block device behind the volume it mounts; the compositor +# reaches the scanout its driver announced, its own endpoint (the mouse-listener +# thread opens /protocol/display like any other client — threads share no +# handles), and the input stream that moves the cursor. +/system/services/fat, /system/services/init, open, block +/system/services/display, /system/services/init, open, scanout +/system/services/display, /system/services/init, open, display +/system/services/display, /system/services/init, open, input +/system/services/display-demo, /system/services/init, open, display + +# --- the same two when the kernel test harness starts them directly --------- +/system/services/display, kernel, open, scanout +/system/services/display, kernel, open, display +/system/services/display, kernel, open, input +/system/services/display-demo, kernel, open, display + +# --- the drivers, and the discovery service --------------------------------- +# Every driver says hello to the manager that started it — one row for the whole +# subtree, because that handshake is what being a driver means. The rest are per +# driver: the storage and HID class drivers talk to their controller, the HID +# drivers publish into the input stream, and the GPU driver announces its scanout +# to the compositor. +/system/drivers/*, /system/services/device-manager, open, device-manager +/system/services/discovery, /system/services/device-manager, open, device-manager +/system/drivers/usb-storage, /system/services/device-manager, open, usb-transfer +/system/drivers/usb-hid-keyboard, /system/services/device-manager, open, usb-transfer +/system/drivers/usb-hid-keyboard, /system/services/device-manager, open, input +/system/drivers/usb-hid-mouse, /system/services/device-manager, open, usb-transfer +/system/drivers/usb-hid-mouse, /system/services/device-manager, open, input +/system/drivers/virtio-gpu, /system/services/device-manager, open, display + +# --- the PS/2 child drivers, one hop further down --------------------------- +# The keyboard and mouse drivers are started by the BUS driver, not by the +# device manager — the one three-deep chain in the tree. Init cannot vouch for +# the bus by acquaintance (it never started it), so the manifest authorizes it +# explicitly, and only for the two contracts its children need. +/system/drivers/ps2-bus, /system/services/device-manager, supervise, ps2-bus +/system/drivers/ps2-bus, /system/services/device-manager, supervise, input +/system/drivers/ps2-keyboard, /system/drivers/ps2-bus, open, ps2-bus +/system/drivers/ps2-keyboard, /system/drivers/ps2-bus, open, input +/system/drivers/ps2-mouse, /system/drivers/ps2-bus, open, ps2-bus +/system/drivers/ps2-mouse, /system/drivers/ps2-bus, open, input + +# --- test fixtures ---------------------------------------------------------- +# The /protocol/test subtree is theirs whole, the way the bind rows give it to +# them. Everything ABOVE that subtree is named one fixture at a time, so a +# fixture reaches a system contract only where a scenario needs it — which is +# what leaves the rest genuinely absent for the rest of them (the protocol-denied +# case asks for one it was not given, and is told there is no such thing). +/test/*, kernel, open, test/* +/test/*, /test/*, open, test/* +/test/*, kernel, open, device-manager +/test/*, /system/services/device-manager, open, device-manager +/test/system/services/input-source, kernel, open, input +/test/system/services/input-test, kernel, open, input + +# The laundering-deputy probe (test/system/services/protocol-registry-test) runs +# a grandchild whose supervisor is a fixture nobody authorized — that is the +# point of it, and its bind must stay refused. It still has to report the verdict +# it got, so its reporting channel, and nothing else, is delegated. +/test/*, /test/*, supervise, test/verdict diff --git a/system/kernel/process.zig b/system/kernel/process.zig index 22ca4af..4104808 100644 --- a/system/kernel/process.zig +++ b/system/kernel/process.zig @@ -332,7 +332,14 @@ fn systemIpcCall(state: *architecture.CpuState) void { /// ipc_reply_wait(handle, reply_ptr, reply_len, receive_ptr, receive_cap) -> receive_len, /// with the sender's badge in the secondary result register (rdx). fn systemIpcReplyWait(state: *architecture.CpuState) void { - const endpoint = ipc.resolveHandle(scheduler.current(), architecture.systemCallArg(state, 0)) orelse return failErr(state, ipc.EBADF); + const t = scheduler.current(); + const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 0)) orelse return failErr(state, ipc.EBADF); + // Receiving is the owner's privilege, the same rule the notification binders + // enforce: a sendable handle means only "you may talk to this". Anything + // else and a mount's backend endpoint — which `fs_resolve` installs in every + // caller's table — would let a stranger dequeue the requests meant for the + // server, taking the capabilities they carry and answering in its name. + if (!ipc.ownedBy(endpoint, t)) return failErr(state, ipc.EPERM); var badge: u64 = 0; var received_cap: u64 = abi.no_cap; const r = ipc.replyWait(endpoint, architecture.systemCallArg(state, 1), architecture.systemCallArg(state, 2), architecture.systemCallArg(state, 3), architecture.systemCallArg(state, 4), architecture.systemCallArg(state, 5), &badge, &received_cap); diff --git a/system/kernel/tests.zig b/system/kernel/tests.zig index 62fc4f9..9d91dd3 100644 --- a/system/kernel/tests.zig +++ b/system/kernel/tests.zig @@ -248,6 +248,8 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void { deviceManagerTest(boot_information); } else if (eql(case, "protocol-registry")) { protocolRegistryTest(boot_information); + } else if (eql(case, "protocol-denied")) { + protocolDeniedTest(boot_information); } else if (eql(case, "reboot")) { rebootTest(); } else { @@ -3843,6 +3845,59 @@ fn protocolRegistryTest(boot_information: *const BootInformation) void { result(); } +/// P3 — restriction stage one (docs/os-development/protocol-namespace.md). The +/// registrar now checks `open` against `/system/configuration/protocol.csv`, and +/// a caller with no grant is told exactly what a caller asking for a name nobody +/// bound is told. +/// +/// The scenario is the assertion's scaffolding: `/protocol` (init in its registry +/// role), the **input service** — which binds a real contract the fixture is +/// deliberately not granted — and the fixture. Without a live provider on the +/// forbidden name, "refused" and "not bound yet" would be the same observation +/// and the case would prove nothing; the fixture reads `/protocol`'s own listing +/// to confirm the name is there before it asks for it. +/// +/// The fixture's `protocol-denied: ok` is the marker; each step prints its own +/// line, which the harness's ordered regex reads. +fn protocolDeniedTest(boot_information: *const BootInformation) void { + log("DANOS-TEST-BEGIN: protocol-denied\n", .{}); + if (boot_information.initial_ramdisk_len == 0) { + check("bootloader handed over an initial_ramdisk", false); + result(); + return; + } + const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len]; + const rd = initial_ramdisk.Reader.init(image) orelse { + check("initial_ramdisk image is valid", false); + result(); + return; + }; + + process.setInitialRamdisk(image); + check("registry (init) spawned", spawnRegistry(rd)); + // The provider of the contract the fixture may NOT reach. It needs no + // hardware: it binds /protocol/input and waits for subscribers. + check("input service spawned", spawnNamed(rd, "input")); + check("protocol-denied-test spawned", spawnNamedWithArg(rd, "protocol-denied-test", "run")); + + const pass_marker = "protocol-denied: ok"; + const fail_marker = "protocol-denied: FAIL"; + scheduler.setPriority(1); + const deadline = architecture.millis() + 20000; + var saw_pass = false; + var saw_fail = false; + while (architecture.millis() < deadline and !saw_pass and !saw_fail) { + if (bufferHas(pass_marker)) saw_pass = true; + if (bufferHas(fail_marker)) saw_fail = true; + scheduler.yield(); + } + scheduler.setPriority(4); + + check("no step of the restriction contract failed", !saw_fail); + check("the fixture completed every restriction assertion", saw_pass); + result(); +} + fn deviceManagerTest(boot_information: *const BootInformation) void { log("DANOS-TEST-BEGIN: device-manager\n", .{}); if (boot_information.initial_ramdisk_len == 0) { diff --git a/system/services/init/init.zig b/system/services/init/init.zig index 91057b4..fcef455 100644 --- a/system/services/init/init.zig +++ b/system/services/init/init.zig @@ -35,6 +35,12 @@ //! stranger's bytes; the only identity on it is the task id the kernel stamps. //! Content never authorizes (`onPowerEvent`), and neither does a name — the //! registrar attests a caller's supervision by task id (`supervisorSatisfies`). +//! - **Absence is the enforcement.** P3: `open` consults the manifest with the +//! same attested identity a `bind` does, and a caller with no grant is told +//! exactly what a caller asking for a name nobody bound is told — `-ENOENT`, +//! and no capability (`onOpen`). Restriction stage one of +//! docs/os-development/protocol-namespace.md: what a process cannot open does +//! not exist for it, so there is no "permission denied" to distinguish. //! - **A capability that arrives is closed unless it is claimed** (`Arrival`), //! because PID 1's thirty-two handle slots are a resource an unauthenticated //! caller would otherwise be able to spend. @@ -153,7 +159,7 @@ const maximum_restarts = 3; /// init holds. const maximum_name = 64; const maximum_bindings = 16; -const maximum_grants = 48; +const maximum_grants = 64; /// One bound contract: the name, the provider's endpoint (a capability init /// holds and hands to whoever opens the name), and the provenance a diagnostic @@ -177,10 +183,26 @@ const Binding = struct { var bindings: [maximum_bindings]Binding = .{Binding{}} ** maximum_bindings; -/// What a grant row permits: claiming a name, or reaching one. `open` rows are -/// parsed and held but not yet enforced — every open resolves in P2, and P3 is -/// the milestone that turns these into refusals (docs/security-track-plan.md). -const Permission = enum { bind, open }; +/// What a grant row permits. +/// +/// - `bind` — claim the name, i.e. provide the contract. +/// - `open` — reach the name, i.e. speak the contract to whoever provides it. +/// - `supervise` — stand in a third task's supervision chain: a task running this +/// binary, under this supervisor, may be the supervising task named by an +/// `open` row for this contract. It grants the *delegate* nothing itself. +/// +/// `supervise` exists because attestation is deliberately one hop deep +/// (`supervisorSatisfies`): init vouches only for tasks it or the kernel started. +/// The driver tree is deeper than that — the device manager starts the PS/2 bus, +/// and the bus starts the keyboard and mouse drivers — so without a way to say +/// "this task is an authorized supervisor", a legitimate grandchild would be +/// indistinguishable from a laundering deputy. Naming the delegate in the +/// manifest is what tells them apart, and it is the same shape as every other +/// row: a binary, the supervisor it must have, and the contract it concerns. +/// Deliberately `open`-only — a delegate may vouch for what its children may +/// *reach*, never for what they may *claim* — so the bind path's attestation is +/// exactly what P2 shipped and every refusal it makes still holds. +const Permission = enum { bind, open, supervise }; /// One row of `/system/configuration/protocol.csv`. Every field may end in `*`, /// which matches any tail — the subtree scoping the design doc describes, and @@ -193,8 +215,9 @@ const Grant = struct { }; /// Roomier than init.csv's: this manifest carries a row per provider per spawn -/// path, its own format documentation, and grows again with the open grants. -var protocol_csv: [8192]u8 = undefined; +/// path, a row per client per contract it reaches, and its own format +/// documentation — which is most of the bytes, and is the point of the file. +var protocol_csv: [16384]u8 = undefined; var grants: [maximum_grants]Grant = .{Grant{}} ** maximum_grants; var grant_count: usize = 0; @@ -225,6 +248,8 @@ fn loadGrants() void { .bind else if (std.mem.eql(u8, permission, "open")) .open + else if (std.mem.eql(u8, permission, "supervise")) + .supervise else continue; // an unreadable row grants nothing rather than something wrong grants[grant_count] = .{ .binary = binary, .supervisor = supervisor, .permission = kind, .name = name }; @@ -393,6 +418,44 @@ fn granted(identity: Identity, permission: Permission, name: []const u8) bool { return false; } +/// Whether `identity` may reach `name` — `granted(.open, …)`, plus the one hop +/// `open` takes that `bind` does not (`Permission.supervise`). +/// +/// The hop is needed because the driver tree is three deep and attestation is +/// one: the PS/2 keyboard driver's supervising task is the PS/2 bus driver, +/// which the device manager started, which init started. Init cannot vouch for +/// the bus by acquaintance — it never met it — so the manifest says so instead, +/// and says it per contract: `ps2-bus` may be the supervisor named in an `open` +/// grant for `ps2-bus` and for `input`, and for nothing else. +fn mayOpen(identity: Identity, name: []const u8) bool { + if (granted(identity, .open, name)) return true; + return delegatedOpen(identity, name); +} + +/// The delegated `open`: the row's supervisor column names the caller's actual +/// supervising task by binary, that task is one init cannot vouch for directly, +/// and a `supervise` row authorizes it for exactly this contract. +/// +/// The delegate itself is attested the ordinary way (`granted` → strict +/// `supervisorSatisfies`), so the chain is still anchored one hop above it in +/// init or the kernel and the recursion stops there. Two hops of manifest, never +/// an unbounded walk — a laundering deputy is refused at the first hop nobody +/// wrote a row for. +fn delegatedOpen(identity: Identity, name: []const u8) bool { + if (identity.supervisor_task == 0) return false; // a kernel-spawned caller needs no delegate + if (identity.supervisor_vouched) return false; // already answered by `granted` above + const delegate = identify(identity.supervisor_task) orelse return false; + if (!granted(delegate, .supervise, name)) return false; + for (grants[0..grant_count]) |grant| { + if (grant.permission != .open) continue; + if (!matches(grant.binary, identity.binary)) continue; + if (!matches(grant.supervisor, identity.supervisor_binary)) continue; + if (!matches(grant.name, name)) continue; + return true; + } + return false; +} + fn findBinding(name: []const u8) ?*Binding { for (&bindings) |*binding| { if (binding.used and std.mem.eql(u8, binding.nameSlice(), name)) return binding; @@ -501,7 +564,7 @@ fn serveRegistry(request_bytes: []const u8, reply: []u8, sender: u32, arrived: * // Only `bind` claims a capability; one attached to anything else is closed by // the turn's `defer` in the loop, along with the ones sent to a request that // was too short to name a verb at all. - if (operation == @intFromEnum(vfs_protocol.Operation.open)) return onOpen(reply, payload); + if (operation == @intFromEnum(vfs_protocol.Operation.open)) return onOpen(reply, sender, payload); if (operation == @intFromEnum(vfs_protocol.Operation.readdir)) return onReaddir(reply, cursor); // Everything else a filesystem answers is meaningless here: `/protocol` holds // contracts, not bytes. @@ -576,12 +639,42 @@ fn onBind(sender: u32, raw_name: []const u8, arrived: *Arrival) i32 { } /// `open(name)` -> the provider's endpoint, delivered as the reply's capability. -/// A name nothing has bound is `-ENOENT`; in P3 an ungranted one becomes the same -/// answer, because absence and refusal are deliberately indistinguishable. -fn onOpen(reply: []u8, raw_name: []const u8) usize { +/// +/// **A refusal and an absence are the same answer, and that is the whole point.** +/// The namespace is the restriction (docs/os-development/protocol-namespace.md): +/// what a process may open is what exists for it, so "you may not have this" and +/// "there is no such thing" collapse into one reply — `-ENOENT`, no payload, no +/// capability. A caller therefore has no oracle: it cannot use `open` to learn +/// that a contract it lacks is bound, and — the reason this matters beyond +/// tidiness — stage two's supervisor can refuse, stall for a human, or substitute +/// a fake without the child being able to tell which happened. +/// +/// Indistinguishable is a claim about *work done*, not only about the bytes, so +/// both questions are asked on every open whatever the first one answers: the +/// process table is refreshed, the caller identified, the grants scanned and the +/// bindings scanned, and only then is the single verdict formed. Nothing here +/// logs, either — `klog_read` is ungated (system/kernel/process.zig), so a line +/// written on one branch is a line the refused caller can read, and a serial line +/// costs milliseconds it could time. The operator's diagnosis is the pair the +/// namespace already publishes on purpose: `readdir` over `/protocol` says what is +/// bound, `/system/configuration/protocol.csv` says who may reach it, and the +/// client's own retry loop says which one it wanted. +/// +/// (Not constant-time in the cryptographic sense, and not claimed to be: the two +/// scans stop at the row they match, and the optimiser is free to sink a pure +/// table walk past a branch that discards it. What is removed is the difference a +/// caller could actually measure or read — a syscall on one branch and not the +/// other, a line in a world-readable log ring, or a serial write costing +/// milliseconds.) +fn onOpen(reply: []u8, sender: u32, raw_name: []const u8) usize { const name = contractName(raw_name) orelse return answer(reply, -envelope.ENOENT, 0, 0); - const binding = findBinding(name) orelse return answer(reply, -envelope.ENOENT, 0, 0); - pending_capability = binding.endpoint; + refreshProcessTable(); + const identity = identify(sender); + const permitted = if (identity) |who| mayOpen(who, name) else false; + const binding = findBinding(name); + if (!permitted) return answer(reply, -envelope.ENOENT, 0, 0); + const found = binding orelse return answer(reply, -envelope.ENOENT, 0, 0); + pending_capability = found.endpoint; return answer(reply, 0, 0, 0); } diff --git a/test/qemu_test.py b/test/qemu_test.py index d7817d0..ed37da5 100644 --- a/test/qemu_test.py +++ b/test/qemu_test.py @@ -869,10 +869,26 @@ CASES = [ r"(?=.*protocol-registry: restarted provider reached)" r"(?=.*protocol-registry: laundering deputy refused)" r"(?=.*protocol-registry: foreign signal binding refused)" - r"(?=.*protocol-registry: foreign timer and exit binding refused)" + r"(?=.*protocol-registry: foreign timer and exit binding refused)(?=.*protocol-registry: foreign receive refused)" r"(?=.*protocol-registry: capability-carrying pings did not exhaust the harness)" r"(?=.*DANOS-TEST-RESULT: PASS)", "fail": r"DANOS-TEST-RESULT: FAIL|protocol-registry: FAIL"}, + # Restriction stage one (docs/os-development/protocol-namespace.md): the + # registrar checks `open` against /system/configuration/protocol.csv, and a + # caller with no grant gets the same answer as a caller naming a contract + # nobody bound. The scenario boots /protocol plus the input service, so the + # forbidden name is genuinely BOUND — the fixture reads the namespace listing + # to prove it — and then compares the refusal with an unbound name field by + # field: status, node, payload length, the whole reply packet, and the + # presence of a capability. All three failure shapes (refused-and-bound, + # granted-and-unbound, neither) must collapse into one answer. + {"name": "protocol-denied", + "expect": r"(?s)(?=.*protocol-denied: granted open succeeded)" + r"(?=.*protocol-denied: ungranted open refused as absent)" + r"(?=.*protocol-denied: refusal is indistinguishable from absence)" + r"(?=.*protocol-denied: ok)" + r"(?=.*DANOS-TEST-RESULT: PASS)", + "fail": r"DANOS-TEST-RESULT: FAIL|protocol-denied: FAIL"}, # Device manager: a ring-3 service enumerates /system/devices, matches the PCI host # bridge to pci-bus, and spawns it — end-to-end proof of discover -> match -> spawn # -> driver-up (the spawned pci-bus logs " functions found"). diff --git a/test/system/services/protocol-denied-test/build.zig b/test/system/services/protocol-denied-test/build.zig new file mode 100644 index 0000000..472938a --- /dev/null +++ b/test/system/services/protocol-denied-test/build.zig @@ -0,0 +1,15 @@ +//! The protocol-denied-test fixture as a binary package (docs/build-packages-plan.md): +//! this file names the binary and EXACTLY the modules its source imports — +//! build-support resolves each name from the domains this zon declares. + +const std = @import("std"); +const build_support = @import("build-support"); + +pub fn build(b: *std.Build) void { + const exe = build_support.userBinary(b, .{ + .name = "protocol-denied-test", + .root_source_file = b.path("protocol-denied-test.zig"), + .imports = &.{ "channel", "envelope", "file-system", "ipc", "logging", "process", "time", "vfs-protocol" }, + }); + b.installArtifact(exe); +} diff --git a/test/system/services/protocol-denied-test/build.zig.zon b/test/system/services/protocol-denied-test/build.zig.zon new file mode 100644 index 0000000..7d2102b --- /dev/null +++ b/test/system/services/protocol-denied-test/build.zig.zon @@ -0,0 +1,19 @@ +.{ + .name = .protocol_denied_test, + .version = "0.0.0", + .fingerprint = 0xab37a6fa7116698f, // Changing this has security and trust implications. + .minimum_zig_version = "0.16.0", + .dependencies = .{ + // build-support supplies the shared recipe; kernel is implicit in + // every binary (the root shim + link script live there). The rest + // are exactly the homes of this binary's declared imports. + .@"build-support" = .{ .path = "../../../../build-support" }, + .kernel = .{ .path = "../../../../library/kernel" }, + // envelope: the errno the registrar answers a refused open with; + // vfs-protocol: the request and reply this fixture compares byte for + // byte, which is why it speaks the wire itself instead of using the + // Channel client. + .protocol = .{ .path = "../../../../library/protocol" }, + }, + .paths = .{""}, +} diff --git a/test/system/services/protocol-denied-test/protocol-denied-test.zig b/test/system/services/protocol-denied-test/protocol-denied-test.zig new file mode 100644 index 0000000..7e0942b --- /dev/null +++ b/test/system/services/protocol-denied-test/protocol-denied-test.zig @@ -0,0 +1,269 @@ +//! protocol-denied-test — restriction stage one's own test fixture +//! (docs/os-development/protocol-namespace.md, "Restriction: per-process +//! namespaces, not ACLs"). One binary, one role, driven by the +//! `protocol-denied` kernel case: +//! +//! - `protocol-denied-test run` — the driver, and the assertions: +//! 1. a contract this binary IS granted opens: the reply says success and +//! carries a capability — the channel itself. The control comes first +//! and is repeated last, because every refusal below would read exactly +//! the same against a registrar that had simply stopped opening things; +//! 2. a contract this binary is NOT granted is refused — and `/protocol`'s +//! own listing is read first to prove the name is genuinely BOUND, so +//! the refusal is a policy decision and not an accident of boot order; +//! 3. the refusal is **indistinguishable from a name that does not exist**. +//! The fixture asks for a name nothing ever bound and compares the two +//! answers field by field — status, node, payload length, the whole +//! reply packet byte for byte, and the presence of a capability. That +//! collapse is the model, not a nicety: "permission denied" and "not +//! found" are one answer, so an open can never be used as an oracle for +//! what exists outside a process's view, and stage two's supervisor can +//! refuse, park for a human, or substitute a fake with the child unable +//! to tell which happened; +//! 4. the third shape — neither granted nor bound — answers identically +//! too, so all three collapse into one rather than two. +//! +//! **What this fixture deliberately does not claim.** Two channels are outside +//! what a ring-3 client can honestly assert: +//! +//! - *The registrar's log.* `klog_read` is ungated, so a line written on one +//! branch and not the other would be readable here — but the ring carries +//! every task's output, so searching it for a contract name proves nothing +//! either way (the input service prints "input:" lines of its own). The +//! guarantee is made at the source instead: `onOpen` in +//! system/services/init/init.zig writes nothing on any branch, and says why. +//! - *Timing.* The difference worth measuring — a syscall or a serial write on +//! one branch — is milliseconds, but this fixture shares four cores and a +//! serial line with a booting system, so a measurement here would be noise +//! wearing an assertion's clothes. `onOpen` asks both questions on every +//! open, whatever the first one answers; that is the claim, and it is a claim +//! about the code, checked by reading it. +//! +//! Prints `protocol-denied: ok` on success, or a `protocol-denied: FAIL` line +//! naming the step. Spawned bare (the initial-ramdisk sweep starts every bundled +//! binary), it exits silently so it cannot derange other tests. + +const std = @import("std"); +const channel = @import("channel"); +const envelope = @import("envelope"); +const file_system = @import("file-system"); +const ipc = @import("ipc"); +const logging = @import("logging"); +const process = @import("process"); +const time = @import("time"); +const vfs_protocol = @import("vfs-protocol"); + +/// The contract this fixture provides and then reaches — under `/protocol/test`, +/// the subtree every `/test/` binary is granted. Binding it ourselves keeps the +/// granted case to one process: the registry does not know or care that the +/// provider on the other end of the channel is us. +const granted_contract = "test/denied-probe"; + +/// A contract this fixture is NOT granted and that the case makes sure IS bound: +/// the input service claims it, and only `input-source` and `input-test` are +/// named against it in /system/configuration/protocol.csv. +const forbidden_contract = "input"; + +/// A contract this fixture IS granted (the `test/*` subtree) and that nothing +/// ever binds. The comparison partner: refusal must look like this. +const absent_contract = "test/never-bound"; + +/// Neither granted nor bound. The third shape, so the collapse is into one +/// answer rather than two. +const forbidden_and_absent_contract = "display"; + +fn fail(step: []const u8) noreturn { + _ = logging.write("protocol-denied: FAIL "); + _ = logging.write(step); + _ = logging.write("\n"); + process.exit(1); +} + +// --- talking to the registrar directly -------------------------------------- +// +// `channel.openEndpoint` folds every failure into null, which is exactly right +// for a client and useless here: the whole assertion is about the *shape* of the +// answer, so this fixture speaks the vfs protocol to the registry itself and +// keeps every byte that came back. + +/// One `open` answer, kept whole. +const Answer = struct { + /// Bytes the registrar replied with — the reply packet's length is itself a + /// channel, so it is compared like any other field. + length: usize = 0, + packet: [vfs_protocol.message_maximum]u8 = .{0} ** vfs_protocol.message_maximum, + /// Whether a capability rode the reply. The one field that actually matters + /// to a client: the capability IS the channel. + capability: bool = false, + /// The reply header, decoded — compared field by field as well as byte for + /// byte, so a failure says *which* field diverged. + reply: vfs_protocol.Reply = .{ .status = 0, .node = 0, .len = 0 }, + + fn bytes(self: *const Answer) []const u8 { + return self.packet[0..self.length]; + } +}; + +/// The registry's endpoint, obtained the way every process obtains it: resolve +/// `/protocol`. The handle is the kernel's, shared with every other user of the +/// mount, so it is never ours to close. +fn registryEndpoint() ?ipc.Handle { + // Patiently, for the same reason `bindPatiently` is patient: the kernel test + // harness starts the registrar and this fixture together, so a first resolve + // can land in the window before init has mounted `/protocol` at all. An + // absent mount is a boot race and worth waiting out; a registrar that + // answers has decided, and that answer is what the assertions below weigh. + var attempt: u32 = 0; + while (attempt < resolve_attempts) : (attempt += 1) { + var relative: [channel.path_maximum]u8 = undefined; + if (file_system.fsResolve(channel.root, 0, &relative)) |route| switch (route) { + .kernel => return null, // a kernel route means something other than the registry owns the name + .backend => |backend| return backend.handle, + }; + time.sleepMillis(resolve_retry_ms); + } + return null; +} + +/// The cadence `channel.bindPatiently` uses, for the same window. +const resolve_attempts: u32 = 200; +const resolve_retry_ms: u64 = 20; + +/// One vfs-protocol request at the registry: the fixed header, then the contract +/// name inline. Names go bare (`input`, not `/input`) — the registrar normalises +/// both, and bare is what `bind` sends. +fn transact(registry: ipc.Handle, operation: vfs_protocol.Operation, name: []const u8, cursor: u64) ?Answer { + var request: [vfs_protocol.message_maximum]u8 = undefined; + if (vfs_protocol.request_size + name.len > request.len) return null; + const header = vfs_protocol.Request{ + .operation = operation, + .node = 0, + .offset = cursor, + .len = @intCast(name.len), + .flags = 0, + }; + @memcpy(request[0..vfs_protocol.request_size], std.mem.asBytes(&header)); + @memcpy(request[vfs_protocol.request_size..][0..name.len], name); + + var answer: Answer = .{}; + const got = ipc.callCap( + registry, + request[0 .. vfs_protocol.request_size + name.len], + &answer.packet, + null, + ) catch return null; + if (got.len < vfs_protocol.reply_size) return null; + answer.length = got.len; + answer.capability = got.cap != null; + answer.reply = std.mem.bytesToValue(vfs_protocol.Reply, answer.packet[0..vfs_protocol.reply_size]); + // A capability we did not ask to keep is a handle slot spent; the assertions + // below only care that one arrived. + if (got.cap) |handle| _ = ipc.close(handle); + return answer; +} + +/// `open(name)`, kept whole. Null only if the registry could not be reached at +/// all — a registrar that answered has decided, and its decision is the subject. +fn openContract(registry: ipc.Handle, name: []const u8) Answer { + return transact(registry, .open, name, 0) orelse fail("the registry stopped answering"); +} + +/// Whether `/protocol` currently lists `name`. The namespace is browsable on +/// purpose (docs/os-development/protocol-namespace.md: `readdir` lists protocol +/// nodes like any others, so the tree stays diagnosable), and that is what lets +/// this fixture prove a refused name is really there — without it, "refused" +/// and "not bound yet" would be the same observation and the test would assert +/// nothing. +fn listed(registry: ipc.Handle, name: []const u8) bool { + var cursor: u64 = 0; + while (cursor < 64) : (cursor += 1) { + const answer = transact(registry, .readdir, "", cursor) orelse return false; + if (answer.reply.status != 0 or answer.reply.len == 0) return false; // end of directory + const payload = answer.packet[vfs_protocol.reply_size..answer.length]; + if (payload.len < vfs_protocol.directory_entry_size) return false; + const entry = std.mem.bytesToValue(vfs_protocol.DirectoryEntry, payload[0..vfs_protocol.directory_entry_size]); + const text = payload[vfs_protocol.directory_entry_size..]; + const length = @min(@as(usize, entry.name_len), text.len); + if (std.mem.eql(u8, text[0..length], name)) return true; + } + return false; +} + +/// Wait until `/protocol` lists `name` — the providers this case needs come up +/// alongside the fixture, and racing them would make the assertions meaningless +/// rather than merely flaky. +fn awaitListed(registry: ipc.Handle, name: []const u8) void { + var attempts: u32 = 0; + while (attempts < 400) : (attempts += 1) { + if (listed(registry, name)) return; + time.sleepMillis(20); + } + _ = logging.write("protocol-denied: FAIL /protocol never listed "); + _ = logging.write(name); + _ = logging.write("\n"); + process.exit(1); +} + +/// Every caller-visible field of two answers, compared. `step` names the pair so +/// a failure says which comparison broke and in which field. +fn expectIdentical(step: []const u8, refused: Answer, absent: Answer) void { + if (refused.reply.status != absent.reply.status) fail(step); // the errno + if (refused.reply.node != absent.reply.node) fail(step); // the node id an open would return + if (refused.reply.len != absent.reply.len) fail(step); // payload bytes promised + if (refused.length != absent.length) fail(step); // reply packet length + if (refused.capability != absent.capability) fail(step); // the channel itself + if (!std.mem.eql(u8, refused.bytes(), absent.bytes())) fail(step); // and every byte of it +} + +fn run() void { + const registry = registryEndpoint() orelse fail("resolve /protocol"); + + // Provide the granted contract ourselves. `bindPatiently` waits out a + // registry that has not mounted `/protocol` yet, which is the one thing + // worth retrying — a registrar that answered has decided. + const provider = ipc.createIpcEndpoint() orelse fail("create the provider endpoint"); + if (!channel.bindPatiently(granted_contract, provider)) fail("binding a granted contract was refused"); + + // Both names must be bound before anything is asked of them, or the + // comparison below would be between two boot races. + awaitListed(registry, granted_contract); + awaitListed(registry, forbidden_contract); + + // 1. The control. A granted, bound contract opens: success, and the + // capability that IS the channel. + const allowed = openContract(registry, granted_contract); + if (allowed.reply.status != 0) fail("a granted open was refused"); + if (!allowed.capability) fail("a granted open carried no channel"); + _ = logging.write("protocol-denied: granted open succeeded\n"); + + // 2. The refusal. `input` is bound — the listing above proved it — and no + // manifest row names this binary against it. + const refused = openContract(registry, forbidden_contract); + if (refused.reply.status != -envelope.ENOENT) fail("an ungranted open did not answer -ENOENT"); + if (refused.capability) fail("an ungranted open carried a channel"); + _ = logging.write("protocol-denied: ungranted open refused as absent\n"); + + // 3. The point of the whole fixture. A name this binary IS granted and that + // nothing has ever bound, answered by the same registrar in the same + // breath — and every caller-visible field of the two answers is the same. + const absent = openContract(registry, absent_contract); + expectIdentical("a refused open differed from a nonexistent name", refused, absent); + + // 4. And the third shape, so the two reasons collapse into one answer rather + // than into two that happen to match: neither granted nor bound. + const neither = openContract(registry, forbidden_and_absent_contract); + expectIdentical("a refused-and-absent open differed from the others", refused, neither); + _ = logging.write("protocol-denied: refusal is indistinguishable from absence\n"); + + // 5. The control again, after the refusals: the registrar is still opening + // what it should, so what steps 2-4 saw was policy and not a registry + // that had wedged. + const again = openContract(registry, granted_contract); + if (again.reply.status != 0 or !again.capability) fail("the granted contract stopped opening"); + _ = logging.write("protocol-denied: ok\n"); +} + +pub fn main(startup: process.Init) void { + const role = startup.arguments.get(1) orelse return; // bare (ramdisk sweep): stay silent + if (std.mem.eql(u8, role, "run")) run(); +} diff --git a/test/system/services/protocol-registry-test/protocol-registry-test.zig b/test/system/services/protocol-registry-test/protocol-registry-test.zig index cfb01e6..14ede1e 100644 --- a/test/system/services/protocol-registry-test/protocol-registry-test.zig +++ b/test/system/services/protocol-registry-test/protocol-registry-test.zig @@ -399,6 +399,22 @@ fn run() void { if (!process.subscribeExits(ticker)) fail("subscribing an endpoint we created to exit events was refused"); _ = logging.write("protocol-registry: foreign timer and exit binding refused\n"); + // 9b. Receiving is the same privilege, and it was the one member of the family + // left unguarded. Every process holds a sendable handle to the registrar's + // mailbox — `fs_resolve` installs one for anyone who asks — and a stranger + // that could *dequeue* there would not merely evade the grants this fixture + // checks: it would take the provider endpoints that ride `bind` requests + // straight out of the queue, and answer other clients' opens in the + // registrar's name. Sending to it stays legal; receiving on it must not be. + // A refusal comes back as a negative errno in the length register, which + // is the whole point: the call returns instead of parking us on someone + // else's queue, where a success would have blocked until a request it was + // never ours to see arrived. + var stolen: [8]u8 = undefined; + const theft = ipc.replyWait(registry, stolen[0..0], &stolen, null); + if (theft.len <= ~@as(usize, 0) - 4095) fail("receiving on the registry's endpoint was allowed"); + _ = logging.write("protocol-registry: foreign receive refused\n"); + // 10. The handle-table storm again, aimed at a **harness-run service** this // time. Step 4 covers PID 1, which runs its own hand-written loop; every // other service in the system — the VFS, the display compositor, the device