Merge feat/power-events: ACPI events + system power (M21)
The QMP harness channel, the SCI + power button published from ring 3, Notify/GPE dispatch in the AML interpreter, and orderly shutdown — init's M17 stop cascade into a ring-3 S5 write. Proven by injecting a real ACPI power-button event; QEMU powers off through the whole chain. # Conflicts: # system/services/init/init.zig
This commit is contained in:
+323
-32
@@ -4,13 +4,11 @@
|
||||
//! grant, a broad irq window, the SCI), and runs the **shared AML module** in
|
||||
//! ring 3 — the same parser and interpreter the kernel uses.
|
||||
//!
|
||||
//! M20.2 (this increment): after parsing, walk the namespace and, for each
|
||||
//! present Device with a hardware id (`_HID`), evaluate its current resource
|
||||
//! settings (`_CRS`) through a ring-3 `Hal` (port I/O over the claimed node),
|
||||
//! register it under the acpi-tables node (its I/O ports and IRQs contained by
|
||||
//! the node's broad grants), and report it to the device manager with its
|
||||
//! EISA-decoded hid as identity. Matching those reports to drivers (ps2-bus)
|
||||
//! and retiring the kernel's own device build follow in M20.3.
|
||||
//! It also owns the **event side** (M21): it registers the domain-named `.power`
|
||||
//! service, binds the SCI (System Control Interrupt), and on a power-button
|
||||
//! fixed event publishes `power_button` to subscribers — and on init's request
|
||||
//! writes S5 to power the machine off. The device discovery (M20) and the event
|
||||
//! handling both run in one `runtime.service.run` loop.
|
||||
|
||||
const std = @import("std");
|
||||
const runtime = @import("runtime");
|
||||
@@ -18,6 +16,7 @@ const aml = @import("aml");
|
||||
const acpi_ids = @import("acpi-ids");
|
||||
const device = runtime.device;
|
||||
const protocol = runtime.device_manager_protocol;
|
||||
const power = runtime.power_protocol;
|
||||
/// AML opcode/prefix bytes by name (`zero_opcode`, `byte_prefix`, …) — so the `_HID`
|
||||
/// integer decode names the opcodes instead of bare 0x0A/0x0B/… (docs/coding-standards.md).
|
||||
const opcodes = aml.opcodes;
|
||||
@@ -31,6 +30,43 @@ fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
|
||||
// window — the Hal routes every port access through this one claim.
|
||||
var node_id: u64 = 0;
|
||||
var io_resource_index: u64 = 0;
|
||||
// The SCI's irq resource index on the node (the len-1 irq, distinct from the
|
||||
// broad [0,256) window), for irqBind / irqAck.
|
||||
var sci_resource_index: u64 = 0;
|
||||
var has_sci = false;
|
||||
|
||||
// PM1 event/control and GPE register ports, read from the FADT copy the kernel
|
||||
// publishes on the node (M21). Port 0 means absent.
|
||||
var pm1a_evt: u16 = 0;
|
||||
var pm1b_evt: u16 = 0;
|
||||
var pm1_evt_len: u8 = 0;
|
||||
var pm1a_cnt: u16 = 0;
|
||||
var pm1b_cnt: u16 = 0;
|
||||
var gpe0_blk: u16 = 0;
|
||||
var gpe0_len: u8 = 0;
|
||||
var gpe1_blk: u16 = 0;
|
||||
var gpe1_len: u8 = 0;
|
||||
var smi_cmd: u16 = 0;
|
||||
var acpi_enable_value: u8 = 0;
|
||||
var s5_slp_typ_a: u8 = 0;
|
||||
var s5_slp_typ_b: u8 = 0;
|
||||
var s5_valid = false;
|
||||
|
||||
// PM1 event-register bits (ACPI): PWRBTN in the status/enable word is bit 8;
|
||||
// the control word's SCI_EN is bit 0; SLP_EN is bit 13.
|
||||
const pwrbtn_bit: u16 = 1 << 8;
|
||||
const sci_en_bit: u32 = 1 << 0;
|
||||
const slp_en: u32 = 1 << 13;
|
||||
|
||||
// The `.power` subscribers: endpoints handed over as capabilities, each
|
||||
// receiving events as buffered messages. Dropped on a failed send. The
|
||||
// subscriber's task id is kept too — a shutdown request is honored only from a
|
||||
// subscriber (init subscribes; a stray process does not), the soft gate that
|
||||
// stands in for "only the system supervisor may power off" without hardcoding
|
||||
// a pid the kernel's idle tasks would have taken.
|
||||
const maximum_subscribers = 8;
|
||||
var subscribers: [maximum_subscribers]?runtime.ipc.Handle = .{null} ** maximum_subscribers;
|
||||
var subscriber_tasks: [maximum_subscribers]u32 = .{0} ** maximum_subscribers;
|
||||
|
||||
// Pass-1 registration record (see main): what pass 2 reports.
|
||||
const Registered = struct { hid: [8]u8 = .{0} ** 8, hid_len: usize = 0, device_id: u64 = 0, resource_count: u64 = 0 };
|
||||
@@ -85,22 +121,34 @@ pub fn main(init: runtime.process.Init) void {
|
||||
return;
|
||||
}
|
||||
|
||||
// Map each memory resource (an AML blob) and note the io_port resource.
|
||||
// Map the node's resources: the AML blobs (bytecode), the FADT (intact
|
||||
// "FACP" header — decision 3), the io_port grant, and the SCI irq.
|
||||
var blocks: [8][]const u8 = undefined;
|
||||
var block_count: usize = 0;
|
||||
var found_io = false;
|
||||
var fadt: ?[]const u8 = null;
|
||||
for (node.resources[0..@intCast(node.resource_count)], 0..) |resource, index| {
|
||||
if (resource.kind == @intFromEnum(device.ResourceKind.io_port) and !found_io) {
|
||||
io_resource_index = index;
|
||||
found_io = true;
|
||||
continue;
|
||||
}
|
||||
if (resource.kind == @intFromEnum(device.ResourceKind.irq) and resource.len == 1) {
|
||||
sci_resource_index = index;
|
||||
has_sci = true;
|
||||
continue;
|
||||
}
|
||||
if (resource.kind != @intFromEnum(device.ResourceKind.memory)) continue;
|
||||
const base = device.mmioMap(node_id, index) orelse continue;
|
||||
const pointer: [*]const u8 = @ptrFromInt(base);
|
||||
blocks[block_count] = pointer[0..@intCast(resource.len)];
|
||||
const bytes = pointer[0..@intCast(resource.len)];
|
||||
if (bytes.len >= 4 and std.mem.eql(u8, bytes[0..4], "FACP")) {
|
||||
fadt = bytes;
|
||||
continue;
|
||||
}
|
||||
if (block_count == blocks.len) continue;
|
||||
blocks[block_count] = bytes;
|
||||
block_count += 1;
|
||||
if (block_count == blocks.len) break;
|
||||
}
|
||||
if (block_count == 0) {
|
||||
_ = runtime.system.write("/system/services/acpi: no AML blobs on the node\n");
|
||||
@@ -124,29 +172,45 @@ pub fn main(init: runtime.process.Init) void {
|
||||
while (true) runtime.system.sleep(1000);
|
||||
}
|
||||
|
||||
// Register + report the present _HID devices (M20.2).
|
||||
var arena = std.heap.ArenaAllocator.init(runtime.allocator());
|
||||
var interpreter = aml.Interpreter.init(&namespace, .{
|
||||
// Register + report the present _HID devices (M20), then set up the power
|
||||
// event side (M21), then serve — all in one harness loop. The interpreter
|
||||
// and namespace outlive this frame (static), so the harness callbacks can
|
||||
// reach them.
|
||||
interpreter_arena = std.heap.ArenaAllocator.init(runtime.allocator());
|
||||
persistent_namespace = namespace;
|
||||
global_interpreter = aml.Interpreter.init(&persistent_namespace, .{
|
||||
.mapMmio = halMapMmio,
|
||||
.pioRead = halPioRead,
|
||||
.pioWrite = halPioWrite,
|
||||
}, arena.allocator());
|
||||
}, interpreter_arena.allocator());
|
||||
|
||||
// Pass 1: register every present _HID device under acpi-tables, remembering
|
||||
// each (hid, device id). Pass 2: report them all. Registering before any
|
||||
// report reaches the manager means a driver it spawns on the first report
|
||||
// already sees the whole set (no keyboard-before-mouse race for ps2-bus).
|
||||
readFadt(fadt);
|
||||
s5_valid = readSleepS5(&persistent_namespace);
|
||||
|
||||
runtime.service.run(power.message_maximum, .{
|
||||
.service = .power,
|
||||
.init = onInit,
|
||||
.on_message = onMessage,
|
||||
.on_notification = onNotification,
|
||||
});
|
||||
}
|
||||
|
||||
// Static so the harness callbacks (which run after main's stack frame is gone)
|
||||
// can reach the namespace and interpreter.
|
||||
var persistent_namespace: aml.Namespace = undefined;
|
||||
var global_interpreter: aml.Interpreter = undefined;
|
||||
var interpreter_arena: std.heap.ArenaAllocator = undefined;
|
||||
|
||||
/// Startup under the harness: register + report the discovered devices to the
|
||||
/// manager (M20), then enable ACPI mode and arm the power button (M21).
|
||||
fn onInit(endpoint: runtime.ipc.Handle) bool {
|
||||
registered_count = 0;
|
||||
walkDevices(namespace.root, &interpreter);
|
||||
walkDevices(persistent_namespace.root, &global_interpreter);
|
||||
|
||||
const manager = runtime.ipc.lookup(.device_manager);
|
||||
var i: usize = 0;
|
||||
while (i < registered_count) : (i += 1) {
|
||||
const entry = registered[i];
|
||||
// Append the _HID's human-readable name when it is a known standard PnP/ACPI
|
||||
// id (e.g. PNP0303 -> "PS/2 Keyboard"), so the boot log says what each
|
||||
// reported device actually is. The description trails the existing fields so
|
||||
// the acpi-report/acpi-ps2 matchers still see "<hid> (device N, M resources)".
|
||||
const hid = entry.hid[0..entry.hid_len];
|
||||
const desc = acpi_ids.description(hid);
|
||||
if (desc.len != 0)
|
||||
@@ -154,12 +218,7 @@ pub fn main(init: runtime.process.Init) void {
|
||||
else
|
||||
writeLine("/system/services/acpi: reported {s} (device {d}, {d} resources)\n", .{ hid, entry.device_id, entry.resource_count });
|
||||
if (manager) |h| {
|
||||
var report = protocol.ChildAdded{
|
||||
.parent = node_id,
|
||||
.bus_address = entry.device_id,
|
||||
.identity = 0,
|
||||
.device_id = entry.device_id,
|
||||
};
|
||||
var report = protocol.ChildAdded{ .parent = node_id, .bus_address = entry.device_id, .identity = 0, .device_id = entry.device_id };
|
||||
@memcpy(report.hid[0..entry.hid_len], entry.hid[0..entry.hid_len]);
|
||||
var reply: [protocol.message_maximum]u8 = undefined;
|
||||
_ = runtime.ipc.call(h, std.mem.asBytes(&report), &reply) catch {};
|
||||
@@ -167,9 +226,241 @@ pub fn main(init: runtime.process.Init) void {
|
||||
}
|
||||
writeLine("/system/services/acpi: reported {d} device(s) to the manager\n", .{registered_count});
|
||||
|
||||
// Stay resident: the claim holds, and the service is here to grow into the
|
||||
// supervised discoverer (M20.3, then the M21 event side on the SCI).
|
||||
while (true) runtime.system.sleep(1000);
|
||||
armPowerButton(endpoint);
|
||||
return true;
|
||||
}
|
||||
|
||||
// --- power event side (M21) ---------------------------------------------------
|
||||
|
||||
/// Read the PM1 event/control and GPE register ports plus the SMI enable pair
|
||||
/// from the FADT copy on the node. Offsets are from the FADT table start (the
|
||||
/// SDT header is the first 36 bytes). Prefers the 32-bit port fields; QEMU's
|
||||
/// FADT populates them.
|
||||
fn readFadt(fadt: ?[]const u8) void {
|
||||
const f = fadt orelse {
|
||||
_ = runtime.system.write("acpi: no FADT on the node — power events off\n");
|
||||
return;
|
||||
};
|
||||
smi_cmd = @truncate(rd32(f, 48));
|
||||
acpi_enable_value = f[52];
|
||||
pm1a_evt = @truncate(rd32(f, 56));
|
||||
pm1b_evt = @truncate(rd32(f, 60));
|
||||
pm1a_cnt = @truncate(rd32(f, 64));
|
||||
pm1b_cnt = @truncate(rd32(f, 68));
|
||||
gpe0_blk = @truncate(rd32(f, 80));
|
||||
gpe1_blk = @truncate(rd32(f, 84));
|
||||
pm1_evt_len = if (f.len > 88) f[88] else 4;
|
||||
gpe0_len = if (f.len > 92) f[92] else 0;
|
||||
gpe1_len = if (f.len > 93) f[93] else 0;
|
||||
}
|
||||
|
||||
fn readSleepS5(ns: *aml.Namespace) bool {
|
||||
const st = aml.sleepState(ns, 5) orelse return false;
|
||||
s5_slp_typ_a = st.slp_typ_a;
|
||||
s5_slp_typ_b = st.slp_typ_b;
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Enable ACPI mode if the firmware isn't already in it, then bind the SCI and
|
||||
/// set PWRBTN_EN so the power button raises an interrupt we can see.
|
||||
fn armPowerButton(endpoint: runtime.ipc.Handle) void {
|
||||
if (pm1a_cnt != 0 and (halPioRead(2, pm1a_cnt) & sci_en_bit) == 0 and smi_cmd != 0) {
|
||||
// Switch to ACPI mode: write ACPI_ENABLE to the SMI command port, then
|
||||
// spin (bounded) until SCI_EN latches.
|
||||
halPioWrite(1, smi_cmd, acpi_enable_value);
|
||||
var tries: u32 = 0;
|
||||
while (tries < 1000 and (halPioRead(2, pm1a_cnt) & sci_en_bit) == 0) : (tries += 1) {
|
||||
runtime.system.sleep(1);
|
||||
}
|
||||
}
|
||||
if (!has_sci) {
|
||||
_ = runtime.system.write("acpi: no SCI resource — power button unavailable\n");
|
||||
return;
|
||||
}
|
||||
if (!device.irqBind(node_id, sci_resource_index, endpoint)) {
|
||||
_ = runtime.system.write("acpi: SCI irq_bind failed\n");
|
||||
return;
|
||||
}
|
||||
// PWRBTN_EN lives in the PM1 enable register at evt_blk + evt_len/2.
|
||||
if (pm1a_evt != 0) {
|
||||
const en_port = pm1a_evt + pm1_evt_len / 2;
|
||||
halPioWrite(2, en_port, @as(u16, @truncate(halPioRead(2, en_port))) | pwrbtn_bit);
|
||||
}
|
||||
if (pm1b_evt != 0) {
|
||||
const en_port = pm1b_evt + pm1_evt_len / 2;
|
||||
halPioWrite(2, en_port, @as(u16, @truncate(halPioRead(2, en_port))) | pwrbtn_bit);
|
||||
}
|
||||
_ = runtime.system.write("acpi: power button armed\n");
|
||||
}
|
||||
|
||||
/// The SCI fired. Read PM1 status; a set PWRBTN_STS is the power button — clear
|
||||
/// it (write-1), publish, log. Any other set status is cleared and logged
|
||||
/// (GPE/Notify dispatch is M21.2). Always re-arm the line.
|
||||
fn onSci() void {
|
||||
var handled = false;
|
||||
inline for (.{ pm1a_evt, pm1b_evt }) |evt_port| {
|
||||
if (evt_port != 0) {
|
||||
const sts: u16 = @truncate(halPioRead(2, evt_port));
|
||||
if (sts & pwrbtn_bit != 0) {
|
||||
halPioWrite(2, evt_port, pwrbtn_bit); // write-1-to-clear
|
||||
handled = true;
|
||||
} else if (sts != 0) {
|
||||
halPioWrite(2, evt_port, sts); // clear whatever else latched
|
||||
}
|
||||
}
|
||||
}
|
||||
if (handled) {
|
||||
_ = runtime.system.write("power: button pressed\n");
|
||||
publishButton();
|
||||
}
|
||||
handleGpe();
|
||||
_ = device.irqAck(node_id, sci_resource_index);
|
||||
}
|
||||
|
||||
/// General-purpose events: for each set+enabled GPE bit, evaluate its `\_GPE`
|
||||
/// handler method (`_Lxx` level / `_Exx` edge), drain the Notify queue the
|
||||
/// method produced, and publish an event per notified device. Then clear the
|
||||
/// status bit. QEMU raises no GPEs on this config, so this path is exercised by
|
||||
/// host unit tests (docs/m21-plan.md decision 5); on real hardware it carries
|
||||
/// battery/AC/lid. The embedded controller's `_Qxx` queries are out of scope.
|
||||
fn handleGpe() void {
|
||||
handleGpeBlock(gpe0_blk, gpe0_len, 0);
|
||||
handleGpeBlock(gpe1_blk, gpe1_len, gpe0_len * 4);
|
||||
}
|
||||
|
||||
fn handleGpeBlock(blk: u16, len: u8, gpe_base: u32) void {
|
||||
if (blk == 0 or len == 0) return;
|
||||
const status_bytes = len / 2; // status half, then enable half
|
||||
var byte_index: u8 = 0;
|
||||
while (byte_index < status_bytes) : (byte_index += 1) {
|
||||
const sts: u8 = @truncate(halPioRead(1, blk + byte_index));
|
||||
const en: u8 = @truncate(halPioRead(1, blk + status_bytes + byte_index));
|
||||
const active = sts & en;
|
||||
if (active == 0) continue;
|
||||
var bit: u3 = 0;
|
||||
while (true) : (bit += 1) {
|
||||
if (active & (@as(u8, 1) << bit) != 0) {
|
||||
dispatchGpe(gpe_base + @as(u32, byte_index) * 8 + bit);
|
||||
}
|
||||
if (bit == 7) break;
|
||||
}
|
||||
halPioWrite(1, blk + byte_index, active); // write-1-to-clear the serviced bits
|
||||
}
|
||||
}
|
||||
|
||||
/// Evaluate the `\_GPE._L%02X` or `_E%02X` handler for GPE number `n`, then
|
||||
/// publish an event for each device it notified.
|
||||
fn dispatchGpe(n: u32) void {
|
||||
const gpe_scope = aml.Namespace.resolve(&persistent_namespace, persistent_namespace.root, true, 0, &.{seg4("_GPE")}) orelse return;
|
||||
var name: [4]u8 = .{ '_', 'L', 0, 0 };
|
||||
writeHex2(name[2..4], n);
|
||||
var method = aml.Namespace.childOf(gpe_scope, name);
|
||||
if (method == null) {
|
||||
name[1] = 'E';
|
||||
method = aml.Namespace.childOf(gpe_scope, name);
|
||||
}
|
||||
const m = method orelse return; // no handler — the status bit was already cleared
|
||||
_ = global_interpreter.evaluate(m, &.{}) catch return;
|
||||
for (global_interpreter.takeNotifications()) |event| publishNotify(event.node, event.code);
|
||||
}
|
||||
|
||||
fn publishNotify(node: *aml.Node, code: u64) void {
|
||||
// Map the notified device's _HID to a domain event where we recognize it.
|
||||
var hid: [8]u8 = .{0} ** 8;
|
||||
if (readHid(node, &global_interpreter)) |h| hid = h;
|
||||
const which: power.Event = if (std.mem.eql(u8, hid[0..7], "PNP0C0A")) .battery else if (std.mem.eql(u8, hid[0..7], "ACPI0003")) .ac else if (std.mem.eql(u8, hid[0..7], "PNP0C0D")) .lid else .notify;
|
||||
var event = power.EventMessage{ .event = @intFromEnum(which), .code = @truncate(code) };
|
||||
event.hid = hid;
|
||||
writeLine("power: notify {s} code {d}\n", .{ hid[0..7], code });
|
||||
publishEvent(std.mem.asBytes(&event));
|
||||
}
|
||||
|
||||
/// Two lowercase hex digits of `n` into `out[0..2]`.
|
||||
fn writeHex2(out: []u8, n: u32) void {
|
||||
const digits = "0123456789ABCDEF";
|
||||
out[0] = digits[(n >> 4) & 0xF];
|
||||
out[1] = digits[n & 0xF];
|
||||
}
|
||||
|
||||
fn publishButton() void {
|
||||
const event = power.EventMessage{ .event = @intFromEnum(power.Event.power_button) };
|
||||
publishEvent(std.mem.asBytes(&event));
|
||||
}
|
||||
|
||||
fn publishEvent(bytes: []const u8) void {
|
||||
for (&subscribers) |*slot| {
|
||||
if (slot.*) |handle| {
|
||||
if (!runtime.ipc.send(handle, bytes)) slot.* = null;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn isSubscriber(task: u32) bool {
|
||||
for (&subscribers, 0..) |*slot, si| {
|
||||
if (slot.* != null and subscriber_tasks[si] == task) return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/// Enter S5 (soft off): write SLP_TYP|SLP_EN to the PM1 control register(s).
|
||||
/// Mirrors the kernel's power.zig sleepValue. Only reached from a PID-1
|
||||
/// shutdown request (M21.3).
|
||||
fn enterS5() void {
|
||||
if (!s5_valid or pm1a_cnt == 0) {
|
||||
_ = runtime.system.write("power: S5 unavailable\n");
|
||||
return;
|
||||
}
|
||||
_ = runtime.system.write("power: entering S5\n");
|
||||
halPioWrite(2, pm1a_cnt, (@as(u32, s5_slp_typ_a & 0x7) << 10) | slp_en);
|
||||
if (pm1b_cnt != 0) halPioWrite(2, pm1b_cnt, (@as(u32, s5_slp_typ_b & 0x7) << 10) | slp_en);
|
||||
// If control returns, the write did not take — say so instead of hanging.
|
||||
runtime.system.sleep(500);
|
||||
_ = runtime.system.write("power: S5 write did not take\n");
|
||||
}
|
||||
|
||||
// --- harness callbacks --------------------------------------------------------
|
||||
|
||||
fn onNotification(badge: u64) void {
|
||||
// The only notification the service binds is the SCI (an IRQ badge).
|
||||
_ = badge;
|
||||
onSci();
|
||||
}
|
||||
|
||||
/// The `.power` protocol: subscribe (endpoint as the call's capability),
|
||||
/// shutdown (PID 1 only). Device discovery uses a different endpoint (the
|
||||
/// device manager's), so nothing here handles ChildAdded.
|
||||
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?runtime.ipc.Handle) usize {
|
||||
if (message.len < 1) return 0;
|
||||
switch (message[0]) {
|
||||
@intFromEnum(power.Operation.subscribe) => {
|
||||
var status: i32 = -1;
|
||||
if (capability) |handle| {
|
||||
for (&subscribers, 0..) |*slot, si| {
|
||||
if (slot.* == null) {
|
||||
slot.* = handle;
|
||||
subscriber_tasks[si] = sender;
|
||||
status = 0;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
const r = power.Reply{ .status = status };
|
||||
@memcpy(reply[0..@sizeOf(power.Reply)], std.mem.asBytes(&r));
|
||||
return @sizeOf(power.Reply);
|
||||
},
|
||||
@intFromEnum(power.Operation.shutdown) => {
|
||||
// Honored only from a power subscriber — init, which has already run
|
||||
// the stop sequence over everything else. The power service is
|
||||
// mechanism (write S5); deciding *when* to shut down and stopping
|
||||
// the rest of the system first is init's policy.
|
||||
const allowed = isSubscriber(sender);
|
||||
const r = power.Reply{ .status = if (allowed) 0 else -1 };
|
||||
@memcpy(reply[0..@sizeOf(power.Reply)], std.mem.asBytes(&r));
|
||||
if (allowed) enterS5();
|
||||
return @sizeOf(power.Reply);
|
||||
},
|
||||
else => return 0,
|
||||
}
|
||||
}
|
||||
|
||||
/// Depth-first walk: register + report each present device with a _HID, then
|
||||
|
||||
+102
-15
@@ -1,29 +1,41 @@
|
||||
//! /system/services/system/services/init: — the first user-space program, PID 1. Built as its own
|
||||
//! freestanding binary (see build.zig), shipped on the boot volume at /system/services/system/services/init:,
|
||||
//! /system/services/init — the first user-space program, PID 1. Built as its own
|
||||
//! freestanding binary (see build.zig), shipped on the boot volume at /system/services/init,
|
||||
//! loaded by the bootloader, and started in ring 3 as a scheduled process by the
|
||||
//! kernel (system/kernel/process.zig). It links against the shared user runtime
|
||||
//! library `runtime` and talks to the kernel only through `runtime`'s system_call wrappers.
|
||||
//!
|
||||
//! It proves the C-convention heap works, then — as PID 1 — acts as the system's
|
||||
//! **service supervisor**: it spawns the user-space services danos brings up at boot
|
||||
//! (the VFS server, the device manager), and settles into a heartbeat so it stays
|
||||
//! alive as the root of user space. Drivers are *not* its job: the device manager
|
||||
//! discovers the hardware and spawns those. This is the service half of the
|
||||
//! service/driver spawn split (docs/driver-model.md).
|
||||
//! (the VFS server, the device manager), and settles into an event loop as the root
|
||||
//! of user space. Drivers are *not* its job: the device manager discovers the
|
||||
//! hardware and spawns those. This is the service half of the service/driver spawn
|
||||
//! split (docs/driver-model.md).
|
||||
//!
|
||||
//! M21: init also owns **orderly shutdown**. It supervises its children (keeping
|
||||
//! their ids and an exit endpoint), subscribes to the power service, and on a
|
||||
//! power-button event runs the stop sequence over its children in reverse order
|
||||
//! before asking the power service to enter S5 — lifecycle (M17) and events (M21)
|
||||
//! composing into a clean poweroff.
|
||||
|
||||
const std = @import("std");
|
||||
const runtime = @import("runtime");
|
||||
const power = runtime.power_protocol;
|
||||
|
||||
/// The system services system/services/init: brings up at boot, in order. This is system/services/init:'s policy — the
|
||||
/// The system services init brings up at boot, in order. This is init's policy — the
|
||||
/// microkernel keeps such choices in user space, not the kernel. Drivers are absent
|
||||
/// on purpose: the device manager owns those. (A future system/services/init: reads this from a
|
||||
/// on purpose: the device manager owns those. (A future init reads this from a
|
||||
/// manifest under /system/services instead of a hardcoded list.)
|
||||
const boot_services = [_][]const u8{ "vfs", "input", "device-manager" };
|
||||
|
||||
var children: [boot_services.len]u32 = .{0} ** boot_services.len;
|
||||
var child_count: usize = 0;
|
||||
var supervision_endpoint: runtime.ipc.Handle = 0;
|
||||
|
||||
pub fn main() void {
|
||||
// Prove the heap end to end: allocate through the runtime allocator (which
|
||||
// mmaps pages from the kernel and carves them with the free list), write into
|
||||
// that heap buffer (exercising the widened debug_write bounds check), and
|
||||
// free it. A fault here would kill system/services/init: before it heartbeats — so the system/services/init:
|
||||
// free it. A fault here would kill init before it heartbeats — so the init
|
||||
// test doubles as the heap regression test. (C code links the same heap via
|
||||
// the extern malloc/free symbols; Zig code uses this allocator.)
|
||||
const gpa = runtime.allocator();
|
||||
@@ -34,19 +46,94 @@ pub fn main() void {
|
||||
gpa.free(buffer);
|
||||
} else |_| {}
|
||||
|
||||
// Bring up the boot services. Best-effort and silent: each service announces its
|
||||
// own readiness (`vfs: ready`, ...), and in an isolation test that runs system/services/init: with
|
||||
// no system/services/init:ial-ramdisk the spawns simply no-op rather than deranging the heartbeat.
|
||||
// One endpoint carries everything init waits on: children's exit
|
||||
// notifications (they are spawned supervised against it), init's own
|
||||
// signals, and power events it subscribes to. All arrive in the loop below.
|
||||
supervision_endpoint = runtime.ipc.createIpcEndpoint() orelse {
|
||||
_ = runtime.system.write("/system/services/init: no endpoint\n");
|
||||
return;
|
||||
};
|
||||
_ = runtime.process.bindSignals(supervision_endpoint);
|
||||
|
||||
// Bring up the boot services, supervised so init can stop them cleanly.
|
||||
// Best-effort and silent: each service announces its own readiness, and in
|
||||
// an isolation test with no initial-ramdisk the spawns simply no-op.
|
||||
for (boot_services) |service| {
|
||||
_ = runtime.system.spawn(service);
|
||||
if (runtime.system.spawnSupervised(service, &.{}, supervision_endpoint)) |id| {
|
||||
children[child_count] = id;
|
||||
child_count += 1;
|
||||
}
|
||||
}
|
||||
|
||||
// Subscribe to power events (retry: the power service registers well after
|
||||
// init starts). Best-effort — without it, a `terminate` signal still
|
||||
// triggers the same shutdown path.
|
||||
subscribePower();
|
||||
|
||||
// A re-arming timer drives the liveness heartbeat: proof PID 1 is alive
|
||||
// (the init test's marker) while the loop stays free to receive signals,
|
||||
// power events, and children's exit notifications.
|
||||
_ = runtime.system.timerOnce(supervision_endpoint, 1000);
|
||||
|
||||
var receive: [power.message_maximum]u8 = undefined;
|
||||
while (true) {
|
||||
_ = runtime.system.write("/system/services/init: heartbeat\n");
|
||||
runtime.system.sleep(1000);
|
||||
const got = runtime.ipc.replyWait(supervision_endpoint, &.{}, &receive, null);
|
||||
if (runtime.process.signalsFrom(got.badge)) |signals| {
|
||||
if (signals.has(.terminate)) shutDown();
|
||||
continue;
|
||||
}
|
||||
if (got.isTimer()) {
|
||||
_ = runtime.system.write("/system/services/init: heartbeat\n");
|
||||
_ = runtime.system.timerOnce(supervision_endpoint, 1000);
|
||||
continue;
|
||||
}
|
||||
if (got.isMessage() and got.len >= 2 and receive[0] == @intFromEnum(power.Operation.event)) {
|
||||
// A power event (the only buffered messages init receives).
|
||||
if (receive[1] == @intFromEnum(power.Event.power_button)) shutDown();
|
||||
continue;
|
||||
}
|
||||
// Child-exit notifications and anything else: keep waiting.
|
||||
if (got.isNotification()) continue;
|
||||
}
|
||||
}
|
||||
|
||||
/// Look up the power service and subscribe our endpoint (handed over as the
|
||||
/// call's capability) so events arrive as buffered messages here.
|
||||
fn subscribePower() void {
|
||||
var handle: ?runtime.ipc.Handle = null;
|
||||
var tries: u32 = 0;
|
||||
while (handle == null and tries < 200) : (tries += 1) {
|
||||
handle = runtime.ipc.lookup(.power);
|
||||
if (handle == null) runtime.system.sleep(20);
|
||||
}
|
||||
// A missing power service is not fatal — init proceeds to its heartbeat and
|
||||
// a `terminate` signal still drives shutdown. Silent so the no-ramdisk init
|
||||
// test's heartbeat marker is the next line written.
|
||||
const h = handle orelse return;
|
||||
const request = power.Subscribe{};
|
||||
var reply: [power.message_maximum]u8 = undefined;
|
||||
_ = runtime.ipc.callCap(h, std.mem.asBytes(&request), &reply, supervision_endpoint) catch {};
|
||||
}
|
||||
|
||||
/// The stop sequence: terminate each child in reverse spawn order (vfs last —
|
||||
/// other services may flush through it), waiting up to a deadline for each to
|
||||
/// exit before killing it, then ask the power service to enter S5.
|
||||
fn shutDown() void {
|
||||
_ = runtime.system.write("/system/services/init: shutting down\n");
|
||||
var i = child_count;
|
||||
while (i > 0) {
|
||||
i -= 1;
|
||||
if (children[i] != 0) runtime.process.stop(children[i], 2000, supervision_endpoint);
|
||||
}
|
||||
if (runtime.ipc.lookup(.power)) |h| {
|
||||
const request = power.Shutdown{};
|
||||
var reply: [power.message_maximum]u8 = undefined;
|
||||
_ = runtime.ipc.call(h, std.mem.asBytes(&request), &reply) catch {};
|
||||
}
|
||||
// If S5 did not take, init has nothing left to do but idle.
|
||||
while (true) runtime.system.sleep(1000);
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start; // pull the runtime entry shim into the image
|
||||
|
||||
@@ -0,0 +1,68 @@
|
||||
//! The power protocol (docs/m21-plan.md): system power's domain-named surface,
|
||||
//! registered under `ServiceId.power`. On x86 the acpi service serves it; on
|
||||
//! ARM a PSCI/mailbox service will register the same id — subscribers never
|
||||
//! learn which firmware they are on (m19-m20-plan.md decision 7). The
|
||||
//! vfs-protocol pattern: extern-struct messages, a version, reserved fields.
|
||||
|
||||
/// The protocol version a client states nowhere yet — reserved for the day a
|
||||
/// handshake needs it; requests carry it so a mismatch can be refused loudly.
|
||||
pub const version: u16 = 1;
|
||||
|
||||
pub const Operation = enum(u8) {
|
||||
/// Subscribe to power events: the subscriber's endpoint rides as the
|
||||
/// call's capability (the input/device-manager pattern); events arrive on
|
||||
/// it as buffered messages carrying an `EventMessage`.
|
||||
subscribe = 1,
|
||||
/// Orderly shutdown's last step: enter S5. Accepted only from PID 1
|
||||
/// (init) — the process that has already run the stop sequence over
|
||||
/// everything else.
|
||||
shutdown = 2,
|
||||
/// The published event payload (never sent *to* the service).
|
||||
event = 3,
|
||||
};
|
||||
|
||||
/// What happened. The vocabulary is hardware-neutral: a lid is a lid whether
|
||||
/// ACPI or a PSCI mailbox reported it.
|
||||
pub const Event = enum(u8) {
|
||||
power_button = 1,
|
||||
lid = 2,
|
||||
ac = 3,
|
||||
battery = 4,
|
||||
/// A device notification that maps to none of the named events — the
|
||||
/// `code` and `hid` fields say which device and what code.
|
||||
notify = 5,
|
||||
};
|
||||
|
||||
pub const Subscribe = extern struct {
|
||||
operation: u8 = @intFromEnum(Operation.subscribe),
|
||||
reserved0: u8 = 0,
|
||||
version: u16 = version,
|
||||
reserved1: u32 = 0,
|
||||
};
|
||||
|
||||
pub const Shutdown = extern struct {
|
||||
operation: u8 = @intFromEnum(Operation.shutdown),
|
||||
reserved0: u8 = 0,
|
||||
version: u16 = version,
|
||||
reserved1: u32 = 0,
|
||||
};
|
||||
|
||||
/// A published event, as the buffered-message payload subscribers receive.
|
||||
pub const EventMessage = extern struct {
|
||||
operation: u8 = @intFromEnum(Operation.event),
|
||||
/// An Event value.
|
||||
event: u8,
|
||||
reserved0: u16 = 0,
|
||||
/// The device notification code (Notify's second argument), or 0.
|
||||
code: u32 = 0,
|
||||
/// The notifying device's hardware id (EISA-decoded), or all zero.
|
||||
hid: [8]u8 = .{0} ** 8,
|
||||
};
|
||||
|
||||
pub const Reply = extern struct {
|
||||
status: i32,
|
||||
reserved: u32 = 0,
|
||||
};
|
||||
|
||||
/// Upper bound on any message in this protocol — sizes endpoint buffers.
|
||||
pub const message_maximum = 64;
|
||||
Reference in New Issue
Block a user