Merge feat/power-events: ACPI events + system power (M21)

The QMP harness channel, the SCI + power button published from ring 3,
Notify/GPE dispatch in the AML interpreter, and orderly shutdown — init's
M17 stop cascade into a ring-3 S5 write. Proven by injecting a real ACPI
power-button event; QEMU powers off through the whole chain.

# Conflicts:
#	system/services/init/init.zig
This commit is contained in:
Daniel Samson
2026-07-13 06:27:18 +01:00
13 changed files with 713 additions and 94 deletions
+1
View File
@@ -177,6 +177,7 @@ pub const ServiceId = enum(u32) {
input = 2,
ps2_bus = 3, // the 8042 owner; child device drivers attach here for raw bytes
device_manager = 4, // the tree, the matcher, the supervisor (docs/device-manager.md)
power = 5, // system power: events (button, lid, battery) + shutdown (docs/m21-plan.md; domain-named per decision 7 — the acpi service registers it on x86, a PSCI service will on ARM)
_,
};
+14
View File
@@ -157,6 +157,13 @@ pub var namespace: ?aml.Namespace = null;
/// Physical address of the DSDT the FADT points at, or 0.
pub var dsdt_physical: u64 = 0;
/// The FADT itself (physical + length), published on the acpi-tables node so
/// the ring-3 acpi service can read the PM1 event and GPE blocks it needs for
/// the event side (docs/m21-plan.md decision 3). Distinguished from the AML
/// blob resources by its intact "FACP" header — the blobs are header-stripped.
var fadt_physical: u64 = 0;
var fadt_length: u64 = 0;
// AML blocks (DSDT + any SSDTs) collected during the table walk, as physical
// address + length of each table's post-header bytecode. Scanned after the walk
// for the sleep-state (`_Sx`) packages.
@@ -373,6 +380,8 @@ pub fn discover(rsdp_physical: u64, memory_regions: []const boot_handoff.MemoryR
// Start clean so a re-run doesn't accumulate stale state.
power_information = .{};
fadt_physical = 0;
fadt_length = 0;
platform_information = .{};
aml_stats = .{};
namespace = null;
@@ -447,6 +456,9 @@ fn publishAcpiTablesNode(device_tree: *DeviceTree) !void {
// SCI (recorded first, len 1) stays distinct so M21 can pick it out.
if (power_information.sci_interrupt != 0) _ = node.addResource(.irq, power_information.sci_interrupt, 1);
_ = node.addResource(.irq, 0, 256);
// The FADT rides along (M21): the service reads the PM1 event / GPE blocks
// from its own copy, telling it apart from the AML blobs by signature.
if (fadt_physical != 0) _ = node.addResource(.memory, fadt_physical, fadt_length);
}
/// The number of Device objects in the namespace built during discovery, or 0.
@@ -482,6 +494,8 @@ fn handleTable(device_tree: *DeviceTree, hal: Hal, sdt_physical: u64) !void {
} else if (std.mem.eql(u8, &sig, &HPET)) {
try parseHpet(device_tree, hal, header);
} else if (std.mem.eql(u8, &sig, &FACP)) {
fadt_physical = sdt_physical;
fadt_length = header.length;
parseFadt(header);
} else if (std.mem.eql(u8, &sig, &SPCR)) {
parseSpcr(header);
+28
View File
@@ -230,3 +230,31 @@ test "interpreter runs a method with args, arithmetic, and control flow" {
const lo = try interpreter.evaluate(tst, &.{.{ .integer = 2 }}); // 2+5=7 !> 10 -> 0
try std.testing.expectEqual(@as(u64, 0), try lo.asInteger());
}
test "interpreter records Notify(device, code)" {
// Device(DEV_) { Name(_HID, 0x030AD041) } // PNP0A03-ish placeholder
// Method(TST_, 0) { Notify(DEV_, 0x80); Return(Zero) }
// Encoded: a Device holding a Name, then a Method issuing Notify on it.
const blob = [_]u8{
0x5B, 0x82, 0x0F, 0x44, 0x45, 0x56, 0x5F, // Device(DEV_) len=0x0F (pkglen + DEV_ + Name)
0x08, 0x5F, 0x48, 0x49, 0x44, 0x0C, 0x41, 0xD0, 0x0A, 0x03, // Name(_HID, DWord 0x030AD041)
0x14, 0x0F, 0x54, 0x53, 0x54, 0x5F, 0x00, // Method(TST_, 0) len=0x0F (pkglen + TST_ + flags + body)
0x86, 0x44, 0x45, 0x56, 0x5F, 0x0A, 0x80, // Notify(DEV_, 0x80)
0xA4, 0x00, // Return(Zero)
};
var arena = std.heap.ArenaAllocator.init(std.testing.allocator);
defer arena.deinit();
var result = try parse(arena.allocator(), &.{&blob});
const namespace = &result.namespace;
const tst = namespace.resolve(namespace.root, false, 0, &.{.{ 'T', 'S', 'T', '_' }}) orelse return error.NoMethod;
const dev = namespace.resolve(namespace.root, false, 0, &.{.{ 'D', 'E', 'V', '_' }}) orelse return error.NoDevice;
var interpreter = Interpreter.init(namespace, .{ .mapMmio = noMap, .pioRead = noRead, .pioWrite = noWrite }, arena.allocator());
_ = try interpreter.evaluate(tst, &.{});
const events = interpreter.takeNotifications();
try std.testing.expectEqual(@as(usize, 1), events.len);
try std.testing.expectEqual(dev, events[0].node);
try std.testing.expectEqual(@as(u64, 0x80), events[0].code);
}
+41
View File
@@ -141,6 +141,9 @@ const Frame = struct {
/// A CreateField binding: a name that indexes into a buffer object.
const BufferField = struct { buffer: *Node, byte_off: usize, bit_width: u32 };
/// One Notify(device, code) the interpreter executed.
pub const NotifyEvent = struct { node: *Node, code: u64 };
pub const Interpreter = struct {
namespace: *Namespace,
hal: Hal,
@@ -149,6 +152,11 @@ pub const Interpreter = struct {
dynamic_overrides: std.AutoHashMapUnmanaged(*Node, Object) = .{},
/// CreateField bindings active for the current evaluation.
fields: std.AutoHashMapUnmanaged(*Node, BufferField) = .{},
/// Notify(device, code) operations the last evaluation executed — a GPE or
/// EC handler tells the OS "look at this device" this way. Bounded; the
/// caller drains it with `takeNotifications` after `evaluate` (M21).
notify_queue: [16]NotifyEvent = undefined,
notify_count: usize = 0,
pub fn init(namespace: *Namespace, hal: Hal, arena: std.mem.Allocator) Interpreter {
return .{ .namespace = namespace, .hal = hal, .arena = arena };
@@ -157,6 +165,7 @@ pub const Interpreter = struct {
/// Evaluate a namespace object: invoke a Method, read a Name's value, or read a
/// Field. Resets per-evaluation runtime state first.
pub fn evaluate(self: *Interpreter, node: *Node, args: []const Object) Error!Object {
self.notify_count = 0;
self.dynamic_overrides.clearRetainingCapacity();
self.fields.clearRetainingCapacity();
return self.invoke(node, args);
@@ -267,6 +276,8 @@ pub const Interpreter = struct {
},
opcode.to_buffer_opcode => try self.passThroughUnary(current, frame),
opcode.notify_opcode => try self.notify(current, frame),
opcode.extended_opcode_prefix => try self.ext(current, frame),
// CreateXField: source, index, name (bit widths differ by op)
@@ -542,6 +553,36 @@ pub const Interpreter = struct {
try self.storeInto(current, frame, value);
}
/// Notify(SuperName, NotifyValue): resolve the named device, evaluate the
/// code, and record the pair for the caller to dispatch. AML control flow
/// continues (Notify returns nothing).
fn notify(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
const lead = current.peek() orelse return error.Truncated;
var target: ?*Node = null;
if (isNameStart(lead)) {
const name_path = try current.nameString();
target = self.namespace.resolve(frame.scope, name_path.rooted, name_path.parents, name_path.slice());
} else {
// A non-name SuperName (Local/Arg holding a reference).
const obj = try self.term(current, frame);
if (obj == .reference) target = obj.reference;
}
const code = try self.evaluateInteger(current, frame);
if (target) |node| {
if (self.notify_count < self.notify_queue.len) {
self.notify_queue[self.notify_count] = .{ .node = node, .code = code };
self.notify_count += 1;
}
}
return .uninitialized;
}
/// The Notify events the last `evaluate` produced. Valid until the next
/// `evaluate` clears the queue.
pub fn takeNotifications(self: *Interpreter) []const NotifyEvent {
return self.notify_queue[0..self.notify_count];
}
fn storeInto(self: *Interpreter, current: *Cursor, frame: *Frame, value: Object) Error!void {
const lead = current.peek() orelse return error.Truncated;
if (isNameStart(lead)) {
+25
View File
@@ -152,6 +152,10 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void {
acpiReportTest(boot_information);
} else if (eql(case, "acpi-ps2")) {
acpiReportTest(boot_information); // same spawn; the harness regex differs
} else if (eql(case, "power-button")) {
acpiReportTest(boot_information); // boot the manager (spawns the acpi service); harness injects the button
} else if (eql(case, "orderly-shutdown")) {
orderlyShutdownTest(boot_information);
} else if (eql(case, "initial-ramdisk")) {
initialRamdiskTest(boot_information);
} else if (eql(case, "vfs")) {
@@ -1929,6 +1933,27 @@ fn pciScanTest(boot_information: *const BootInformation) void {
result();
}
/// M21.3 capstone: orderly shutdown. Boot init with the initial-ramdisk
/// published, so init spawns the full service tree (vfs, input, device-manager
/// -> discovery/acpi); the harness injects a real power-button event via QMP;
/// the acpi service publishes it; init runs the stop sequence over its children
/// and asks the power service for S5; the machine powers off (QEMU exits). The
/// kernel test only spawns init — the ordered chain is the harness assertion.
fn orderlyShutdownTest(boot_information: *const BootInformation) void {
log("DANOS-TEST-BEGIN: orderly-shutdown\n", .{});
if (boot_information.init_len == 0 or boot_information.initial_ramdisk_len == 0) {
check("bootloader handed over init and the initial_ramdisk", false);
result();
return;
}
const ramdisk = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
process.setInitialRamdisk(ramdisk);
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.init_base)))[0..boot_information.init_len];
const spawned = if (process.spawnProcess(image, 4, &.{"/system/services/init"})) true else |_| false;
check("init spawned as PID root of user space", spawned);
result();
}
/// M20.2: the acpi service registers + reports its _HID devices. Boot normally
/// (the manager spawns discovery); the harness's expect regex requires the two
/// PS/2 nodes among the service's report lines, each with its _CRS resources —
+323 -32
View File
@@ -4,13 +4,11 @@
//! grant, a broad irq window, the SCI), and runs the **shared AML module** in
//! ring 3 — the same parser and interpreter the kernel uses.
//!
//! M20.2 (this increment): after parsing, walk the namespace and, for each
//! present Device with a hardware id (`_HID`), evaluate its current resource
//! settings (`_CRS`) through a ring-3 `Hal` (port I/O over the claimed node),
//! register it under the acpi-tables node (its I/O ports and IRQs contained by
//! the node's broad grants), and report it to the device manager with its
//! EISA-decoded hid as identity. Matching those reports to drivers (ps2-bus)
//! and retiring the kernel's own device build follow in M20.3.
//! It also owns the **event side** (M21): it registers the domain-named `.power`
//! service, binds the SCI (System Control Interrupt), and on a power-button
//! fixed event publishes `power_button` to subscribers — and on init's request
//! writes S5 to power the machine off. The device discovery (M20) and the event
//! handling both run in one `runtime.service.run` loop.
const std = @import("std");
const runtime = @import("runtime");
@@ -18,6 +16,7 @@ const aml = @import("aml");
const acpi_ids = @import("acpi-ids");
const device = runtime.device;
const protocol = runtime.device_manager_protocol;
const power = runtime.power_protocol;
/// AML opcode/prefix bytes by name (`zero_opcode`, `byte_prefix`, …) — so the `_HID`
/// integer decode names the opcodes instead of bare 0x0A/0x0B/… (docs/coding-standards.md).
const opcodes = aml.opcodes;
@@ -31,6 +30,43 @@ fn writeLine(comptime fmt: []const u8, arguments: anytype) void {
// window — the Hal routes every port access through this one claim.
var node_id: u64 = 0;
var io_resource_index: u64 = 0;
// The SCI's irq resource index on the node (the len-1 irq, distinct from the
// broad [0,256) window), for irqBind / irqAck.
var sci_resource_index: u64 = 0;
var has_sci = false;
// PM1 event/control and GPE register ports, read from the FADT copy the kernel
// publishes on the node (M21). Port 0 means absent.
var pm1a_evt: u16 = 0;
var pm1b_evt: u16 = 0;
var pm1_evt_len: u8 = 0;
var pm1a_cnt: u16 = 0;
var pm1b_cnt: u16 = 0;
var gpe0_blk: u16 = 0;
var gpe0_len: u8 = 0;
var gpe1_blk: u16 = 0;
var gpe1_len: u8 = 0;
var smi_cmd: u16 = 0;
var acpi_enable_value: u8 = 0;
var s5_slp_typ_a: u8 = 0;
var s5_slp_typ_b: u8 = 0;
var s5_valid = false;
// PM1 event-register bits (ACPI): PWRBTN in the status/enable word is bit 8;
// the control word's SCI_EN is bit 0; SLP_EN is bit 13.
const pwrbtn_bit: u16 = 1 << 8;
const sci_en_bit: u32 = 1 << 0;
const slp_en: u32 = 1 << 13;
// The `.power` subscribers: endpoints handed over as capabilities, each
// receiving events as buffered messages. Dropped on a failed send. The
// subscriber's task id is kept too — a shutdown request is honored only from a
// subscriber (init subscribes; a stray process does not), the soft gate that
// stands in for "only the system supervisor may power off" without hardcoding
// a pid the kernel's idle tasks would have taken.
const maximum_subscribers = 8;
var subscribers: [maximum_subscribers]?runtime.ipc.Handle = .{null} ** maximum_subscribers;
var subscriber_tasks: [maximum_subscribers]u32 = .{0} ** maximum_subscribers;
// Pass-1 registration record (see main): what pass 2 reports.
const Registered = struct { hid: [8]u8 = .{0} ** 8, hid_len: usize = 0, device_id: u64 = 0, resource_count: u64 = 0 };
@@ -85,22 +121,34 @@ pub fn main(init: runtime.process.Init) void {
return;
}
// Map each memory resource (an AML blob) and note the io_port resource.
// Map the node's resources: the AML blobs (bytecode), the FADT (intact
// "FACP" header — decision 3), the io_port grant, and the SCI irq.
var blocks: [8][]const u8 = undefined;
var block_count: usize = 0;
var found_io = false;
var fadt: ?[]const u8 = null;
for (node.resources[0..@intCast(node.resource_count)], 0..) |resource, index| {
if (resource.kind == @intFromEnum(device.ResourceKind.io_port) and !found_io) {
io_resource_index = index;
found_io = true;
continue;
}
if (resource.kind == @intFromEnum(device.ResourceKind.irq) and resource.len == 1) {
sci_resource_index = index;
has_sci = true;
continue;
}
if (resource.kind != @intFromEnum(device.ResourceKind.memory)) continue;
const base = device.mmioMap(node_id, index) orelse continue;
const pointer: [*]const u8 = @ptrFromInt(base);
blocks[block_count] = pointer[0..@intCast(resource.len)];
const bytes = pointer[0..@intCast(resource.len)];
if (bytes.len >= 4 and std.mem.eql(u8, bytes[0..4], "FACP")) {
fadt = bytes;
continue;
}
if (block_count == blocks.len) continue;
blocks[block_count] = bytes;
block_count += 1;
if (block_count == blocks.len) break;
}
if (block_count == 0) {
_ = runtime.system.write("/system/services/acpi: no AML blobs on the node\n");
@@ -124,29 +172,45 @@ pub fn main(init: runtime.process.Init) void {
while (true) runtime.system.sleep(1000);
}
// Register + report the present _HID devices (M20.2).
var arena = std.heap.ArenaAllocator.init(runtime.allocator());
var interpreter = aml.Interpreter.init(&namespace, .{
// Register + report the present _HID devices (M20), then set up the power
// event side (M21), then serve — all in one harness loop. The interpreter
// and namespace outlive this frame (static), so the harness callbacks can
// reach them.
interpreter_arena = std.heap.ArenaAllocator.init(runtime.allocator());
persistent_namespace = namespace;
global_interpreter = aml.Interpreter.init(&persistent_namespace, .{
.mapMmio = halMapMmio,
.pioRead = halPioRead,
.pioWrite = halPioWrite,
}, arena.allocator());
}, interpreter_arena.allocator());
// Pass 1: register every present _HID device under acpi-tables, remembering
// each (hid, device id). Pass 2: report them all. Registering before any
// report reaches the manager means a driver it spawns on the first report
// already sees the whole set (no keyboard-before-mouse race for ps2-bus).
readFadt(fadt);
s5_valid = readSleepS5(&persistent_namespace);
runtime.service.run(power.message_maximum, .{
.service = .power,
.init = onInit,
.on_message = onMessage,
.on_notification = onNotification,
});
}
// Static so the harness callbacks (which run after main's stack frame is gone)
// can reach the namespace and interpreter.
var persistent_namespace: aml.Namespace = undefined;
var global_interpreter: aml.Interpreter = undefined;
var interpreter_arena: std.heap.ArenaAllocator = undefined;
/// Startup under the harness: register + report the discovered devices to the
/// manager (M20), then enable ACPI mode and arm the power button (M21).
fn onInit(endpoint: runtime.ipc.Handle) bool {
registered_count = 0;
walkDevices(namespace.root, &interpreter);
walkDevices(persistent_namespace.root, &global_interpreter);
const manager = runtime.ipc.lookup(.device_manager);
var i: usize = 0;
while (i < registered_count) : (i += 1) {
const entry = registered[i];
// Append the _HID's human-readable name when it is a known standard PnP/ACPI
// id (e.g. PNP0303 -> "PS/2 Keyboard"), so the boot log says what each
// reported device actually is. The description trails the existing fields so
// the acpi-report/acpi-ps2 matchers still see "<hid> (device N, M resources)".
const hid = entry.hid[0..entry.hid_len];
const desc = acpi_ids.description(hid);
if (desc.len != 0)
@@ -154,12 +218,7 @@ pub fn main(init: runtime.process.Init) void {
else
writeLine("/system/services/acpi: reported {s} (device {d}, {d} resources)\n", .{ hid, entry.device_id, entry.resource_count });
if (manager) |h| {
var report = protocol.ChildAdded{
.parent = node_id,
.bus_address = entry.device_id,
.identity = 0,
.device_id = entry.device_id,
};
var report = protocol.ChildAdded{ .parent = node_id, .bus_address = entry.device_id, .identity = 0, .device_id = entry.device_id };
@memcpy(report.hid[0..entry.hid_len], entry.hid[0..entry.hid_len]);
var reply: [protocol.message_maximum]u8 = undefined;
_ = runtime.ipc.call(h, std.mem.asBytes(&report), &reply) catch {};
@@ -167,9 +226,241 @@ pub fn main(init: runtime.process.Init) void {
}
writeLine("/system/services/acpi: reported {d} device(s) to the manager\n", .{registered_count});
// Stay resident: the claim holds, and the service is here to grow into the
// supervised discoverer (M20.3, then the M21 event side on the SCI).
while (true) runtime.system.sleep(1000);
armPowerButton(endpoint);
return true;
}
// --- power event side (M21) ---------------------------------------------------
/// Read the PM1 event/control and GPE register ports plus the SMI enable pair
/// from the FADT copy on the node. Offsets are from the FADT table start (the
/// SDT header is the first 36 bytes). Prefers the 32-bit port fields; QEMU's
/// FADT populates them.
fn readFadt(fadt: ?[]const u8) void {
const f = fadt orelse {
_ = runtime.system.write("acpi: no FADT on the node — power events off\n");
return;
};
smi_cmd = @truncate(rd32(f, 48));
acpi_enable_value = f[52];
pm1a_evt = @truncate(rd32(f, 56));
pm1b_evt = @truncate(rd32(f, 60));
pm1a_cnt = @truncate(rd32(f, 64));
pm1b_cnt = @truncate(rd32(f, 68));
gpe0_blk = @truncate(rd32(f, 80));
gpe1_blk = @truncate(rd32(f, 84));
pm1_evt_len = if (f.len > 88) f[88] else 4;
gpe0_len = if (f.len > 92) f[92] else 0;
gpe1_len = if (f.len > 93) f[93] else 0;
}
fn readSleepS5(ns: *aml.Namespace) bool {
const st = aml.sleepState(ns, 5) orelse return false;
s5_slp_typ_a = st.slp_typ_a;
s5_slp_typ_b = st.slp_typ_b;
return true;
}
/// Enable ACPI mode if the firmware isn't already in it, then bind the SCI and
/// set PWRBTN_EN so the power button raises an interrupt we can see.
fn armPowerButton(endpoint: runtime.ipc.Handle) void {
if (pm1a_cnt != 0 and (halPioRead(2, pm1a_cnt) & sci_en_bit) == 0 and smi_cmd != 0) {
// Switch to ACPI mode: write ACPI_ENABLE to the SMI command port, then
// spin (bounded) until SCI_EN latches.
halPioWrite(1, smi_cmd, acpi_enable_value);
var tries: u32 = 0;
while (tries < 1000 and (halPioRead(2, pm1a_cnt) & sci_en_bit) == 0) : (tries += 1) {
runtime.system.sleep(1);
}
}
if (!has_sci) {
_ = runtime.system.write("acpi: no SCI resource — power button unavailable\n");
return;
}
if (!device.irqBind(node_id, sci_resource_index, endpoint)) {
_ = runtime.system.write("acpi: SCI irq_bind failed\n");
return;
}
// PWRBTN_EN lives in the PM1 enable register at evt_blk + evt_len/2.
if (pm1a_evt != 0) {
const en_port = pm1a_evt + pm1_evt_len / 2;
halPioWrite(2, en_port, @as(u16, @truncate(halPioRead(2, en_port))) | pwrbtn_bit);
}
if (pm1b_evt != 0) {
const en_port = pm1b_evt + pm1_evt_len / 2;
halPioWrite(2, en_port, @as(u16, @truncate(halPioRead(2, en_port))) | pwrbtn_bit);
}
_ = runtime.system.write("acpi: power button armed\n");
}
/// The SCI fired. Read PM1 status; a set PWRBTN_STS is the power button — clear
/// it (write-1), publish, log. Any other set status is cleared and logged
/// (GPE/Notify dispatch is M21.2). Always re-arm the line.
fn onSci() void {
var handled = false;
inline for (.{ pm1a_evt, pm1b_evt }) |evt_port| {
if (evt_port != 0) {
const sts: u16 = @truncate(halPioRead(2, evt_port));
if (sts & pwrbtn_bit != 0) {
halPioWrite(2, evt_port, pwrbtn_bit); // write-1-to-clear
handled = true;
} else if (sts != 0) {
halPioWrite(2, evt_port, sts); // clear whatever else latched
}
}
}
if (handled) {
_ = runtime.system.write("power: button pressed\n");
publishButton();
}
handleGpe();
_ = device.irqAck(node_id, sci_resource_index);
}
/// General-purpose events: for each set+enabled GPE bit, evaluate its `\_GPE`
/// handler method (`_Lxx` level / `_Exx` edge), drain the Notify queue the
/// method produced, and publish an event per notified device. Then clear the
/// status bit. QEMU raises no GPEs on this config, so this path is exercised by
/// host unit tests (docs/m21-plan.md decision 5); on real hardware it carries
/// battery/AC/lid. The embedded controller's `_Qxx` queries are out of scope.
fn handleGpe() void {
handleGpeBlock(gpe0_blk, gpe0_len, 0);
handleGpeBlock(gpe1_blk, gpe1_len, gpe0_len * 4);
}
fn handleGpeBlock(blk: u16, len: u8, gpe_base: u32) void {
if (blk == 0 or len == 0) return;
const status_bytes = len / 2; // status half, then enable half
var byte_index: u8 = 0;
while (byte_index < status_bytes) : (byte_index += 1) {
const sts: u8 = @truncate(halPioRead(1, blk + byte_index));
const en: u8 = @truncate(halPioRead(1, blk + status_bytes + byte_index));
const active = sts & en;
if (active == 0) continue;
var bit: u3 = 0;
while (true) : (bit += 1) {
if (active & (@as(u8, 1) << bit) != 0) {
dispatchGpe(gpe_base + @as(u32, byte_index) * 8 + bit);
}
if (bit == 7) break;
}
halPioWrite(1, blk + byte_index, active); // write-1-to-clear the serviced bits
}
}
/// Evaluate the `\_GPE._L%02X` or `_E%02X` handler for GPE number `n`, then
/// publish an event for each device it notified.
fn dispatchGpe(n: u32) void {
const gpe_scope = aml.Namespace.resolve(&persistent_namespace, persistent_namespace.root, true, 0, &.{seg4("_GPE")}) orelse return;
var name: [4]u8 = .{ '_', 'L', 0, 0 };
writeHex2(name[2..4], n);
var method = aml.Namespace.childOf(gpe_scope, name);
if (method == null) {
name[1] = 'E';
method = aml.Namespace.childOf(gpe_scope, name);
}
const m = method orelse return; // no handler — the status bit was already cleared
_ = global_interpreter.evaluate(m, &.{}) catch return;
for (global_interpreter.takeNotifications()) |event| publishNotify(event.node, event.code);
}
fn publishNotify(node: *aml.Node, code: u64) void {
// Map the notified device's _HID to a domain event where we recognize it.
var hid: [8]u8 = .{0} ** 8;
if (readHid(node, &global_interpreter)) |h| hid = h;
const which: power.Event = if (std.mem.eql(u8, hid[0..7], "PNP0C0A")) .battery else if (std.mem.eql(u8, hid[0..7], "ACPI0003")) .ac else if (std.mem.eql(u8, hid[0..7], "PNP0C0D")) .lid else .notify;
var event = power.EventMessage{ .event = @intFromEnum(which), .code = @truncate(code) };
event.hid = hid;
writeLine("power: notify {s} code {d}\n", .{ hid[0..7], code });
publishEvent(std.mem.asBytes(&event));
}
/// Two lowercase hex digits of `n` into `out[0..2]`.
fn writeHex2(out: []u8, n: u32) void {
const digits = "0123456789ABCDEF";
out[0] = digits[(n >> 4) & 0xF];
out[1] = digits[n & 0xF];
}
fn publishButton() void {
const event = power.EventMessage{ .event = @intFromEnum(power.Event.power_button) };
publishEvent(std.mem.asBytes(&event));
}
fn publishEvent(bytes: []const u8) void {
for (&subscribers) |*slot| {
if (slot.*) |handle| {
if (!runtime.ipc.send(handle, bytes)) slot.* = null;
}
}
}
fn isSubscriber(task: u32) bool {
for (&subscribers, 0..) |*slot, si| {
if (slot.* != null and subscriber_tasks[si] == task) return true;
}
return false;
}
/// Enter S5 (soft off): write SLP_TYP|SLP_EN to the PM1 control register(s).
/// Mirrors the kernel's power.zig sleepValue. Only reached from a PID-1
/// shutdown request (M21.3).
fn enterS5() void {
if (!s5_valid or pm1a_cnt == 0) {
_ = runtime.system.write("power: S5 unavailable\n");
return;
}
_ = runtime.system.write("power: entering S5\n");
halPioWrite(2, pm1a_cnt, (@as(u32, s5_slp_typ_a & 0x7) << 10) | slp_en);
if (pm1b_cnt != 0) halPioWrite(2, pm1b_cnt, (@as(u32, s5_slp_typ_b & 0x7) << 10) | slp_en);
// If control returns, the write did not take — say so instead of hanging.
runtime.system.sleep(500);
_ = runtime.system.write("power: S5 write did not take\n");
}
// --- harness callbacks --------------------------------------------------------
fn onNotification(badge: u64) void {
// The only notification the service binds is the SCI (an IRQ badge).
_ = badge;
onSci();
}
/// The `.power` protocol: subscribe (endpoint as the call's capability),
/// shutdown (PID 1 only). Device discovery uses a different endpoint (the
/// device manager's), so nothing here handles ChildAdded.
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?runtime.ipc.Handle) usize {
if (message.len < 1) return 0;
switch (message[0]) {
@intFromEnum(power.Operation.subscribe) => {
var status: i32 = -1;
if (capability) |handle| {
for (&subscribers, 0..) |*slot, si| {
if (slot.* == null) {
slot.* = handle;
subscriber_tasks[si] = sender;
status = 0;
break;
}
}
}
const r = power.Reply{ .status = status };
@memcpy(reply[0..@sizeOf(power.Reply)], std.mem.asBytes(&r));
return @sizeOf(power.Reply);
},
@intFromEnum(power.Operation.shutdown) => {
// Honored only from a power subscriber — init, which has already run
// the stop sequence over everything else. The power service is
// mechanism (write S5); deciding *when* to shut down and stopping
// the rest of the system first is init's policy.
const allowed = isSubscriber(sender);
const r = power.Reply{ .status = if (allowed) 0 else -1 };
@memcpy(reply[0..@sizeOf(power.Reply)], std.mem.asBytes(&r));
if (allowed) enterS5();
return @sizeOf(power.Reply);
},
else => return 0,
}
}
/// Depth-first walk: register + report each present device with a _HID, then
+102 -15
View File
@@ -1,29 +1,41 @@
//! /system/services/system/services/init: — the first user-space program, PID 1. Built as its own
//! freestanding binary (see build.zig), shipped on the boot volume at /system/services/system/services/init:,
//! /system/services/init — the first user-space program, PID 1. Built as its own
//! freestanding binary (see build.zig), shipped on the boot volume at /system/services/init,
//! loaded by the bootloader, and started in ring 3 as a scheduled process by the
//! kernel (system/kernel/process.zig). It links against the shared user runtime
//! library `runtime` and talks to the kernel only through `runtime`'s system_call wrappers.
//!
//! It proves the C-convention heap works, then — as PID 1 — acts as the system's
//! **service supervisor**: it spawns the user-space services danos brings up at boot
//! (the VFS server, the device manager), and settles into a heartbeat so it stays
//! alive as the root of user space. Drivers are *not* its job: the device manager
//! discovers the hardware and spawns those. This is the service half of the
//! service/driver spawn split (docs/driver-model.md).
//! (the VFS server, the device manager), and settles into an event loop as the root
//! of user space. Drivers are *not* its job: the device manager discovers the
//! hardware and spawns those. This is the service half of the service/driver spawn
//! split (docs/driver-model.md).
//!
//! M21: init also owns **orderly shutdown**. It supervises its children (keeping
//! their ids and an exit endpoint), subscribes to the power service, and on a
//! power-button event runs the stop sequence over its children in reverse order
//! before asking the power service to enter S5 — lifecycle (M17) and events (M21)
//! composing into a clean poweroff.
const std = @import("std");
const runtime = @import("runtime");
const power = runtime.power_protocol;
/// The system services system/services/init: brings up at boot, in order. This is system/services/init:'s policy — the
/// The system services init brings up at boot, in order. This is init's policy — the
/// microkernel keeps such choices in user space, not the kernel. Drivers are absent
/// on purpose: the device manager owns those. (A future system/services/init: reads this from a
/// on purpose: the device manager owns those. (A future init reads this from a
/// manifest under /system/services instead of a hardcoded list.)
const boot_services = [_][]const u8{ "vfs", "input", "device-manager" };
var children: [boot_services.len]u32 = .{0} ** boot_services.len;
var child_count: usize = 0;
var supervision_endpoint: runtime.ipc.Handle = 0;
pub fn main() void {
// Prove the heap end to end: allocate through the runtime allocator (which
// mmaps pages from the kernel and carves them with the free list), write into
// that heap buffer (exercising the widened debug_write bounds check), and
// free it. A fault here would kill system/services/init: before it heartbeats — so the system/services/init:
// free it. A fault here would kill init before it heartbeats — so the init
// test doubles as the heap regression test. (C code links the same heap via
// the extern malloc/free symbols; Zig code uses this allocator.)
const gpa = runtime.allocator();
@@ -34,19 +46,94 @@ pub fn main() void {
gpa.free(buffer);
} else |_| {}
// Bring up the boot services. Best-effort and silent: each service announces its
// own readiness (`vfs: ready`, ...), and in an isolation test that runs system/services/init: with
// no system/services/init:ial-ramdisk the spawns simply no-op rather than deranging the heartbeat.
// One endpoint carries everything init waits on: children's exit
// notifications (they are spawned supervised against it), init's own
// signals, and power events it subscribes to. All arrive in the loop below.
supervision_endpoint = runtime.ipc.createIpcEndpoint() orelse {
_ = runtime.system.write("/system/services/init: no endpoint\n");
return;
};
_ = runtime.process.bindSignals(supervision_endpoint);
// Bring up the boot services, supervised so init can stop them cleanly.
// Best-effort and silent: each service announces its own readiness, and in
// an isolation test with no initial-ramdisk the spawns simply no-op.
for (boot_services) |service| {
_ = runtime.system.spawn(service);
if (runtime.system.spawnSupervised(service, &.{}, supervision_endpoint)) |id| {
children[child_count] = id;
child_count += 1;
}
}
// Subscribe to power events (retry: the power service registers well after
// init starts). Best-effort — without it, a `terminate` signal still
// triggers the same shutdown path.
subscribePower();
// A re-arming timer drives the liveness heartbeat: proof PID 1 is alive
// (the init test's marker) while the loop stays free to receive signals,
// power events, and children's exit notifications.
_ = runtime.system.timerOnce(supervision_endpoint, 1000);
var receive: [power.message_maximum]u8 = undefined;
while (true) {
_ = runtime.system.write("/system/services/init: heartbeat\n");
runtime.system.sleep(1000);
const got = runtime.ipc.replyWait(supervision_endpoint, &.{}, &receive, null);
if (runtime.process.signalsFrom(got.badge)) |signals| {
if (signals.has(.terminate)) shutDown();
continue;
}
if (got.isTimer()) {
_ = runtime.system.write("/system/services/init: heartbeat\n");
_ = runtime.system.timerOnce(supervision_endpoint, 1000);
continue;
}
if (got.isMessage() and got.len >= 2 and receive[0] == @intFromEnum(power.Operation.event)) {
// A power event (the only buffered messages init receives).
if (receive[1] == @intFromEnum(power.Event.power_button)) shutDown();
continue;
}
// Child-exit notifications and anything else: keep waiting.
if (got.isNotification()) continue;
}
}
/// Look up the power service and subscribe our endpoint (handed over as the
/// call's capability) so events arrive as buffered messages here.
fn subscribePower() void {
var handle: ?runtime.ipc.Handle = null;
var tries: u32 = 0;
while (handle == null and tries < 200) : (tries += 1) {
handle = runtime.ipc.lookup(.power);
if (handle == null) runtime.system.sleep(20);
}
// A missing power service is not fatal — init proceeds to its heartbeat and
// a `terminate` signal still drives shutdown. Silent so the no-ramdisk init
// test's heartbeat marker is the next line written.
const h = handle orelse return;
const request = power.Subscribe{};
var reply: [power.message_maximum]u8 = undefined;
_ = runtime.ipc.callCap(h, std.mem.asBytes(&request), &reply, supervision_endpoint) catch {};
}
/// The stop sequence: terminate each child in reverse spawn order (vfs last —
/// other services may flush through it), waiting up to a deadline for each to
/// exit before killing it, then ask the power service to enter S5.
fn shutDown() void {
_ = runtime.system.write("/system/services/init: shutting down\n");
var i = child_count;
while (i > 0) {
i -= 1;
if (children[i] != 0) runtime.process.stop(children[i], 2000, supervision_endpoint);
}
if (runtime.ipc.lookup(.power)) |h| {
const request = power.Shutdown{};
var reply: [power.message_maximum]u8 = undefined;
_ = runtime.ipc.call(h, std.mem.asBytes(&request), &reply) catch {};
}
// If S5 did not take, init has nothing left to do but idle.
while (true) runtime.system.sleep(1000);
}
pub const panic = runtime.panic;
comptime {
_ = &runtime.start._start; // pull the runtime entry shim into the image
+68
View File
@@ -0,0 +1,68 @@
//! The power protocol (docs/m21-plan.md): system power's domain-named surface,
//! registered under `ServiceId.power`. On x86 the acpi service serves it; on
//! ARM a PSCI/mailbox service will register the same id — subscribers never
//! learn which firmware they are on (m19-m20-plan.md decision 7). The
//! vfs-protocol pattern: extern-struct messages, a version, reserved fields.
/// The protocol version a client states nowhere yet — reserved for the day a
/// handshake needs it; requests carry it so a mismatch can be refused loudly.
pub const version: u16 = 1;
pub const Operation = enum(u8) {
/// Subscribe to power events: the subscriber's endpoint rides as the
/// call's capability (the input/device-manager pattern); events arrive on
/// it as buffered messages carrying an `EventMessage`.
subscribe = 1,
/// Orderly shutdown's last step: enter S5. Accepted only from PID 1
/// (init) — the process that has already run the stop sequence over
/// everything else.
shutdown = 2,
/// The published event payload (never sent *to* the service).
event = 3,
};
/// What happened. The vocabulary is hardware-neutral: a lid is a lid whether
/// ACPI or a PSCI mailbox reported it.
pub const Event = enum(u8) {
power_button = 1,
lid = 2,
ac = 3,
battery = 4,
/// A device notification that maps to none of the named events — the
/// `code` and `hid` fields say which device and what code.
notify = 5,
};
pub const Subscribe = extern struct {
operation: u8 = @intFromEnum(Operation.subscribe),
reserved0: u8 = 0,
version: u16 = version,
reserved1: u32 = 0,
};
pub const Shutdown = extern struct {
operation: u8 = @intFromEnum(Operation.shutdown),
reserved0: u8 = 0,
version: u16 = version,
reserved1: u32 = 0,
};
/// A published event, as the buffered-message payload subscribers receive.
pub const EventMessage = extern struct {
operation: u8 = @intFromEnum(Operation.event),
/// An Event value.
event: u8,
reserved0: u16 = 0,
/// The device notification code (Notify's second argument), or 0.
code: u32 = 0,
/// The notifying device's hardware id (EISA-decoded), or all zero.
hid: [8]u8 = .{0} ** 8,
};
pub const Reply = extern struct {
status: i32,
reserved: u32 = 0,
};
/// Upper bound on any message in this protocol — sizes endpoint buffers.
pub const message_maximum = 64;