const std = @import("std"); const danos = @import("danos"); const arch = @import("arch"); const console = @import("console.zig"); const log = @import("log.zig"); const pmm = @import("pmm.zig"); const heap = @import("heap.zig"); const scheduler = @import("scheduler.zig"); const platform = @import("platform"); const tests = @import("tests.zig"); const build_options = @import("build_options"); const BootInfo = danos.BootInfo; /// The calling convention used to enter the kernel. Pinned to SysV explicitly: /// the bootloader is built for the UEFI target, whose C convention is Microsoft /// x64 (first argument in RCX), while the kernel is SysV (first argument in /// RDI). Both sides reference this so the `boot_info` pointer lands in the /// register the other expects. `danos.kernel_abi` re-exports it to the loader. pub const kernel_abi = danos.kernel_abi; // POST/checkpoint codes emitted to I/O port 0x80 at boot milestones — the // last-resort progress signal on a machine with no text output at all. const cp_entry = 0x10; const cp_paging = 0x20; const cp_heap = 0x30; const cp_discovery = 0x40; const cp_scheduler = 0x50; const cp_timer = 0x60; const cp_running = 0x70; const cp_exception = 0xE0; const cp_panic = 0xEE; /// Physical address of the low page reserved at boot for the AP trampoline (0 = none /// was available). Claimed right after the frame allocator comes up, before paging /// and the heap consume the scarce sub-1 MiB frames. var ap_trampoline_page: u64 = 0; /// Kernel entry point. The bootloader jumps here after `ExitBootServices` with a /// pointer to the handoff data. There is no runtime, no stack unwinding, and no /// caller to return to, so this never returns. export fn _start(boot_info: *const BootInfo) callconv(kernel_abi) noreturn { kmain(boot_info); } fn kmain(boot_info: *const BootInfo) noreturn { // The **log** is the machine-readable diagnostic stream: it fans out to every // *diagnostic* channel that exists (serial, the 0xE9 debug console, and later a // file on a ramdisk/USB/SSD), so a message survives as long as any is present. // A headless, serial-less machine still boots correctly — it just goes quiet, // with port-0x80 checkpoints as the only progress signal. arch.serialInit(); log.addSink(arch.serialWrite); if (arch.debugconPresent()) log.addSink(arch.debugconWrite); // The **framebuffer** is deliberately *not* a log sink. It's a separate output // surface — a bootstrap text console today, a graphics device driver later — so // we never assume the OS is text-based. Only a few user-facing status lines // (via `status`) and panics are mirrored to it; the verbose log stays out. const fb = boot_info.framebuffer; console.init(fb); log.checkpoint(cp_entry); // Catch CPU exceptions before doing anything that might fault: install our // reporter, then bring up the GDT + IDT. arch.setFaultHandler(onException); arch.init(); status("danos: initialising kernel...\n"); log.write(if (console.present()) "danos: framebuffer console online (bootstrap; graphics driver later)\n" else "danos: no framebuffer (headless) -> logging to serial/debugcon only\n"); log.write("danos: cpu tables online (GDT, IDT, TSS)\n"); log.print(" resolution : {d}x{d}\n", .{ fb.width, fb.height }); log.print(" pitch : {d} bytes\n", .{fb.pitch}); log.print(" format : {s}\n", .{@tagName(fb.format)}); log.print(" framebuffer: 0x{x:0>16}\n", .{fb.base}); log.print (" footprint : {d} MiB\n", .{(fb.pitch * fb.height) / (1024 * 1024)}); // Summarise the physical memory the loader handed us. The array is danos's // own MemoryRegion, so this is a plain slice — no firmware layout in sight. const regions = @as([*]const danos.MemoryRegion, @ptrFromInt(boot_info.memory_map.regions))[0..boot_info.memory_map.len]; var usable_pages: u64 = 0; var reserved_pages: u64 = 0; // reserved RAM only — MMIO is device space, not RAM for (regions) |r| { switch (r.kind) { .usable => usable_pages += r.pages, .reserved, .acpi_tables, .acpi_nvs => reserved_pages += r.pages, .mmio => {}, } } const total_pages = usable_pages + reserved_pages; const total_bytes = total_pages * danos.page_size; const gib = 1 << 30; log.write("\ndanos: physical memory\n"); log.print(" total RAM : {d}.{d:0>2} GiB ({d} MiB) - RAM the firmware reported\n", .{ total_bytes / gib, (total_bytes % gib) * 100 / gib, mib(total_pages) }); log.print(" usable : {d} MiB - free RAM (incl. reclaimed boot-services memory)\n", .{mib(usable_pages)}); log.print(" reserved : {d} MiB - kernel image, boot stack, ACPI, runtime services\n", .{mib(reserved_pages)}); log.print(" regions : {d} - entries in the firmware memory map\n", .{regions.len}); // Bring up the physical frame allocator over that map, and prove it works: // allocate three frames, then hand them back. pmm.init(boot_info.memory_map); // Claim the AP trampoline's low (<1 MiB) page *now*, before paging and the heap // draw down sub-1 MiB frames (the allocator scans upward from frame 0). Held // until SMP bring-up; 0 means none was available (we stay uniprocessor). ap_trampoline_page = pmm.allocBelow(0x100000) orelse 0; const s1 = pmm.stats(); log.print("\ndanos: frame allocator online\n", .{}); log.print(" free frames: {d} ({d} MiB)\n", .{ s1.free_frames, mib(s1.free_frames) }); const f0 = pmm.alloc(); const f1 = pmm.alloc(); const f2 = pmm.alloc(); log.print(" alloc x3 : 0x{x} 0x{x} 0x{x}\n", .{ f0 orelse 0, f1 orelse 0, f2 orelse 0 }); if (f0) |p| pmm.free(p); if (f1) |p| pmm.free(p); if (f2) |p| pmm.free(p); log.print(" after free : {d} frames free\n", .{pmm.stats().free_frames}); // Switch off the firmware's page tables onto our own (with real permissions). arch.enablePaging(pmm.alloc, boot_info); log.checkpoint(cp_paging); log.print("\ndanos: paging enabled\n", .{}); log.print(" page tables: CR3 = 0x{x:0>16}\n", .{arch.readCr3()}); log.print(" kernel segs: {d} (mapped with W^X permissions)\n", .{boot_info.kernel_segment_count}); // Bring up the kernel heap (dynamic allocation), built on the VMM. heap.init(); log.checkpoint(cp_heap); log.write("\ndanos: kernel heap online\n"); // Measure the amount of resources the kernel is actually using const s2 = pmm.stats(); log.print(" Kernel footprint: {d} KiB\n", .{kib(s1.free_frames - s2.free_frames)}); // Enumerate hardware from the firmware tables (ACPI here) into a generic // device tree, then list it. Discovery walks ACPI memory directly (identity- // mapped) and maps PCIe config space on demand via the VMM. A failure here is // not fatal yet — log it and carry on. const hal = platform.Hal{ .mapMmio = arch.mapPage, .pioRead = arch.pioRead, .pioWrite = arch.pioWrite, }; if (platform.discover(boot_info, heap.allocator(), hal)) |devtree| { var dt = devtree; log.write("\ndanos: device discovery online\n"); dt.dump(log.write); // Power register map extracted from the FADT + AML, for confidence it parsed. const pw = platform.powerInfo(); log.write("danos: power\n"); log.print(" pm1a_cnt : {s} 0x{x} (width {d})\n", .{ if (pw.pm1a_cnt.mmio) "mmio" else "io", pw.pm1a_cnt.address, pw.pm1a_cnt.width }); if (pw.s5) |s| { log.print(" S5 slp_typ : a={d} b={d}\n", .{ s.slp_typ_a, s.slp_typ_b }); } else { log.write(" S5 slp_typ : (not found)\n"); } log.print(" reset : supported={} {s} 0x{x} val 0x{x}\n", .{ pw.reset_supported, if (pw.reset.mmio) "mmio" else "io", pw.reset.address, pw.reset_value }); // AML namespace parse integrity: consumed should equal total. const am = platform.amlStats(); log.print(" aml : {d} namespace nodes, parsed {d}/{d} bytes\n", .{ am.nodes, am.consumed, am.total }); // Feed the arch layer the discovered addresses/facts so it makes no legacy // assumptions — the point of all this on UEFI Class 3 firmware. MMIO bases // (HPET, I/O APIC) come from the device tree; scalar facts from ACPI. const pinfo = platform.platformInfo(); const hpet_base: u64 = if (dt.firstOfClass(.timer)) |t| (if (t.firstResource(.memory)) |r| r.start else 0) else 0; var ioapic_base: u64 = 0; var ioapic_gsi: u32 = 0; if (dt.firstOfClass(.interrupt_controller)) |ic| { if (ic.firstResource(.memory)) |r| ioapic_base = r.start; if (ic.firstResource(.irq)) |r| ioapic_gsi = @intCast(r.start); } var isos: [16]arch.IsoEntry = undefined; const iso_n = @min(pinfo.override_count, isos.len); for (0..iso_n) |i| isos[i] = .{ .source = pinfo.overrides[i].source, .gsi = pinfo.overrides[i].gsi, .flags = pinfo.overrides[i].flags, }; const pm_timer: ?arch.PmTimer = if (pinfo.pm_timer.present()) .{ .mmio = pinfo.pm_timer.mmio, .address = pinfo.pm_timer.address, .is_32bit = pinfo.pm_timer_32bit } else null; arch.configurePlatform(.{ .pic_present = pinfo.pic_present, .hpet_base = hpet_base, .pm_timer = pm_timer, .ioapic_base = ioapic_base, .ioapic_gsi_base = ioapic_gsi, .overrides = isos[0..iso_n], }); if (pinfo.spcr_uart) |u| arch.serialReconfigure(u.mmio, u.address); log.write("danos: platform\n"); log.print(" 8259 PIC : {s}\n", .{if (pinfo.pic_present) "present" else "absent"}); log.print(" lapic base : 0x{x}\n", .{pinfo.lapic_base}); log.print(" hpet base : 0x{x}\n", .{hpet_base}); log.print(" pm timer : {s} 0x{x} ({s})\n", .{ if (pinfo.pm_timer.mmio) "mmio" else "io", pinfo.pm_timer.address, if (pinfo.pm_timer_32bit) "32-bit" else "24-bit" }); if (pinfo.spcr_uart) |u| { log.print(" console UART: {s} 0x{x} (SPCR type {d})\n", .{ if (u.mmio) "mmio" else "io", u.address, pinfo.spcr_kind }); } else { log.write(" console UART: none in SPCR -> legacy COM1\n"); } log.print(" ioapic : base 0x{x}, {d} inputs (masked); entry0 low 0x{x}\n", .{ ioapic_base, arch.ioapicEntryCount(), arch.ioapicEntryLow(0) }); const cores = platform.cpus(); log.print(" cpus : {d} usable core(s); 1 running (BSP), {d} AP(s) parked (SMP bring-up pending)\n", .{ cores.len, if (cores.len > 0) cores.len - 1 else 0 }); if (platform.cpusDropped() > 0) log.print(" cpus : WARNING {d} core(s) beyond pool cap dropped\n", .{platform.cpusDropped()}); } else |err| { log.print("\ndanos: device discovery failed: {s}\n", .{@errorName(err)}); } log.checkpoint(cp_discovery); // Register the current context as the first task before enabling preemption. scheduler.init(4); log.checkpoint(cp_scheduler); log.write("\ndanos: scheduler online\n"); // Start the timer and unmask interrupts — the kernel now has a heartbeat, and // the timer preempts among tasks. arch.startTimer(); arch.enableInterrupts(); log.checkpoint(cp_timer); log.print("danos: timer online ({d} Hz tick; LAPIC {d} MHz, TSC {d} MHz; calibrated via {s})\n", .{ arch.timer_hz, arch.lapicHz() / 1_000_000, arch.tscHz() / 1_000_000, arch.timerCalibrationSource() }); // Wake the other cores (application processors). A no-op on a single-core // machine; on SMP each AP climbs to long mode and reports in (docs/smp.md). bringUpSecondaries(); // In a test build (`zig build -Dtest-case=`), run that case and stop. // Normal builds fall through to the idle halt. if (build_options.test_case) |case| { tests.run(case, boot_info); arch.halt(); } log.checkpoint(cp_running); status("kernel initialised.\n"); // TODO: init process status("\nnothing left to do; halting CPU.\n"); arch.halt(); } /// Wake the application processors the firmware left parked. Allocates the low /// trampoline page (and makes it executable), then wakes each non-boot core in turn, /// handing it a fresh kernel stack and its per-CPU slot. Cores that don't report in /// are left parked — the running system is unaffected. See docs/smp.md. fn bringUpSecondaries() void { const cores = platform.cpus(); if (cores.len <= 1) return; // A low (<1 MiB) frame was reserved at boot for the real-mode trampoline (a SIPI // vector addresses it). It's kept for the system's life — armed only during a // wake, inert (zeroed, non-executable) otherwise — so cores can be re-woken later. if (ap_trampoline_page == 0) { log.write("danos: smp: no low page for the AP trampoline; staying uniprocessor\n"); return; } arch.setTrampolinePage(ap_trampoline_page); arch.setSecondaryEntry(scheduler.secondaryMain); // where a woken core joins the run loop log.print("\ndanos: bringing up {d} application processor(s)\n", .{cores.len - 1}); const max_wake_attempts = 3; // a core that misses the first INIT-SIPI-SIPI gets retried for (cores[1..], 1..) |core, index| { const stack = heap.allocator().alloc(u8, 16 * 1024) catch { log.print(" cpu apic_id {d}: no stack; skipped\n", .{core.apic_id}); continue; }; const stack_top = (@intFromPtr(stack.ptr) + stack.len) & ~@as(usize, 15); const pc = scheduler.prepareSecondary(index, core.apic_id); var attempt: u32 = 1; while (attempt <= max_wake_attempts) : (attempt += 1) { if (arch.startSecondary(core.apic_id, stack_top, @intFromPtr(pc), index)) { pc.online = true; log.print(" cpu apic_id {d}: online (attempt {d})\n", .{ core.apic_id, attempt }); break; } if (attempt == max_wake_attempts) log.print(" cpu apic_id {d}: no response after {d} attempts (parked)\n", .{ core.apic_id, max_wake_attempts }); } } log.print("danos: {d}/{d} cores online\n", .{ scheduler.onlineCount(), cores.len }); } /// A user-facing status line: to the diagnostic `log` *and* the on-screen console /// (if a framebuffer is present). The verbose log uses `log.*` directly and never /// touches the framebuffer. fn status(msg: []const u8) void { log.write(msg); console.write(msg); } fn statusPrint(comptime fmt: []const u8, args: anytype) void { var buf: [256]u8 = undefined; status(std.fmt.bufPrint(&buf, fmt, args) catch return); } /// Frames (4 KiB pages) to whole MiB. fn mib(pages: u64) u64 { return pages * danos.page_size / (1024 * 1024); } fn kib(frames: u64) u64 { return frames * danos.page_size / (1024); } /// Report a CPU exception and halt. There's no fault recovery yet, so any /// exception is terminal — but it reports what and where (to every output sink, /// plus a POST code and a persistent breadcrumb) instead of silently resetting. fn onException(state: *const arch.CpuState) noreturn { log.checkpoint(cp_exception); // A fault is user-facing enough to paint on screen too (via statusPrint), on // top of the diagnostic log. statusPrint("\nCPU EXCEPTION: {s} (vector {d})\n", .{ arch.vectorName(state.vector), state.vector }); statusPrint(" error code : 0x{x}\n", .{state.error_code}); statusPrint(" RIP : 0x{x:0>16}\n", .{state.rip}); statusPrint(" RSP : 0x{x:0>16}\n", .{state.rsp}); if (state.vector == 14) statusPrint(" CR2 (addr) : 0x{x:0>16}\n", .{arch.readCr2()}); var buf: [128]u8 = undefined; log.recordPanic(std.fmt.bufPrint(&buf, "CPU exception {s} (vector {d}) at RIP 0x{x}", .{ arch.vectorName(state.vector), state.vector, state.rip }) catch "cpu exception"); arch.halt(); } /// Freestanding has no OS to receive a panic. Emit it to every output sink, drop a /// POST code + a persistent breadcrumb (so a post-mortem can recover it even with /// no live console), then halt. Assumes no console — the sinks self-guard. pub const panic = std.debug.FullPanic(struct { fn panic(msg: []const u8, first_trace_addr: ?usize) noreturn { _ = first_trace_addr; log.checkpoint(cp_panic); log.recordPanic(msg); status("\nKERNEL PANIC: "); status(msg); status("\n"); arch.halt(); } }.panic);