moving kernel code to kernel/

This commit is contained in:
2026-07-05 10:19:11 +01:00
parent 7c3cffb337
commit 6f6ccc8bc9
37 changed files with 63 additions and 61 deletions
+202
View File
@@ -0,0 +1,202 @@
//! Local APIC and its timer — the source of device interrupts.
//!
//! Modern x86 routes interrupts through the per-CPU Local APIC (the legacy 8259
//! PIC is remapped out of the way and masked). The LAPIC also has a built-in
//! timer, which is the simplest device interrupt to bring up: it needs no
//! external routing, just a vector and a count. We use it as danos's heartbeat.
//!
//! The LAPIC is memory-mapped (default physical 0xFEE00000, inside our identity
//! map). Every interrupt must be acknowledged with an end-of-interrupt write, or
//! the LAPIC won't deliver the next one.
const io = @import("io.zig");
/// IDT vector the timer fires on (in the device range, >= 32).
pub const timer_vector = 32;
/// Spurious-interrupt vector. Low nibble 0xF by convention; also in our gate
/// range so a stray spurious interrupt lands on a valid (no-op) handler.
const spurious_vector = 47;
// LAPIC register offsets.
const reg_spurious = 0x0F0;
const reg_eoi = 0x0B0;
const reg_lvt_timer = 0x320;
const reg_timer_initial = 0x380;
const reg_timer_current = 0x390;
const reg_timer_divide = 0x3E0;
const lvt_masked = 1 << 16;
const lvt_periodic = 1 << 17;
const timer_divide_16 = 0x3;
const ia32_apic_base_msr = 0x1B;
/// LAPIC MMIO base. A runtime var (not a constant) both because we read it from
/// the MSR and so register writes compile to normal stores rather than a
/// `mov moffs`, which the self-hosted backend can't encode.
var base: usize = 0xFEE00000;
var tick_count: u64 = 0;
/// LAPIC timer counts per millisecond, measured against the PIT (see calibrate).
/// At divide-by-16, this is the effective counting rate.
var ticks_per_ms: u32 = 0;
/// The periodic-interrupt frequency the timer is armed at, once initTimer runs.
var timer_hz: u32 = 0;
/// TSC (Time Stamp Counter) calibration: cycles per second, and the count at boot.
/// The TSC is a per-core cycle counter, giving a ~nanosecond high-resolution
/// monotonic clock — far finer than the millisecond timer tick.
var tsc_hz: u64 = 0;
var tsc_base: u64 = 0;
/// Read the 64-bit Time Stamp Counter.
fn rdtsc() u64 {
var low: u32 = undefined;
var high: u32 = undefined;
asm volatile ("rdtsc"
: [low] "={eax}" (low),
[high] "={edx}" (high),
);
return (@as(u64, high) << 32) | low;
}
fn read(reg: u32) u32 {
return @as(*volatile u32, @ptrFromInt(base + reg)).*;
}
fn write(reg: u32, value: u32) void {
@as(*volatile u32, @ptrFromInt(base + reg)).* = value;
}
/// Move the legacy 8259 PIC's vectors to 0x20-0x2F (clear of the CPU exception
/// vectors) and mask every line, so it can't deliver interrupts behind the APIC.
fn remapAndMaskPic() void {
io.outb(0x20, 0x11); // start init (cascade mode)
io.outb(0xA0, 0x11);
io.outb(0x21, 0x20); // master offset 0x20
io.outb(0xA1, 0x28); // slave offset 0x28
io.outb(0x21, 0x04); // tell master about slave on IRQ2
io.outb(0xA1, 0x02);
io.outb(0x21, 0x01); // 8086 mode
io.outb(0xA1, 0x01);
io.outb(0x21, 0xFF); // mask all
io.outb(0xA1, 0xFF);
}
/// Enable the Local APIC: mask the PIC, set the global-enable MSR bit, and
/// software-enable the APIC via its spurious-vector register.
pub fn init() void {
remapAndMaskPic();
const msr = io.rdmsr(ia32_apic_base_msr);
base = @intCast(msr & 0xFFFFF000); // physical base is bits 12+
io.wrmsr(ia32_apic_base_msr, msr | (1 << 11)); // global enable
write(reg_spurious, 0x100 | spurious_vector); // bit 8 = software enable
}
/// Measure the LAPIC timer's and the TSC's rates against the PIT (channel 2, which
/// can be polled without interrupts). We run the LAPIC timer one-shot from its max
/// count and snapshot the TSC while the PIT counts out a known 10 ms, then see how
/// far each got. This gives real time, which the RTOS timing guarantees depend on.
pub fn calibrate() void {
const pit_hz = 1_193_182;
const calib_ms = 10;
const pit_count: u16 = @intCast(pit_hz / 1000 * calib_ms);
// LAPIC timer: divide 16, masked (no interrupt — we just want the count),
// counting down from the maximum.
write(reg_timer_divide, timer_divide_16);
write(reg_lvt_timer, lvt_masked);
write(reg_timer_initial, 0xFFFFFFFF);
// PIT channel 2, mode 0 (interrupt on terminal count): load the count with the
// gate low, then raise the gate to start it counting.
io.outb(0x61, io.inb(0x61) & 0xFC); // speaker off, gate low
io.outb(0x43, 0xB0); // channel 2, lo/hi byte, mode 0
io.outb(0x42, @truncate(pit_count));
io.outb(0x42, @truncate(pit_count >> 8));
const tsc_start = rdtsc();
io.outb(0x61, (io.inb(0x61) & 0xFC) | 0x01); // gate high -> start
while (io.inb(0x61) & 0x20 == 0) {} // poll channel-2 output until terminal count
const tsc_end = rdtsc();
const elapsed = 0xFFFFFFFF - read(reg_timer_current);
write(reg_timer_initial, 0); // stop the timer
ticks_per_ms = elapsed / calib_ms;
tsc_hz = (tsc_end -% tsc_start) * (1000 / calib_ms); // cycles/10ms -> cycles/s
tsc_base = rdtsc(); // the clock's zero point (boot)
}
/// Arm the LAPIC timer to fire on `timer_vector` at `hz` (periodic). Requires
/// calibrate() to have run.
pub fn initTimer(hz: u32) void {
timer_hz = hz;
const count = @as(u64, ticks_per_ms) * 1000 / hz; // counts per (1/hz) second
write(reg_timer_divide, timer_divide_16);
write(reg_lvt_timer, timer_vector | lvt_periodic);
write(reg_timer_initial, @intCast(count));
}
/// Configured periodic-interrupt frequency (Hz).
pub fn frequencyHz() u32 {
return timer_hz;
}
/// Measured LAPIC timer frequency (Hz), for reporting/sanity checks.
pub fn lapicHz() u64 {
return @as(u64, ticks_per_ms) * 1000;
}
/// Measured TSC frequency (Hz).
pub fn tscHz() u64 {
return tsc_hz;
}
// Monotonic high-resolution clock, from the TSC. A function per resolution, each
// scaling the cycle delta directly at its unit (the 128-bit intermediate avoids
// overflow across a long uptime). nanos() resolves to a few ns; millis() is what
// the scheduler uses for sleep deadlines.
pub fn nanos() u64 {
if (tsc_hz == 0) return 0;
return @intCast(@as(u128, rdtsc() -% tsc_base) * 1_000_000_000 / tsc_hz);
}
pub fn micros() u64 {
if (tsc_hz == 0) return 0;
return @intCast(@as(u128, rdtsc() -% tsc_base) * 1_000_000 / tsc_hz);
}
pub fn millis() u64 {
if (tsc_hz == 0) return 0;
return @intCast(@as(u128, rdtsc() -% tsc_base) * 1_000 / tsc_hz);
}
/// Acknowledge the current interrupt so the LAPIC will deliver the next one.
pub fn eoi() void {
write(reg_eoi, 0);
}
/// Optional callback run each tick (the scheduler registers it for preemption).
var on_tick: ?*const fn () void = null;
pub fn setTickHook(hook: *const fn () void) void {
on_tick = hook;
}
/// The timer interrupt handler: advance the monotonic tick count, then run the
/// tick hook (which may switch tasks). The interrupt is already acknowledged by
/// the dispatcher before we get here, so a task switch here doesn't stall it.
pub fn timerTick() void {
tick_count +%= 1;
if (on_tick) |hook| hook();
}
/// Number of timer ticks so far. Volatile load: the count is bumped
/// asynchronously by the interrupt handler, so callers must re-read memory.
pub fn ticks() u64 {
return @as(*const volatile u64, &tick_count).*;
}
+192
View File
@@ -0,0 +1,192 @@
//! x86_64 CPU operations. This is the "arch" module: the generic kernel imports
//! it as `@import("arch")` and never names x86_64 directly, so a second
//! architecture is added by pointing that module at a different directory in
//! build.zig — no change to the generic code. Keep everything CPU-specific here
//! (halt, the descriptor tables, later paging), and nothing generic.
const danos = @import("danos");
const gdt = @import("gdt.zig");
const tss = @import("tss.zig");
const idt = @import("idt.zig");
const paging = @import("paging.zig");
const serial = @import("serial.zig");
const apic = @import("apic.zig");
/// The saved register/trap frame passed to a fault handler.
pub const CpuState = idt.CpuState;
/// Bring up the serial port (the kernel's machine-readable log). No dependencies,
/// so it can be the very first thing called.
pub fn serialInit() void {
serial.init();
}
/// Write bytes to the serial port.
pub fn serialWrite(bytes: []const u8) void {
serial.write(bytes);
}
/// Set up the CPU's descriptor tables: our own GDT, the TSS (with an interrupt
/// stack for double faults), then the IDT with exception handlers. After this a
/// CPU fault is reported instead of triple-faulting. Install the fault handler
/// (setFaultHandler) first so early faults are caught.
pub fn init() void {
gdt.init();
tss.init();
idt.init();
}
/// Build the kernel's own page tables (with real permissions) and switch onto
/// them. Needs the frame allocator and the boot info (for the memory map and the
/// kernel's segment layout). Call once the frame allocator is up.
pub fn enablePaging(allocFrame: *const fn () ?u64, boot_info: *const danos.BootInfo) void {
paging.init(allocFrame, boot_info);
}
/// Map a page into the kernel address space (non-executable). For the heap, etc.
pub fn mapPage(virt: u64, phys: u64, writable: bool) void {
paging.map(virt, phys, writable);
}
/// Remove a kernel mapping.
pub fn unmapPage(virt: u64) void {
paging.unmap(virt);
}
/// CR3 holds the physical address of the active top-level page table.
pub fn readCr3() u64 {
return asm volatile ("mov %%cr3, %[out]"
: [out] "=r" (-> u64),
);
}
/// Kernel tick rate: 1000 Hz (1 ms), the scheduler's time quantum.
pub const timer_hz = 1000;
/// Enable the Local APIC, calibrate its timer against the PIT, and start it firing
/// at `timer_hz` — the kernel's real-time heartbeat. Interrupts still have to be
/// unmasked with enableInterrupts() to be delivered.
pub fn startTimer() void {
apic.init();
apic.calibrate();
idt.setHandler(apic.timer_vector, apic.timerTick);
apic.initTimer(timer_hz);
}
/// Number of timer ticks since startTimer().
pub fn ticks() u64 {
return apic.ticks();
}
// Monotonic high-resolution clock (from the TSC), one function per resolution.
pub fn nanos() u64 {
return apic.nanos();
}
pub fn micros() u64 {
return apic.micros();
}
pub fn millis() u64 {
return apic.millis();
}
/// Measured LAPIC timer / TSC frequencies in Hz (from calibration).
pub fn lapicHz() u64 {
return apic.lapicHz();
}
pub fn tscHz() u64 {
return apic.tscHz();
}
/// Unmask maskable interrupts (`sti`) so device interrupts get delivered.
pub fn enableInterrupts() void {
asm volatile ("sti");
}
/// Mask maskable interrupts (`cli`).
pub fn disableInterrupts() void {
asm volatile ("cli");
}
/// Disable interrupts and return the previous flags, so a nested critical section
/// can restore the caller's state rather than blindly re-enabling. Pairs with
/// restoreInterrupts.
pub fn saveInterrupts() u64 {
var flags: u64 = undefined;
asm volatile (
\\pushfq
\\pop %[f]
\\cli
: [f] "=r" (flags),
:
: .{ .memory = true }
);
return flags;
}
/// Re-enable interrupts only if they were enabled when `flags` was captured.
pub fn restoreInterrupts(flags: u64) void {
if (flags & 0x200 != 0) asm volatile ("sti" ::: .{ .memory = true }); // bit 9 = IF
}
/// Register a callback the timer interrupt invokes each tick (e.g. the scheduler).
pub fn setTickHook(hook: *const fn () void) void {
apic.setTickHook(hook);
}
// --- context switching (for the scheduler) -------------------------------
/// Save the current task's registers/stack and resume `new_rsp`; the old stack
/// pointer is written to `old_rsp`. Defined in isr.s.
extern fn switch_context(old_rsp: *usize, new_rsp: usize) callconv(.c) void;
pub fn switchContext(old_rsp: *usize, new_rsp: usize) void {
switch_context(old_rsp, new_rsp);
}
/// Build the initial stack for a new task so that switching to it lands in
/// `task_trampoline`, which then calls `entry`. Returns the saved stack pointer.
/// The layout must match switch_context's push order (callee-saved, then the
/// return address on top); `entry` is smuggled in via the r15 slot.
pub fn initTaskStack(stack_top: usize, entry: usize) usize {
const trampoline = @extern(*const anyopaque, .{ .name = "task_trampoline" });
var sp = stack_top;
const push = struct {
fn f(p: *usize, value: usize) void {
p.* -= @sizeOf(usize);
@as(*usize, @ptrFromInt(p.*)).* = value;
}
}.f;
push(&sp, @intFromPtr(trampoline)); // return address for switch_context's `ret`
push(&sp, 0); // rbx
push(&sp, 0); // rbp
push(&sp, 0); // r12
push(&sp, 0); // r13
push(&sp, 0); // r14
push(&sp, entry); // r15 -> task entry, read by task_trampoline
return sp;
}
/// Route CPU exceptions to `handler`, which receives the trap frame and does not
/// return. Until set, faults just halt the core.
pub fn setFaultHandler(handler: *const fn (*const CpuState) noreturn) void {
idt.on_fault = handler;
}
/// A human-readable name for a CPU exception vector.
pub fn vectorName(vector: u64) []const u8 {
return idt.vectorName(vector);
}
/// CR2 holds the faulting linear address after a page fault (#PF, vector 14).
pub fn readCr2() u64 {
return asm volatile ("mov %%cr2, %[out]"
: [out] "=r" (-> u64),
);
}
/// Park the core forever. `hlt` drops it into a low-power idle until the next
/// interrupt; the loop re-halts on every wake so the stop is permanent. See
/// docs/halting.md for the full reasoning.
pub fn halt() noreturn {
while (true) asm volatile ("hlt");
}
+54
View File
@@ -0,0 +1,54 @@
//! Global Descriptor Table. In long mode segmentation is mostly vestigial, but
//! the CPU still needs valid code/data segment descriptors, and the IDT's gates
//! reference a code selector — so we install our own flat GDT with known
//! selectors (0x08 kernel code, 0x10 kernel data) rather than trusting whatever
//! the firmware left in place.
/// Selectors into the table below (index * 8).
pub const kernel_code = 0x08;
pub const kernel_data = 0x10;
pub const tss_selector = 0x18;
/// Flat 64-bit descriptors. Base/limit are ignored in long mode; what matters is
/// the access byte and, for code, the long-mode (L) flag.
/// code: present, ring 0, executable, readable, L=1 -> 0x00AF9A00_0000FFFF
/// data: present, ring 0, writable -> 0x00CF9200_0000FFFF
/// The last two slots hold one 16-byte TSS descriptor, filled in by setTss.
var table = [_]u64{
0, // null descriptor (required)
0x00AF9A000000FFFF, // kernel code (0x08)
0x00CF92000000FFFF, // kernel data (0x10)
0, // TSS descriptor low (0x18)
0, // TSS descriptor high
};
/// Fill the 64-bit TSS system descriptor (two GDT slots) so the task register can
/// point at our TSS. Type 0x89 = present, ring 0, available 64-bit TSS.
pub fn setTss(base: u64, limit: u64) void {
table[3] = (limit & 0xFFFF) |
((base & 0xFFFF) << 16) |
(((base >> 16) & 0xFF) << 32) |
(@as(u64, 0x89) << 40) |
(((limit >> 16) & 0xF) << 48) |
(((base >> 24) & 0xFF) << 56);
table[4] = (base >> 32) & 0xFFFFFFFF;
}
/// The operand `lgdt` wants: table byte-length minus one, then its address.
const Descriptor = packed struct {
limit: u16,
base: u64,
};
/// Loads the GDT and reloads the segment registers (including CS). Defined in
/// isr.s — it uses the selectors 0x08 (code) and 0x10 (data) that match `table`.
extern fn gdt_flush(descriptor: *const Descriptor) callconv(.c) void;
/// Install our GDT and switch onto its segments.
pub fn init() void {
const descriptor = Descriptor{
.limit = @sizeOf(@TypeOf(table)) - 1,
.base = @intFromPtr(&table),
};
gdt_flush(&descriptor);
}
+155
View File
@@ -0,0 +1,155 @@
//! Interrupt Descriptor Table, CPU-exception handlers, and device-interrupt
//! dispatch. Without this, any fault (a stray pointer, a bad page-table entry)
//! triple-faults and silently resets the machine. With it, the CPU vectors into
//! our stubs, which capture the register state and hand it to a dispatcher.
//!
//! Vectors split in two: 0-31 are CPU exceptions (terminal — reported and
//! halted); 32+ are device interrupts (a registered handler runs, the APIC is
//! acknowledged, and we return to the interrupted code).
const gdt = @import("gdt.zig");
const tss = @import("tss.zig");
const apic = @import("apic.zig");
/// Highest vector we install a gate/stub for (exceptions 0-31 plus the device
/// range 32-47, which covers the timer and the spurious vector).
const gate_count = 48;
/// A device-interrupt handler. It doesn't get the trap frame (a timer or keyboard
/// handler doesn't need the interrupted registers); add that if one ever does.
pub const Handler = *const fn () void;
var handlers = [_]?Handler{null} ** 256;
/// Register `handler` for a device-interrupt `vector` (>= 32).
pub fn setHandler(vector: usize, handler: Handler) void {
handlers[vector] = handler;
}
/// The register + trap frame the ISR stubs build on the stack, laid out so the
/// lowest address (where RSP points when we call the handler) is the first field.
/// See the push order in `isrCommon` below.
pub const CpuState = extern struct {
r15: u64,
r14: u64,
r13: u64,
r12: u64,
r11: u64,
r10: u64,
r9: u64,
r8: u64,
rbp: u64,
rdi: u64,
rsi: u64,
rdx: u64,
rcx: u64,
rbx: u64,
rax: u64,
vector: u64, // pushed by the per-vector stub
error_code: u64, // real one from the CPU, or 0 pushed by the stub
rip: u64, // from here down: pushed by the CPU on entry
cs: u64,
rflags: u64,
rsp: u64,
ss: u64,
};
/// Where a fault is reported. The kernel overrides this (see setFaultHandler) with
/// something that prints to the console; until then, just stop.
pub var on_fault: *const fn (*const CpuState) noreturn = defaultFault;
fn defaultFault(_: *const CpuState) noreturn {
while (true) asm volatile ("hlt");
}
/// Names for the 32 defined exception vectors, for readable output.
const names = [_][]const u8{
"divide error", "debug",
"NMI", "breakpoint",
"overflow", "bound range exceeded",
"invalid opcode", "device not available",
"double fault", "coprocessor segment overrun",
"invalid TSS", "segment not present",
"stack-segment fault", "general protection fault",
"page fault", "reserved (15)",
"x87 floating-point", "alignment check",
"machine check", "SIMD floating-point",
"virtualization", "control protection",
"reserved (22)", "reserved (23)",
"reserved (24)", "reserved (25)",
"reserved (26)", "reserved (27)",
"hypervisor injection", "VMM communication",
"security exception", "reserved (31)",
};
pub fn vectorName(vector: u64) []const u8 {
return if (vector < names.len) names[vector] else "unknown";
}
/// A 64-bit IDT gate descriptor (16 bytes).
const Gate = packed struct {
offset_low: u16,
selector: u16,
ist: u8, // interrupt-stack-table index; 0 = use the current stack
flags: u8, // present, DPL, gate type
offset_mid: u16,
offset_high: u32,
reserved: u32 = 0,
};
var idt = [_]Gate{std.mem.zeroes(Gate)} ** 256;
const Descriptor = packed struct {
limit: u16,
base: u64,
};
/// Loads the IDT (`lidt`). Defined in isr.s.
extern fn idt_flush(descriptor: *const Descriptor) callconv(.c) void;
fn setGate(vector: usize, handler: u64) void {
idt[vector] = .{
.offset_low = @truncate(handler),
.selector = gdt.kernel_code,
.ist = 0,
.flags = 0x8E, // present, ring 0, 64-bit interrupt gate
.offset_mid = @truncate(handler >> 16),
.offset_high = @truncate(handler >> 32),
};
}
/// Point every installed vector at its stub (isr.s) and load the IDT.
pub fn init() void {
@setEvalBranchQuota(20000); // comptimePrint across all the gates adds up
inline for (0..gate_count) |vector| {
const stub = @extern(*const anyopaque, .{ .name = std.fmt.comptimePrint("isr{d}", .{vector}) });
setGate(vector, @intFromPtr(stub));
}
// Run the double-fault handler (vector 8) on IST1: a #DF usually means the
// current stack is unusable, so it needs a guaranteed-good one. See tss.zig.
idt[8].ist = tss.double_fault_ist;
const descriptor = Descriptor{
.limit = @sizeOf(@TypeOf(idt)) - 1,
.base = @intFromPtr(&idt),
};
idt_flush(&descriptor);
}
/// Called by isr_common (isr.s) with a pointer to the trap frame. Exported so the
/// assembly stubs can `call` it by name. Exceptions are terminal; device
/// interrupts run their handler, get acknowledged, and return.
export fn interruptDispatch(state: *const CpuState) callconv(.c) void {
if (state.vector < 32) {
on_fault(state); // CPU exception — never returns
} else if (handlers[state.vector]) |handler| {
// Acknowledge before running the handler: a handler that switches tasks
// (the scheduler) may not return promptly, and the LAPIC mustn't wait on
// it to deliver the next interrupt. Fine for edge-triggered sources like
// the timer; a level-triggered device would need EOI after handling.
apic.eoi();
handler();
}
// else: spurious/unhandled device interrupt — don't acknowledge it
}
const std = @import("std");
+38
View File
@@ -0,0 +1,38 @@
//! x86 port I/O and model-specific registers — the low-level primitives the
//! serial port and the APIC talk to hardware through.
pub fn outb(port: u16, value: u8) void {
asm volatile ("outb %[value], %[port]"
:
: [value] "{al}" (value),
[port] "{dx}" (port),
);
}
pub fn inb(port: u16) u8 {
return asm volatile ("inb %[port], %[value]"
: [value] "={al}" (-> u8),
: [port] "{dx}" (port),
);
}
/// Read a model-specific register (returns edx:eax combined).
pub fn rdmsr(msr: u32) u64 {
var low: u32 = undefined;
var high: u32 = undefined;
asm volatile ("rdmsr"
: [low] "={eax}" (low),
[high] "={edx}" (high),
: [msr] "{ecx}" (msr),
);
return (@as(u64, high) << 32) | low;
}
pub fn wrmsr(msr: u32, value: u64) void {
asm volatile ("wrmsr"
:
: [msr] "{ecx}" (msr),
[low] "{eax}" (@as(u32, @truncate(value))),
[high] "{edx}" (@as(u32, @truncate(value >> 32))),
);
}
+180
View File
@@ -0,0 +1,180 @@
# x86_64 low-level entry code: the CPU-exception stubs, plus the GDT/IDT load
# helpers. Kept in a dedicated assembly file rather than inline asm because these
# need real labels and cross-symbol jumps/calls (isr_common, exceptionHandler),
# and because `lgdt`/`lidt` memory operands aren't expressible in Zig inline asm.
#
# Each exception vector normalises the stack to a uniform trap frame — a dummy
# error code where the CPU pushes none, then the vector number — and jumps to the
# shared tail, which saves the general registers and calls the Zig handler with a
# pointer to the frame (matching src/arch/x86_64/idt.zig's CpuState).
.text
# gdt_flush(rdi = *GDT descriptor): load the GDT, reload the data segment
# registers to the data selector, and reload CS to the code selector. CS can't be
# set with mov, so we far-return through the caller's own return address.
.global gdt_flush
gdt_flush:
lgdt (%rdi)
mov $0x10, %ax # kernel data selector
mov %ax, %ds
mov %ax, %es
mov %ax, %ss
mov %ax, %fs
mov %ax, %gs
pop %rax # caller's return address
push $0x08 # kernel code selector (new CS)
push %rax # return address (new RIP)
lretq
# idt_flush(rdi = *IDT descriptor): load the IDT.
.global idt_flush
idt_flush:
lidt (%rdi)
ret
# load_tr(di = TSS selector): load the task register.
.global load_tr
load_tr:
ltr %di
ret
# switch_context(rdi = &old_task.rsp, rsi = new_task.rsp)
# Cooperative context switch: save the callee-saved registers on the current
# stack, stash the stack pointer in the old task, load the new task's stack
# pointer, restore its callee-saved registers, and return into it. Caller-saved
# registers are the compiler's responsibility (this looks like a normal call).
.global switch_context
switch_context:
push %rbx
push %rbp
push %r12
push %r13
push %r14
push %r15
mov %rsp, (%rdi) # save old stack pointer into old_task.rsp
mov %rsi, %rsp # switch to the new task's stack
pop %r15
pop %r14
pop %r13
pop %r12
pop %rbp
pop %rbx
ret # return into the new task's saved instruction pointer
# task_trampoline: the first thing a freshly-spawned task runs. init_task_stack
# leaves its entry function in r15. New tasks start with interrupts enabled.
.global task_trampoline
task_trampoline:
sti
call *%r15 # call the task entry (fn() void)
1: hlt # if the entry returns, idle (still preemptible)
jmp 1b
# Stub for a vector the CPU does NOT push an error code for: push a dummy 0.
.macro STUB_NOERR vec
.global isr\vec
isr\vec:
pushq $0
pushq $\vec
jmp isr_common
.endm
# Stub for a vector the CPU DOES push an error code for: leave it in place.
.macro STUB_ERR vec
.global isr\vec
isr\vec:
pushq $\vec
jmp isr_common
.endm
STUB_NOERR 0
STUB_NOERR 1
STUB_NOERR 2
STUB_NOERR 3
STUB_NOERR 4
STUB_NOERR 5
STUB_NOERR 6
STUB_NOERR 7
STUB_ERR 8
STUB_NOERR 9
STUB_ERR 10
STUB_ERR 11
STUB_ERR 12
STUB_ERR 13
STUB_ERR 14
STUB_NOERR 15
STUB_NOERR 16
STUB_ERR 17
STUB_NOERR 18
STUB_NOERR 19
STUB_NOERR 20
STUB_ERR 21
STUB_NOERR 22
STUB_NOERR 23
STUB_NOERR 24
STUB_NOERR 25
STUB_NOERR 26
STUB_NOERR 27
STUB_NOERR 28
STUB_NOERR 29
STUB_NOERR 30
STUB_NOERR 31
# Device-interrupt vectors (timer, spurious, room for more). None push an error
# code, so they all use the dummy-zero form.
STUB_NOERR 32
STUB_NOERR 33
STUB_NOERR 34
STUB_NOERR 35
STUB_NOERR 36
STUB_NOERR 37
STUB_NOERR 38
STUB_NOERR 39
STUB_NOERR 40
STUB_NOERR 41
STUB_NOERR 42
STUB_NOERR 43
STUB_NOERR 44
STUB_NOERR 45
STUB_NOERR 46
STUB_NOERR 47
.extern interruptDispatch
# Shared tail. Register push order here defines the CpuState field order.
isr_common:
push %rax
push %rbx
push %rcx
push %rdx
push %rsi
push %rdi
push %rbp
push %r8
push %r9
push %r10
push %r11
push %r12
push %r13
push %r14
push %r15
mov %rsp, %rdi # first argument: pointer to the trap frame
call interruptDispatch
pop %r15
pop %r14
pop %r13
pop %r12
pop %r11
pop %r10
pop %r9
pop %r8
pop %rbp
pop %rdi
pop %rsi
pop %rdx
pop %rcx
pop %rbx
pop %rax
add $16, %rsp # drop the vector and error code
iretq
+47
View File
@@ -0,0 +1,47 @@
/* Kernel link layout.
*
* The kernel is linked at a fixed low physical address (set by `image_base` in
* build.zig). UEFI runs with memory identity-mapped, so the bootloader can load
* each PT_LOAD segment to the physical address matching its virtual address and
* jump straight to _start — no page tables to build yet. (Moving to a
* higher-half virtual base is a later step, once the bootloader sets up paging.)
*/
ENTRY(_start)
/* One loadable segment per permission set, so the loader can map .text as R+X,
* .rodata as R, and .data/.bss as R+W. FLAGS bits: 1=X, 2=W, 4=R. */
PHDRS {
text PT_LOAD FLAGS(5); /* R + X */
rodata PT_LOAD FLAGS(4); /* R */
data PT_LOAD FLAGS(6); /* R + W */
}
SECTIONS {
.text ALIGN(4K) : {
*(.text .text.*)
} :text
.rodata ALIGN(4K) : {
*(.rodata .rodata.*)
} :rodata
.data ALIGN(4K) : {
*(.data .data.*)
} :data
/* .bss occupies memory but not file space. The loader zeroes it via the
* gap between each PT_LOAD segment's file size and memory size, so no
* boundary symbols are needed here. (Zig's self-hosted linker also does not
* yet honour linker-script symbol assignments.) */
.bss ALIGN(4K) : {
*(.bss .bss.*)
*(COMMON)
} :data
/DISCARD/ : {
*(.comment)
*(.note .note.*)
*(.eh_frame .eh_frame_hdr)
}
}
+153
View File
@@ -0,0 +1,153 @@
//! The kernel's page tables and virtual memory manager.
//!
//! Builds our own 4-level page tables and switches CR3 onto them, replacing the
//! firmware's. Unlike the earlier bootstrap this maps with real permissions:
//! RAM is identity-mapped read-write + no-execute, the kernel's own segments get
//! their ELF permissions (code R+X, rodata R, data R+W+NX), and page 0 is left
//! unmapped as a null guard. It also exposes map/unmap for on-demand mapping,
//! which the kernel heap will build on.
//!
//! Everything is 4 KiB pages — precise and simple; the extra table memory is
//! negligible against available RAM.
const danos = @import("danos");
const io = @import("io.zig");
const page_size = danos.page_size;
// Page-table entry bits.
const present: u64 = 1 << 0;
const writable: u64 = 1 << 1;
const no_execute: u64 = 1 << 63;
const addr_mask: u64 = 0x000F_FFFF_FFFF_F000;
// ELF segment flags (p_flags).
const pf_x: u32 = 1;
const pf_w: u32 = 2;
// State kept after init so map()/unmap() can serve later callers (e.g. the heap).
var kernel_pml4: u64 = 0;
var alloc_frame: *const fn () ?u64 = undefined;
fn tableAt(phys: u64) *[512]u64 {
return @ptrFromInt(phys);
}
fn allocTable() u64 {
const frame = alloc_frame() orelse @panic("paging: out of memory building page tables");
@memset(tableAt(frame)[0..], 0);
return frame;
}
/// Return the table an entry points at, creating it if empty. Intermediate
/// entries are writable and executable so the leaf's bits govern (a page is
/// writable only if every level is; non-executable if any level is).
fn descend(entry: *u64) u64 {
if (entry.* & present != 0) return entry.* & addr_mask;
const frame = allocTable();
entry.* = frame | present | writable;
return frame;
}
/// Map one 4 KiB page `virt` -> `phys` with `flags` (present is added).
fn mapPage(pml4: u64, virt: u64, phys: u64, flags: u64) void {
const pml4e = &tableAt(pml4)[(virt >> 39) & 0x1FF];
const pdpt = descend(pml4e);
const pdpte = &tableAt(pdpt)[(virt >> 30) & 0x1FF];
const pd = descend(pdpte);
const pde = &tableAt(pd)[(virt >> 21) & 0x1FF];
const pt = descend(pde);
tableAt(pt)[(virt >> 12) & 0x1FF] = (phys & addr_mask) | flags | present;
}
/// Identity-map [base, base+len) with `flags`, rounded out to whole pages.
fn mapRangeIdentity(pml4: u64, base: u64, len: u64, flags: u64) void {
var addr = base & ~@as(u64, page_size - 1);
const end = base + len;
while (addr < end) : (addr += page_size) {
if (addr == 0) continue; // leave page 0 unmapped: the null guard
mapPage(pml4, addr, addr, flags);
}
}
fn regions(mm: danos.MemoryMap) []const danos.MemoryRegion {
return @as([*]const danos.MemoryRegion, @ptrFromInt(mm.regions))[0..mm.len];
}
/// Enable the NX bit in the page-table format (EFER.NXE). Must happen before we
/// load a CR3 whose entries set the NX bit, or those bits are reserved and fault.
fn enableNx() void {
const efer_msr = 0xC0000080;
io.wrmsr(efer_msr, io.rdmsr(efer_msr) | (1 << 11));
}
/// Build the address space and switch onto it.
pub fn init(allocFrame: *const fn () ?u64, boot_info: *const danos.BootInfo) void {
alloc_frame = allocFrame;
enableNx();
const pml4 = allocTable();
// 1. All RAM identity-mapped RW + NX. Non-RAM (MMIO) is skipped and stays
// unmapped unless mapped explicitly below.
for (regions(boot_info.memory_map)) |r| {
if (r.kind == .mmio) continue;
mapRangeIdentity(pml4, r.base, r.pages * page_size, present | writable | no_execute);
}
// 2. The framebuffer and the Local APIC (device memory we need), RW + NX.
const fb = boot_info.framebuffer;
mapRangeIdentity(pml4, fb.base, @as(u64, fb.height) * fb.pitch, present | writable | no_execute);
mapPage(pml4, 0xFEE00000, 0xFEE00000, present | writable | no_execute);
// 3. Overlay the kernel's own segments with their real ELF permissions,
// replacing the blanket RW+NX from step 1: code becomes R+X, rodata R,
// data R+W+NX. This is the W^X guarantee.
for (boot_info.kernel_segments[0..boot_info.kernel_segment_count]) |seg| {
var flags: u64 = present;
if (seg.flags & pf_w != 0) flags |= writable;
if (seg.flags & pf_x == 0) flags |= no_execute;
var addr = seg.virt;
const end = seg.virt + seg.pages * page_size;
while (addr < end) : (addr += page_size) mapPage(pml4, addr, addr, flags);
}
kernel_pml4 = pml4;
asm volatile ("mov %[pml4], %%cr3"
:
: [pml4] "r" (pml4),
: .{ .memory = true }
);
}
/// Map a page into the kernel address space on demand (for the heap, etc.).
/// `writable_page` controls W; pages are always mapped non-executable.
pub fn map(virt: u64, phys: u64, writable_page: bool) void {
var flags: u64 = present | no_execute;
if (writable_page) flags |= writable;
mapPage(kernel_pml4, virt, phys, flags);
invalidate(virt);
}
/// Remove a mapping and flush it from the TLB.
pub fn unmap(virt: u64) void {
const pml4e = tableAt(kernel_pml4)[(virt >> 39) & 0x1FF];
if (pml4e & present == 0) return;
const pdpte = tableAt(pml4e & addr_mask)[(virt >> 30) & 0x1FF];
if (pdpte & present == 0) return;
const pde = tableAt(pdpte & addr_mask)[(virt >> 21) & 0x1FF];
if (pde & present == 0) return;
tableAt(pde & addr_mask)[(virt >> 12) & 0x1FF] = 0;
invalidate(virt);
}
fn invalidate(virt: u64) void {
// invlpg needs its operand via a register-indirect memory reference that Zig
// inline asm won't form directly, so stage the address in a register first.
asm volatile (
\\mov %[v], %%rax
\\invlpg (%%rax)
:
: [v] "r" (virt),
: .{ .rax = true, .memory = true }
);
}
+46
View File
@@ -0,0 +1,46 @@
//! COM1 serial port (16550 UART) — the kernel's machine-readable output channel.
//! Unlike the framebuffer console, serial text can be captured to a file by QEMU
//! (`-serial file:...`), which is what the test harness asserts on. Each
//! architecture has its own UART; this is the x86 one, driven by port I/O.
const port = 0x3F8; // COM1 base
fn outb(p: u16, value: u8) void {
asm volatile ("outb %[value], %[p]"
:
: [value] "{al}" (value),
[p] "{dx}" (p),
);
}
fn inb(p: u16) u8 {
return asm volatile ("inb %[p], %[value]"
: [value] "={al}" (-> u8),
: [p] "{dx}" (p),
);
}
/// Configure the UART: 38400 baud, 8N1, FIFO on. Safe to call before anything
/// else; it has no dependencies.
pub fn init() void {
outb(port + 1, 0x00); // disable interrupts
outb(port + 3, 0x80); // enable DLAB (set baud divisor)
outb(port + 0, 0x03); // divisor low: 38400 baud
outb(port + 1, 0x00); // divisor high
outb(port + 3, 0x03); // 8 bits, no parity, one stop bit; DLAB off
outb(port + 2, 0xC7); // enable + clear FIFO, 14-byte threshold
outb(port + 4, 0x0B); // RTS/DSR set
}
fn writeByte(c: u8) void {
while (inb(port + 5) & 0x20 == 0) {} // wait until the transmit holding register is empty
outb(port, c);
}
/// Write bytes, translating LF to CRLF so terminals and logs line up.
pub fn write(bytes: []const u8) void {
for (bytes) |c| {
if (c == '\n') writeByte('\r');
writeByte(c);
}
}
+48
View File
@@ -0,0 +1,48 @@
//! Task State Segment and its interrupt stack. In long mode the TSS's main job
//! is the Interrupt Stack Table: an IDT gate can name an IST entry, and the CPU
//! switches to that stack when the exception fires — no matter how broken the
//! interrupted stack was. We use IST1 for the double-fault handler, so a fault
//! that happens *because* the current stack is unusable still lands on solid
//! ground instead of triple-faulting.
const gdt = @import("gdt.zig");
/// x86_64 TSS. `packed` because several 64-bit fields sit at 4-byte-unaligned
/// offsets (rsp0 at byte 4), which a normal struct would pad away.
const Tss = packed struct {
reserved0: u32 = 0,
rsp0: u64 = 0,
rsp1: u64 = 0,
rsp2: u64 = 0,
reserved1: u64 = 0,
ist1: u64 = 0,
ist2: u64 = 0,
ist3: u64 = 0,
ist4: u64 = 0,
ist5: u64 = 0,
ist6: u64 = 0,
ist7: u64 = 0,
reserved2: u64 = 0,
reserved3: u16 = 0,
iomap_base: u16 = 0,
};
/// The IST slot (1-based, as the IDT gate encodes it) used for critical faults.
pub const double_fault_ist = 1;
var tss: Tss align(16) = .{};
/// Dedicated stack for IST1. Static so it needs no allocator and is always valid.
var ist1_stack: [16 * 1024]u8 align(16) = undefined;
/// Loads the task register with the TSS selector. Defined in isr.s.
extern fn load_tr(selector: u16) callconv(.c) void;
/// Point IST1 at its stack, publish the TSS through the GDT, and load it into the
/// task register. Requires the GDT to already be loaded (gdt.init first).
pub fn init() void {
tss.ist1 = @intFromPtr(&ist1_stack) + ist1_stack.len; // stacks grow down
tss.iomap_base = @sizeOf(Tss); // == limit: no I/O permission bitmap
gdt.setTss(@intFromPtr(&tss), @sizeOf(Tss) - 1);
load_tr(gdt.tss_selector);
}
+141
View File
@@ -0,0 +1,141 @@
//! A framebuffer text console: draws glyphs from an embedded PSF2 font directly
//! into the linear framebuffer the bootloader handed us. No firmware, no driver
//! — just pixels. This is the kernel's first output device.
const std = @import("std");
const danos = @import("danos");
const arch = @import("arch");
/// The console font, embedded at compile time. cp850-8x16, PSF2 format:
/// a 32-byte header, then 256 glyphs of 16 bytes each (one byte per 8-pixel
/// row). We index glyphs straight by byte value, so ASCII maps 1:1.
const font = @embedFile("font.psf");
const glyph_w = 8;
const glyph_h = 16;
const glyph_bytes = glyph_h; // 8 pixels wide => 1 byte per row
const glyph_data = 32; // PSF2 header size
pub const Console = struct {
fb: danos.Framebuffer,
cols: u32,
rows: u32,
col: u32 = 0,
row: u32 = 0,
fg: u32 = 0x00c8_c8c8, // light grey
bg: u32 = 0x0000_0000, // black
pub fn init(fb: danos.Framebuffer) Console {
return .{
.fb = fb,
.cols = fb.width / glyph_w,
.rows = fb.height / glyph_h,
};
}
/// Fill the whole screen with the background colour and home the cursor.
pub fn clear(self: *Console) void {
var y: u32 = 0;
while (y < self.fb.height) : (y += 1) self.fillRow(y, self.bg);
self.col = 0;
self.row = 0;
}
pub fn write(self: *Console, bytes: []const u8) void {
// Mirror everything to the serial port so it's captured in logs / tests.
arch.serialWrite(bytes);
for (bytes) |c| self.putChar(c);
}
/// Formatted output, e.g. `con.print("x={d}\n", .{x})`. Silently truncates
/// past 256 bytes — this is a debug console, not a general writer.
pub fn print(self: *Console, comptime fmt: []const u8, args: anytype) void {
var buf: [256]u8 = undefined;
self.write(std.fmt.bufPrint(&buf, fmt, args) catch return);
}
pub fn putChar(self: *Console, ch: u8) void {
switch (ch) {
'\n' => self.newline(),
'\r' => self.col = 0,
else => {
if (self.col >= self.cols) self.newline();
self.drawGlyph(ch, self.col * glyph_w, self.row * glyph_h);
self.col += 1;
},
}
}
fn newline(self: *Console) void {
self.col = 0;
if (self.row + 1 >= self.rows) {
self.scroll();
} else {
self.row += 1;
}
}
fn drawGlyph(self: *Console, ch: u8, px: u32, py: u32) void {
const rows = font[glyph_data + @as(usize, ch) * glyph_bytes ..][0..glyph_bytes];
var gy: u32 = 0;
while (gy < glyph_h) : (gy += 1) {
const bits = rows[gy];
var gx: u32 = 0;
while (gx < glyph_w) : (gx += 1) {
// Leftmost pixel is the high bit.
const on = (bits >> @as(u3, @intCast(7 - gx))) & 1 != 0;
self.pixel(px + gx, py + gy, if (on) self.fg else self.bg);
}
}
}
/// Shift the visible text up one glyph row and clear the freed bottom row,
/// leaving the cursor on that now-blank last line.
fn scroll(self: *Console) void {
const visible = self.rows * glyph_h;
var y: u32 = 0;
while (y + glyph_h < visible) : (y += 1) self.copyRow(y, y + glyph_h);
while (y < visible) : (y += 1) self.fillRow(y, self.bg);
self.row = self.rows - 1;
}
inline fn rowPtr(self: *Console, y: u32) [*]volatile u32 {
const base: [*]volatile u8 = @ptrFromInt(self.fb.base);
return @ptrCast(@alignCast(base + y * self.fb.pitch));
}
inline fn pixel(self: *Console, x: u32, y: u32, color: u32) void {
self.rowPtr(y)[x] = color;
}
fn fillRow(self: *Console, y: u32, color: u32) void {
const row = self.rowPtr(y);
var x: u32 = 0;
while (x < self.fb.width) : (x += 1) row[x] = color;
}
fn copyRow(self: *Console, dst_y: u32, src_y: u32) void {
const dst = self.rowPtr(dst_y);
const src = self.rowPtr(src_y);
var x: u32 = 0;
while (x < self.fb.width) : (x += 1) dst[x] = src[x];
}
};
pub const SerialConsole = struct {
/// Serial-only output: goes to the machine-readable log but *not* the framebuffer,
/// so debug and test detail stays out of the on-screen console. These are free
/// functions, not `Console` methods, because serial has no dependency on the
/// framebuffer — they work even before `con` is initialised.
pub fn debugWrite(bytes: []const u8) void {
arch.serialWrite(bytes);
}
/// Formatted serial-only output, e.g. `debugPrint("x={d}\n", .{x})`. Truncates
/// past 256 bytes, like `Console.print`.
pub fn debugPrint(comptime fmt: []const u8, args: anytype) void {
var buf: [256]u8 = undefined;
debugWrite(std.fmt.bufPrint(&buf, fmt, args) catch return);
}
};
Binary file not shown.
+174
View File
@@ -0,0 +1,174 @@
//! The kernel heap: dynamic allocation for the kernel.
//!
//! Where the frame allocator ([pmm]) hands out fixed 4 KiB physical frames, the
//! heap hands out arbitrary byte-sized blocks from a virtual region, growing on
//! demand by mapping fresh frames into it (arch.mapPage) — the first real user of
//! the VMM (see docs/paging.md).
//!
//! The algorithm is a first-fit free list: an address-ordered singly linked list
//! of free blocks, split on allocation and coalesced with neighbours on free. It
//! is exposed as a std.mem.Allocator, so the kernel can use std containers.
//!
//! Not yet concurrency-safe: it assumes a single caller and no allocation from
//! interrupt handlers (ours don't). A lock comes with threads/SMP.
const std = @import("std");
const danos = @import("danos");
const arch = @import("arch");
const pmm = @import("pmm.zig");
const page_size = danos.page_size;
/// Virtual base of the heap: the start of the higher half, which is unmapped and
/// well clear of the identity-mapped low half. (Canonical on x86_64; an arch that
/// splits the address space differently would choose its own.)
const heap_base: usize = 0xFFFF_8000_0000_0000;
/// Cap on heap growth for now.
const heap_max: usize = 64 * 1024 * 1024;
/// A block header, placed at the start of every block. While the block is free it
/// also links into the free list via `next`.
const Block = extern struct {
size: usize, // total block size in bytes, including this header; a multiple of 16
next: ?*Block, // free-list link (only meaningful while free)
};
const header_size = @sizeOf(Block); // 16
const min_block = header_size + 16; // smallest block worth splitting off
var free_list: ?*Block = null;
var heap_end: usize = heap_base; // [heap_base, heap_end) is currently mapped
fn alignUp(value: usize, alignment: usize) usize {
return (value + alignment - 1) & ~(alignment - 1);
}
fn payloadOf(block: *Block) [*]u8 {
return @ptrFromInt(@intFromPtr(block) + header_size);
}
/// Bring the heap up with an initial mapped region.
pub fn init() void {
free_list = null;
heap_end = heap_base;
_ = grow(page_size);
}
/// Map more pages onto the end of the heap and add them as a free block. Returns
/// false if out of heap virtual space or out of physical frames.
fn grow(min_bytes: usize) bool {
const start = heap_end;
const bytes = alignUp(min_bytes, page_size);
if (start + bytes > heap_base + heap_max) return false;
var virt = start;
while (virt < start + bytes) : (virt += page_size) {
const frame = pmm.alloc() orelse return false;
arch.mapPage(virt, frame, true);
}
heap_end = start + bytes;
const block: *Block = @ptrFromInt(start);
block.size = bytes;
insertFree(block); // coalesces with the previous tail block if adjacent
return true;
}
/// Insert a block into the address-ordered free list, coalescing with the
/// physically adjacent free blocks on either side.
fn insertFree(block: *Block) void {
var prev: ?*Block = null;
var cur = free_list;
while (cur) |c| : (cur = c.next) {
if (@intFromPtr(c) > @intFromPtr(block)) break;
prev = c;
}
block.next = cur;
if (prev) |p| p.next = block else free_list = block;
// Merge forward into `cur` if they're contiguous.
if (cur) |c| {
if (@intFromPtr(block) + block.size == @intFromPtr(c)) {
block.size += c.size;
block.next = c.next;
}
}
// Merge `prev` forward into `block` if they're contiguous.
if (prev) |p| {
if (@intFromPtr(p) + p.size == @intFromPtr(block)) {
p.size += block.size;
p.next = block.next;
}
}
}
/// Allocate `len` bytes (16-byte aligned), or null if out of memory.
fn rawAlloc(len: usize) ?[*]u8 {
const need = alignUp(header_size + len, 16);
var attempts: u32 = 0;
while (attempts < 2) : (attempts += 1) {
var prev: ?*Block = null;
var cur = free_list;
while (cur) |block| : ({
prev = block;
cur = block.next;
}) {
if (block.size < need) continue;
if (block.size >= need + min_block) {
// Split: carve `need` off the front, leave the rest free.
const rest: *Block = @ptrFromInt(@intFromPtr(block) + need);
rest.size = block.size - need;
rest.next = block.next;
if (prev) |p| p.next = rest else free_list = rest;
block.size = need;
} else {
// Take the whole block.
if (prev) |p| p.next = block.next else free_list = block.next;
}
return payloadOf(block);
}
// Nothing fit: grow and try once more.
if (!grow(need)) return null;
}
return null;
}
fn rawFree(ptr: [*]u8) void {
const block: *Block = @ptrFromInt(@intFromPtr(ptr) - header_size);
insertFree(block);
}
// --- std.mem.Allocator interface -----------------------------------------
pub fn allocator() std.mem.Allocator {
return .{ .ptr = undefined, .vtable = &vtable };
}
const vtable = std.mem.Allocator.VTable{
.alloc = allocImpl,
.resize = resizeImpl,
.remap = remapImpl,
.free = freeImpl,
};
fn allocImpl(_: *anyopaque, len: usize, alignment: std.mem.Alignment, _: usize) ?[*]u8 {
// Blocks are 16-byte aligned; larger alignments aren't supported yet.
if (alignment.toByteUnits() > 16) return null;
return rawAlloc(len);
}
fn resizeImpl(_: *anyopaque, _: []u8, _: std.mem.Alignment, _: usize, _: usize) bool {
return false; // no in-place resize; the caller reallocates
}
fn remapImpl(_: *anyopaque, _: []u8, _: std.mem.Alignment, _: usize, _: usize) ?[*]u8 {
return null;
}
fn freeImpl(_: *anyopaque, memory: []u8, _: std.mem.Alignment, _: usize) void {
rawFree(memory.ptr);
}
+54
View File
@@ -0,0 +1,54 @@
//! Inter-process communication: message-passing channels.
//!
//! IPC is the backbone of a microkernel ([vision](../docs/vision.md)): once
//! drivers and services live in separate address spaces, a message is how they
//! talk. This first form is a **bounded blocking channel** — a ring buffer of
//! messages with a producer/consumer rendezvous, built on the scheduler's
//! [wait queues](scheduling.md). `send` blocks when the channel is full, `recv`
//! blocks when it's empty; neither busy-waits.
//!
//! For now both endpoints are kernel threads sharing the kernel address space.
//! When user mode arrives, the same primitive carries messages across the
//! isolation boundary (with the payload copied between address spaces).
const arch = @import("arch");
const sched = @import("sched.zig");
/// A bounded blocking channel of `capacity` messages of type `T`.
pub fn Channel(comptime T: type, comptime capacity: usize) type {
return struct {
const Self = @This();
buffer: [capacity]T = undefined,
head: usize = 0, // next slot to read
tail: usize = 0, // next slot to write
count: usize = 0,
not_full: sched.WaitQueue = .{}, // senders wait here
not_empty: sched.WaitQueue = .{}, // receivers wait here
/// Send a message, blocking while the channel is full.
pub fn send(self: *Self, msg: T) void {
const flags = arch.saveInterrupts();
// Recheck the condition in a loop: a wakeup only means "try again"
// (another waiter may have taken the slot first).
while (self.count == capacity) sched.waitLocked(&self.not_full);
self.buffer[self.tail] = msg;
self.tail = (self.tail + 1) % capacity;
self.count += 1;
sched.wakeLocked(&self.not_empty); // a receiver can now proceed
arch.restoreInterrupts(flags);
}
/// Receive a message, blocking while the channel is empty.
pub fn recv(self: *Self) T {
const flags = arch.saveInterrupts();
while (self.count == 0) sched.waitLocked(&self.not_empty);
const msg = self.buffer[self.head];
self.head = (self.head + 1) % capacity;
self.count -= 1;
sched.wakeLocked(&self.not_full); // a sender can now proceed
arch.restoreInterrupts(flags);
return msg;
}
};
}
+168
View File
@@ -0,0 +1,168 @@
const std = @import("std");
const danos = @import("danos");
const arch = @import("arch");
const console = @import("console.zig");
const pmm = @import("pmm.zig");
const heap = @import("heap.zig");
const sched = @import("sched.zig");
const tests = @import("tests.zig");
const build_options = @import("build_options");
const BootInfo = danos.BootInfo;
/// The calling convention used to enter the kernel. Pinned to SysV explicitly:
/// the bootloader is built for the UEFI target, whose C convention is Microsoft
/// x64 (first argument in RCX), while the kernel is SysV (first argument in
/// RDI). Both sides reference this so the `boot_info` pointer lands in the
/// register the other expects. `danos.kernel_abi` re-exports it to the loader.
pub const kernel_abi = danos.kernel_abi;
/// The system console, valid once `kmain` has initialised it. Global so the
/// panic handler can reach it too.
var con: console.Console = undefined;
var con_ready = false;
/// Kernel entry point. The bootloader jumps here after `ExitBootServices` with a
/// pointer to the handoff data. There is no runtime, no stack unwinding, and no
/// caller to return to, so this never returns.
export fn _start(boot_info: *const BootInfo) callconv(kernel_abi) noreturn {
kmain(boot_info);
}
fn kmain(boot_info: *const BootInfo) noreturn {
arch.serialInit(); // machine-readable log; console mirrors to it
const fb = boot_info.framebuffer;
const serial0 = console.SerialConsole;
con = console.Console.init(fb);
con.clear();
con_ready = true;
// Catch CPU exceptions before doing anything that might fault: install our
// reporter, then bring up the GDT + IDT.
arch.setFaultHandler(onException);
arch.init();
con.write("danos: initalizing kernel...");
serial0.debugWrite("danos: framebuffer console online\n");
serial0.debugWrite("danos: cpu tables online (GDT, IDT, TSS)\n");
serial0.debugPrint(" resolution : {d}x{d}\n", .{ fb.width, fb.height });
serial0.debugPrint(" pitch : {d} bytes\n", .{fb.pitch});
serial0.debugPrint(" format : {s}\n", .{@tagName(fb.format)});
serial0.debugPrint(" framebuffer: 0x{x:0>16}\n", .{fb.base});
serial0.debugPrint (" footdebugPrint : {d} MiB\n", .{(fb.pitch * fb.height) / (1024 * 1024)});
// Summarise the physical memory the loader handed us. The array is danos's
// own MemoryRegion, so this is a plain slice — no firmware layout in sight.
const regions = @as([*]const danos.MemoryRegion, @ptrFromInt(boot_info.memory_map.regions))[0..boot_info.memory_map.len];
var usable_pages: u64 = 0;
var reserved_pages: u64 = 0; // reserved RAM only — MMIO is device space, not RAM
for (regions) |r| {
switch (r.kind) {
.usable => usable_pages += r.pages,
.reserved, .acpi_tables, .acpi_nvs => reserved_pages += r.pages,
.mmio => {},
}
}
const total_pages = usable_pages + reserved_pages;
const total_bytes = total_pages * danos.page_size;
const gib = 1 << 30;
serial0.debugWrite("\ndanos: physical memory\n");
serial0.debugPrint(" total RAM : {d}.{d:0>2} GiB ({d} MiB) - RAM the firmware reported\n", .{ total_bytes / gib, (total_bytes % gib) * 100 / gib, mib(total_pages) });
serial0.debugPrint(" usable : {d} MiB - free RAM (incl. reclaimed boot-services memory)\n", .{mib(usable_pages)});
serial0.debugPrint(" reserved : {d} MiB - kernel image, boot stack, ACPI, runtime services\n", .{mib(reserved_pages)});
serial0.debugPrint(" regions : {d} - entries in the firmware memory map\n", .{regions.len});
// Bring up the physical frame allocator over that map, and prove it works:
// allocate three frames, then hand them back.
pmm.init(boot_info.memory_map);
const s1 = pmm.stats();
serial0.debugPrint("\ndanos: frame allocator online\n", .{});
serial0.debugPrint(" free frames: {d} ({d} MiB)\n", .{ s1.free_frames, mib(s1.free_frames) });
const f0 = pmm.alloc();
const f1 = pmm.alloc();
const f2 = pmm.alloc();
serial0.debugPrint(" alloc x3 : 0x{x} 0x{x} 0x{x}\n", .{ f0 orelse 0, f1 orelse 0, f2 orelse 0 });
if (f0) |p| pmm.free(p);
if (f1) |p| pmm.free(p);
if (f2) |p| pmm.free(p);
serial0.debugPrint(" after free : {d} frames free\n", .{pmm.stats().free_frames});
// Switch off the firmware's page tables onto our own (with real permissions).
arch.enablePaging(pmm.alloc, boot_info);
serial0.debugPrint("\ndanos: paging enabled\n", .{});
serial0.debugPrint(" page tables: CR3 = 0x{x:0>16}\n", .{arch.readCr3()});
serial0.debugPrint(" kernel segs: {d} (mapped with W^X permissions)\n", .{boot_info.kernel_segment_count});
// Bring up the kernel heap (dynamic allocation), built on the VMM.
heap.init();
serial0.debugWrite("\ndanos: kernel heap online\n");
// Measure the amount of resources the kernel is actually using
const s2 = pmm.stats();
serial0.debugPrint(" Kernel FootdebugPrint: {d} KiB\n", .{kib(s1.free_frames - s2.free_frames)});
// Register the current context as the first task before enabling preemption.
sched.init(4);
serial0.debugWrite("\ndanos: scheduler online\n");
// Start the timer and unmask interrupts — the kernel now has a heartbeat, and
// the timer preempts among tasks.
arch.startTimer();
arch.enableInterrupts();
serial0.debugPrint("danos: timer online ({d} Hz tick; LAPIC {d} MHz, TSC {d} MHz measured)\n", .{ arch.timer_hz, arch.lapicHz() / 1_000_000, arch.tscHz() / 1_000_000 });
// In a test build (`zig build -Dtest-case=<name>`), run that case and stop.
// Normal builds fall through to the idle halt.
if (build_options.test_case) |case| {
tests.run(case, boot_info);
arch.halt();
}
con.write("kernel initialised.\n");
// TODO: init process
con.write("\nnothing left to do; halting CPU.\n");
arch.halt();
}
/// Frames (4 KiB pages) to whole MiB.
fn mib(pages: u64) u64 {
return pages * danos.page_size / (1024 * 1024);
}
fn kib(frames: u64) u64 {
return frames * danos.page_size / (1024);
}
/// Report a CPU exception in red and halt. There's no fault recovery yet, so any
/// exception is terminal — but now it debugPrints what and where instead of silently
/// resetting the machine.
fn onException(state: *const arch.CpuState) noreturn {
if (con_ready) {
con.fg = 0x00ff_5555;
con.print("\nCPU EXCEPTION: {s} (vector {d})\n", .{ arch.vectorName(state.vector), state.vector });
con.print(" error code : 0x{x}\n", .{state.error_code});
con.print(" RIP : 0x{x:0>16}\n", .{state.rip});
con.print(" RSP : 0x{x:0>16}\n", .{state.rsp});
if (state.vector == 14) con.print(" CR2 (addr) : 0x{x:0>16}\n", .{arch.readCr2()});
}
arch.halt();
}
/// Freestanding has no OS to receive a panic. debugPrint it to the console (if it is
/// up yet) in red, then halt.
pub const panic = std.debug.FullPanic(struct {
fn panic(msg: []const u8, first_trace_addr: ?usize) noreturn {
_ = first_trace_addr;
if (con_ready) {
con.fg = 0x00ff_5555;
con.write("\nKERNEL PANIC: ");
con.write(msg);
con.write("\n");
}
arch.halt();
}
}.panic);
+152
View File
@@ -0,0 +1,152 @@
//! Physical memory manager: a bitmap frame allocator. It hands out and reclaims
//! 4 KiB physical frames — the primitive every later memory feature (page
//! tables, the heap) is built on top of.
//!
//! This is generic kernel code: it works on the neutral `danos.MemoryRegion`
//! array the loader hands over (see docs/memory-map.md), so it carries no UEFI
//! and nothing architecture-specific beyond the 4 KiB page.
const std = @import("std");
const danos = @import("danos");
const page_size = danos.page_size;
/// One bit per frame, covering physical RAM from 0 up to the highest usable
/// address: 1 = used/unavailable, 0 = free. The bitmap itself lives in a frame
/// we carve out of usable memory during init.
var bitmap: []u8 = &.{};
var total_frames: usize = 0;
var used_frames: usize = 0;
/// Where the next allocation scan begins, so we don't rescan from frame 0 every
/// time. Pulled back on free() so reclaimed low frames get reused.
var next_hint: usize = 0;
pub const Stats = struct {
total_frames: usize,
used_frames: usize,
free_frames: usize,
};
pub fn stats() Stats {
return .{
.total_frames = total_frames,
.used_frames = used_frames,
.free_frames = total_frames - used_frames,
};
}
inline fn bit(frame: usize) u3 {
return @intCast(frame & 7);
}
inline fn isUsed(frame: usize) bool {
return (bitmap[frame >> 3] >> bit(frame)) & 1 != 0;
}
inline fn setUsed(frame: usize) void {
bitmap[frame >> 3] |= @as(u8, 1) << bit(frame);
}
inline fn setFree(frame: usize) void {
bitmap[frame >> 3] &= ~(@as(u8, 1) << bit(frame));
}
fn regions(map: danos.MemoryMap) []const danos.MemoryRegion {
return @as([*]const danos.MemoryRegion, @ptrFromInt(map.regions))[0..map.len];
}
/// Build the allocator from the loader's memory map. Relies on the firmware's
/// identity mapping still being in effect (a physical address is usable directly
/// as a pointer) — true until the kernel installs its own page tables.
pub fn init(map: danos.MemoryMap) void {
const regs = regions(map);
// 1. Size the bitmap to cover every frame up to the highest RAM address —
// including reserved RAM, so those frames are trackable (e.g. to free the
// boot buffers later). Only MMIO (device address space) is excluded.
// Everything starts unallocatable; usable regions are freed below.
var highest: u64 = 0;
for (regs) |r| {
if (r.kind == .mmio) continue;
const end = r.base + r.pages * page_size;
if (end > highest) highest = end;
}
total_frames = @intCast(highest / page_size);
if (total_frames == 0) @panic("pmm: no usable memory");
const bitmap_bytes = (total_frames + 7) / 8;
const bitmap_pages = (bitmap_bytes + page_size - 1) / page_size;
// 2. Park the bitmap in the first usable region large enough to hold it.
// Start at least one page in, so we never place it on frame 0 (which is
// kept reserved as the "none" address, and is an awkward pointer besides).
var storage: ?u64 = null;
for (regs) |r| {
if (r.kind != .usable) continue;
const base = if (r.base == 0) page_size else r.base;
const skipped = (base - r.base) / page_size;
if (r.pages - skipped >= bitmap_pages) {
storage = base;
break;
}
}
const bitmap_base = storage orelse @panic("pmm: no region large enough for the frame bitmap");
bitmap = @as([*]u8, @ptrFromInt(bitmap_base))[0..bitmap_bytes];
// 3. Start with everything marked used, then free the usable regions. Doing
// it this way means every gap, reserved span and MMIO hole is unallocatable
// by default — we only ever hand back memory the firmware called usable.
@memset(bitmap, 0xff);
used_frames = total_frames;
for (regs) |r| {
if (r.kind != .usable) continue;
var f: usize = @intCast(r.base / page_size);
const end = f + @as(usize, @intCast(r.pages));
while (f < end and f < total_frames) : (f += 1) {
setFree(f);
used_frames -= 1;
}
}
// 4. Take back the frames the bitmap occupies, and frame 0 — so a 0 result
// stays reserved to mean "no frame".
reserve(bitmap_base, bitmap_pages);
reserve(0, 1);
}
/// Mark `count` frames from physical `base` as used, counting only those that
/// were actually free.
fn reserve(base: u64, count: usize) void {
var f: usize = @intCast(base / page_size);
const end = f + count;
while (f < end and f < total_frames) : (f += 1) {
if (!isUsed(f)) {
setUsed(f);
used_frames += 1;
}
}
}
/// Allocate one physical frame, or null if none are free. The address is
/// page-aligned; the frame's contents are undefined.
pub fn alloc() ?u64 {
var scanned: usize = 0;
var f = next_hint;
while (scanned < total_frames) : (scanned += 1) {
if (f >= total_frames) f = 0;
if (!isUsed(f)) {
setUsed(f);
used_frames += 1;
next_hint = f + 1;
return @as(u64, f) * page_size;
}
f += 1;
}
return null; // out of physical memory
}
/// Return a frame obtained from alloc() to the pool. Bogus or double frees are
/// ignored rather than corrupting the count.
pub fn free(addr: u64) void {
const f: usize = @intCast(addr / page_size);
if (f >= total_frames or !isUsed(f)) return;
setFree(f);
used_frames -= 1;
if (f < next_hint) next_hint = f;
}
+251
View File
@@ -0,0 +1,251 @@
//! The scheduler: fixed-priority preemptive multitasking.
//!
//! Tasks are kernel threads (ring 0, each with its own stack). The **highest-
//! priority ready task always runs**; within a priority level, tasks round-robin.
//! Selection is O(1) — a bitmap of non-empty priority levels plus a FIFO queue per
//! level — which keeps scheduling deterministic, as a real-time kernel needs (see
//! docs/vision.md).
//!
//! Switching happens both cooperatively (`yield`) and preemptively (the timer
//! calls `tick`). See docs/scheduling.md for the interrupt-flag discipline that
//! makes those two paths coexist.
const std = @import("std");
const arch = @import("arch");
const heap = @import("heap.zig");
/// Priority level: 0 (lowest) .. 7 (highest). 8 levels total.
pub const Priority = u3;
const num_priorities = 8;
const stack_size = 16 * 1024; // each task's kernel stack is 16 KiB
const max_tasks = 16; // the maximum number of tasks alive at once is 16 in a static sized pool
const State = enum { free, ready, running, blocked };
const Task = struct {
id: u32 = 0,
state: State = .free,
priority: Priority = 0,
rsp: usize = 0, // saved stack pointer, valid while not running
stack: []u8 = &.{},
wake_at: u64 = 0, // uptime (ms) to wake a sleeping task; 0 = not sleeping
next: ?*Task = null, // ready-queue link
};
var tasks = [_]Task{.{}} ** max_tasks;
var current: *Task = undefined;
var next_id: u32 = 1;
// Per-priority FIFO ready queues, and a bitmap of which levels are non-empty.
var ready_head: [num_priorities]?*Task = .{null} ** num_priorities;
var ready_tail: [num_priorities]?*Task = .{null} ** num_priorities;
var ready_bitmap: u8 = 0;
var preemption_enabled = true;
/// Register the currently-running kernel context as the first task, spawn the
/// idle task, and hook the timer for preemption.
pub fn init(boot_priority: Priority) void {
tasks[0] = .{ .id = 0, .state = .running, .priority = boot_priority };
current = &tasks[0];
spawn(idle, 0); // lowest priority, always runnable — runs when nothing else is
arch.setTickHook(tick);
}
/// The idle task: run when every other task is blocked or sleeping. `hlt` waits
/// for the next interrupt at near-zero power (see docs/halting.md).
fn idle() void {
while (true) asm volatile ("hlt");
}
fn enqueue(t: *Task) void {
t.next = null;
const p: usize = t.priority;
if (ready_tail[p]) |tail| tail.next = t else ready_head[p] = t;
ready_tail[p] = t;
ready_bitmap |= levelBit(t.priority);
}
fn dequeueHighest() ?*Task {
if (ready_bitmap == 0) return null;
const level: Priority = @intCast(num_priorities - 1 - @clz(ready_bitmap));
const t = ready_head[level].?;
ready_head[level] = t.next;
if (ready_head[level] == null) {
ready_tail[level] = null;
ready_bitmap &= ~levelBit(level);
}
t.next = null;
return t;
}
fn levelBit(p: Priority) u8 {
return @as(u8, 1) << p;
}
/// Create a task that runs `entry` at `priority`. It becomes ready immediately.
pub fn spawn(entry: *const fn () void, priority: Priority) void {
const t = freeSlot() orelse @panic("sched: task table full");
const stack = heap.allocator().alloc(u8, stack_size) catch @panic("sched: no memory for task stack");
t.* = .{ .id = next_id, .state = .ready, .priority = priority, .stack = stack };
next_id += 1;
const top = @intFromPtr(stack.ptr) + stack.len;
t.rsp = arch.initTaskStack(top, @intFromPtr(entry));
enqueue(t);
}
fn freeSlot() ?*Task {
for (&tasks) |*t| {
if (t.state == .free) return t;
}
return null;
}
/// Pick the highest-priority ready task and switch to it. Interrupts must be
/// disabled by the caller.
fn schedule() void {
const prev = current;
if (prev.state == .running) {
prev.state = .ready;
enqueue(prev); // back of its level's queue (round-robin)
}
const next = dequeueHighest() orelse {
prev.state = .running; // nothing else ready — keep running
return;
};
next.state = .running;
current = next;
if (next != prev) arch.switchContext(&prev.rsp, next.rsp);
}
/// Voluntarily give up the CPU to the next ready task.
pub fn yield() void {
const flags = arch.saveInterrupts();
schedule();
arch.restoreInterrupts(flags);
}
/// Block the current task for `ms` milliseconds, then let it become runnable
/// again. The idle task (or other work) runs in the meantime.
pub fn sleep(ms: u64) void {
const flags = arch.saveInterrupts();
current.wake_at = arch.millis() + ms;
current.state = .blocked;
schedule(); // current is blocked, so schedule() won't re-enqueue it
arch.restoreInterrupts(flags);
}
// --- event-based blocking -------------------------------------------------
//
// A WaitQueue is a set of tasks blocked waiting for something (a resource, a
// message). Tasks link into it through the same `next` field the ready queues
// use — a task is in exactly one queue at a time. These are the primitive locks,
// semaphores and IPC channels are built on.
pub const WaitQueue = struct {
head: ?*Task = null,
};
/// Block the current task on `wq` and switch away. Precondition: interrupts are
/// disabled (the caller holds them, so a condition can be checked and the block
/// committed atomically). On return — when woken — interrupts are still disabled.
pub fn waitLocked(wq: *WaitQueue) void {
current.state = .blocked;
current.next = wq.head;
wq.head = current;
schedule();
}
/// Move the highest-priority waiter on `wq` (if any) to the ready queue.
/// Precondition: interrupts disabled. Does not preempt — the caller decides.
pub fn wakeLocked(wq: *WaitQueue) void {
// Find the highest-priority waiter (bounded scan) and unlink it.
var best_prev: ?*Task = null;
var best: ?*Task = null;
var prev: ?*Task = null;
var cur = wq.head;
while (cur) |t| : ({
prev = t;
cur = t.next;
}) {
if (best == null or t.priority > best.?.priority) {
best = t;
best_prev = prev;
}
}
const t = best orelse return;
if (best_prev) |p| p.next = t.next else wq.head = t.next;
t.state = .ready;
enqueue(t);
}
/// Block on `wq` (a self-contained critical section).
pub fn wait(wq: *WaitQueue) void {
const flags = arch.saveInterrupts();
waitLocked(wq);
arch.restoreInterrupts(flags);
}
/// Wake the highest-priority waiter on `wq`, preempting if it outranks us.
pub fn wake(wq: *WaitQueue) void {
const flags = arch.saveInterrupts();
wakeLocked(wq);
// If a higher-priority task is now ready, run it immediately.
if (highestReadyPriority()) |p| {
if (p > current.priority) schedule();
}
arch.restoreInterrupts(flags);
}
fn highestReadyPriority() ?Priority {
if (ready_bitmap == 0) return null;
return @intCast(num_priorities - 1 - @clz(ready_bitmap));
}
/// Wake any sleeping task whose deadline has passed. Bounded by the task count,
/// so it stays deterministic. Called from the timer tick (interrupts disabled).
fn wakeExpired() void {
const now = arch.millis();
for (&tasks) |*t| {
if (t.state == .blocked and t.wake_at != 0 and now >= t.wake_at) {
t.wake_at = 0;
t.state = .ready;
enqueue(t);
}
}
}
/// Called from the timer interrupt (interrupts already disabled): wake due
/// sleepers, then preempt.
pub fn tick() void {
wakeExpired();
if (preemption_enabled) schedule();
}
/// Enable or disable timer-driven preemption (cooperative-only when off).
pub fn setPreemption(enabled: bool) void {
preemption_enabled = enabled;
}
/// End the current task and switch away for good; never returns. The task's stack
/// is leaked for now (no reaper yet).
pub fn exit() noreturn {
arch.disableInterrupts();
current.state = .free;
const next = dequeueHighest() orelse @panic("sched: no task left to run");
next.state = .running;
current = next;
var discard: usize = 0;
arch.switchContext(&discard, next.rsp);
unreachable;
}
pub fn currentId() u32 {
return current.id;
}
/// Change the running task's priority (takes effect next time it's enqueued).
pub fn setPriority(p: Priority) void {
current.priority = p;
}
+462
View File
@@ -0,0 +1,462 @@
//! In-kernel test cases, run at the end of bring-up when the kernel is built with
//! `-Dtest-case=<name>`. Each case writes structured markers to the serial port
//! that the QEMU harness (test/qemu_test.py) asserts on:
//!
//! [PASS]/[FAIL] <check> per assertion
//! DANOS-TEST-RESULT: PASS|FAIL overall, for non-faulting cases
//!
//! Faulting cases (fault-ud, fault-pf, fault-df) deliberately don't return a
//! result line — they trigger a CPU exception, and the harness asserts on the
//! exception report the handler prints (which also reaches serial).
const std = @import("std");
const danos = @import("danos");
const arch = @import("arch");
const pmm = @import("pmm.zig");
const heap = @import("heap.zig");
const sched = @import("sched.zig");
const ipc = @import("ipc.zig");
/// Formatted write straight to serial, independent of the framebuffer console.
fn log(comptime fmt: []const u8, args: anytype) void {
var buf: [128]u8 = undefined;
arch.serialWrite(std.fmt.bufPrint(&buf, fmt, args) catch return);
}
var passed: u32 = 0;
var failed: u32 = 0;
fn check(name: []const u8, ok: bool) void {
if (ok) {
passed += 1;
log("[PASS] {s}\n", .{name});
} else {
failed += 1;
log("[FAIL] {s}\n", .{name});
}
}
/// Emit the overall result line the harness matches, then the done sentinel.
fn result() void {
log("DANOS-TEST-RESULT: {s} ({d} passed, {d} failed)\n", .{
if (failed == 0) "PASS" else "FAIL",
passed,
failed,
});
log("DANOS-TEST-DONE\n", .{});
}
pub fn run(case: []const u8, boot_info: *const BootInfo) void {
if (eql(case, "smoke")) {
smoke(boot_info);
} else if (eql(case, "timer")) {
timer();
} else if (eql(case, "clock")) {
clock();
} else if (eql(case, "vmm")) {
vmm();
} else if (eql(case, "heap")) {
heapTest();
} else if (eql(case, "sched")) {
schedTest();
} else if (eql(case, "priority")) {
priorityTest();
} else if (eql(case, "sleep")) {
sleepTest();
} else if (eql(case, "event")) {
eventTest();
} else if (eql(case, "ipc")) {
ipcTest();
} else if (eql(case, "fault-ud")) {
faultInvalidOpcode();
} else if (eql(case, "fault-pf")) {
faultPageFault();
} else if (eql(case, "fault-df")) {
faultDoubleFault();
} else if (eql(case, "fault-nx")) {
faultNoExecute();
} else if (eql(case, "fault-null")) {
faultNull();
} else {
log("DANOS-TEST-RESULT: FAIL (unknown case '{s}')\n", .{case});
}
}
const BootInfo = danos.BootInfo;
fn eql(a: []const u8, b: []const u8) bool {
return std.mem.eql(u8, a, b);
}
/// Non-destructive checks of the memory map and frame allocator.
fn smoke(boot_info: *const BootInfo) void {
log("DANOS-TEST-BEGIN: smoke\n", .{});
// The memory map has some usable RAM.
const mm = boot_info.memory_map;
const regions = @as([*]const danos.MemoryRegion, @ptrFromInt(mm.regions))[0..mm.len];
var usable: u64 = 0;
for (regions) |r| {
if (r.kind == .usable) usable += r.pages;
}
check("memory map reports usable RAM", usable > 0);
// The frame allocator hands out distinct, page-aligned frames.
const a = pmm.alloc();
const b = pmm.alloc();
check("alloc returns a frame", a != null);
check("alloc returns distinct frames", a != null and b != null and a.? != b.?);
check("frames are page-aligned", (a orelse 1) % danos.page_size == 0);
// Freeing restores the count.
const before = pmm.stats().free_frames;
if (a) |p| pmm.free(p);
if (b) |p| pmm.free(p);
check("free returns frames to the pool", pmm.stats().free_frames == before + 2);
// Paging is active on our own tables (CR3 is non-zero and page-aligned).
const cr3 = arch.readCr3();
check("paging active (CR3 set)", cr3 != 0 and cr3 % danos.page_size == 0);
result();
}
/// Verify device interrupts fire and return: the timer tick counter must advance
/// on its own. Interrupts are already enabled by kmain before tests run.
fn timer() void {
log("DANOS-TEST-BEGIN: timer\n", .{});
const start = arch.ticks();
// Busy-wait for the counter to advance. arch.ticks() is a volatile load, so
// the compiler re-reads it each iteration and sees the interrupt's update.
// The cap is only a safety net; the harness timeout is the real backstop.
var spins: u64 = 0;
while (arch.ticks() == start and spins < 5_000_000_000) spins +%= 1;
check("timer interrupts advance the tick count", arch.ticks() > start);
result();
}
/// Verify the on-demand VMM: map a fresh frame at an unused virtual address, and
/// check it's writable and reads back.
fn vmm() void {
log("DANOS-TEST-BEGIN: vmm\n", .{});
const frame = pmm.alloc();
check("frame available to map", frame != null);
if (frame) |phys| {
var virt: u64 = 0x0000_4000_0000_0000; // canonical, well clear of everything mapped
arch.mapPage(virt, phys, true);
const p: *volatile u64 = @ptrFromInt(virt);
p.* = 0xdead_c0de_cafe_babe;
check("mapped page is writable and reads back", p.* == 0xdead_c0de_cafe_babe);
arch.unmapPage(virt);
pmm.free(phys);
virt += 0;
}
result();
}
/// Exercise the kernel heap: basic alloc/write/free, reuse, growth beyond the
/// initial region, and a std container backed by it.
fn heapTest() void {
log("DANOS-TEST-BEGIN: heap\n", .{});
const a = heap.allocator();
// Allocate, write a pattern, read it back, free.
const buf = a.alloc(u8, 4096) catch null;
check("alloc 4096 bytes", buf != null);
if (buf) |b| {
@memset(b, 0xAB);
check("heap memory is writable and reads back", b[0] == 0xAB and b[4095] == 0xAB);
a.free(b);
}
// Freeing then re-allocating the same size should reuse the block.
const p1 = a.alloc(u64, 8) catch null;
const addr1 = if (p1) |p| @intFromPtr(p.ptr) else 0;
if (p1) |p| a.free(p);
const p2 = a.alloc(u64, 8) catch null;
const addr2 = if (p2) |p| @intFromPtr(p.ptr) else 0;
check("freed block is reused", addr1 != 0 and addr1 == addr2);
if (p2) |p| a.free(p);
// Force growth past the initial page and check every block is usable.
var blocks: [64]?[]u8 = .{null} ** 64;
var ok = true;
for (&blocks, 0..) |*slot, i| {
const b = a.alloc(u8, 4096) catch null;
slot.* = b;
if (b) |bb| @memset(bb, @intCast(i & 0xff)) else {
ok = false;
}
}
for (blocks, 0..) |slot, i| {
if (slot) |bb| {
if (bb[0] != @as(u8, @intCast(i & 0xff)) or bb[4095] != @as(u8, @intCast(i & 0xff))) ok = false;
}
}
check("many allocations (heap growth) stay valid", ok);
for (blocks) |slot| {
if (slot) |bb| a.free(bb);
}
// A std container backed by the kernel heap.
var list: std.ArrayList(u32) = .empty;
var sum: u64 = 0;
var expected: u64 = 0;
var i: u32 = 0;
var list_ok = true;
while (i < 1000) : (i += 1) {
list.append(a, i) catch {
list_ok = false;
};
expected += i;
}
for (list.items) |v| sum += v;
list.deinit(a);
check("std.ArrayList on the kernel heap", list_ok and sum == expected);
result();
}
/// Verify the calibrated clocks: sane measured frequencies, monotonic uptime that
/// advances with real ticks, and — the point of the TSC clock — nanosecond
/// resolution far finer than the 1 ms tick, with the unit functions consistent.
fn clock() void {
log("DANOS-TEST-BEGIN: clock\n", .{});
const lapic = arch.lapicHz();
check("LAPIC frequency measured", lapic > 1_000_000 and lapic < 100_000_000_000);
const tsc = arch.tscHz();
check("TSC frequency measured", tsc > 100_000_000 and tsc < 100_000_000_000);
// Uptime advances over ~5 real ticks (1000 Hz => 1 tick == 1 ms).
const start_ticks = arch.ticks();
const start_ms = arch.millis();
var spins: u64 = 0;
while (arch.ticks() < start_ticks + 5 and spins < 5_000_000_000) spins +%= 1;
const elapsed_ms = arch.millis() - start_ms;
check("uptime advances with ticks", elapsed_ms >= 5 and elapsed_ms < 100);
// Sub-millisecond resolution: spin until nanos() first advances, then confirm
// that first step happened within a millisecond — so nanos() resolves finer
// than the 1 ms tick (a tick clock's smallest step *is* 1 ms). Spinning to the
// first change is robust to QEMU's coarse TSC update granularity.
const n1 = arch.nanos();
var s2: u64 = 0;
while (arch.nanos() == n1 and s2 < 10_000_000) s2 +%= 1;
const n2 = arch.nanos();
check("nanos() has sub-millisecond resolution", n2 > n1 and (n2 - n1) < 1_000_000);
// The unit functions agree (within rounding).
const ns = arch.nanos();
check("nanos/micros/millis are consistent", diffWithin(arch.micros(), ns / 1000, 1000) and diffWithin(arch.millis(), ns / 1_000_000, 2));
result();
}
fn diffWithin(a: u64, b: u64, tol: u64) bool {
return if (a > b) a - b <= tol else b - a <= tol;
}
// --- scheduler tests ------------------------------------------------------
var counters = [_]u64{0} ** 3;
fn spin0() void {
const p: *volatile u64 = &counters[0];
while (true) p.* = p.* +% 1;
}
fn spin1() void {
const p: *volatile u64 = &counters[1];
while (true) p.* = p.* +% 1;
}
fn spin2() void {
const p: *volatile u64 = &counters[2];
while (true) p.* = p.* +% 1;
}
/// Preemption: spawn three tasks that busy-loop *without* yielding. If they all
/// make progress, the timer must be preempting between them (and the context
/// switch works) — because nothing yields voluntarily.
fn schedTest() void {
log("DANOS-TEST-BEGIN: sched\n", .{});
counters = .{ 0, 0, 0 };
sched.spawn(spin0, 4);
sched.spawn(spin1, 4);
sched.spawn(spin2, 4);
const c0: *volatile u64 = &counters[0];
const c1: *volatile u64 = &counters[1];
const c2: *volatile u64 = &counters[2];
var spins: u64 = 0;
while ((c0.* == 0 or c1.* == 0 or c2.* == 0) and spins < 5_000_000_000) spins +%= 1;
check("all three non-yielding tasks made progress (preemption)", c0.* > 0 and c1.* > 0 and c2.* > 0);
result();
}
var run_order = [_]u8{0} ** 4;
var run_n: usize = 0;
fn recordExit(priority: u8) void {
run_order[run_n] = priority;
run_n += 1;
sched.exit();
}
fn taskHigh() void {
recordExit(6);
}
fn taskMid() void {
recordExit(4);
}
fn taskLow() void {
recordExit(2);
}
/// Fixed priority: with preemption off (deterministic), spawn tasks at three
/// priorities and let them run cooperatively. They must run highest-first.
fn priorityTest() void {
log("DANOS-TEST-BEGIN: priority\n", .{});
sched.setPreemption(false);
sched.setPriority(1); // above the idle task (0), below the workers — runs last
run_n = 0;
sched.spawn(taskLow, 2);
sched.spawn(taskMid, 4);
sched.spawn(taskHigh, 6);
while (run_n < 3) sched.yield(); // regain control only once the workers are done
check("tasks ran highest-priority first", run_order[0] == 6 and run_order[1] == 4 and run_order[2] == 2);
sched.setPriority(4);
sched.setPreemption(true);
result();
}
var event_wq: sched.WaitQueue = .{};
var event_stage: u32 = 0;
fn eventWaiter() void {
event_stage = 1; // reached the wait
sched.wait(&event_wq); // block until woken
event_stage = 3; // woken and resumed
sched.exit();
}
/// Event-based blocking: a task blocks on a wait queue and is woken. The waiter is
/// higher priority, so waking it preempts us and it runs to completion at once.
fn eventTest() void {
log("DANOS-TEST-BEGIN: event\n", .{});
event_stage = 0;
sched.spawn(eventWaiter, 6); // higher priority than this task (4)
var spins: u64 = 0;
while (event_stage != 1 and spins < 1_000_000_000) : (spins += 1) sched.yield();
check("waiter reached the wait and blocked", event_stage == 1);
sched.wake(&event_wq);
check("wake resumed the blocked waiter (preempting)", event_stage == 3);
result();
}
var channel: ipc.Channel(u64, 4) = .{};
var recv_sum: u64 = 0;
var recv_count: u64 = 0;
fn producer() void {
var i: u64 = 1;
while (i <= 100) : (i += 1) channel.send(i);
sched.exit();
}
fn consumer() void {
var n: u64 = 0;
while (n < 100) : (n += 1) {
recv_sum += channel.recv();
recv_count += 1;
}
sched.exit();
}
/// IPC: a producer and consumer pass 100 messages through a 4-slot channel. The
/// small buffer forces the channel full and empty repeatedly, exercising both the
/// blocking-send and blocking-recv paths. The messages must arrive intact.
fn ipcTest() void {
log("DANOS-TEST-BEGIN: ipc\n", .{});
channel = .{};
recv_sum = 0;
recv_count = 0;
sched.spawn(consumer, 5); // above this task (4) so they run and we observe after
sched.spawn(producer, 5);
var spins: u64 = 0;
while (recv_count < 100 and spins < 2_000_000_000) : (spins += 1) sched.yield();
check("all 100 messages received", recv_count == 100);
check("messages arrived intact (sum 1..100 == 5050)", recv_sum == 5050);
result();
}
/// Blocking: sleep(50) should block this task for about 50 ms (measured on the
/// calibrated clock) — not busy-wait — while the idle task runs.
fn sleepTest() void {
log("DANOS-TEST-BEGIN: sleep\n", .{});
const t0 = arch.millis();
sched.sleep(50);
const elapsed = arch.millis() - t0;
check("sleep(50) blocked for ~50 ms", elapsed >= 50 and elapsed <= 70);
result();
}
fn faultInvalidOpcode() void {
log("DANOS-TEST-BEGIN: fault-ud\n", .{});
asm volatile ("ud2");
}
/// Verify NX: fetching an instruction from a data page (mapped no-execute) faults.
fn faultNoExecute() void {
log("DANOS-TEST-BEGIN: fault-nx\n", .{});
var scratch: u64 = 0xC3; // a lone `ret` — harmless if NX somehow let it run
const f: *const fn () void = @ptrFromInt(@intFromPtr(&scratch));
f(); // instruction fetch from an NX page -> #PF before it executes
log("DANOS-TEST-RESULT: FAIL (NX not enforced)\n", .{});
}
/// Verify the null guard: dereferencing address 0 (page 0 left unmapped) faults.
fn faultNull() void {
log("DANOS-TEST-BEGIN: fault-null\n", .{});
// Launder the address through empty asm so the compiler no longer knows it's
// 0 (otherwise it folds a null-pointer safety panic instead of doing the real
// access). `allowzero` skips the same null check on the cast. The write then
// hits the unmapped page 0 and takes a real hardware #PF.
var addr: u64 = 0;
addr = asm ("" : [ret] "=r" (-> u64) : [in] "0" (addr));
const p: *allowzero volatile u64 = @ptrFromInt(addr);
p.* = 1;
}
fn faultPageFault() void {
log("DANOS-TEST-BEGIN: fault-pf\n", .{});
// Runtime address so the backend emits a register store (not a `mov moffs`,
// which the self-hosted x86_64 backend can't encode).
var addr: u64 = 0xdeadbeef000; // well above all mapped RAM
const p: *volatile u64 = @ptrFromInt(addr);
p.* = 1;
addr += 0;
}
fn faultDoubleFault() void {
log("DANOS-TEST-BEGIN: fault-df\n", .{});
arch.disableInterrupts(); // so only the ud2 delivery (not a timer tick) triggers the #DF
// Point RSP at unmapped memory, then fault: the CPU can't push the fault
// frame, which escalates to #DF — survivable only because #DF runs on IST1.
var bad_sp: u64 = 0x5000000000;
asm volatile (
\\mov %[sp], %%rsp
\\ud2
:
: [sp] "r" (bad_sp),
: .{ .memory = true }
);
bad_sp += 0;
}