moving kernel code to kernel/
This commit is contained in:
@@ -0,0 +1,202 @@
|
||||
//! Local APIC and its timer — the source of device interrupts.
|
||||
//!
|
||||
//! Modern x86 routes interrupts through the per-CPU Local APIC (the legacy 8259
|
||||
//! PIC is remapped out of the way and masked). The LAPIC also has a built-in
|
||||
//! timer, which is the simplest device interrupt to bring up: it needs no
|
||||
//! external routing, just a vector and a count. We use it as danos's heartbeat.
|
||||
//!
|
||||
//! The LAPIC is memory-mapped (default physical 0xFEE00000, inside our identity
|
||||
//! map). Every interrupt must be acknowledged with an end-of-interrupt write, or
|
||||
//! the LAPIC won't deliver the next one.
|
||||
|
||||
const io = @import("io.zig");
|
||||
|
||||
/// IDT vector the timer fires on (in the device range, >= 32).
|
||||
pub const timer_vector = 32;
|
||||
/// Spurious-interrupt vector. Low nibble 0xF by convention; also in our gate
|
||||
/// range so a stray spurious interrupt lands on a valid (no-op) handler.
|
||||
const spurious_vector = 47;
|
||||
|
||||
// LAPIC register offsets.
|
||||
const reg_spurious = 0x0F0;
|
||||
const reg_eoi = 0x0B0;
|
||||
const reg_lvt_timer = 0x320;
|
||||
const reg_timer_initial = 0x380;
|
||||
const reg_timer_current = 0x390;
|
||||
const reg_timer_divide = 0x3E0;
|
||||
|
||||
const lvt_masked = 1 << 16;
|
||||
const lvt_periodic = 1 << 17;
|
||||
const timer_divide_16 = 0x3;
|
||||
|
||||
const ia32_apic_base_msr = 0x1B;
|
||||
|
||||
/// LAPIC MMIO base. A runtime var (not a constant) both because we read it from
|
||||
/// the MSR and so register writes compile to normal stores rather than a
|
||||
/// `mov moffs`, which the self-hosted backend can't encode.
|
||||
var base: usize = 0xFEE00000;
|
||||
|
||||
var tick_count: u64 = 0;
|
||||
|
||||
/// LAPIC timer counts per millisecond, measured against the PIT (see calibrate).
|
||||
/// At divide-by-16, this is the effective counting rate.
|
||||
var ticks_per_ms: u32 = 0;
|
||||
/// The periodic-interrupt frequency the timer is armed at, once initTimer runs.
|
||||
var timer_hz: u32 = 0;
|
||||
|
||||
/// TSC (Time Stamp Counter) calibration: cycles per second, and the count at boot.
|
||||
/// The TSC is a per-core cycle counter, giving a ~nanosecond high-resolution
|
||||
/// monotonic clock — far finer than the millisecond timer tick.
|
||||
var tsc_hz: u64 = 0;
|
||||
var tsc_base: u64 = 0;
|
||||
|
||||
/// Read the 64-bit Time Stamp Counter.
|
||||
fn rdtsc() u64 {
|
||||
var low: u32 = undefined;
|
||||
var high: u32 = undefined;
|
||||
asm volatile ("rdtsc"
|
||||
: [low] "={eax}" (low),
|
||||
[high] "={edx}" (high),
|
||||
);
|
||||
return (@as(u64, high) << 32) | low;
|
||||
}
|
||||
|
||||
fn read(reg: u32) u32 {
|
||||
return @as(*volatile u32, @ptrFromInt(base + reg)).*;
|
||||
}
|
||||
fn write(reg: u32, value: u32) void {
|
||||
@as(*volatile u32, @ptrFromInt(base + reg)).* = value;
|
||||
}
|
||||
|
||||
/// Move the legacy 8259 PIC's vectors to 0x20-0x2F (clear of the CPU exception
|
||||
/// vectors) and mask every line, so it can't deliver interrupts behind the APIC.
|
||||
fn remapAndMaskPic() void {
|
||||
io.outb(0x20, 0x11); // start init (cascade mode)
|
||||
io.outb(0xA0, 0x11);
|
||||
io.outb(0x21, 0x20); // master offset 0x20
|
||||
io.outb(0xA1, 0x28); // slave offset 0x28
|
||||
io.outb(0x21, 0x04); // tell master about slave on IRQ2
|
||||
io.outb(0xA1, 0x02);
|
||||
io.outb(0x21, 0x01); // 8086 mode
|
||||
io.outb(0xA1, 0x01);
|
||||
io.outb(0x21, 0xFF); // mask all
|
||||
io.outb(0xA1, 0xFF);
|
||||
}
|
||||
|
||||
/// Enable the Local APIC: mask the PIC, set the global-enable MSR bit, and
|
||||
/// software-enable the APIC via its spurious-vector register.
|
||||
pub fn init() void {
|
||||
remapAndMaskPic();
|
||||
|
||||
const msr = io.rdmsr(ia32_apic_base_msr);
|
||||
base = @intCast(msr & 0xFFFFF000); // physical base is bits 12+
|
||||
io.wrmsr(ia32_apic_base_msr, msr | (1 << 11)); // global enable
|
||||
|
||||
write(reg_spurious, 0x100 | spurious_vector); // bit 8 = software enable
|
||||
}
|
||||
|
||||
/// Measure the LAPIC timer's and the TSC's rates against the PIT (channel 2, which
|
||||
/// can be polled without interrupts). We run the LAPIC timer one-shot from its max
|
||||
/// count and snapshot the TSC while the PIT counts out a known 10 ms, then see how
|
||||
/// far each got. This gives real time, which the RTOS timing guarantees depend on.
|
||||
pub fn calibrate() void {
|
||||
const pit_hz = 1_193_182;
|
||||
const calib_ms = 10;
|
||||
const pit_count: u16 = @intCast(pit_hz / 1000 * calib_ms);
|
||||
|
||||
// LAPIC timer: divide 16, masked (no interrupt — we just want the count),
|
||||
// counting down from the maximum.
|
||||
write(reg_timer_divide, timer_divide_16);
|
||||
write(reg_lvt_timer, lvt_masked);
|
||||
write(reg_timer_initial, 0xFFFFFFFF);
|
||||
|
||||
// PIT channel 2, mode 0 (interrupt on terminal count): load the count with the
|
||||
// gate low, then raise the gate to start it counting.
|
||||
io.outb(0x61, io.inb(0x61) & 0xFC); // speaker off, gate low
|
||||
io.outb(0x43, 0xB0); // channel 2, lo/hi byte, mode 0
|
||||
io.outb(0x42, @truncate(pit_count));
|
||||
io.outb(0x42, @truncate(pit_count >> 8));
|
||||
|
||||
const tsc_start = rdtsc();
|
||||
io.outb(0x61, (io.inb(0x61) & 0xFC) | 0x01); // gate high -> start
|
||||
while (io.inb(0x61) & 0x20 == 0) {} // poll channel-2 output until terminal count
|
||||
const tsc_end = rdtsc();
|
||||
|
||||
const elapsed = 0xFFFFFFFF - read(reg_timer_current);
|
||||
write(reg_timer_initial, 0); // stop the timer
|
||||
|
||||
ticks_per_ms = elapsed / calib_ms;
|
||||
tsc_hz = (tsc_end -% tsc_start) * (1000 / calib_ms); // cycles/10ms -> cycles/s
|
||||
tsc_base = rdtsc(); // the clock's zero point (boot)
|
||||
}
|
||||
|
||||
/// Arm the LAPIC timer to fire on `timer_vector` at `hz` (periodic). Requires
|
||||
/// calibrate() to have run.
|
||||
pub fn initTimer(hz: u32) void {
|
||||
timer_hz = hz;
|
||||
const count = @as(u64, ticks_per_ms) * 1000 / hz; // counts per (1/hz) second
|
||||
write(reg_timer_divide, timer_divide_16);
|
||||
write(reg_lvt_timer, timer_vector | lvt_periodic);
|
||||
write(reg_timer_initial, @intCast(count));
|
||||
}
|
||||
|
||||
/// Configured periodic-interrupt frequency (Hz).
|
||||
pub fn frequencyHz() u32 {
|
||||
return timer_hz;
|
||||
}
|
||||
|
||||
/// Measured LAPIC timer frequency (Hz), for reporting/sanity checks.
|
||||
pub fn lapicHz() u64 {
|
||||
return @as(u64, ticks_per_ms) * 1000;
|
||||
}
|
||||
|
||||
/// Measured TSC frequency (Hz).
|
||||
pub fn tscHz() u64 {
|
||||
return tsc_hz;
|
||||
}
|
||||
|
||||
// Monotonic high-resolution clock, from the TSC. A function per resolution, each
|
||||
// scaling the cycle delta directly at its unit (the 128-bit intermediate avoids
|
||||
// overflow across a long uptime). nanos() resolves to a few ns; millis() is what
|
||||
// the scheduler uses for sleep deadlines.
|
||||
|
||||
pub fn nanos() u64 {
|
||||
if (tsc_hz == 0) return 0;
|
||||
return @intCast(@as(u128, rdtsc() -% tsc_base) * 1_000_000_000 / tsc_hz);
|
||||
}
|
||||
|
||||
pub fn micros() u64 {
|
||||
if (tsc_hz == 0) return 0;
|
||||
return @intCast(@as(u128, rdtsc() -% tsc_base) * 1_000_000 / tsc_hz);
|
||||
}
|
||||
|
||||
pub fn millis() u64 {
|
||||
if (tsc_hz == 0) return 0;
|
||||
return @intCast(@as(u128, rdtsc() -% tsc_base) * 1_000 / tsc_hz);
|
||||
}
|
||||
|
||||
/// Acknowledge the current interrupt so the LAPIC will deliver the next one.
|
||||
pub fn eoi() void {
|
||||
write(reg_eoi, 0);
|
||||
}
|
||||
|
||||
/// Optional callback run each tick (the scheduler registers it for preemption).
|
||||
var on_tick: ?*const fn () void = null;
|
||||
|
||||
pub fn setTickHook(hook: *const fn () void) void {
|
||||
on_tick = hook;
|
||||
}
|
||||
|
||||
/// The timer interrupt handler: advance the monotonic tick count, then run the
|
||||
/// tick hook (which may switch tasks). The interrupt is already acknowledged by
|
||||
/// the dispatcher before we get here, so a task switch here doesn't stall it.
|
||||
pub fn timerTick() void {
|
||||
tick_count +%= 1;
|
||||
if (on_tick) |hook| hook();
|
||||
}
|
||||
|
||||
/// Number of timer ticks so far. Volatile load: the count is bumped
|
||||
/// asynchronously by the interrupt handler, so callers must re-read memory.
|
||||
pub fn ticks() u64 {
|
||||
return @as(*const volatile u64, &tick_count).*;
|
||||
}
|
||||
@@ -0,0 +1,192 @@
|
||||
//! x86_64 CPU operations. This is the "arch" module: the generic kernel imports
|
||||
//! it as `@import("arch")` and never names x86_64 directly, so a second
|
||||
//! architecture is added by pointing that module at a different directory in
|
||||
//! build.zig — no change to the generic code. Keep everything CPU-specific here
|
||||
//! (halt, the descriptor tables, later paging), and nothing generic.
|
||||
|
||||
const danos = @import("danos");
|
||||
const gdt = @import("gdt.zig");
|
||||
const tss = @import("tss.zig");
|
||||
const idt = @import("idt.zig");
|
||||
const paging = @import("paging.zig");
|
||||
const serial = @import("serial.zig");
|
||||
const apic = @import("apic.zig");
|
||||
|
||||
/// The saved register/trap frame passed to a fault handler.
|
||||
pub const CpuState = idt.CpuState;
|
||||
|
||||
/// Bring up the serial port (the kernel's machine-readable log). No dependencies,
|
||||
/// so it can be the very first thing called.
|
||||
pub fn serialInit() void {
|
||||
serial.init();
|
||||
}
|
||||
|
||||
/// Write bytes to the serial port.
|
||||
pub fn serialWrite(bytes: []const u8) void {
|
||||
serial.write(bytes);
|
||||
}
|
||||
|
||||
/// Set up the CPU's descriptor tables: our own GDT, the TSS (with an interrupt
|
||||
/// stack for double faults), then the IDT with exception handlers. After this a
|
||||
/// CPU fault is reported instead of triple-faulting. Install the fault handler
|
||||
/// (setFaultHandler) first so early faults are caught.
|
||||
pub fn init() void {
|
||||
gdt.init();
|
||||
tss.init();
|
||||
idt.init();
|
||||
}
|
||||
|
||||
/// Build the kernel's own page tables (with real permissions) and switch onto
|
||||
/// them. Needs the frame allocator and the boot info (for the memory map and the
|
||||
/// kernel's segment layout). Call once the frame allocator is up.
|
||||
pub fn enablePaging(allocFrame: *const fn () ?u64, boot_info: *const danos.BootInfo) void {
|
||||
paging.init(allocFrame, boot_info);
|
||||
}
|
||||
|
||||
/// Map a page into the kernel address space (non-executable). For the heap, etc.
|
||||
pub fn mapPage(virt: u64, phys: u64, writable: bool) void {
|
||||
paging.map(virt, phys, writable);
|
||||
}
|
||||
|
||||
/// Remove a kernel mapping.
|
||||
pub fn unmapPage(virt: u64) void {
|
||||
paging.unmap(virt);
|
||||
}
|
||||
|
||||
/// CR3 holds the physical address of the active top-level page table.
|
||||
pub fn readCr3() u64 {
|
||||
return asm volatile ("mov %%cr3, %[out]"
|
||||
: [out] "=r" (-> u64),
|
||||
);
|
||||
}
|
||||
|
||||
/// Kernel tick rate: 1000 Hz (1 ms), the scheduler's time quantum.
|
||||
pub const timer_hz = 1000;
|
||||
|
||||
/// Enable the Local APIC, calibrate its timer against the PIT, and start it firing
|
||||
/// at `timer_hz` — the kernel's real-time heartbeat. Interrupts still have to be
|
||||
/// unmasked with enableInterrupts() to be delivered.
|
||||
pub fn startTimer() void {
|
||||
apic.init();
|
||||
apic.calibrate();
|
||||
idt.setHandler(apic.timer_vector, apic.timerTick);
|
||||
apic.initTimer(timer_hz);
|
||||
}
|
||||
|
||||
/// Number of timer ticks since startTimer().
|
||||
pub fn ticks() u64 {
|
||||
return apic.ticks();
|
||||
}
|
||||
|
||||
// Monotonic high-resolution clock (from the TSC), one function per resolution.
|
||||
pub fn nanos() u64 {
|
||||
return apic.nanos();
|
||||
}
|
||||
pub fn micros() u64 {
|
||||
return apic.micros();
|
||||
}
|
||||
pub fn millis() u64 {
|
||||
return apic.millis();
|
||||
}
|
||||
|
||||
/// Measured LAPIC timer / TSC frequencies in Hz (from calibration).
|
||||
pub fn lapicHz() u64 {
|
||||
return apic.lapicHz();
|
||||
}
|
||||
pub fn tscHz() u64 {
|
||||
return apic.tscHz();
|
||||
}
|
||||
|
||||
/// Unmask maskable interrupts (`sti`) so device interrupts get delivered.
|
||||
pub fn enableInterrupts() void {
|
||||
asm volatile ("sti");
|
||||
}
|
||||
|
||||
/// Mask maskable interrupts (`cli`).
|
||||
pub fn disableInterrupts() void {
|
||||
asm volatile ("cli");
|
||||
}
|
||||
|
||||
/// Disable interrupts and return the previous flags, so a nested critical section
|
||||
/// can restore the caller's state rather than blindly re-enabling. Pairs with
|
||||
/// restoreInterrupts.
|
||||
pub fn saveInterrupts() u64 {
|
||||
var flags: u64 = undefined;
|
||||
asm volatile (
|
||||
\\pushfq
|
||||
\\pop %[f]
|
||||
\\cli
|
||||
: [f] "=r" (flags),
|
||||
:
|
||||
: .{ .memory = true }
|
||||
);
|
||||
return flags;
|
||||
}
|
||||
|
||||
/// Re-enable interrupts only if they were enabled when `flags` was captured.
|
||||
pub fn restoreInterrupts(flags: u64) void {
|
||||
if (flags & 0x200 != 0) asm volatile ("sti" ::: .{ .memory = true }); // bit 9 = IF
|
||||
}
|
||||
|
||||
/// Register a callback the timer interrupt invokes each tick (e.g. the scheduler).
|
||||
pub fn setTickHook(hook: *const fn () void) void {
|
||||
apic.setTickHook(hook);
|
||||
}
|
||||
|
||||
// --- context switching (for the scheduler) -------------------------------
|
||||
|
||||
/// Save the current task's registers/stack and resume `new_rsp`; the old stack
|
||||
/// pointer is written to `old_rsp`. Defined in isr.s.
|
||||
extern fn switch_context(old_rsp: *usize, new_rsp: usize) callconv(.c) void;
|
||||
|
||||
pub fn switchContext(old_rsp: *usize, new_rsp: usize) void {
|
||||
switch_context(old_rsp, new_rsp);
|
||||
}
|
||||
|
||||
/// Build the initial stack for a new task so that switching to it lands in
|
||||
/// `task_trampoline`, which then calls `entry`. Returns the saved stack pointer.
|
||||
/// The layout must match switch_context's push order (callee-saved, then the
|
||||
/// return address on top); `entry` is smuggled in via the r15 slot.
|
||||
pub fn initTaskStack(stack_top: usize, entry: usize) usize {
|
||||
const trampoline = @extern(*const anyopaque, .{ .name = "task_trampoline" });
|
||||
var sp = stack_top;
|
||||
const push = struct {
|
||||
fn f(p: *usize, value: usize) void {
|
||||
p.* -= @sizeOf(usize);
|
||||
@as(*usize, @ptrFromInt(p.*)).* = value;
|
||||
}
|
||||
}.f;
|
||||
push(&sp, @intFromPtr(trampoline)); // return address for switch_context's `ret`
|
||||
push(&sp, 0); // rbx
|
||||
push(&sp, 0); // rbp
|
||||
push(&sp, 0); // r12
|
||||
push(&sp, 0); // r13
|
||||
push(&sp, 0); // r14
|
||||
push(&sp, entry); // r15 -> task entry, read by task_trampoline
|
||||
return sp;
|
||||
}
|
||||
|
||||
/// Route CPU exceptions to `handler`, which receives the trap frame and does not
|
||||
/// return. Until set, faults just halt the core.
|
||||
pub fn setFaultHandler(handler: *const fn (*const CpuState) noreturn) void {
|
||||
idt.on_fault = handler;
|
||||
}
|
||||
|
||||
/// A human-readable name for a CPU exception vector.
|
||||
pub fn vectorName(vector: u64) []const u8 {
|
||||
return idt.vectorName(vector);
|
||||
}
|
||||
|
||||
/// CR2 holds the faulting linear address after a page fault (#PF, vector 14).
|
||||
pub fn readCr2() u64 {
|
||||
return asm volatile ("mov %%cr2, %[out]"
|
||||
: [out] "=r" (-> u64),
|
||||
);
|
||||
}
|
||||
|
||||
/// Park the core forever. `hlt` drops it into a low-power idle until the next
|
||||
/// interrupt; the loop re-halts on every wake so the stop is permanent. See
|
||||
/// docs/halting.md for the full reasoning.
|
||||
pub fn halt() noreturn {
|
||||
while (true) asm volatile ("hlt");
|
||||
}
|
||||
@@ -0,0 +1,54 @@
|
||||
//! Global Descriptor Table. In long mode segmentation is mostly vestigial, but
|
||||
//! the CPU still needs valid code/data segment descriptors, and the IDT's gates
|
||||
//! reference a code selector — so we install our own flat GDT with known
|
||||
//! selectors (0x08 kernel code, 0x10 kernel data) rather than trusting whatever
|
||||
//! the firmware left in place.
|
||||
|
||||
/// Selectors into the table below (index * 8).
|
||||
pub const kernel_code = 0x08;
|
||||
pub const kernel_data = 0x10;
|
||||
pub const tss_selector = 0x18;
|
||||
|
||||
/// Flat 64-bit descriptors. Base/limit are ignored in long mode; what matters is
|
||||
/// the access byte and, for code, the long-mode (L) flag.
|
||||
/// code: present, ring 0, executable, readable, L=1 -> 0x00AF9A00_0000FFFF
|
||||
/// data: present, ring 0, writable -> 0x00CF9200_0000FFFF
|
||||
/// The last two slots hold one 16-byte TSS descriptor, filled in by setTss.
|
||||
var table = [_]u64{
|
||||
0, // null descriptor (required)
|
||||
0x00AF9A000000FFFF, // kernel code (0x08)
|
||||
0x00CF92000000FFFF, // kernel data (0x10)
|
||||
0, // TSS descriptor low (0x18)
|
||||
0, // TSS descriptor high
|
||||
};
|
||||
|
||||
/// Fill the 64-bit TSS system descriptor (two GDT slots) so the task register can
|
||||
/// point at our TSS. Type 0x89 = present, ring 0, available 64-bit TSS.
|
||||
pub fn setTss(base: u64, limit: u64) void {
|
||||
table[3] = (limit & 0xFFFF) |
|
||||
((base & 0xFFFF) << 16) |
|
||||
(((base >> 16) & 0xFF) << 32) |
|
||||
(@as(u64, 0x89) << 40) |
|
||||
(((limit >> 16) & 0xF) << 48) |
|
||||
(((base >> 24) & 0xFF) << 56);
|
||||
table[4] = (base >> 32) & 0xFFFFFFFF;
|
||||
}
|
||||
|
||||
/// The operand `lgdt` wants: table byte-length minus one, then its address.
|
||||
const Descriptor = packed struct {
|
||||
limit: u16,
|
||||
base: u64,
|
||||
};
|
||||
|
||||
/// Loads the GDT and reloads the segment registers (including CS). Defined in
|
||||
/// isr.s — it uses the selectors 0x08 (code) and 0x10 (data) that match `table`.
|
||||
extern fn gdt_flush(descriptor: *const Descriptor) callconv(.c) void;
|
||||
|
||||
/// Install our GDT and switch onto its segments.
|
||||
pub fn init() void {
|
||||
const descriptor = Descriptor{
|
||||
.limit = @sizeOf(@TypeOf(table)) - 1,
|
||||
.base = @intFromPtr(&table),
|
||||
};
|
||||
gdt_flush(&descriptor);
|
||||
}
|
||||
@@ -0,0 +1,155 @@
|
||||
//! Interrupt Descriptor Table, CPU-exception handlers, and device-interrupt
|
||||
//! dispatch. Without this, any fault (a stray pointer, a bad page-table entry)
|
||||
//! triple-faults and silently resets the machine. With it, the CPU vectors into
|
||||
//! our stubs, which capture the register state and hand it to a dispatcher.
|
||||
//!
|
||||
//! Vectors split in two: 0-31 are CPU exceptions (terminal — reported and
|
||||
//! halted); 32+ are device interrupts (a registered handler runs, the APIC is
|
||||
//! acknowledged, and we return to the interrupted code).
|
||||
|
||||
const gdt = @import("gdt.zig");
|
||||
const tss = @import("tss.zig");
|
||||
const apic = @import("apic.zig");
|
||||
|
||||
/// Highest vector we install a gate/stub for (exceptions 0-31 plus the device
|
||||
/// range 32-47, which covers the timer and the spurious vector).
|
||||
const gate_count = 48;
|
||||
|
||||
/// A device-interrupt handler. It doesn't get the trap frame (a timer or keyboard
|
||||
/// handler doesn't need the interrupted registers); add that if one ever does.
|
||||
pub const Handler = *const fn () void;
|
||||
|
||||
var handlers = [_]?Handler{null} ** 256;
|
||||
|
||||
/// Register `handler` for a device-interrupt `vector` (>= 32).
|
||||
pub fn setHandler(vector: usize, handler: Handler) void {
|
||||
handlers[vector] = handler;
|
||||
}
|
||||
|
||||
/// The register + trap frame the ISR stubs build on the stack, laid out so the
|
||||
/// lowest address (where RSP points when we call the handler) is the first field.
|
||||
/// See the push order in `isrCommon` below.
|
||||
pub const CpuState = extern struct {
|
||||
r15: u64,
|
||||
r14: u64,
|
||||
r13: u64,
|
||||
r12: u64,
|
||||
r11: u64,
|
||||
r10: u64,
|
||||
r9: u64,
|
||||
r8: u64,
|
||||
rbp: u64,
|
||||
rdi: u64,
|
||||
rsi: u64,
|
||||
rdx: u64,
|
||||
rcx: u64,
|
||||
rbx: u64,
|
||||
rax: u64,
|
||||
vector: u64, // pushed by the per-vector stub
|
||||
error_code: u64, // real one from the CPU, or 0 pushed by the stub
|
||||
rip: u64, // from here down: pushed by the CPU on entry
|
||||
cs: u64,
|
||||
rflags: u64,
|
||||
rsp: u64,
|
||||
ss: u64,
|
||||
};
|
||||
|
||||
/// Where a fault is reported. The kernel overrides this (see setFaultHandler) with
|
||||
/// something that prints to the console; until then, just stop.
|
||||
pub var on_fault: *const fn (*const CpuState) noreturn = defaultFault;
|
||||
|
||||
fn defaultFault(_: *const CpuState) noreturn {
|
||||
while (true) asm volatile ("hlt");
|
||||
}
|
||||
|
||||
/// Names for the 32 defined exception vectors, for readable output.
|
||||
const names = [_][]const u8{
|
||||
"divide error", "debug",
|
||||
"NMI", "breakpoint",
|
||||
"overflow", "bound range exceeded",
|
||||
"invalid opcode", "device not available",
|
||||
"double fault", "coprocessor segment overrun",
|
||||
"invalid TSS", "segment not present",
|
||||
"stack-segment fault", "general protection fault",
|
||||
"page fault", "reserved (15)",
|
||||
"x87 floating-point", "alignment check",
|
||||
"machine check", "SIMD floating-point",
|
||||
"virtualization", "control protection",
|
||||
"reserved (22)", "reserved (23)",
|
||||
"reserved (24)", "reserved (25)",
|
||||
"reserved (26)", "reserved (27)",
|
||||
"hypervisor injection", "VMM communication",
|
||||
"security exception", "reserved (31)",
|
||||
};
|
||||
|
||||
pub fn vectorName(vector: u64) []const u8 {
|
||||
return if (vector < names.len) names[vector] else "unknown";
|
||||
}
|
||||
|
||||
/// A 64-bit IDT gate descriptor (16 bytes).
|
||||
const Gate = packed struct {
|
||||
offset_low: u16,
|
||||
selector: u16,
|
||||
ist: u8, // interrupt-stack-table index; 0 = use the current stack
|
||||
flags: u8, // present, DPL, gate type
|
||||
offset_mid: u16,
|
||||
offset_high: u32,
|
||||
reserved: u32 = 0,
|
||||
};
|
||||
|
||||
var idt = [_]Gate{std.mem.zeroes(Gate)} ** 256;
|
||||
|
||||
const Descriptor = packed struct {
|
||||
limit: u16,
|
||||
base: u64,
|
||||
};
|
||||
|
||||
/// Loads the IDT (`lidt`). Defined in isr.s.
|
||||
extern fn idt_flush(descriptor: *const Descriptor) callconv(.c) void;
|
||||
|
||||
fn setGate(vector: usize, handler: u64) void {
|
||||
idt[vector] = .{
|
||||
.offset_low = @truncate(handler),
|
||||
.selector = gdt.kernel_code,
|
||||
.ist = 0,
|
||||
.flags = 0x8E, // present, ring 0, 64-bit interrupt gate
|
||||
.offset_mid = @truncate(handler >> 16),
|
||||
.offset_high = @truncate(handler >> 32),
|
||||
};
|
||||
}
|
||||
|
||||
/// Point every installed vector at its stub (isr.s) and load the IDT.
|
||||
pub fn init() void {
|
||||
@setEvalBranchQuota(20000); // comptimePrint across all the gates adds up
|
||||
inline for (0..gate_count) |vector| {
|
||||
const stub = @extern(*const anyopaque, .{ .name = std.fmt.comptimePrint("isr{d}", .{vector}) });
|
||||
setGate(vector, @intFromPtr(stub));
|
||||
}
|
||||
// Run the double-fault handler (vector 8) on IST1: a #DF usually means the
|
||||
// current stack is unusable, so it needs a guaranteed-good one. See tss.zig.
|
||||
idt[8].ist = tss.double_fault_ist;
|
||||
const descriptor = Descriptor{
|
||||
.limit = @sizeOf(@TypeOf(idt)) - 1,
|
||||
.base = @intFromPtr(&idt),
|
||||
};
|
||||
idt_flush(&descriptor);
|
||||
}
|
||||
|
||||
/// Called by isr_common (isr.s) with a pointer to the trap frame. Exported so the
|
||||
/// assembly stubs can `call` it by name. Exceptions are terminal; device
|
||||
/// interrupts run their handler, get acknowledged, and return.
|
||||
export fn interruptDispatch(state: *const CpuState) callconv(.c) void {
|
||||
if (state.vector < 32) {
|
||||
on_fault(state); // CPU exception — never returns
|
||||
} else if (handlers[state.vector]) |handler| {
|
||||
// Acknowledge before running the handler: a handler that switches tasks
|
||||
// (the scheduler) may not return promptly, and the LAPIC mustn't wait on
|
||||
// it to deliver the next interrupt. Fine for edge-triggered sources like
|
||||
// the timer; a level-triggered device would need EOI after handling.
|
||||
apic.eoi();
|
||||
handler();
|
||||
}
|
||||
// else: spurious/unhandled device interrupt — don't acknowledge it
|
||||
}
|
||||
|
||||
const std = @import("std");
|
||||
@@ -0,0 +1,38 @@
|
||||
//! x86 port I/O and model-specific registers — the low-level primitives the
|
||||
//! serial port and the APIC talk to hardware through.
|
||||
|
||||
pub fn outb(port: u16, value: u8) void {
|
||||
asm volatile ("outb %[value], %[port]"
|
||||
:
|
||||
: [value] "{al}" (value),
|
||||
[port] "{dx}" (port),
|
||||
);
|
||||
}
|
||||
|
||||
pub fn inb(port: u16) u8 {
|
||||
return asm volatile ("inb %[port], %[value]"
|
||||
: [value] "={al}" (-> u8),
|
||||
: [port] "{dx}" (port),
|
||||
);
|
||||
}
|
||||
|
||||
/// Read a model-specific register (returns edx:eax combined).
|
||||
pub fn rdmsr(msr: u32) u64 {
|
||||
var low: u32 = undefined;
|
||||
var high: u32 = undefined;
|
||||
asm volatile ("rdmsr"
|
||||
: [low] "={eax}" (low),
|
||||
[high] "={edx}" (high),
|
||||
: [msr] "{ecx}" (msr),
|
||||
);
|
||||
return (@as(u64, high) << 32) | low;
|
||||
}
|
||||
|
||||
pub fn wrmsr(msr: u32, value: u64) void {
|
||||
asm volatile ("wrmsr"
|
||||
:
|
||||
: [msr] "{ecx}" (msr),
|
||||
[low] "{eax}" (@as(u32, @truncate(value))),
|
||||
[high] "{edx}" (@as(u32, @truncate(value >> 32))),
|
||||
);
|
||||
}
|
||||
@@ -0,0 +1,180 @@
|
||||
# x86_64 low-level entry code: the CPU-exception stubs, plus the GDT/IDT load
|
||||
# helpers. Kept in a dedicated assembly file rather than inline asm because these
|
||||
# need real labels and cross-symbol jumps/calls (isr_common, exceptionHandler),
|
||||
# and because `lgdt`/`lidt` memory operands aren't expressible in Zig inline asm.
|
||||
#
|
||||
# Each exception vector normalises the stack to a uniform trap frame — a dummy
|
||||
# error code where the CPU pushes none, then the vector number — and jumps to the
|
||||
# shared tail, which saves the general registers and calls the Zig handler with a
|
||||
# pointer to the frame (matching src/arch/x86_64/idt.zig's CpuState).
|
||||
|
||||
.text
|
||||
|
||||
# gdt_flush(rdi = *GDT descriptor): load the GDT, reload the data segment
|
||||
# registers to the data selector, and reload CS to the code selector. CS can't be
|
||||
# set with mov, so we far-return through the caller's own return address.
|
||||
.global gdt_flush
|
||||
gdt_flush:
|
||||
lgdt (%rdi)
|
||||
mov $0x10, %ax # kernel data selector
|
||||
mov %ax, %ds
|
||||
mov %ax, %es
|
||||
mov %ax, %ss
|
||||
mov %ax, %fs
|
||||
mov %ax, %gs
|
||||
pop %rax # caller's return address
|
||||
push $0x08 # kernel code selector (new CS)
|
||||
push %rax # return address (new RIP)
|
||||
lretq
|
||||
|
||||
# idt_flush(rdi = *IDT descriptor): load the IDT.
|
||||
.global idt_flush
|
||||
idt_flush:
|
||||
lidt (%rdi)
|
||||
ret
|
||||
|
||||
# load_tr(di = TSS selector): load the task register.
|
||||
.global load_tr
|
||||
load_tr:
|
||||
ltr %di
|
||||
ret
|
||||
|
||||
# switch_context(rdi = &old_task.rsp, rsi = new_task.rsp)
|
||||
# Cooperative context switch: save the callee-saved registers on the current
|
||||
# stack, stash the stack pointer in the old task, load the new task's stack
|
||||
# pointer, restore its callee-saved registers, and return into it. Caller-saved
|
||||
# registers are the compiler's responsibility (this looks like a normal call).
|
||||
.global switch_context
|
||||
switch_context:
|
||||
push %rbx
|
||||
push %rbp
|
||||
push %r12
|
||||
push %r13
|
||||
push %r14
|
||||
push %r15
|
||||
mov %rsp, (%rdi) # save old stack pointer into old_task.rsp
|
||||
mov %rsi, %rsp # switch to the new task's stack
|
||||
pop %r15
|
||||
pop %r14
|
||||
pop %r13
|
||||
pop %r12
|
||||
pop %rbp
|
||||
pop %rbx
|
||||
ret # return into the new task's saved instruction pointer
|
||||
|
||||
# task_trampoline: the first thing a freshly-spawned task runs. init_task_stack
|
||||
# leaves its entry function in r15. New tasks start with interrupts enabled.
|
||||
.global task_trampoline
|
||||
task_trampoline:
|
||||
sti
|
||||
call *%r15 # call the task entry (fn() void)
|
||||
1: hlt # if the entry returns, idle (still preemptible)
|
||||
jmp 1b
|
||||
|
||||
# Stub for a vector the CPU does NOT push an error code for: push a dummy 0.
|
||||
.macro STUB_NOERR vec
|
||||
.global isr\vec
|
||||
isr\vec:
|
||||
pushq $0
|
||||
pushq $\vec
|
||||
jmp isr_common
|
||||
.endm
|
||||
|
||||
# Stub for a vector the CPU DOES push an error code for: leave it in place.
|
||||
.macro STUB_ERR vec
|
||||
.global isr\vec
|
||||
isr\vec:
|
||||
pushq $\vec
|
||||
jmp isr_common
|
||||
.endm
|
||||
|
||||
STUB_NOERR 0
|
||||
STUB_NOERR 1
|
||||
STUB_NOERR 2
|
||||
STUB_NOERR 3
|
||||
STUB_NOERR 4
|
||||
STUB_NOERR 5
|
||||
STUB_NOERR 6
|
||||
STUB_NOERR 7
|
||||
STUB_ERR 8
|
||||
STUB_NOERR 9
|
||||
STUB_ERR 10
|
||||
STUB_ERR 11
|
||||
STUB_ERR 12
|
||||
STUB_ERR 13
|
||||
STUB_ERR 14
|
||||
STUB_NOERR 15
|
||||
STUB_NOERR 16
|
||||
STUB_ERR 17
|
||||
STUB_NOERR 18
|
||||
STUB_NOERR 19
|
||||
STUB_NOERR 20
|
||||
STUB_ERR 21
|
||||
STUB_NOERR 22
|
||||
STUB_NOERR 23
|
||||
STUB_NOERR 24
|
||||
STUB_NOERR 25
|
||||
STUB_NOERR 26
|
||||
STUB_NOERR 27
|
||||
STUB_NOERR 28
|
||||
STUB_NOERR 29
|
||||
STUB_NOERR 30
|
||||
STUB_NOERR 31
|
||||
|
||||
# Device-interrupt vectors (timer, spurious, room for more). None push an error
|
||||
# code, so they all use the dummy-zero form.
|
||||
STUB_NOERR 32
|
||||
STUB_NOERR 33
|
||||
STUB_NOERR 34
|
||||
STUB_NOERR 35
|
||||
STUB_NOERR 36
|
||||
STUB_NOERR 37
|
||||
STUB_NOERR 38
|
||||
STUB_NOERR 39
|
||||
STUB_NOERR 40
|
||||
STUB_NOERR 41
|
||||
STUB_NOERR 42
|
||||
STUB_NOERR 43
|
||||
STUB_NOERR 44
|
||||
STUB_NOERR 45
|
||||
STUB_NOERR 46
|
||||
STUB_NOERR 47
|
||||
|
||||
.extern interruptDispatch
|
||||
|
||||
# Shared tail. Register push order here defines the CpuState field order.
|
||||
isr_common:
|
||||
push %rax
|
||||
push %rbx
|
||||
push %rcx
|
||||
push %rdx
|
||||
push %rsi
|
||||
push %rdi
|
||||
push %rbp
|
||||
push %r8
|
||||
push %r9
|
||||
push %r10
|
||||
push %r11
|
||||
push %r12
|
||||
push %r13
|
||||
push %r14
|
||||
push %r15
|
||||
mov %rsp, %rdi # first argument: pointer to the trap frame
|
||||
call interruptDispatch
|
||||
pop %r15
|
||||
pop %r14
|
||||
pop %r13
|
||||
pop %r12
|
||||
pop %r11
|
||||
pop %r10
|
||||
pop %r9
|
||||
pop %r8
|
||||
pop %rbp
|
||||
pop %rdi
|
||||
pop %rsi
|
||||
pop %rdx
|
||||
pop %rcx
|
||||
pop %rbx
|
||||
pop %rax
|
||||
add $16, %rsp # drop the vector and error code
|
||||
iretq
|
||||
@@ -0,0 +1,47 @@
|
||||
/* Kernel link layout.
|
||||
*
|
||||
* The kernel is linked at a fixed low physical address (set by `image_base` in
|
||||
* build.zig). UEFI runs with memory identity-mapped, so the bootloader can load
|
||||
* each PT_LOAD segment to the physical address matching its virtual address and
|
||||
* jump straight to _start — no page tables to build yet. (Moving to a
|
||||
* higher-half virtual base is a later step, once the bootloader sets up paging.)
|
||||
*/
|
||||
|
||||
ENTRY(_start)
|
||||
|
||||
/* One loadable segment per permission set, so the loader can map .text as R+X,
|
||||
* .rodata as R, and .data/.bss as R+W. FLAGS bits: 1=X, 2=W, 4=R. */
|
||||
PHDRS {
|
||||
text PT_LOAD FLAGS(5); /* R + X */
|
||||
rodata PT_LOAD FLAGS(4); /* R */
|
||||
data PT_LOAD FLAGS(6); /* R + W */
|
||||
}
|
||||
|
||||
SECTIONS {
|
||||
.text ALIGN(4K) : {
|
||||
*(.text .text.*)
|
||||
} :text
|
||||
|
||||
.rodata ALIGN(4K) : {
|
||||
*(.rodata .rodata.*)
|
||||
} :rodata
|
||||
|
||||
.data ALIGN(4K) : {
|
||||
*(.data .data.*)
|
||||
} :data
|
||||
|
||||
/* .bss occupies memory but not file space. The loader zeroes it via the
|
||||
* gap between each PT_LOAD segment's file size and memory size, so no
|
||||
* boundary symbols are needed here. (Zig's self-hosted linker also does not
|
||||
* yet honour linker-script symbol assignments.) */
|
||||
.bss ALIGN(4K) : {
|
||||
*(.bss .bss.*)
|
||||
*(COMMON)
|
||||
} :data
|
||||
|
||||
/DISCARD/ : {
|
||||
*(.comment)
|
||||
*(.note .note.*)
|
||||
*(.eh_frame .eh_frame_hdr)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,153 @@
|
||||
//! The kernel's page tables and virtual memory manager.
|
||||
//!
|
||||
//! Builds our own 4-level page tables and switches CR3 onto them, replacing the
|
||||
//! firmware's. Unlike the earlier bootstrap this maps with real permissions:
|
||||
//! RAM is identity-mapped read-write + no-execute, the kernel's own segments get
|
||||
//! their ELF permissions (code R+X, rodata R, data R+W+NX), and page 0 is left
|
||||
//! unmapped as a null guard. It also exposes map/unmap for on-demand mapping,
|
||||
//! which the kernel heap will build on.
|
||||
//!
|
||||
//! Everything is 4 KiB pages — precise and simple; the extra table memory is
|
||||
//! negligible against available RAM.
|
||||
|
||||
const danos = @import("danos");
|
||||
const io = @import("io.zig");
|
||||
|
||||
const page_size = danos.page_size;
|
||||
|
||||
// Page-table entry bits.
|
||||
const present: u64 = 1 << 0;
|
||||
const writable: u64 = 1 << 1;
|
||||
const no_execute: u64 = 1 << 63;
|
||||
const addr_mask: u64 = 0x000F_FFFF_FFFF_F000;
|
||||
|
||||
// ELF segment flags (p_flags).
|
||||
const pf_x: u32 = 1;
|
||||
const pf_w: u32 = 2;
|
||||
|
||||
// State kept after init so map()/unmap() can serve later callers (e.g. the heap).
|
||||
var kernel_pml4: u64 = 0;
|
||||
var alloc_frame: *const fn () ?u64 = undefined;
|
||||
|
||||
fn tableAt(phys: u64) *[512]u64 {
|
||||
return @ptrFromInt(phys);
|
||||
}
|
||||
|
||||
fn allocTable() u64 {
|
||||
const frame = alloc_frame() orelse @panic("paging: out of memory building page tables");
|
||||
@memset(tableAt(frame)[0..], 0);
|
||||
return frame;
|
||||
}
|
||||
|
||||
/// Return the table an entry points at, creating it if empty. Intermediate
|
||||
/// entries are writable and executable so the leaf's bits govern (a page is
|
||||
/// writable only if every level is; non-executable if any level is).
|
||||
fn descend(entry: *u64) u64 {
|
||||
if (entry.* & present != 0) return entry.* & addr_mask;
|
||||
const frame = allocTable();
|
||||
entry.* = frame | present | writable;
|
||||
return frame;
|
||||
}
|
||||
|
||||
/// Map one 4 KiB page `virt` -> `phys` with `flags` (present is added).
|
||||
fn mapPage(pml4: u64, virt: u64, phys: u64, flags: u64) void {
|
||||
const pml4e = &tableAt(pml4)[(virt >> 39) & 0x1FF];
|
||||
const pdpt = descend(pml4e);
|
||||
const pdpte = &tableAt(pdpt)[(virt >> 30) & 0x1FF];
|
||||
const pd = descend(pdpte);
|
||||
const pde = &tableAt(pd)[(virt >> 21) & 0x1FF];
|
||||
const pt = descend(pde);
|
||||
tableAt(pt)[(virt >> 12) & 0x1FF] = (phys & addr_mask) | flags | present;
|
||||
}
|
||||
|
||||
/// Identity-map [base, base+len) with `flags`, rounded out to whole pages.
|
||||
fn mapRangeIdentity(pml4: u64, base: u64, len: u64, flags: u64) void {
|
||||
var addr = base & ~@as(u64, page_size - 1);
|
||||
const end = base + len;
|
||||
while (addr < end) : (addr += page_size) {
|
||||
if (addr == 0) continue; // leave page 0 unmapped: the null guard
|
||||
mapPage(pml4, addr, addr, flags);
|
||||
}
|
||||
}
|
||||
|
||||
fn regions(mm: danos.MemoryMap) []const danos.MemoryRegion {
|
||||
return @as([*]const danos.MemoryRegion, @ptrFromInt(mm.regions))[0..mm.len];
|
||||
}
|
||||
|
||||
/// Enable the NX bit in the page-table format (EFER.NXE). Must happen before we
|
||||
/// load a CR3 whose entries set the NX bit, or those bits are reserved and fault.
|
||||
fn enableNx() void {
|
||||
const efer_msr = 0xC0000080;
|
||||
io.wrmsr(efer_msr, io.rdmsr(efer_msr) | (1 << 11));
|
||||
}
|
||||
|
||||
/// Build the address space and switch onto it.
|
||||
pub fn init(allocFrame: *const fn () ?u64, boot_info: *const danos.BootInfo) void {
|
||||
alloc_frame = allocFrame;
|
||||
enableNx();
|
||||
const pml4 = allocTable();
|
||||
|
||||
// 1. All RAM identity-mapped RW + NX. Non-RAM (MMIO) is skipped and stays
|
||||
// unmapped unless mapped explicitly below.
|
||||
for (regions(boot_info.memory_map)) |r| {
|
||||
if (r.kind == .mmio) continue;
|
||||
mapRangeIdentity(pml4, r.base, r.pages * page_size, present | writable | no_execute);
|
||||
}
|
||||
|
||||
// 2. The framebuffer and the Local APIC (device memory we need), RW + NX.
|
||||
const fb = boot_info.framebuffer;
|
||||
mapRangeIdentity(pml4, fb.base, @as(u64, fb.height) * fb.pitch, present | writable | no_execute);
|
||||
mapPage(pml4, 0xFEE00000, 0xFEE00000, present | writable | no_execute);
|
||||
|
||||
// 3. Overlay the kernel's own segments with their real ELF permissions,
|
||||
// replacing the blanket RW+NX from step 1: code becomes R+X, rodata R,
|
||||
// data R+W+NX. This is the W^X guarantee.
|
||||
for (boot_info.kernel_segments[0..boot_info.kernel_segment_count]) |seg| {
|
||||
var flags: u64 = present;
|
||||
if (seg.flags & pf_w != 0) flags |= writable;
|
||||
if (seg.flags & pf_x == 0) flags |= no_execute;
|
||||
var addr = seg.virt;
|
||||
const end = seg.virt + seg.pages * page_size;
|
||||
while (addr < end) : (addr += page_size) mapPage(pml4, addr, addr, flags);
|
||||
}
|
||||
|
||||
kernel_pml4 = pml4;
|
||||
asm volatile ("mov %[pml4], %%cr3"
|
||||
:
|
||||
: [pml4] "r" (pml4),
|
||||
: .{ .memory = true }
|
||||
);
|
||||
}
|
||||
|
||||
/// Map a page into the kernel address space on demand (for the heap, etc.).
|
||||
/// `writable_page` controls W; pages are always mapped non-executable.
|
||||
pub fn map(virt: u64, phys: u64, writable_page: bool) void {
|
||||
var flags: u64 = present | no_execute;
|
||||
if (writable_page) flags |= writable;
|
||||
mapPage(kernel_pml4, virt, phys, flags);
|
||||
invalidate(virt);
|
||||
}
|
||||
|
||||
/// Remove a mapping and flush it from the TLB.
|
||||
pub fn unmap(virt: u64) void {
|
||||
const pml4e = tableAt(kernel_pml4)[(virt >> 39) & 0x1FF];
|
||||
if (pml4e & present == 0) return;
|
||||
const pdpte = tableAt(pml4e & addr_mask)[(virt >> 30) & 0x1FF];
|
||||
if (pdpte & present == 0) return;
|
||||
const pde = tableAt(pdpte & addr_mask)[(virt >> 21) & 0x1FF];
|
||||
if (pde & present == 0) return;
|
||||
tableAt(pde & addr_mask)[(virt >> 12) & 0x1FF] = 0;
|
||||
invalidate(virt);
|
||||
}
|
||||
|
||||
fn invalidate(virt: u64) void {
|
||||
// invlpg needs its operand via a register-indirect memory reference that Zig
|
||||
// inline asm won't form directly, so stage the address in a register first.
|
||||
asm volatile (
|
||||
\\mov %[v], %%rax
|
||||
\\invlpg (%%rax)
|
||||
:
|
||||
: [v] "r" (virt),
|
||||
: .{ .rax = true, .memory = true }
|
||||
);
|
||||
}
|
||||
@@ -0,0 +1,46 @@
|
||||
//! COM1 serial port (16550 UART) — the kernel's machine-readable output channel.
|
||||
//! Unlike the framebuffer console, serial text can be captured to a file by QEMU
|
||||
//! (`-serial file:...`), which is what the test harness asserts on. Each
|
||||
//! architecture has its own UART; this is the x86 one, driven by port I/O.
|
||||
|
||||
const port = 0x3F8; // COM1 base
|
||||
|
||||
fn outb(p: u16, value: u8) void {
|
||||
asm volatile ("outb %[value], %[p]"
|
||||
:
|
||||
: [value] "{al}" (value),
|
||||
[p] "{dx}" (p),
|
||||
);
|
||||
}
|
||||
|
||||
fn inb(p: u16) u8 {
|
||||
return asm volatile ("inb %[p], %[value]"
|
||||
: [value] "={al}" (-> u8),
|
||||
: [p] "{dx}" (p),
|
||||
);
|
||||
}
|
||||
|
||||
/// Configure the UART: 38400 baud, 8N1, FIFO on. Safe to call before anything
|
||||
/// else; it has no dependencies.
|
||||
pub fn init() void {
|
||||
outb(port + 1, 0x00); // disable interrupts
|
||||
outb(port + 3, 0x80); // enable DLAB (set baud divisor)
|
||||
outb(port + 0, 0x03); // divisor low: 38400 baud
|
||||
outb(port + 1, 0x00); // divisor high
|
||||
outb(port + 3, 0x03); // 8 bits, no parity, one stop bit; DLAB off
|
||||
outb(port + 2, 0xC7); // enable + clear FIFO, 14-byte threshold
|
||||
outb(port + 4, 0x0B); // RTS/DSR set
|
||||
}
|
||||
|
||||
fn writeByte(c: u8) void {
|
||||
while (inb(port + 5) & 0x20 == 0) {} // wait until the transmit holding register is empty
|
||||
outb(port, c);
|
||||
}
|
||||
|
||||
/// Write bytes, translating LF to CRLF so terminals and logs line up.
|
||||
pub fn write(bytes: []const u8) void {
|
||||
for (bytes) |c| {
|
||||
if (c == '\n') writeByte('\r');
|
||||
writeByte(c);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,48 @@
|
||||
//! Task State Segment and its interrupt stack. In long mode the TSS's main job
|
||||
//! is the Interrupt Stack Table: an IDT gate can name an IST entry, and the CPU
|
||||
//! switches to that stack when the exception fires — no matter how broken the
|
||||
//! interrupted stack was. We use IST1 for the double-fault handler, so a fault
|
||||
//! that happens *because* the current stack is unusable still lands on solid
|
||||
//! ground instead of triple-faulting.
|
||||
|
||||
const gdt = @import("gdt.zig");
|
||||
|
||||
/// x86_64 TSS. `packed` because several 64-bit fields sit at 4-byte-unaligned
|
||||
/// offsets (rsp0 at byte 4), which a normal struct would pad away.
|
||||
const Tss = packed struct {
|
||||
reserved0: u32 = 0,
|
||||
rsp0: u64 = 0,
|
||||
rsp1: u64 = 0,
|
||||
rsp2: u64 = 0,
|
||||
reserved1: u64 = 0,
|
||||
ist1: u64 = 0,
|
||||
ist2: u64 = 0,
|
||||
ist3: u64 = 0,
|
||||
ist4: u64 = 0,
|
||||
ist5: u64 = 0,
|
||||
ist6: u64 = 0,
|
||||
ist7: u64 = 0,
|
||||
reserved2: u64 = 0,
|
||||
reserved3: u16 = 0,
|
||||
iomap_base: u16 = 0,
|
||||
};
|
||||
|
||||
/// The IST slot (1-based, as the IDT gate encodes it) used for critical faults.
|
||||
pub const double_fault_ist = 1;
|
||||
|
||||
var tss: Tss align(16) = .{};
|
||||
|
||||
/// Dedicated stack for IST1. Static so it needs no allocator and is always valid.
|
||||
var ist1_stack: [16 * 1024]u8 align(16) = undefined;
|
||||
|
||||
/// Loads the task register with the TSS selector. Defined in isr.s.
|
||||
extern fn load_tr(selector: u16) callconv(.c) void;
|
||||
|
||||
/// Point IST1 at its stack, publish the TSS through the GDT, and load it into the
|
||||
/// task register. Requires the GDT to already be loaded (gdt.init first).
|
||||
pub fn init() void {
|
||||
tss.ist1 = @intFromPtr(&ist1_stack) + ist1_stack.len; // stacks grow down
|
||||
tss.iomap_base = @sizeOf(Tss); // == limit: no I/O permission bitmap
|
||||
gdt.setTss(@intFromPtr(&tss), @sizeOf(Tss) - 1);
|
||||
load_tr(gdt.tss_selector);
|
||||
}
|
||||
@@ -0,0 +1,141 @@
|
||||
//! A framebuffer text console: draws glyphs from an embedded PSF2 font directly
|
||||
//! into the linear framebuffer the bootloader handed us. No firmware, no driver
|
||||
//! — just pixels. This is the kernel's first output device.
|
||||
|
||||
const std = @import("std");
|
||||
const danos = @import("danos");
|
||||
const arch = @import("arch");
|
||||
|
||||
/// The console font, embedded at compile time. cp850-8x16, PSF2 format:
|
||||
/// a 32-byte header, then 256 glyphs of 16 bytes each (one byte per 8-pixel
|
||||
/// row). We index glyphs straight by byte value, so ASCII maps 1:1.
|
||||
const font = @embedFile("font.psf");
|
||||
const glyph_w = 8;
|
||||
const glyph_h = 16;
|
||||
const glyph_bytes = glyph_h; // 8 pixels wide => 1 byte per row
|
||||
const glyph_data = 32; // PSF2 header size
|
||||
|
||||
pub const Console = struct {
|
||||
fb: danos.Framebuffer,
|
||||
cols: u32,
|
||||
rows: u32,
|
||||
col: u32 = 0,
|
||||
row: u32 = 0,
|
||||
fg: u32 = 0x00c8_c8c8, // light grey
|
||||
bg: u32 = 0x0000_0000, // black
|
||||
|
||||
pub fn init(fb: danos.Framebuffer) Console {
|
||||
return .{
|
||||
.fb = fb,
|
||||
.cols = fb.width / glyph_w,
|
||||
.rows = fb.height / glyph_h,
|
||||
};
|
||||
}
|
||||
|
||||
/// Fill the whole screen with the background colour and home the cursor.
|
||||
pub fn clear(self: *Console) void {
|
||||
var y: u32 = 0;
|
||||
while (y < self.fb.height) : (y += 1) self.fillRow(y, self.bg);
|
||||
self.col = 0;
|
||||
self.row = 0;
|
||||
}
|
||||
|
||||
pub fn write(self: *Console, bytes: []const u8) void {
|
||||
// Mirror everything to the serial port so it's captured in logs / tests.
|
||||
arch.serialWrite(bytes);
|
||||
for (bytes) |c| self.putChar(c);
|
||||
}
|
||||
|
||||
/// Formatted output, e.g. `con.print("x={d}\n", .{x})`. Silently truncates
|
||||
/// past 256 bytes — this is a debug console, not a general writer.
|
||||
pub fn print(self: *Console, comptime fmt: []const u8, args: anytype) void {
|
||||
var buf: [256]u8 = undefined;
|
||||
self.write(std.fmt.bufPrint(&buf, fmt, args) catch return);
|
||||
}
|
||||
|
||||
pub fn putChar(self: *Console, ch: u8) void {
|
||||
switch (ch) {
|
||||
'\n' => self.newline(),
|
||||
'\r' => self.col = 0,
|
||||
else => {
|
||||
if (self.col >= self.cols) self.newline();
|
||||
self.drawGlyph(ch, self.col * glyph_w, self.row * glyph_h);
|
||||
self.col += 1;
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
fn newline(self: *Console) void {
|
||||
self.col = 0;
|
||||
if (self.row + 1 >= self.rows) {
|
||||
self.scroll();
|
||||
} else {
|
||||
self.row += 1;
|
||||
}
|
||||
}
|
||||
|
||||
fn drawGlyph(self: *Console, ch: u8, px: u32, py: u32) void {
|
||||
const rows = font[glyph_data + @as(usize, ch) * glyph_bytes ..][0..glyph_bytes];
|
||||
var gy: u32 = 0;
|
||||
while (gy < glyph_h) : (gy += 1) {
|
||||
const bits = rows[gy];
|
||||
var gx: u32 = 0;
|
||||
while (gx < glyph_w) : (gx += 1) {
|
||||
// Leftmost pixel is the high bit.
|
||||
const on = (bits >> @as(u3, @intCast(7 - gx))) & 1 != 0;
|
||||
self.pixel(px + gx, py + gy, if (on) self.fg else self.bg);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Shift the visible text up one glyph row and clear the freed bottom row,
|
||||
/// leaving the cursor on that now-blank last line.
|
||||
fn scroll(self: *Console) void {
|
||||
const visible = self.rows * glyph_h;
|
||||
var y: u32 = 0;
|
||||
while (y + glyph_h < visible) : (y += 1) self.copyRow(y, y + glyph_h);
|
||||
while (y < visible) : (y += 1) self.fillRow(y, self.bg);
|
||||
self.row = self.rows - 1;
|
||||
}
|
||||
|
||||
inline fn rowPtr(self: *Console, y: u32) [*]volatile u32 {
|
||||
const base: [*]volatile u8 = @ptrFromInt(self.fb.base);
|
||||
return @ptrCast(@alignCast(base + y * self.fb.pitch));
|
||||
}
|
||||
|
||||
inline fn pixel(self: *Console, x: u32, y: u32, color: u32) void {
|
||||
self.rowPtr(y)[x] = color;
|
||||
}
|
||||
|
||||
fn fillRow(self: *Console, y: u32, color: u32) void {
|
||||
const row = self.rowPtr(y);
|
||||
var x: u32 = 0;
|
||||
while (x < self.fb.width) : (x += 1) row[x] = color;
|
||||
}
|
||||
|
||||
fn copyRow(self: *Console, dst_y: u32, src_y: u32) void {
|
||||
const dst = self.rowPtr(dst_y);
|
||||
const src = self.rowPtr(src_y);
|
||||
var x: u32 = 0;
|
||||
while (x < self.fb.width) : (x += 1) dst[x] = src[x];
|
||||
}
|
||||
};
|
||||
|
||||
pub const SerialConsole = struct {
|
||||
/// Serial-only output: goes to the machine-readable log but *not* the framebuffer,
|
||||
/// so debug and test detail stays out of the on-screen console. These are free
|
||||
/// functions, not `Console` methods, because serial has no dependency on the
|
||||
/// framebuffer — they work even before `con` is initialised.
|
||||
pub fn debugWrite(bytes: []const u8) void {
|
||||
arch.serialWrite(bytes);
|
||||
}
|
||||
|
||||
/// Formatted serial-only output, e.g. `debugPrint("x={d}\n", .{x})`. Truncates
|
||||
/// past 256 bytes, like `Console.print`.
|
||||
pub fn debugPrint(comptime fmt: []const u8, args: anytype) void {
|
||||
var buf: [256]u8 = undefined;
|
||||
debugWrite(std.fmt.bufPrint(&buf, fmt, args) catch return);
|
||||
}
|
||||
|
||||
};
|
||||
|
||||
Binary file not shown.
@@ -0,0 +1,174 @@
|
||||
//! The kernel heap: dynamic allocation for the kernel.
|
||||
//!
|
||||
//! Where the frame allocator ([pmm]) hands out fixed 4 KiB physical frames, the
|
||||
//! heap hands out arbitrary byte-sized blocks from a virtual region, growing on
|
||||
//! demand by mapping fresh frames into it (arch.mapPage) — the first real user of
|
||||
//! the VMM (see docs/paging.md).
|
||||
//!
|
||||
//! The algorithm is a first-fit free list: an address-ordered singly linked list
|
||||
//! of free blocks, split on allocation and coalesced with neighbours on free. It
|
||||
//! is exposed as a std.mem.Allocator, so the kernel can use std containers.
|
||||
//!
|
||||
//! Not yet concurrency-safe: it assumes a single caller and no allocation from
|
||||
//! interrupt handlers (ours don't). A lock comes with threads/SMP.
|
||||
|
||||
const std = @import("std");
|
||||
const danos = @import("danos");
|
||||
const arch = @import("arch");
|
||||
const pmm = @import("pmm.zig");
|
||||
|
||||
const page_size = danos.page_size;
|
||||
|
||||
/// Virtual base of the heap: the start of the higher half, which is unmapped and
|
||||
/// well clear of the identity-mapped low half. (Canonical on x86_64; an arch that
|
||||
/// splits the address space differently would choose its own.)
|
||||
const heap_base: usize = 0xFFFF_8000_0000_0000;
|
||||
/// Cap on heap growth for now.
|
||||
const heap_max: usize = 64 * 1024 * 1024;
|
||||
|
||||
/// A block header, placed at the start of every block. While the block is free it
|
||||
/// also links into the free list via `next`.
|
||||
const Block = extern struct {
|
||||
size: usize, // total block size in bytes, including this header; a multiple of 16
|
||||
next: ?*Block, // free-list link (only meaningful while free)
|
||||
};
|
||||
|
||||
const header_size = @sizeOf(Block); // 16
|
||||
const min_block = header_size + 16; // smallest block worth splitting off
|
||||
|
||||
var free_list: ?*Block = null;
|
||||
var heap_end: usize = heap_base; // [heap_base, heap_end) is currently mapped
|
||||
|
||||
fn alignUp(value: usize, alignment: usize) usize {
|
||||
return (value + alignment - 1) & ~(alignment - 1);
|
||||
}
|
||||
|
||||
fn payloadOf(block: *Block) [*]u8 {
|
||||
return @ptrFromInt(@intFromPtr(block) + header_size);
|
||||
}
|
||||
|
||||
/// Bring the heap up with an initial mapped region.
|
||||
pub fn init() void {
|
||||
free_list = null;
|
||||
heap_end = heap_base;
|
||||
_ = grow(page_size);
|
||||
}
|
||||
|
||||
/// Map more pages onto the end of the heap and add them as a free block. Returns
|
||||
/// false if out of heap virtual space or out of physical frames.
|
||||
fn grow(min_bytes: usize) bool {
|
||||
const start = heap_end;
|
||||
const bytes = alignUp(min_bytes, page_size);
|
||||
if (start + bytes > heap_base + heap_max) return false;
|
||||
|
||||
var virt = start;
|
||||
while (virt < start + bytes) : (virt += page_size) {
|
||||
const frame = pmm.alloc() orelse return false;
|
||||
arch.mapPage(virt, frame, true);
|
||||
}
|
||||
heap_end = start + bytes;
|
||||
|
||||
const block: *Block = @ptrFromInt(start);
|
||||
block.size = bytes;
|
||||
insertFree(block); // coalesces with the previous tail block if adjacent
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Insert a block into the address-ordered free list, coalescing with the
|
||||
/// physically adjacent free blocks on either side.
|
||||
fn insertFree(block: *Block) void {
|
||||
var prev: ?*Block = null;
|
||||
var cur = free_list;
|
||||
while (cur) |c| : (cur = c.next) {
|
||||
if (@intFromPtr(c) > @intFromPtr(block)) break;
|
||||
prev = c;
|
||||
}
|
||||
|
||||
block.next = cur;
|
||||
if (prev) |p| p.next = block else free_list = block;
|
||||
|
||||
// Merge forward into `cur` if they're contiguous.
|
||||
if (cur) |c| {
|
||||
if (@intFromPtr(block) + block.size == @intFromPtr(c)) {
|
||||
block.size += c.size;
|
||||
block.next = c.next;
|
||||
}
|
||||
}
|
||||
// Merge `prev` forward into `block` if they're contiguous.
|
||||
if (prev) |p| {
|
||||
if (@intFromPtr(p) + p.size == @intFromPtr(block)) {
|
||||
p.size += block.size;
|
||||
p.next = block.next;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Allocate `len` bytes (16-byte aligned), or null if out of memory.
|
||||
fn rawAlloc(len: usize) ?[*]u8 {
|
||||
const need = alignUp(header_size + len, 16);
|
||||
|
||||
var attempts: u32 = 0;
|
||||
while (attempts < 2) : (attempts += 1) {
|
||||
var prev: ?*Block = null;
|
||||
var cur = free_list;
|
||||
while (cur) |block| : ({
|
||||
prev = block;
|
||||
cur = block.next;
|
||||
}) {
|
||||
if (block.size < need) continue;
|
||||
|
||||
if (block.size >= need + min_block) {
|
||||
// Split: carve `need` off the front, leave the rest free.
|
||||
const rest: *Block = @ptrFromInt(@intFromPtr(block) + need);
|
||||
rest.size = block.size - need;
|
||||
rest.next = block.next;
|
||||
if (prev) |p| p.next = rest else free_list = rest;
|
||||
block.size = need;
|
||||
} else {
|
||||
// Take the whole block.
|
||||
if (prev) |p| p.next = block.next else free_list = block.next;
|
||||
}
|
||||
return payloadOf(block);
|
||||
}
|
||||
|
||||
// Nothing fit: grow and try once more.
|
||||
if (!grow(need)) return null;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
fn rawFree(ptr: [*]u8) void {
|
||||
const block: *Block = @ptrFromInt(@intFromPtr(ptr) - header_size);
|
||||
insertFree(block);
|
||||
}
|
||||
|
||||
// --- std.mem.Allocator interface -----------------------------------------
|
||||
|
||||
pub fn allocator() std.mem.Allocator {
|
||||
return .{ .ptr = undefined, .vtable = &vtable };
|
||||
}
|
||||
|
||||
const vtable = std.mem.Allocator.VTable{
|
||||
.alloc = allocImpl,
|
||||
.resize = resizeImpl,
|
||||
.remap = remapImpl,
|
||||
.free = freeImpl,
|
||||
};
|
||||
|
||||
fn allocImpl(_: *anyopaque, len: usize, alignment: std.mem.Alignment, _: usize) ?[*]u8 {
|
||||
// Blocks are 16-byte aligned; larger alignments aren't supported yet.
|
||||
if (alignment.toByteUnits() > 16) return null;
|
||||
return rawAlloc(len);
|
||||
}
|
||||
|
||||
fn resizeImpl(_: *anyopaque, _: []u8, _: std.mem.Alignment, _: usize, _: usize) bool {
|
||||
return false; // no in-place resize; the caller reallocates
|
||||
}
|
||||
|
||||
fn remapImpl(_: *anyopaque, _: []u8, _: std.mem.Alignment, _: usize, _: usize) ?[*]u8 {
|
||||
return null;
|
||||
}
|
||||
|
||||
fn freeImpl(_: *anyopaque, memory: []u8, _: std.mem.Alignment, _: usize) void {
|
||||
rawFree(memory.ptr);
|
||||
}
|
||||
@@ -0,0 +1,54 @@
|
||||
//! Inter-process communication: message-passing channels.
|
||||
//!
|
||||
//! IPC is the backbone of a microkernel ([vision](../docs/vision.md)): once
|
||||
//! drivers and services live in separate address spaces, a message is how they
|
||||
//! talk. This first form is a **bounded blocking channel** — a ring buffer of
|
||||
//! messages with a producer/consumer rendezvous, built on the scheduler's
|
||||
//! [wait queues](scheduling.md). `send` blocks when the channel is full, `recv`
|
||||
//! blocks when it's empty; neither busy-waits.
|
||||
//!
|
||||
//! For now both endpoints are kernel threads sharing the kernel address space.
|
||||
//! When user mode arrives, the same primitive carries messages across the
|
||||
//! isolation boundary (with the payload copied between address spaces).
|
||||
|
||||
const arch = @import("arch");
|
||||
const sched = @import("sched.zig");
|
||||
|
||||
/// A bounded blocking channel of `capacity` messages of type `T`.
|
||||
pub fn Channel(comptime T: type, comptime capacity: usize) type {
|
||||
return struct {
|
||||
const Self = @This();
|
||||
|
||||
buffer: [capacity]T = undefined,
|
||||
head: usize = 0, // next slot to read
|
||||
tail: usize = 0, // next slot to write
|
||||
count: usize = 0,
|
||||
not_full: sched.WaitQueue = .{}, // senders wait here
|
||||
not_empty: sched.WaitQueue = .{}, // receivers wait here
|
||||
|
||||
/// Send a message, blocking while the channel is full.
|
||||
pub fn send(self: *Self, msg: T) void {
|
||||
const flags = arch.saveInterrupts();
|
||||
// Recheck the condition in a loop: a wakeup only means "try again"
|
||||
// (another waiter may have taken the slot first).
|
||||
while (self.count == capacity) sched.waitLocked(&self.not_full);
|
||||
self.buffer[self.tail] = msg;
|
||||
self.tail = (self.tail + 1) % capacity;
|
||||
self.count += 1;
|
||||
sched.wakeLocked(&self.not_empty); // a receiver can now proceed
|
||||
arch.restoreInterrupts(flags);
|
||||
}
|
||||
|
||||
/// Receive a message, blocking while the channel is empty.
|
||||
pub fn recv(self: *Self) T {
|
||||
const flags = arch.saveInterrupts();
|
||||
while (self.count == 0) sched.waitLocked(&self.not_empty);
|
||||
const msg = self.buffer[self.head];
|
||||
self.head = (self.head + 1) % capacity;
|
||||
self.count -= 1;
|
||||
sched.wakeLocked(&self.not_full); // a sender can now proceed
|
||||
arch.restoreInterrupts(flags);
|
||||
return msg;
|
||||
}
|
||||
};
|
||||
}
|
||||
@@ -0,0 +1,168 @@
|
||||
const std = @import("std");
|
||||
const danos = @import("danos");
|
||||
const arch = @import("arch");
|
||||
const console = @import("console.zig");
|
||||
const pmm = @import("pmm.zig");
|
||||
const heap = @import("heap.zig");
|
||||
const sched = @import("sched.zig");
|
||||
const tests = @import("tests.zig");
|
||||
const build_options = @import("build_options");
|
||||
const BootInfo = danos.BootInfo;
|
||||
|
||||
/// The calling convention used to enter the kernel. Pinned to SysV explicitly:
|
||||
/// the bootloader is built for the UEFI target, whose C convention is Microsoft
|
||||
/// x64 (first argument in RCX), while the kernel is SysV (first argument in
|
||||
/// RDI). Both sides reference this so the `boot_info` pointer lands in the
|
||||
/// register the other expects. `danos.kernel_abi` re-exports it to the loader.
|
||||
pub const kernel_abi = danos.kernel_abi;
|
||||
|
||||
/// The system console, valid once `kmain` has initialised it. Global so the
|
||||
/// panic handler can reach it too.
|
||||
var con: console.Console = undefined;
|
||||
var con_ready = false;
|
||||
|
||||
/// Kernel entry point. The bootloader jumps here after `ExitBootServices` with a
|
||||
/// pointer to the handoff data. There is no runtime, no stack unwinding, and no
|
||||
/// caller to return to, so this never returns.
|
||||
export fn _start(boot_info: *const BootInfo) callconv(kernel_abi) noreturn {
|
||||
kmain(boot_info);
|
||||
}
|
||||
|
||||
fn kmain(boot_info: *const BootInfo) noreturn {
|
||||
arch.serialInit(); // machine-readable log; console mirrors to it
|
||||
|
||||
const fb = boot_info.framebuffer;
|
||||
const serial0 = console.SerialConsole;
|
||||
con = console.Console.init(fb);
|
||||
con.clear();
|
||||
con_ready = true;
|
||||
|
||||
// Catch CPU exceptions before doing anything that might fault: install our
|
||||
// reporter, then bring up the GDT + IDT.
|
||||
arch.setFaultHandler(onException);
|
||||
arch.init();
|
||||
|
||||
con.write("danos: initalizing kernel...");
|
||||
|
||||
serial0.debugWrite("danos: framebuffer console online\n");
|
||||
serial0.debugWrite("danos: cpu tables online (GDT, IDT, TSS)\n");
|
||||
serial0.debugPrint(" resolution : {d}x{d}\n", .{ fb.width, fb.height });
|
||||
serial0.debugPrint(" pitch : {d} bytes\n", .{fb.pitch});
|
||||
serial0.debugPrint(" format : {s}\n", .{@tagName(fb.format)});
|
||||
serial0.debugPrint(" framebuffer: 0x{x:0>16}\n", .{fb.base});
|
||||
serial0.debugPrint (" footdebugPrint : {d} MiB\n", .{(fb.pitch * fb.height) / (1024 * 1024)});
|
||||
|
||||
// Summarise the physical memory the loader handed us. The array is danos's
|
||||
// own MemoryRegion, so this is a plain slice — no firmware layout in sight.
|
||||
const regions = @as([*]const danos.MemoryRegion, @ptrFromInt(boot_info.memory_map.regions))[0..boot_info.memory_map.len];
|
||||
var usable_pages: u64 = 0;
|
||||
var reserved_pages: u64 = 0; // reserved RAM only — MMIO is device space, not RAM
|
||||
for (regions) |r| {
|
||||
switch (r.kind) {
|
||||
.usable => usable_pages += r.pages,
|
||||
.reserved, .acpi_tables, .acpi_nvs => reserved_pages += r.pages,
|
||||
.mmio => {},
|
||||
}
|
||||
}
|
||||
const total_pages = usable_pages + reserved_pages;
|
||||
const total_bytes = total_pages * danos.page_size;
|
||||
const gib = 1 << 30;
|
||||
|
||||
serial0.debugWrite("\ndanos: physical memory\n");
|
||||
serial0.debugPrint(" total RAM : {d}.{d:0>2} GiB ({d} MiB) - RAM the firmware reported\n", .{ total_bytes / gib, (total_bytes % gib) * 100 / gib, mib(total_pages) });
|
||||
serial0.debugPrint(" usable : {d} MiB - free RAM (incl. reclaimed boot-services memory)\n", .{mib(usable_pages)});
|
||||
serial0.debugPrint(" reserved : {d} MiB - kernel image, boot stack, ACPI, runtime services\n", .{mib(reserved_pages)});
|
||||
serial0.debugPrint(" regions : {d} - entries in the firmware memory map\n", .{regions.len});
|
||||
|
||||
// Bring up the physical frame allocator over that map, and prove it works:
|
||||
// allocate three frames, then hand them back.
|
||||
pmm.init(boot_info.memory_map);
|
||||
const s1 = pmm.stats();
|
||||
serial0.debugPrint("\ndanos: frame allocator online\n", .{});
|
||||
serial0.debugPrint(" free frames: {d} ({d} MiB)\n", .{ s1.free_frames, mib(s1.free_frames) });
|
||||
const f0 = pmm.alloc();
|
||||
const f1 = pmm.alloc();
|
||||
const f2 = pmm.alloc();
|
||||
serial0.debugPrint(" alloc x3 : 0x{x} 0x{x} 0x{x}\n", .{ f0 orelse 0, f1 orelse 0, f2 orelse 0 });
|
||||
if (f0) |p| pmm.free(p);
|
||||
if (f1) |p| pmm.free(p);
|
||||
if (f2) |p| pmm.free(p);
|
||||
serial0.debugPrint(" after free : {d} frames free\n", .{pmm.stats().free_frames});
|
||||
|
||||
// Switch off the firmware's page tables onto our own (with real permissions).
|
||||
arch.enablePaging(pmm.alloc, boot_info);
|
||||
serial0.debugPrint("\ndanos: paging enabled\n", .{});
|
||||
serial0.debugPrint(" page tables: CR3 = 0x{x:0>16}\n", .{arch.readCr3()});
|
||||
serial0.debugPrint(" kernel segs: {d} (mapped with W^X permissions)\n", .{boot_info.kernel_segment_count});
|
||||
|
||||
// Bring up the kernel heap (dynamic allocation), built on the VMM.
|
||||
heap.init();
|
||||
serial0.debugWrite("\ndanos: kernel heap online\n");
|
||||
// Measure the amount of resources the kernel is actually using
|
||||
const s2 = pmm.stats();
|
||||
serial0.debugPrint(" Kernel FootdebugPrint: {d} KiB\n", .{kib(s1.free_frames - s2.free_frames)});
|
||||
|
||||
// Register the current context as the first task before enabling preemption.
|
||||
sched.init(4);
|
||||
serial0.debugWrite("\ndanos: scheduler online\n");
|
||||
|
||||
// Start the timer and unmask interrupts — the kernel now has a heartbeat, and
|
||||
// the timer preempts among tasks.
|
||||
arch.startTimer();
|
||||
arch.enableInterrupts();
|
||||
serial0.debugPrint("danos: timer online ({d} Hz tick; LAPIC {d} MHz, TSC {d} MHz measured)\n", .{ arch.timer_hz, arch.lapicHz() / 1_000_000, arch.tscHz() / 1_000_000 });
|
||||
|
||||
// In a test build (`zig build -Dtest-case=<name>`), run that case and stop.
|
||||
// Normal builds fall through to the idle halt.
|
||||
if (build_options.test_case) |case| {
|
||||
tests.run(case, boot_info);
|
||||
arch.halt();
|
||||
}
|
||||
|
||||
con.write("kernel initialised.\n");
|
||||
|
||||
// TODO: init process
|
||||
|
||||
con.write("\nnothing left to do; halting CPU.\n");
|
||||
|
||||
arch.halt();
|
||||
}
|
||||
|
||||
/// Frames (4 KiB pages) to whole MiB.
|
||||
fn mib(pages: u64) u64 {
|
||||
return pages * danos.page_size / (1024 * 1024);
|
||||
}
|
||||
|
||||
fn kib(frames: u64) u64 {
|
||||
return frames * danos.page_size / (1024);
|
||||
}
|
||||
|
||||
/// Report a CPU exception in red and halt. There's no fault recovery yet, so any
|
||||
/// exception is terminal — but now it debugPrints what and where instead of silently
|
||||
/// resetting the machine.
|
||||
fn onException(state: *const arch.CpuState) noreturn {
|
||||
if (con_ready) {
|
||||
con.fg = 0x00ff_5555;
|
||||
con.print("\nCPU EXCEPTION: {s} (vector {d})\n", .{ arch.vectorName(state.vector), state.vector });
|
||||
con.print(" error code : 0x{x}\n", .{state.error_code});
|
||||
con.print(" RIP : 0x{x:0>16}\n", .{state.rip});
|
||||
con.print(" RSP : 0x{x:0>16}\n", .{state.rsp});
|
||||
if (state.vector == 14) con.print(" CR2 (addr) : 0x{x:0>16}\n", .{arch.readCr2()});
|
||||
}
|
||||
arch.halt();
|
||||
}
|
||||
|
||||
/// Freestanding has no OS to receive a panic. debugPrint it to the console (if it is
|
||||
/// up yet) in red, then halt.
|
||||
pub const panic = std.debug.FullPanic(struct {
|
||||
fn panic(msg: []const u8, first_trace_addr: ?usize) noreturn {
|
||||
_ = first_trace_addr;
|
||||
if (con_ready) {
|
||||
con.fg = 0x00ff_5555;
|
||||
con.write("\nKERNEL PANIC: ");
|
||||
con.write(msg);
|
||||
con.write("\n");
|
||||
}
|
||||
arch.halt();
|
||||
}
|
||||
}.panic);
|
||||
@@ -0,0 +1,152 @@
|
||||
//! Physical memory manager: a bitmap frame allocator. It hands out and reclaims
|
||||
//! 4 KiB physical frames — the primitive every later memory feature (page
|
||||
//! tables, the heap) is built on top of.
|
||||
//!
|
||||
//! This is generic kernel code: it works on the neutral `danos.MemoryRegion`
|
||||
//! array the loader hands over (see docs/memory-map.md), so it carries no UEFI
|
||||
//! and nothing architecture-specific beyond the 4 KiB page.
|
||||
|
||||
const std = @import("std");
|
||||
const danos = @import("danos");
|
||||
|
||||
const page_size = danos.page_size;
|
||||
|
||||
/// One bit per frame, covering physical RAM from 0 up to the highest usable
|
||||
/// address: 1 = used/unavailable, 0 = free. The bitmap itself lives in a frame
|
||||
/// we carve out of usable memory during init.
|
||||
var bitmap: []u8 = &.{};
|
||||
var total_frames: usize = 0;
|
||||
var used_frames: usize = 0;
|
||||
/// Where the next allocation scan begins, so we don't rescan from frame 0 every
|
||||
/// time. Pulled back on free() so reclaimed low frames get reused.
|
||||
var next_hint: usize = 0;
|
||||
|
||||
pub const Stats = struct {
|
||||
total_frames: usize,
|
||||
used_frames: usize,
|
||||
free_frames: usize,
|
||||
};
|
||||
|
||||
pub fn stats() Stats {
|
||||
return .{
|
||||
.total_frames = total_frames,
|
||||
.used_frames = used_frames,
|
||||
.free_frames = total_frames - used_frames,
|
||||
};
|
||||
}
|
||||
|
||||
inline fn bit(frame: usize) u3 {
|
||||
return @intCast(frame & 7);
|
||||
}
|
||||
inline fn isUsed(frame: usize) bool {
|
||||
return (bitmap[frame >> 3] >> bit(frame)) & 1 != 0;
|
||||
}
|
||||
inline fn setUsed(frame: usize) void {
|
||||
bitmap[frame >> 3] |= @as(u8, 1) << bit(frame);
|
||||
}
|
||||
inline fn setFree(frame: usize) void {
|
||||
bitmap[frame >> 3] &= ~(@as(u8, 1) << bit(frame));
|
||||
}
|
||||
|
||||
fn regions(map: danos.MemoryMap) []const danos.MemoryRegion {
|
||||
return @as([*]const danos.MemoryRegion, @ptrFromInt(map.regions))[0..map.len];
|
||||
}
|
||||
|
||||
/// Build the allocator from the loader's memory map. Relies on the firmware's
|
||||
/// identity mapping still being in effect (a physical address is usable directly
|
||||
/// as a pointer) — true until the kernel installs its own page tables.
|
||||
pub fn init(map: danos.MemoryMap) void {
|
||||
const regs = regions(map);
|
||||
|
||||
// 1. Size the bitmap to cover every frame up to the highest RAM address —
|
||||
// including reserved RAM, so those frames are trackable (e.g. to free the
|
||||
// boot buffers later). Only MMIO (device address space) is excluded.
|
||||
// Everything starts unallocatable; usable regions are freed below.
|
||||
var highest: u64 = 0;
|
||||
for (regs) |r| {
|
||||
if (r.kind == .mmio) continue;
|
||||
const end = r.base + r.pages * page_size;
|
||||
if (end > highest) highest = end;
|
||||
}
|
||||
total_frames = @intCast(highest / page_size);
|
||||
if (total_frames == 0) @panic("pmm: no usable memory");
|
||||
const bitmap_bytes = (total_frames + 7) / 8;
|
||||
const bitmap_pages = (bitmap_bytes + page_size - 1) / page_size;
|
||||
|
||||
// 2. Park the bitmap in the first usable region large enough to hold it.
|
||||
// Start at least one page in, so we never place it on frame 0 (which is
|
||||
// kept reserved as the "none" address, and is an awkward pointer besides).
|
||||
var storage: ?u64 = null;
|
||||
for (regs) |r| {
|
||||
if (r.kind != .usable) continue;
|
||||
const base = if (r.base == 0) page_size else r.base;
|
||||
const skipped = (base - r.base) / page_size;
|
||||
if (r.pages - skipped >= bitmap_pages) {
|
||||
storage = base;
|
||||
break;
|
||||
}
|
||||
}
|
||||
const bitmap_base = storage orelse @panic("pmm: no region large enough for the frame bitmap");
|
||||
bitmap = @as([*]u8, @ptrFromInt(bitmap_base))[0..bitmap_bytes];
|
||||
|
||||
// 3. Start with everything marked used, then free the usable regions. Doing
|
||||
// it this way means every gap, reserved span and MMIO hole is unallocatable
|
||||
// by default — we only ever hand back memory the firmware called usable.
|
||||
@memset(bitmap, 0xff);
|
||||
used_frames = total_frames;
|
||||
for (regs) |r| {
|
||||
if (r.kind != .usable) continue;
|
||||
var f: usize = @intCast(r.base / page_size);
|
||||
const end = f + @as(usize, @intCast(r.pages));
|
||||
while (f < end and f < total_frames) : (f += 1) {
|
||||
setFree(f);
|
||||
used_frames -= 1;
|
||||
}
|
||||
}
|
||||
|
||||
// 4. Take back the frames the bitmap occupies, and frame 0 — so a 0 result
|
||||
// stays reserved to mean "no frame".
|
||||
reserve(bitmap_base, bitmap_pages);
|
||||
reserve(0, 1);
|
||||
}
|
||||
|
||||
/// Mark `count` frames from physical `base` as used, counting only those that
|
||||
/// were actually free.
|
||||
fn reserve(base: u64, count: usize) void {
|
||||
var f: usize = @intCast(base / page_size);
|
||||
const end = f + count;
|
||||
while (f < end and f < total_frames) : (f += 1) {
|
||||
if (!isUsed(f)) {
|
||||
setUsed(f);
|
||||
used_frames += 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Allocate one physical frame, or null if none are free. The address is
|
||||
/// page-aligned; the frame's contents are undefined.
|
||||
pub fn alloc() ?u64 {
|
||||
var scanned: usize = 0;
|
||||
var f = next_hint;
|
||||
while (scanned < total_frames) : (scanned += 1) {
|
||||
if (f >= total_frames) f = 0;
|
||||
if (!isUsed(f)) {
|
||||
setUsed(f);
|
||||
used_frames += 1;
|
||||
next_hint = f + 1;
|
||||
return @as(u64, f) * page_size;
|
||||
}
|
||||
f += 1;
|
||||
}
|
||||
return null; // out of physical memory
|
||||
}
|
||||
|
||||
/// Return a frame obtained from alloc() to the pool. Bogus or double frees are
|
||||
/// ignored rather than corrupting the count.
|
||||
pub fn free(addr: u64) void {
|
||||
const f: usize = @intCast(addr / page_size);
|
||||
if (f >= total_frames or !isUsed(f)) return;
|
||||
setFree(f);
|
||||
used_frames -= 1;
|
||||
if (f < next_hint) next_hint = f;
|
||||
}
|
||||
@@ -0,0 +1,251 @@
|
||||
//! The scheduler: fixed-priority preemptive multitasking.
|
||||
//!
|
||||
//! Tasks are kernel threads (ring 0, each with its own stack). The **highest-
|
||||
//! priority ready task always runs**; within a priority level, tasks round-robin.
|
||||
//! Selection is O(1) — a bitmap of non-empty priority levels plus a FIFO queue per
|
||||
//! level — which keeps scheduling deterministic, as a real-time kernel needs (see
|
||||
//! docs/vision.md).
|
||||
//!
|
||||
//! Switching happens both cooperatively (`yield`) and preemptively (the timer
|
||||
//! calls `tick`). See docs/scheduling.md for the interrupt-flag discipline that
|
||||
//! makes those two paths coexist.
|
||||
|
||||
const std = @import("std");
|
||||
const arch = @import("arch");
|
||||
const heap = @import("heap.zig");
|
||||
|
||||
/// Priority level: 0 (lowest) .. 7 (highest). 8 levels total.
|
||||
pub const Priority = u3;
|
||||
const num_priorities = 8;
|
||||
|
||||
const stack_size = 16 * 1024; // each task's kernel stack is 16 KiB
|
||||
const max_tasks = 16; // the maximum number of tasks alive at once is 16 in a static sized pool
|
||||
|
||||
const State = enum { free, ready, running, blocked };
|
||||
|
||||
const Task = struct {
|
||||
id: u32 = 0,
|
||||
state: State = .free,
|
||||
priority: Priority = 0,
|
||||
rsp: usize = 0, // saved stack pointer, valid while not running
|
||||
stack: []u8 = &.{},
|
||||
wake_at: u64 = 0, // uptime (ms) to wake a sleeping task; 0 = not sleeping
|
||||
next: ?*Task = null, // ready-queue link
|
||||
};
|
||||
|
||||
var tasks = [_]Task{.{}} ** max_tasks;
|
||||
var current: *Task = undefined;
|
||||
var next_id: u32 = 1;
|
||||
|
||||
// Per-priority FIFO ready queues, and a bitmap of which levels are non-empty.
|
||||
var ready_head: [num_priorities]?*Task = .{null} ** num_priorities;
|
||||
var ready_tail: [num_priorities]?*Task = .{null} ** num_priorities;
|
||||
var ready_bitmap: u8 = 0;
|
||||
|
||||
var preemption_enabled = true;
|
||||
|
||||
/// Register the currently-running kernel context as the first task, spawn the
|
||||
/// idle task, and hook the timer for preemption.
|
||||
pub fn init(boot_priority: Priority) void {
|
||||
tasks[0] = .{ .id = 0, .state = .running, .priority = boot_priority };
|
||||
current = &tasks[0];
|
||||
spawn(idle, 0); // lowest priority, always runnable — runs when nothing else is
|
||||
arch.setTickHook(tick);
|
||||
}
|
||||
|
||||
/// The idle task: run when every other task is blocked or sleeping. `hlt` waits
|
||||
/// for the next interrupt at near-zero power (see docs/halting.md).
|
||||
fn idle() void {
|
||||
while (true) asm volatile ("hlt");
|
||||
}
|
||||
|
||||
fn enqueue(t: *Task) void {
|
||||
t.next = null;
|
||||
const p: usize = t.priority;
|
||||
if (ready_tail[p]) |tail| tail.next = t else ready_head[p] = t;
|
||||
ready_tail[p] = t;
|
||||
ready_bitmap |= levelBit(t.priority);
|
||||
}
|
||||
|
||||
fn dequeueHighest() ?*Task {
|
||||
if (ready_bitmap == 0) return null;
|
||||
const level: Priority = @intCast(num_priorities - 1 - @clz(ready_bitmap));
|
||||
const t = ready_head[level].?;
|
||||
ready_head[level] = t.next;
|
||||
if (ready_head[level] == null) {
|
||||
ready_tail[level] = null;
|
||||
ready_bitmap &= ~levelBit(level);
|
||||
}
|
||||
t.next = null;
|
||||
return t;
|
||||
}
|
||||
|
||||
fn levelBit(p: Priority) u8 {
|
||||
return @as(u8, 1) << p;
|
||||
}
|
||||
|
||||
/// Create a task that runs `entry` at `priority`. It becomes ready immediately.
|
||||
pub fn spawn(entry: *const fn () void, priority: Priority) void {
|
||||
const t = freeSlot() orelse @panic("sched: task table full");
|
||||
const stack = heap.allocator().alloc(u8, stack_size) catch @panic("sched: no memory for task stack");
|
||||
t.* = .{ .id = next_id, .state = .ready, .priority = priority, .stack = stack };
|
||||
next_id += 1;
|
||||
const top = @intFromPtr(stack.ptr) + stack.len;
|
||||
t.rsp = arch.initTaskStack(top, @intFromPtr(entry));
|
||||
enqueue(t);
|
||||
}
|
||||
|
||||
fn freeSlot() ?*Task {
|
||||
for (&tasks) |*t| {
|
||||
if (t.state == .free) return t;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Pick the highest-priority ready task and switch to it. Interrupts must be
|
||||
/// disabled by the caller.
|
||||
fn schedule() void {
|
||||
const prev = current;
|
||||
if (prev.state == .running) {
|
||||
prev.state = .ready;
|
||||
enqueue(prev); // back of its level's queue (round-robin)
|
||||
}
|
||||
const next = dequeueHighest() orelse {
|
||||
prev.state = .running; // nothing else ready — keep running
|
||||
return;
|
||||
};
|
||||
next.state = .running;
|
||||
current = next;
|
||||
if (next != prev) arch.switchContext(&prev.rsp, next.rsp);
|
||||
}
|
||||
|
||||
/// Voluntarily give up the CPU to the next ready task.
|
||||
pub fn yield() void {
|
||||
const flags = arch.saveInterrupts();
|
||||
schedule();
|
||||
arch.restoreInterrupts(flags);
|
||||
}
|
||||
|
||||
/// Block the current task for `ms` milliseconds, then let it become runnable
|
||||
/// again. The idle task (or other work) runs in the meantime.
|
||||
pub fn sleep(ms: u64) void {
|
||||
const flags = arch.saveInterrupts();
|
||||
current.wake_at = arch.millis() + ms;
|
||||
current.state = .blocked;
|
||||
schedule(); // current is blocked, so schedule() won't re-enqueue it
|
||||
arch.restoreInterrupts(flags);
|
||||
}
|
||||
|
||||
// --- event-based blocking -------------------------------------------------
|
||||
//
|
||||
// A WaitQueue is a set of tasks blocked waiting for something (a resource, a
|
||||
// message). Tasks link into it through the same `next` field the ready queues
|
||||
// use — a task is in exactly one queue at a time. These are the primitive locks,
|
||||
// semaphores and IPC channels are built on.
|
||||
|
||||
pub const WaitQueue = struct {
|
||||
head: ?*Task = null,
|
||||
};
|
||||
|
||||
/// Block the current task on `wq` and switch away. Precondition: interrupts are
|
||||
/// disabled (the caller holds them, so a condition can be checked and the block
|
||||
/// committed atomically). On return — when woken — interrupts are still disabled.
|
||||
pub fn waitLocked(wq: *WaitQueue) void {
|
||||
current.state = .blocked;
|
||||
current.next = wq.head;
|
||||
wq.head = current;
|
||||
schedule();
|
||||
}
|
||||
|
||||
/// Move the highest-priority waiter on `wq` (if any) to the ready queue.
|
||||
/// Precondition: interrupts disabled. Does not preempt — the caller decides.
|
||||
pub fn wakeLocked(wq: *WaitQueue) void {
|
||||
// Find the highest-priority waiter (bounded scan) and unlink it.
|
||||
var best_prev: ?*Task = null;
|
||||
var best: ?*Task = null;
|
||||
var prev: ?*Task = null;
|
||||
var cur = wq.head;
|
||||
while (cur) |t| : ({
|
||||
prev = t;
|
||||
cur = t.next;
|
||||
}) {
|
||||
if (best == null or t.priority > best.?.priority) {
|
||||
best = t;
|
||||
best_prev = prev;
|
||||
}
|
||||
}
|
||||
const t = best orelse return;
|
||||
if (best_prev) |p| p.next = t.next else wq.head = t.next;
|
||||
t.state = .ready;
|
||||
enqueue(t);
|
||||
}
|
||||
|
||||
/// Block on `wq` (a self-contained critical section).
|
||||
pub fn wait(wq: *WaitQueue) void {
|
||||
const flags = arch.saveInterrupts();
|
||||
waitLocked(wq);
|
||||
arch.restoreInterrupts(flags);
|
||||
}
|
||||
|
||||
/// Wake the highest-priority waiter on `wq`, preempting if it outranks us.
|
||||
pub fn wake(wq: *WaitQueue) void {
|
||||
const flags = arch.saveInterrupts();
|
||||
wakeLocked(wq);
|
||||
// If a higher-priority task is now ready, run it immediately.
|
||||
if (highestReadyPriority()) |p| {
|
||||
if (p > current.priority) schedule();
|
||||
}
|
||||
arch.restoreInterrupts(flags);
|
||||
}
|
||||
|
||||
fn highestReadyPriority() ?Priority {
|
||||
if (ready_bitmap == 0) return null;
|
||||
return @intCast(num_priorities - 1 - @clz(ready_bitmap));
|
||||
}
|
||||
|
||||
/// Wake any sleeping task whose deadline has passed. Bounded by the task count,
|
||||
/// so it stays deterministic. Called from the timer tick (interrupts disabled).
|
||||
fn wakeExpired() void {
|
||||
const now = arch.millis();
|
||||
for (&tasks) |*t| {
|
||||
if (t.state == .blocked and t.wake_at != 0 and now >= t.wake_at) {
|
||||
t.wake_at = 0;
|
||||
t.state = .ready;
|
||||
enqueue(t);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Called from the timer interrupt (interrupts already disabled): wake due
|
||||
/// sleepers, then preempt.
|
||||
pub fn tick() void {
|
||||
wakeExpired();
|
||||
if (preemption_enabled) schedule();
|
||||
}
|
||||
|
||||
/// Enable or disable timer-driven preemption (cooperative-only when off).
|
||||
pub fn setPreemption(enabled: bool) void {
|
||||
preemption_enabled = enabled;
|
||||
}
|
||||
|
||||
/// End the current task and switch away for good; never returns. The task's stack
|
||||
/// is leaked for now (no reaper yet).
|
||||
pub fn exit() noreturn {
|
||||
arch.disableInterrupts();
|
||||
current.state = .free;
|
||||
const next = dequeueHighest() orelse @panic("sched: no task left to run");
|
||||
next.state = .running;
|
||||
current = next;
|
||||
var discard: usize = 0;
|
||||
arch.switchContext(&discard, next.rsp);
|
||||
unreachable;
|
||||
}
|
||||
|
||||
pub fn currentId() u32 {
|
||||
return current.id;
|
||||
}
|
||||
|
||||
/// Change the running task's priority (takes effect next time it's enqueued).
|
||||
pub fn setPriority(p: Priority) void {
|
||||
current.priority = p;
|
||||
}
|
||||
@@ -0,0 +1,462 @@
|
||||
//! In-kernel test cases, run at the end of bring-up when the kernel is built with
|
||||
//! `-Dtest-case=<name>`. Each case writes structured markers to the serial port
|
||||
//! that the QEMU harness (test/qemu_test.py) asserts on:
|
||||
//!
|
||||
//! [PASS]/[FAIL] <check> per assertion
|
||||
//! DANOS-TEST-RESULT: PASS|FAIL overall, for non-faulting cases
|
||||
//!
|
||||
//! Faulting cases (fault-ud, fault-pf, fault-df) deliberately don't return a
|
||||
//! result line — they trigger a CPU exception, and the harness asserts on the
|
||||
//! exception report the handler prints (which also reaches serial).
|
||||
|
||||
const std = @import("std");
|
||||
const danos = @import("danos");
|
||||
const arch = @import("arch");
|
||||
const pmm = @import("pmm.zig");
|
||||
const heap = @import("heap.zig");
|
||||
const sched = @import("sched.zig");
|
||||
const ipc = @import("ipc.zig");
|
||||
|
||||
/// Formatted write straight to serial, independent of the framebuffer console.
|
||||
fn log(comptime fmt: []const u8, args: anytype) void {
|
||||
var buf: [128]u8 = undefined;
|
||||
arch.serialWrite(std.fmt.bufPrint(&buf, fmt, args) catch return);
|
||||
}
|
||||
|
||||
var passed: u32 = 0;
|
||||
var failed: u32 = 0;
|
||||
|
||||
fn check(name: []const u8, ok: bool) void {
|
||||
if (ok) {
|
||||
passed += 1;
|
||||
log("[PASS] {s}\n", .{name});
|
||||
} else {
|
||||
failed += 1;
|
||||
log("[FAIL] {s}\n", .{name});
|
||||
}
|
||||
}
|
||||
|
||||
/// Emit the overall result line the harness matches, then the done sentinel.
|
||||
fn result() void {
|
||||
log("DANOS-TEST-RESULT: {s} ({d} passed, {d} failed)\n", .{
|
||||
if (failed == 0) "PASS" else "FAIL",
|
||||
passed,
|
||||
failed,
|
||||
});
|
||||
log("DANOS-TEST-DONE\n", .{});
|
||||
}
|
||||
|
||||
pub fn run(case: []const u8, boot_info: *const BootInfo) void {
|
||||
if (eql(case, "smoke")) {
|
||||
smoke(boot_info);
|
||||
} else if (eql(case, "timer")) {
|
||||
timer();
|
||||
} else if (eql(case, "clock")) {
|
||||
clock();
|
||||
} else if (eql(case, "vmm")) {
|
||||
vmm();
|
||||
} else if (eql(case, "heap")) {
|
||||
heapTest();
|
||||
} else if (eql(case, "sched")) {
|
||||
schedTest();
|
||||
} else if (eql(case, "priority")) {
|
||||
priorityTest();
|
||||
} else if (eql(case, "sleep")) {
|
||||
sleepTest();
|
||||
} else if (eql(case, "event")) {
|
||||
eventTest();
|
||||
} else if (eql(case, "ipc")) {
|
||||
ipcTest();
|
||||
} else if (eql(case, "fault-ud")) {
|
||||
faultInvalidOpcode();
|
||||
} else if (eql(case, "fault-pf")) {
|
||||
faultPageFault();
|
||||
} else if (eql(case, "fault-df")) {
|
||||
faultDoubleFault();
|
||||
} else if (eql(case, "fault-nx")) {
|
||||
faultNoExecute();
|
||||
} else if (eql(case, "fault-null")) {
|
||||
faultNull();
|
||||
} else {
|
||||
log("DANOS-TEST-RESULT: FAIL (unknown case '{s}')\n", .{case});
|
||||
}
|
||||
}
|
||||
|
||||
const BootInfo = danos.BootInfo;
|
||||
|
||||
fn eql(a: []const u8, b: []const u8) bool {
|
||||
return std.mem.eql(u8, a, b);
|
||||
}
|
||||
|
||||
/// Non-destructive checks of the memory map and frame allocator.
|
||||
fn smoke(boot_info: *const BootInfo) void {
|
||||
log("DANOS-TEST-BEGIN: smoke\n", .{});
|
||||
|
||||
// The memory map has some usable RAM.
|
||||
const mm = boot_info.memory_map;
|
||||
const regions = @as([*]const danos.MemoryRegion, @ptrFromInt(mm.regions))[0..mm.len];
|
||||
var usable: u64 = 0;
|
||||
for (regions) |r| {
|
||||
if (r.kind == .usable) usable += r.pages;
|
||||
}
|
||||
check("memory map reports usable RAM", usable > 0);
|
||||
|
||||
// The frame allocator hands out distinct, page-aligned frames.
|
||||
const a = pmm.alloc();
|
||||
const b = pmm.alloc();
|
||||
check("alloc returns a frame", a != null);
|
||||
check("alloc returns distinct frames", a != null and b != null and a.? != b.?);
|
||||
check("frames are page-aligned", (a orelse 1) % danos.page_size == 0);
|
||||
|
||||
// Freeing restores the count.
|
||||
const before = pmm.stats().free_frames;
|
||||
if (a) |p| pmm.free(p);
|
||||
if (b) |p| pmm.free(p);
|
||||
check("free returns frames to the pool", pmm.stats().free_frames == before + 2);
|
||||
|
||||
// Paging is active on our own tables (CR3 is non-zero and page-aligned).
|
||||
const cr3 = arch.readCr3();
|
||||
check("paging active (CR3 set)", cr3 != 0 and cr3 % danos.page_size == 0);
|
||||
|
||||
result();
|
||||
}
|
||||
|
||||
/// Verify device interrupts fire and return: the timer tick counter must advance
|
||||
/// on its own. Interrupts are already enabled by kmain before tests run.
|
||||
fn timer() void {
|
||||
log("DANOS-TEST-BEGIN: timer\n", .{});
|
||||
const start = arch.ticks();
|
||||
// Busy-wait for the counter to advance. arch.ticks() is a volatile load, so
|
||||
// the compiler re-reads it each iteration and sees the interrupt's update.
|
||||
// The cap is only a safety net; the harness timeout is the real backstop.
|
||||
var spins: u64 = 0;
|
||||
while (arch.ticks() == start and spins < 5_000_000_000) spins +%= 1;
|
||||
check("timer interrupts advance the tick count", arch.ticks() > start);
|
||||
|
||||
result();
|
||||
}
|
||||
|
||||
/// Verify the on-demand VMM: map a fresh frame at an unused virtual address, and
|
||||
/// check it's writable and reads back.
|
||||
fn vmm() void {
|
||||
log("DANOS-TEST-BEGIN: vmm\n", .{});
|
||||
const frame = pmm.alloc();
|
||||
check("frame available to map", frame != null);
|
||||
if (frame) |phys| {
|
||||
var virt: u64 = 0x0000_4000_0000_0000; // canonical, well clear of everything mapped
|
||||
arch.mapPage(virt, phys, true);
|
||||
const p: *volatile u64 = @ptrFromInt(virt);
|
||||
p.* = 0xdead_c0de_cafe_babe;
|
||||
check("mapped page is writable and reads back", p.* == 0xdead_c0de_cafe_babe);
|
||||
arch.unmapPage(virt);
|
||||
pmm.free(phys);
|
||||
virt += 0;
|
||||
}
|
||||
result();
|
||||
}
|
||||
|
||||
/// Exercise the kernel heap: basic alloc/write/free, reuse, growth beyond the
|
||||
/// initial region, and a std container backed by it.
|
||||
fn heapTest() void {
|
||||
log("DANOS-TEST-BEGIN: heap\n", .{});
|
||||
const a = heap.allocator();
|
||||
|
||||
// Allocate, write a pattern, read it back, free.
|
||||
const buf = a.alloc(u8, 4096) catch null;
|
||||
check("alloc 4096 bytes", buf != null);
|
||||
if (buf) |b| {
|
||||
@memset(b, 0xAB);
|
||||
check("heap memory is writable and reads back", b[0] == 0xAB and b[4095] == 0xAB);
|
||||
a.free(b);
|
||||
}
|
||||
|
||||
// Freeing then re-allocating the same size should reuse the block.
|
||||
const p1 = a.alloc(u64, 8) catch null;
|
||||
const addr1 = if (p1) |p| @intFromPtr(p.ptr) else 0;
|
||||
if (p1) |p| a.free(p);
|
||||
const p2 = a.alloc(u64, 8) catch null;
|
||||
const addr2 = if (p2) |p| @intFromPtr(p.ptr) else 0;
|
||||
check("freed block is reused", addr1 != 0 and addr1 == addr2);
|
||||
if (p2) |p| a.free(p);
|
||||
|
||||
// Force growth past the initial page and check every block is usable.
|
||||
var blocks: [64]?[]u8 = .{null} ** 64;
|
||||
var ok = true;
|
||||
for (&blocks, 0..) |*slot, i| {
|
||||
const b = a.alloc(u8, 4096) catch null;
|
||||
slot.* = b;
|
||||
if (b) |bb| @memset(bb, @intCast(i & 0xff)) else {
|
||||
ok = false;
|
||||
}
|
||||
}
|
||||
for (blocks, 0..) |slot, i| {
|
||||
if (slot) |bb| {
|
||||
if (bb[0] != @as(u8, @intCast(i & 0xff)) or bb[4095] != @as(u8, @intCast(i & 0xff))) ok = false;
|
||||
}
|
||||
}
|
||||
check("many allocations (heap growth) stay valid", ok);
|
||||
for (blocks) |slot| {
|
||||
if (slot) |bb| a.free(bb);
|
||||
}
|
||||
|
||||
// A std container backed by the kernel heap.
|
||||
var list: std.ArrayList(u32) = .empty;
|
||||
var sum: u64 = 0;
|
||||
var expected: u64 = 0;
|
||||
var i: u32 = 0;
|
||||
var list_ok = true;
|
||||
while (i < 1000) : (i += 1) {
|
||||
list.append(a, i) catch {
|
||||
list_ok = false;
|
||||
};
|
||||
expected += i;
|
||||
}
|
||||
for (list.items) |v| sum += v;
|
||||
list.deinit(a);
|
||||
check("std.ArrayList on the kernel heap", list_ok and sum == expected);
|
||||
|
||||
result();
|
||||
}
|
||||
|
||||
/// Verify the calibrated clocks: sane measured frequencies, monotonic uptime that
|
||||
/// advances with real ticks, and — the point of the TSC clock — nanosecond
|
||||
/// resolution far finer than the 1 ms tick, with the unit functions consistent.
|
||||
fn clock() void {
|
||||
log("DANOS-TEST-BEGIN: clock\n", .{});
|
||||
|
||||
const lapic = arch.lapicHz();
|
||||
check("LAPIC frequency measured", lapic > 1_000_000 and lapic < 100_000_000_000);
|
||||
const tsc = arch.tscHz();
|
||||
check("TSC frequency measured", tsc > 100_000_000 and tsc < 100_000_000_000);
|
||||
|
||||
// Uptime advances over ~5 real ticks (1000 Hz => 1 tick == 1 ms).
|
||||
const start_ticks = arch.ticks();
|
||||
const start_ms = arch.millis();
|
||||
var spins: u64 = 0;
|
||||
while (arch.ticks() < start_ticks + 5 and spins < 5_000_000_000) spins +%= 1;
|
||||
const elapsed_ms = arch.millis() - start_ms;
|
||||
check("uptime advances with ticks", elapsed_ms >= 5 and elapsed_ms < 100);
|
||||
|
||||
// Sub-millisecond resolution: spin until nanos() first advances, then confirm
|
||||
// that first step happened within a millisecond — so nanos() resolves finer
|
||||
// than the 1 ms tick (a tick clock's smallest step *is* 1 ms). Spinning to the
|
||||
// first change is robust to QEMU's coarse TSC update granularity.
|
||||
const n1 = arch.nanos();
|
||||
var s2: u64 = 0;
|
||||
while (arch.nanos() == n1 and s2 < 10_000_000) s2 +%= 1;
|
||||
const n2 = arch.nanos();
|
||||
check("nanos() has sub-millisecond resolution", n2 > n1 and (n2 - n1) < 1_000_000);
|
||||
|
||||
// The unit functions agree (within rounding).
|
||||
const ns = arch.nanos();
|
||||
check("nanos/micros/millis are consistent", diffWithin(arch.micros(), ns / 1000, 1000) and diffWithin(arch.millis(), ns / 1_000_000, 2));
|
||||
|
||||
result();
|
||||
}
|
||||
|
||||
fn diffWithin(a: u64, b: u64, tol: u64) bool {
|
||||
return if (a > b) a - b <= tol else b - a <= tol;
|
||||
}
|
||||
|
||||
// --- scheduler tests ------------------------------------------------------
|
||||
|
||||
var counters = [_]u64{0} ** 3;
|
||||
|
||||
fn spin0() void {
|
||||
const p: *volatile u64 = &counters[0];
|
||||
while (true) p.* = p.* +% 1;
|
||||
}
|
||||
fn spin1() void {
|
||||
const p: *volatile u64 = &counters[1];
|
||||
while (true) p.* = p.* +% 1;
|
||||
}
|
||||
fn spin2() void {
|
||||
const p: *volatile u64 = &counters[2];
|
||||
while (true) p.* = p.* +% 1;
|
||||
}
|
||||
|
||||
/// Preemption: spawn three tasks that busy-loop *without* yielding. If they all
|
||||
/// make progress, the timer must be preempting between them (and the context
|
||||
/// switch works) — because nothing yields voluntarily.
|
||||
fn schedTest() void {
|
||||
log("DANOS-TEST-BEGIN: sched\n", .{});
|
||||
counters = .{ 0, 0, 0 };
|
||||
sched.spawn(spin0, 4);
|
||||
sched.spawn(spin1, 4);
|
||||
sched.spawn(spin2, 4);
|
||||
|
||||
const c0: *volatile u64 = &counters[0];
|
||||
const c1: *volatile u64 = &counters[1];
|
||||
const c2: *volatile u64 = &counters[2];
|
||||
var spins: u64 = 0;
|
||||
while ((c0.* == 0 or c1.* == 0 or c2.* == 0) and spins < 5_000_000_000) spins +%= 1;
|
||||
|
||||
check("all three non-yielding tasks made progress (preemption)", c0.* > 0 and c1.* > 0 and c2.* > 0);
|
||||
result();
|
||||
}
|
||||
|
||||
var run_order = [_]u8{0} ** 4;
|
||||
var run_n: usize = 0;
|
||||
|
||||
fn recordExit(priority: u8) void {
|
||||
run_order[run_n] = priority;
|
||||
run_n += 1;
|
||||
sched.exit();
|
||||
}
|
||||
fn taskHigh() void {
|
||||
recordExit(6);
|
||||
}
|
||||
fn taskMid() void {
|
||||
recordExit(4);
|
||||
}
|
||||
fn taskLow() void {
|
||||
recordExit(2);
|
||||
}
|
||||
|
||||
/// Fixed priority: with preemption off (deterministic), spawn tasks at three
|
||||
/// priorities and let them run cooperatively. They must run highest-first.
|
||||
fn priorityTest() void {
|
||||
log("DANOS-TEST-BEGIN: priority\n", .{});
|
||||
sched.setPreemption(false);
|
||||
sched.setPriority(1); // above the idle task (0), below the workers — runs last
|
||||
run_n = 0;
|
||||
|
||||
sched.spawn(taskLow, 2);
|
||||
sched.spawn(taskMid, 4);
|
||||
sched.spawn(taskHigh, 6);
|
||||
|
||||
while (run_n < 3) sched.yield(); // regain control only once the workers are done
|
||||
|
||||
check("tasks ran highest-priority first", run_order[0] == 6 and run_order[1] == 4 and run_order[2] == 2);
|
||||
|
||||
sched.setPriority(4);
|
||||
sched.setPreemption(true);
|
||||
result();
|
||||
}
|
||||
|
||||
var event_wq: sched.WaitQueue = .{};
|
||||
var event_stage: u32 = 0;
|
||||
|
||||
fn eventWaiter() void {
|
||||
event_stage = 1; // reached the wait
|
||||
sched.wait(&event_wq); // block until woken
|
||||
event_stage = 3; // woken and resumed
|
||||
sched.exit();
|
||||
}
|
||||
|
||||
/// Event-based blocking: a task blocks on a wait queue and is woken. The waiter is
|
||||
/// higher priority, so waking it preempts us and it runs to completion at once.
|
||||
fn eventTest() void {
|
||||
log("DANOS-TEST-BEGIN: event\n", .{});
|
||||
event_stage = 0;
|
||||
sched.spawn(eventWaiter, 6); // higher priority than this task (4)
|
||||
|
||||
var spins: u64 = 0;
|
||||
while (event_stage != 1 and spins < 1_000_000_000) : (spins += 1) sched.yield();
|
||||
check("waiter reached the wait and blocked", event_stage == 1);
|
||||
|
||||
sched.wake(&event_wq);
|
||||
check("wake resumed the blocked waiter (preempting)", event_stage == 3);
|
||||
result();
|
||||
}
|
||||
|
||||
var channel: ipc.Channel(u64, 4) = .{};
|
||||
var recv_sum: u64 = 0;
|
||||
var recv_count: u64 = 0;
|
||||
|
||||
fn producer() void {
|
||||
var i: u64 = 1;
|
||||
while (i <= 100) : (i += 1) channel.send(i);
|
||||
sched.exit();
|
||||
}
|
||||
fn consumer() void {
|
||||
var n: u64 = 0;
|
||||
while (n < 100) : (n += 1) {
|
||||
recv_sum += channel.recv();
|
||||
recv_count += 1;
|
||||
}
|
||||
sched.exit();
|
||||
}
|
||||
|
||||
/// IPC: a producer and consumer pass 100 messages through a 4-slot channel. The
|
||||
/// small buffer forces the channel full and empty repeatedly, exercising both the
|
||||
/// blocking-send and blocking-recv paths. The messages must arrive intact.
|
||||
fn ipcTest() void {
|
||||
log("DANOS-TEST-BEGIN: ipc\n", .{});
|
||||
channel = .{};
|
||||
recv_sum = 0;
|
||||
recv_count = 0;
|
||||
sched.spawn(consumer, 5); // above this task (4) so they run and we observe after
|
||||
sched.spawn(producer, 5);
|
||||
|
||||
var spins: u64 = 0;
|
||||
while (recv_count < 100 and spins < 2_000_000_000) : (spins += 1) sched.yield();
|
||||
|
||||
check("all 100 messages received", recv_count == 100);
|
||||
check("messages arrived intact (sum 1..100 == 5050)", recv_sum == 5050);
|
||||
result();
|
||||
}
|
||||
|
||||
/// Blocking: sleep(50) should block this task for about 50 ms (measured on the
|
||||
/// calibrated clock) — not busy-wait — while the idle task runs.
|
||||
fn sleepTest() void {
|
||||
log("DANOS-TEST-BEGIN: sleep\n", .{});
|
||||
const t0 = arch.millis();
|
||||
sched.sleep(50);
|
||||
const elapsed = arch.millis() - t0;
|
||||
check("sleep(50) blocked for ~50 ms", elapsed >= 50 and elapsed <= 70);
|
||||
result();
|
||||
}
|
||||
|
||||
fn faultInvalidOpcode() void {
|
||||
log("DANOS-TEST-BEGIN: fault-ud\n", .{});
|
||||
asm volatile ("ud2");
|
||||
}
|
||||
|
||||
/// Verify NX: fetching an instruction from a data page (mapped no-execute) faults.
|
||||
fn faultNoExecute() void {
|
||||
log("DANOS-TEST-BEGIN: fault-nx\n", .{});
|
||||
var scratch: u64 = 0xC3; // a lone `ret` — harmless if NX somehow let it run
|
||||
const f: *const fn () void = @ptrFromInt(@intFromPtr(&scratch));
|
||||
f(); // instruction fetch from an NX page -> #PF before it executes
|
||||
log("DANOS-TEST-RESULT: FAIL (NX not enforced)\n", .{});
|
||||
}
|
||||
|
||||
/// Verify the null guard: dereferencing address 0 (page 0 left unmapped) faults.
|
||||
fn faultNull() void {
|
||||
log("DANOS-TEST-BEGIN: fault-null\n", .{});
|
||||
// Launder the address through empty asm so the compiler no longer knows it's
|
||||
// 0 (otherwise it folds a null-pointer safety panic instead of doing the real
|
||||
// access). `allowzero` skips the same null check on the cast. The write then
|
||||
// hits the unmapped page 0 and takes a real hardware #PF.
|
||||
var addr: u64 = 0;
|
||||
addr = asm ("" : [ret] "=r" (-> u64) : [in] "0" (addr));
|
||||
const p: *allowzero volatile u64 = @ptrFromInt(addr);
|
||||
p.* = 1;
|
||||
}
|
||||
|
||||
fn faultPageFault() void {
|
||||
log("DANOS-TEST-BEGIN: fault-pf\n", .{});
|
||||
// Runtime address so the backend emits a register store (not a `mov moffs`,
|
||||
// which the self-hosted x86_64 backend can't encode).
|
||||
var addr: u64 = 0xdeadbeef000; // well above all mapped RAM
|
||||
const p: *volatile u64 = @ptrFromInt(addr);
|
||||
p.* = 1;
|
||||
addr += 0;
|
||||
}
|
||||
|
||||
fn faultDoubleFault() void {
|
||||
log("DANOS-TEST-BEGIN: fault-df\n", .{});
|
||||
arch.disableInterrupts(); // so only the ud2 delivery (not a timer tick) triggers the #DF
|
||||
// Point RSP at unmapped memory, then fault: the CPU can't push the fault
|
||||
// frame, which escalates to #DF — survivable only because #DF runs on IST1.
|
||||
var bad_sp: u64 = 0x5000000000;
|
||||
asm volatile (
|
||||
\\mov %[sp], %%rsp
|
||||
\\ud2
|
||||
:
|
||||
: [sp] "r" (bad_sp),
|
||||
: .{ .memory = true }
|
||||
);
|
||||
bad_sp += 0;
|
||||
}
|
||||
Reference in New Issue
Block a user