moving kernel code to kernel/

This commit is contained in:
2026-07-05 10:19:11 +01:00
parent 7c3cffb337
commit 6f6ccc8bc9
37 changed files with 63 additions and 61 deletions
+202
View File
@@ -0,0 +1,202 @@
//! Local APIC and its timer — the source of device interrupts.
//!
//! Modern x86 routes interrupts through the per-CPU Local APIC (the legacy 8259
//! PIC is remapped out of the way and masked). The LAPIC also has a built-in
//! timer, which is the simplest device interrupt to bring up: it needs no
//! external routing, just a vector and a count. We use it as danos's heartbeat.
//!
//! The LAPIC is memory-mapped (default physical 0xFEE00000, inside our identity
//! map). Every interrupt must be acknowledged with an end-of-interrupt write, or
//! the LAPIC won't deliver the next one.
const io = @import("io.zig");
/// IDT vector the timer fires on (in the device range, >= 32).
pub const timer_vector = 32;
/// Spurious-interrupt vector. Low nibble 0xF by convention; also in our gate
/// range so a stray spurious interrupt lands on a valid (no-op) handler.
const spurious_vector = 47;
// LAPIC register offsets.
const reg_spurious = 0x0F0;
const reg_eoi = 0x0B0;
const reg_lvt_timer = 0x320;
const reg_timer_initial = 0x380;
const reg_timer_current = 0x390;
const reg_timer_divide = 0x3E0;
const lvt_masked = 1 << 16;
const lvt_periodic = 1 << 17;
const timer_divide_16 = 0x3;
const ia32_apic_base_msr = 0x1B;
/// LAPIC MMIO base. A runtime var (not a constant) both because we read it from
/// the MSR and so register writes compile to normal stores rather than a
/// `mov moffs`, which the self-hosted backend can't encode.
var base: usize = 0xFEE00000;
var tick_count: u64 = 0;
/// LAPIC timer counts per millisecond, measured against the PIT (see calibrate).
/// At divide-by-16, this is the effective counting rate.
var ticks_per_ms: u32 = 0;
/// The periodic-interrupt frequency the timer is armed at, once initTimer runs.
var timer_hz: u32 = 0;
/// TSC (Time Stamp Counter) calibration: cycles per second, and the count at boot.
/// The TSC is a per-core cycle counter, giving a ~nanosecond high-resolution
/// monotonic clock — far finer than the millisecond timer tick.
var tsc_hz: u64 = 0;
var tsc_base: u64 = 0;
/// Read the 64-bit Time Stamp Counter.
fn rdtsc() u64 {
var low: u32 = undefined;
var high: u32 = undefined;
asm volatile ("rdtsc"
: [low] "={eax}" (low),
[high] "={edx}" (high),
);
return (@as(u64, high) << 32) | low;
}
fn read(reg: u32) u32 {
return @as(*volatile u32, @ptrFromInt(base + reg)).*;
}
fn write(reg: u32, value: u32) void {
@as(*volatile u32, @ptrFromInt(base + reg)).* = value;
}
/// Move the legacy 8259 PIC's vectors to 0x20-0x2F (clear of the CPU exception
/// vectors) and mask every line, so it can't deliver interrupts behind the APIC.
fn remapAndMaskPic() void {
io.outb(0x20, 0x11); // start init (cascade mode)
io.outb(0xA0, 0x11);
io.outb(0x21, 0x20); // master offset 0x20
io.outb(0xA1, 0x28); // slave offset 0x28
io.outb(0x21, 0x04); // tell master about slave on IRQ2
io.outb(0xA1, 0x02);
io.outb(0x21, 0x01); // 8086 mode
io.outb(0xA1, 0x01);
io.outb(0x21, 0xFF); // mask all
io.outb(0xA1, 0xFF);
}
/// Enable the Local APIC: mask the PIC, set the global-enable MSR bit, and
/// software-enable the APIC via its spurious-vector register.
pub fn init() void {
remapAndMaskPic();
const msr = io.rdmsr(ia32_apic_base_msr);
base = @intCast(msr & 0xFFFFF000); // physical base is bits 12+
io.wrmsr(ia32_apic_base_msr, msr | (1 << 11)); // global enable
write(reg_spurious, 0x100 | spurious_vector); // bit 8 = software enable
}
/// Measure the LAPIC timer's and the TSC's rates against the PIT (channel 2, which
/// can be polled without interrupts). We run the LAPIC timer one-shot from its max
/// count and snapshot the TSC while the PIT counts out a known 10 ms, then see how
/// far each got. This gives real time, which the RTOS timing guarantees depend on.
pub fn calibrate() void {
const pit_hz = 1_193_182;
const calib_ms = 10;
const pit_count: u16 = @intCast(pit_hz / 1000 * calib_ms);
// LAPIC timer: divide 16, masked (no interrupt — we just want the count),
// counting down from the maximum.
write(reg_timer_divide, timer_divide_16);
write(reg_lvt_timer, lvt_masked);
write(reg_timer_initial, 0xFFFFFFFF);
// PIT channel 2, mode 0 (interrupt on terminal count): load the count with the
// gate low, then raise the gate to start it counting.
io.outb(0x61, io.inb(0x61) & 0xFC); // speaker off, gate low
io.outb(0x43, 0xB0); // channel 2, lo/hi byte, mode 0
io.outb(0x42, @truncate(pit_count));
io.outb(0x42, @truncate(pit_count >> 8));
const tsc_start = rdtsc();
io.outb(0x61, (io.inb(0x61) & 0xFC) | 0x01); // gate high -> start
while (io.inb(0x61) & 0x20 == 0) {} // poll channel-2 output until terminal count
const tsc_end = rdtsc();
const elapsed = 0xFFFFFFFF - read(reg_timer_current);
write(reg_timer_initial, 0); // stop the timer
ticks_per_ms = elapsed / calib_ms;
tsc_hz = (tsc_end -% tsc_start) * (1000 / calib_ms); // cycles/10ms -> cycles/s
tsc_base = rdtsc(); // the clock's zero point (boot)
}
/// Arm the LAPIC timer to fire on `timer_vector` at `hz` (periodic). Requires
/// calibrate() to have run.
pub fn initTimer(hz: u32) void {
timer_hz = hz;
const count = @as(u64, ticks_per_ms) * 1000 / hz; // counts per (1/hz) second
write(reg_timer_divide, timer_divide_16);
write(reg_lvt_timer, timer_vector | lvt_periodic);
write(reg_timer_initial, @intCast(count));
}
/// Configured periodic-interrupt frequency (Hz).
pub fn frequencyHz() u32 {
return timer_hz;
}
/// Measured LAPIC timer frequency (Hz), for reporting/sanity checks.
pub fn lapicHz() u64 {
return @as(u64, ticks_per_ms) * 1000;
}
/// Measured TSC frequency (Hz).
pub fn tscHz() u64 {
return tsc_hz;
}
// Monotonic high-resolution clock, from the TSC. A function per resolution, each
// scaling the cycle delta directly at its unit (the 128-bit intermediate avoids
// overflow across a long uptime). nanos() resolves to a few ns; millis() is what
// the scheduler uses for sleep deadlines.
pub fn nanos() u64 {
if (tsc_hz == 0) return 0;
return @intCast(@as(u128, rdtsc() -% tsc_base) * 1_000_000_000 / tsc_hz);
}
pub fn micros() u64 {
if (tsc_hz == 0) return 0;
return @intCast(@as(u128, rdtsc() -% tsc_base) * 1_000_000 / tsc_hz);
}
pub fn millis() u64 {
if (tsc_hz == 0) return 0;
return @intCast(@as(u128, rdtsc() -% tsc_base) * 1_000 / tsc_hz);
}
/// Acknowledge the current interrupt so the LAPIC will deliver the next one.
pub fn eoi() void {
write(reg_eoi, 0);
}
/// Optional callback run each tick (the scheduler registers it for preemption).
var on_tick: ?*const fn () void = null;
pub fn setTickHook(hook: *const fn () void) void {
on_tick = hook;
}
/// The timer interrupt handler: advance the monotonic tick count, then run the
/// tick hook (which may switch tasks). The interrupt is already acknowledged by
/// the dispatcher before we get here, so a task switch here doesn't stall it.
pub fn timerTick() void {
tick_count +%= 1;
if (on_tick) |hook| hook();
}
/// Number of timer ticks so far. Volatile load: the count is bumped
/// asynchronously by the interrupt handler, so callers must re-read memory.
pub fn ticks() u64 {
return @as(*const volatile u64, &tick_count).*;
}
+192
View File
@@ -0,0 +1,192 @@
//! x86_64 CPU operations. This is the "arch" module: the generic kernel imports
//! it as `@import("arch")` and never names x86_64 directly, so a second
//! architecture is added by pointing that module at a different directory in
//! build.zig — no change to the generic code. Keep everything CPU-specific here
//! (halt, the descriptor tables, later paging), and nothing generic.
const danos = @import("danos");
const gdt = @import("gdt.zig");
const tss = @import("tss.zig");
const idt = @import("idt.zig");
const paging = @import("paging.zig");
const serial = @import("serial.zig");
const apic = @import("apic.zig");
/// The saved register/trap frame passed to a fault handler.
pub const CpuState = idt.CpuState;
/// Bring up the serial port (the kernel's machine-readable log). No dependencies,
/// so it can be the very first thing called.
pub fn serialInit() void {
serial.init();
}
/// Write bytes to the serial port.
pub fn serialWrite(bytes: []const u8) void {
serial.write(bytes);
}
/// Set up the CPU's descriptor tables: our own GDT, the TSS (with an interrupt
/// stack for double faults), then the IDT with exception handlers. After this a
/// CPU fault is reported instead of triple-faulting. Install the fault handler
/// (setFaultHandler) first so early faults are caught.
pub fn init() void {
gdt.init();
tss.init();
idt.init();
}
/// Build the kernel's own page tables (with real permissions) and switch onto
/// them. Needs the frame allocator and the boot info (for the memory map and the
/// kernel's segment layout). Call once the frame allocator is up.
pub fn enablePaging(allocFrame: *const fn () ?u64, boot_info: *const danos.BootInfo) void {
paging.init(allocFrame, boot_info);
}
/// Map a page into the kernel address space (non-executable). For the heap, etc.
pub fn mapPage(virt: u64, phys: u64, writable: bool) void {
paging.map(virt, phys, writable);
}
/// Remove a kernel mapping.
pub fn unmapPage(virt: u64) void {
paging.unmap(virt);
}
/// CR3 holds the physical address of the active top-level page table.
pub fn readCr3() u64 {
return asm volatile ("mov %%cr3, %[out]"
: [out] "=r" (-> u64),
);
}
/// Kernel tick rate: 1000 Hz (1 ms), the scheduler's time quantum.
pub const timer_hz = 1000;
/// Enable the Local APIC, calibrate its timer against the PIT, and start it firing
/// at `timer_hz` — the kernel's real-time heartbeat. Interrupts still have to be
/// unmasked with enableInterrupts() to be delivered.
pub fn startTimer() void {
apic.init();
apic.calibrate();
idt.setHandler(apic.timer_vector, apic.timerTick);
apic.initTimer(timer_hz);
}
/// Number of timer ticks since startTimer().
pub fn ticks() u64 {
return apic.ticks();
}
// Monotonic high-resolution clock (from the TSC), one function per resolution.
pub fn nanos() u64 {
return apic.nanos();
}
pub fn micros() u64 {
return apic.micros();
}
pub fn millis() u64 {
return apic.millis();
}
/// Measured LAPIC timer / TSC frequencies in Hz (from calibration).
pub fn lapicHz() u64 {
return apic.lapicHz();
}
pub fn tscHz() u64 {
return apic.tscHz();
}
/// Unmask maskable interrupts (`sti`) so device interrupts get delivered.
pub fn enableInterrupts() void {
asm volatile ("sti");
}
/// Mask maskable interrupts (`cli`).
pub fn disableInterrupts() void {
asm volatile ("cli");
}
/// Disable interrupts and return the previous flags, so a nested critical section
/// can restore the caller's state rather than blindly re-enabling. Pairs with
/// restoreInterrupts.
pub fn saveInterrupts() u64 {
var flags: u64 = undefined;
asm volatile (
\\pushfq
\\pop %[f]
\\cli
: [f] "=r" (flags),
:
: .{ .memory = true }
);
return flags;
}
/// Re-enable interrupts only if they were enabled when `flags` was captured.
pub fn restoreInterrupts(flags: u64) void {
if (flags & 0x200 != 0) asm volatile ("sti" ::: .{ .memory = true }); // bit 9 = IF
}
/// Register a callback the timer interrupt invokes each tick (e.g. the scheduler).
pub fn setTickHook(hook: *const fn () void) void {
apic.setTickHook(hook);
}
// --- context switching (for the scheduler) -------------------------------
/// Save the current task's registers/stack and resume `new_rsp`; the old stack
/// pointer is written to `old_rsp`. Defined in isr.s.
extern fn switch_context(old_rsp: *usize, new_rsp: usize) callconv(.c) void;
pub fn switchContext(old_rsp: *usize, new_rsp: usize) void {
switch_context(old_rsp, new_rsp);
}
/// Build the initial stack for a new task so that switching to it lands in
/// `task_trampoline`, which then calls `entry`. Returns the saved stack pointer.
/// The layout must match switch_context's push order (callee-saved, then the
/// return address on top); `entry` is smuggled in via the r15 slot.
pub fn initTaskStack(stack_top: usize, entry: usize) usize {
const trampoline = @extern(*const anyopaque, .{ .name = "task_trampoline" });
var sp = stack_top;
const push = struct {
fn f(p: *usize, value: usize) void {
p.* -= @sizeOf(usize);
@as(*usize, @ptrFromInt(p.*)).* = value;
}
}.f;
push(&sp, @intFromPtr(trampoline)); // return address for switch_context's `ret`
push(&sp, 0); // rbx
push(&sp, 0); // rbp
push(&sp, 0); // r12
push(&sp, 0); // r13
push(&sp, 0); // r14
push(&sp, entry); // r15 -> task entry, read by task_trampoline
return sp;
}
/// Route CPU exceptions to `handler`, which receives the trap frame and does not
/// return. Until set, faults just halt the core.
pub fn setFaultHandler(handler: *const fn (*const CpuState) noreturn) void {
idt.on_fault = handler;
}
/// A human-readable name for a CPU exception vector.
pub fn vectorName(vector: u64) []const u8 {
return idt.vectorName(vector);
}
/// CR2 holds the faulting linear address after a page fault (#PF, vector 14).
pub fn readCr2() u64 {
return asm volatile ("mov %%cr2, %[out]"
: [out] "=r" (-> u64),
);
}
/// Park the core forever. `hlt` drops it into a low-power idle until the next
/// interrupt; the loop re-halts on every wake so the stop is permanent. See
/// docs/halting.md for the full reasoning.
pub fn halt() noreturn {
while (true) asm volatile ("hlt");
}
+54
View File
@@ -0,0 +1,54 @@
//! Global Descriptor Table. In long mode segmentation is mostly vestigial, but
//! the CPU still needs valid code/data segment descriptors, and the IDT's gates
//! reference a code selector — so we install our own flat GDT with known
//! selectors (0x08 kernel code, 0x10 kernel data) rather than trusting whatever
//! the firmware left in place.
/// Selectors into the table below (index * 8).
pub const kernel_code = 0x08;
pub const kernel_data = 0x10;
pub const tss_selector = 0x18;
/// Flat 64-bit descriptors. Base/limit are ignored in long mode; what matters is
/// the access byte and, for code, the long-mode (L) flag.
/// code: present, ring 0, executable, readable, L=1 -> 0x00AF9A00_0000FFFF
/// data: present, ring 0, writable -> 0x00CF9200_0000FFFF
/// The last two slots hold one 16-byte TSS descriptor, filled in by setTss.
var table = [_]u64{
0, // null descriptor (required)
0x00AF9A000000FFFF, // kernel code (0x08)
0x00CF92000000FFFF, // kernel data (0x10)
0, // TSS descriptor low (0x18)
0, // TSS descriptor high
};
/// Fill the 64-bit TSS system descriptor (two GDT slots) so the task register can
/// point at our TSS. Type 0x89 = present, ring 0, available 64-bit TSS.
pub fn setTss(base: u64, limit: u64) void {
table[3] = (limit & 0xFFFF) |
((base & 0xFFFF) << 16) |
(((base >> 16) & 0xFF) << 32) |
(@as(u64, 0x89) << 40) |
(((limit >> 16) & 0xF) << 48) |
(((base >> 24) & 0xFF) << 56);
table[4] = (base >> 32) & 0xFFFFFFFF;
}
/// The operand `lgdt` wants: table byte-length minus one, then its address.
const Descriptor = packed struct {
limit: u16,
base: u64,
};
/// Loads the GDT and reloads the segment registers (including CS). Defined in
/// isr.s — it uses the selectors 0x08 (code) and 0x10 (data) that match `table`.
extern fn gdt_flush(descriptor: *const Descriptor) callconv(.c) void;
/// Install our GDT and switch onto its segments.
pub fn init() void {
const descriptor = Descriptor{
.limit = @sizeOf(@TypeOf(table)) - 1,
.base = @intFromPtr(&table),
};
gdt_flush(&descriptor);
}
+155
View File
@@ -0,0 +1,155 @@
//! Interrupt Descriptor Table, CPU-exception handlers, and device-interrupt
//! dispatch. Without this, any fault (a stray pointer, a bad page-table entry)
//! triple-faults and silently resets the machine. With it, the CPU vectors into
//! our stubs, which capture the register state and hand it to a dispatcher.
//!
//! Vectors split in two: 0-31 are CPU exceptions (terminal — reported and
//! halted); 32+ are device interrupts (a registered handler runs, the APIC is
//! acknowledged, and we return to the interrupted code).
const gdt = @import("gdt.zig");
const tss = @import("tss.zig");
const apic = @import("apic.zig");
/// Highest vector we install a gate/stub for (exceptions 0-31 plus the device
/// range 32-47, which covers the timer and the spurious vector).
const gate_count = 48;
/// A device-interrupt handler. It doesn't get the trap frame (a timer or keyboard
/// handler doesn't need the interrupted registers); add that if one ever does.
pub const Handler = *const fn () void;
var handlers = [_]?Handler{null} ** 256;
/// Register `handler` for a device-interrupt `vector` (>= 32).
pub fn setHandler(vector: usize, handler: Handler) void {
handlers[vector] = handler;
}
/// The register + trap frame the ISR stubs build on the stack, laid out so the
/// lowest address (where RSP points when we call the handler) is the first field.
/// See the push order in `isrCommon` below.
pub const CpuState = extern struct {
r15: u64,
r14: u64,
r13: u64,
r12: u64,
r11: u64,
r10: u64,
r9: u64,
r8: u64,
rbp: u64,
rdi: u64,
rsi: u64,
rdx: u64,
rcx: u64,
rbx: u64,
rax: u64,
vector: u64, // pushed by the per-vector stub
error_code: u64, // real one from the CPU, or 0 pushed by the stub
rip: u64, // from here down: pushed by the CPU on entry
cs: u64,
rflags: u64,
rsp: u64,
ss: u64,
};
/// Where a fault is reported. The kernel overrides this (see setFaultHandler) with
/// something that prints to the console; until then, just stop.
pub var on_fault: *const fn (*const CpuState) noreturn = defaultFault;
fn defaultFault(_: *const CpuState) noreturn {
while (true) asm volatile ("hlt");
}
/// Names for the 32 defined exception vectors, for readable output.
const names = [_][]const u8{
"divide error", "debug",
"NMI", "breakpoint",
"overflow", "bound range exceeded",
"invalid opcode", "device not available",
"double fault", "coprocessor segment overrun",
"invalid TSS", "segment not present",
"stack-segment fault", "general protection fault",
"page fault", "reserved (15)",
"x87 floating-point", "alignment check",
"machine check", "SIMD floating-point",
"virtualization", "control protection",
"reserved (22)", "reserved (23)",
"reserved (24)", "reserved (25)",
"reserved (26)", "reserved (27)",
"hypervisor injection", "VMM communication",
"security exception", "reserved (31)",
};
pub fn vectorName(vector: u64) []const u8 {
return if (vector < names.len) names[vector] else "unknown";
}
/// A 64-bit IDT gate descriptor (16 bytes).
const Gate = packed struct {
offset_low: u16,
selector: u16,
ist: u8, // interrupt-stack-table index; 0 = use the current stack
flags: u8, // present, DPL, gate type
offset_mid: u16,
offset_high: u32,
reserved: u32 = 0,
};
var idt = [_]Gate{std.mem.zeroes(Gate)} ** 256;
const Descriptor = packed struct {
limit: u16,
base: u64,
};
/// Loads the IDT (`lidt`). Defined in isr.s.
extern fn idt_flush(descriptor: *const Descriptor) callconv(.c) void;
fn setGate(vector: usize, handler: u64) void {
idt[vector] = .{
.offset_low = @truncate(handler),
.selector = gdt.kernel_code,
.ist = 0,
.flags = 0x8E, // present, ring 0, 64-bit interrupt gate
.offset_mid = @truncate(handler >> 16),
.offset_high = @truncate(handler >> 32),
};
}
/// Point every installed vector at its stub (isr.s) and load the IDT.
pub fn init() void {
@setEvalBranchQuota(20000); // comptimePrint across all the gates adds up
inline for (0..gate_count) |vector| {
const stub = @extern(*const anyopaque, .{ .name = std.fmt.comptimePrint("isr{d}", .{vector}) });
setGate(vector, @intFromPtr(stub));
}
// Run the double-fault handler (vector 8) on IST1: a #DF usually means the
// current stack is unusable, so it needs a guaranteed-good one. See tss.zig.
idt[8].ist = tss.double_fault_ist;
const descriptor = Descriptor{
.limit = @sizeOf(@TypeOf(idt)) - 1,
.base = @intFromPtr(&idt),
};
idt_flush(&descriptor);
}
/// Called by isr_common (isr.s) with a pointer to the trap frame. Exported so the
/// assembly stubs can `call` it by name. Exceptions are terminal; device
/// interrupts run their handler, get acknowledged, and return.
export fn interruptDispatch(state: *const CpuState) callconv(.c) void {
if (state.vector < 32) {
on_fault(state); // CPU exception — never returns
} else if (handlers[state.vector]) |handler| {
// Acknowledge before running the handler: a handler that switches tasks
// (the scheduler) may not return promptly, and the LAPIC mustn't wait on
// it to deliver the next interrupt. Fine for edge-triggered sources like
// the timer; a level-triggered device would need EOI after handling.
apic.eoi();
handler();
}
// else: spurious/unhandled device interrupt — don't acknowledge it
}
const std = @import("std");
+38
View File
@@ -0,0 +1,38 @@
//! x86 port I/O and model-specific registers — the low-level primitives the
//! serial port and the APIC talk to hardware through.
pub fn outb(port: u16, value: u8) void {
asm volatile ("outb %[value], %[port]"
:
: [value] "{al}" (value),
[port] "{dx}" (port),
);
}
pub fn inb(port: u16) u8 {
return asm volatile ("inb %[port], %[value]"
: [value] "={al}" (-> u8),
: [port] "{dx}" (port),
);
}
/// Read a model-specific register (returns edx:eax combined).
pub fn rdmsr(msr: u32) u64 {
var low: u32 = undefined;
var high: u32 = undefined;
asm volatile ("rdmsr"
: [low] "={eax}" (low),
[high] "={edx}" (high),
: [msr] "{ecx}" (msr),
);
return (@as(u64, high) << 32) | low;
}
pub fn wrmsr(msr: u32, value: u64) void {
asm volatile ("wrmsr"
:
: [msr] "{ecx}" (msr),
[low] "{eax}" (@as(u32, @truncate(value))),
[high] "{edx}" (@as(u32, @truncate(value >> 32))),
);
}
+180
View File
@@ -0,0 +1,180 @@
# x86_64 low-level entry code: the CPU-exception stubs, plus the GDT/IDT load
# helpers. Kept in a dedicated assembly file rather than inline asm because these
# need real labels and cross-symbol jumps/calls (isr_common, exceptionHandler),
# and because `lgdt`/`lidt` memory operands aren't expressible in Zig inline asm.
#
# Each exception vector normalises the stack to a uniform trap frame — a dummy
# error code where the CPU pushes none, then the vector number — and jumps to the
# shared tail, which saves the general registers and calls the Zig handler with a
# pointer to the frame (matching src/arch/x86_64/idt.zig's CpuState).
.text
# gdt_flush(rdi = *GDT descriptor): load the GDT, reload the data segment
# registers to the data selector, and reload CS to the code selector. CS can't be
# set with mov, so we far-return through the caller's own return address.
.global gdt_flush
gdt_flush:
lgdt (%rdi)
mov $0x10, %ax # kernel data selector
mov %ax, %ds
mov %ax, %es
mov %ax, %ss
mov %ax, %fs
mov %ax, %gs
pop %rax # caller's return address
push $0x08 # kernel code selector (new CS)
push %rax # return address (new RIP)
lretq
# idt_flush(rdi = *IDT descriptor): load the IDT.
.global idt_flush
idt_flush:
lidt (%rdi)
ret
# load_tr(di = TSS selector): load the task register.
.global load_tr
load_tr:
ltr %di
ret
# switch_context(rdi = &old_task.rsp, rsi = new_task.rsp)
# Cooperative context switch: save the callee-saved registers on the current
# stack, stash the stack pointer in the old task, load the new task's stack
# pointer, restore its callee-saved registers, and return into it. Caller-saved
# registers are the compiler's responsibility (this looks like a normal call).
.global switch_context
switch_context:
push %rbx
push %rbp
push %r12
push %r13
push %r14
push %r15
mov %rsp, (%rdi) # save old stack pointer into old_task.rsp
mov %rsi, %rsp # switch to the new task's stack
pop %r15
pop %r14
pop %r13
pop %r12
pop %rbp
pop %rbx
ret # return into the new task's saved instruction pointer
# task_trampoline: the first thing a freshly-spawned task runs. init_task_stack
# leaves its entry function in r15. New tasks start with interrupts enabled.
.global task_trampoline
task_trampoline:
sti
call *%r15 # call the task entry (fn() void)
1: hlt # if the entry returns, idle (still preemptible)
jmp 1b
# Stub for a vector the CPU does NOT push an error code for: push a dummy 0.
.macro STUB_NOERR vec
.global isr\vec
isr\vec:
pushq $0
pushq $\vec
jmp isr_common
.endm
# Stub for a vector the CPU DOES push an error code for: leave it in place.
.macro STUB_ERR vec
.global isr\vec
isr\vec:
pushq $\vec
jmp isr_common
.endm
STUB_NOERR 0
STUB_NOERR 1
STUB_NOERR 2
STUB_NOERR 3
STUB_NOERR 4
STUB_NOERR 5
STUB_NOERR 6
STUB_NOERR 7
STUB_ERR 8
STUB_NOERR 9
STUB_ERR 10
STUB_ERR 11
STUB_ERR 12
STUB_ERR 13
STUB_ERR 14
STUB_NOERR 15
STUB_NOERR 16
STUB_ERR 17
STUB_NOERR 18
STUB_NOERR 19
STUB_NOERR 20
STUB_ERR 21
STUB_NOERR 22
STUB_NOERR 23
STUB_NOERR 24
STUB_NOERR 25
STUB_NOERR 26
STUB_NOERR 27
STUB_NOERR 28
STUB_NOERR 29
STUB_NOERR 30
STUB_NOERR 31
# Device-interrupt vectors (timer, spurious, room for more). None push an error
# code, so they all use the dummy-zero form.
STUB_NOERR 32
STUB_NOERR 33
STUB_NOERR 34
STUB_NOERR 35
STUB_NOERR 36
STUB_NOERR 37
STUB_NOERR 38
STUB_NOERR 39
STUB_NOERR 40
STUB_NOERR 41
STUB_NOERR 42
STUB_NOERR 43
STUB_NOERR 44
STUB_NOERR 45
STUB_NOERR 46
STUB_NOERR 47
.extern interruptDispatch
# Shared tail. Register push order here defines the CpuState field order.
isr_common:
push %rax
push %rbx
push %rcx
push %rdx
push %rsi
push %rdi
push %rbp
push %r8
push %r9
push %r10
push %r11
push %r12
push %r13
push %r14
push %r15
mov %rsp, %rdi # first argument: pointer to the trap frame
call interruptDispatch
pop %r15
pop %r14
pop %r13
pop %r12
pop %r11
pop %r10
pop %r9
pop %r8
pop %rbp
pop %rdi
pop %rsi
pop %rdx
pop %rcx
pop %rbx
pop %rax
add $16, %rsp # drop the vector and error code
iretq
+47
View File
@@ -0,0 +1,47 @@
/* Kernel link layout.
*
* The kernel is linked at a fixed low physical address (set by `image_base` in
* build.zig). UEFI runs with memory identity-mapped, so the bootloader can load
* each PT_LOAD segment to the physical address matching its virtual address and
* jump straight to _start — no page tables to build yet. (Moving to a
* higher-half virtual base is a later step, once the bootloader sets up paging.)
*/
ENTRY(_start)
/* One loadable segment per permission set, so the loader can map .text as R+X,
* .rodata as R, and .data/.bss as R+W. FLAGS bits: 1=X, 2=W, 4=R. */
PHDRS {
text PT_LOAD FLAGS(5); /* R + X */
rodata PT_LOAD FLAGS(4); /* R */
data PT_LOAD FLAGS(6); /* R + W */
}
SECTIONS {
.text ALIGN(4K) : {
*(.text .text.*)
} :text
.rodata ALIGN(4K) : {
*(.rodata .rodata.*)
} :rodata
.data ALIGN(4K) : {
*(.data .data.*)
} :data
/* .bss occupies memory but not file space. The loader zeroes it via the
* gap between each PT_LOAD segment's file size and memory size, so no
* boundary symbols are needed here. (Zig's self-hosted linker also does not
* yet honour linker-script symbol assignments.) */
.bss ALIGN(4K) : {
*(.bss .bss.*)
*(COMMON)
} :data
/DISCARD/ : {
*(.comment)
*(.note .note.*)
*(.eh_frame .eh_frame_hdr)
}
}
+153
View File
@@ -0,0 +1,153 @@
//! The kernel's page tables and virtual memory manager.
//!
//! Builds our own 4-level page tables and switches CR3 onto them, replacing the
//! firmware's. Unlike the earlier bootstrap this maps with real permissions:
//! RAM is identity-mapped read-write + no-execute, the kernel's own segments get
//! their ELF permissions (code R+X, rodata R, data R+W+NX), and page 0 is left
//! unmapped as a null guard. It also exposes map/unmap for on-demand mapping,
//! which the kernel heap will build on.
//!
//! Everything is 4 KiB pages — precise and simple; the extra table memory is
//! negligible against available RAM.
const danos = @import("danos");
const io = @import("io.zig");
const page_size = danos.page_size;
// Page-table entry bits.
const present: u64 = 1 << 0;
const writable: u64 = 1 << 1;
const no_execute: u64 = 1 << 63;
const addr_mask: u64 = 0x000F_FFFF_FFFF_F000;
// ELF segment flags (p_flags).
const pf_x: u32 = 1;
const pf_w: u32 = 2;
// State kept after init so map()/unmap() can serve later callers (e.g. the heap).
var kernel_pml4: u64 = 0;
var alloc_frame: *const fn () ?u64 = undefined;
fn tableAt(phys: u64) *[512]u64 {
return @ptrFromInt(phys);
}
fn allocTable() u64 {
const frame = alloc_frame() orelse @panic("paging: out of memory building page tables");
@memset(tableAt(frame)[0..], 0);
return frame;
}
/// Return the table an entry points at, creating it if empty. Intermediate
/// entries are writable and executable so the leaf's bits govern (a page is
/// writable only if every level is; non-executable if any level is).
fn descend(entry: *u64) u64 {
if (entry.* & present != 0) return entry.* & addr_mask;
const frame = allocTable();
entry.* = frame | present | writable;
return frame;
}
/// Map one 4 KiB page `virt` -> `phys` with `flags` (present is added).
fn mapPage(pml4: u64, virt: u64, phys: u64, flags: u64) void {
const pml4e = &tableAt(pml4)[(virt >> 39) & 0x1FF];
const pdpt = descend(pml4e);
const pdpte = &tableAt(pdpt)[(virt >> 30) & 0x1FF];
const pd = descend(pdpte);
const pde = &tableAt(pd)[(virt >> 21) & 0x1FF];
const pt = descend(pde);
tableAt(pt)[(virt >> 12) & 0x1FF] = (phys & addr_mask) | flags | present;
}
/// Identity-map [base, base+len) with `flags`, rounded out to whole pages.
fn mapRangeIdentity(pml4: u64, base: u64, len: u64, flags: u64) void {
var addr = base & ~@as(u64, page_size - 1);
const end = base + len;
while (addr < end) : (addr += page_size) {
if (addr == 0) continue; // leave page 0 unmapped: the null guard
mapPage(pml4, addr, addr, flags);
}
}
fn regions(mm: danos.MemoryMap) []const danos.MemoryRegion {
return @as([*]const danos.MemoryRegion, @ptrFromInt(mm.regions))[0..mm.len];
}
/// Enable the NX bit in the page-table format (EFER.NXE). Must happen before we
/// load a CR3 whose entries set the NX bit, or those bits are reserved and fault.
fn enableNx() void {
const efer_msr = 0xC0000080;
io.wrmsr(efer_msr, io.rdmsr(efer_msr) | (1 << 11));
}
/// Build the address space and switch onto it.
pub fn init(allocFrame: *const fn () ?u64, boot_info: *const danos.BootInfo) void {
alloc_frame = allocFrame;
enableNx();
const pml4 = allocTable();
// 1. All RAM identity-mapped RW + NX. Non-RAM (MMIO) is skipped and stays
// unmapped unless mapped explicitly below.
for (regions(boot_info.memory_map)) |r| {
if (r.kind == .mmio) continue;
mapRangeIdentity(pml4, r.base, r.pages * page_size, present | writable | no_execute);
}
// 2. The framebuffer and the Local APIC (device memory we need), RW + NX.
const fb = boot_info.framebuffer;
mapRangeIdentity(pml4, fb.base, @as(u64, fb.height) * fb.pitch, present | writable | no_execute);
mapPage(pml4, 0xFEE00000, 0xFEE00000, present | writable | no_execute);
// 3. Overlay the kernel's own segments with their real ELF permissions,
// replacing the blanket RW+NX from step 1: code becomes R+X, rodata R,
// data R+W+NX. This is the W^X guarantee.
for (boot_info.kernel_segments[0..boot_info.kernel_segment_count]) |seg| {
var flags: u64 = present;
if (seg.flags & pf_w != 0) flags |= writable;
if (seg.flags & pf_x == 0) flags |= no_execute;
var addr = seg.virt;
const end = seg.virt + seg.pages * page_size;
while (addr < end) : (addr += page_size) mapPage(pml4, addr, addr, flags);
}
kernel_pml4 = pml4;
asm volatile ("mov %[pml4], %%cr3"
:
: [pml4] "r" (pml4),
: .{ .memory = true }
);
}
/// Map a page into the kernel address space on demand (for the heap, etc.).
/// `writable_page` controls W; pages are always mapped non-executable.
pub fn map(virt: u64, phys: u64, writable_page: bool) void {
var flags: u64 = present | no_execute;
if (writable_page) flags |= writable;
mapPage(kernel_pml4, virt, phys, flags);
invalidate(virt);
}
/// Remove a mapping and flush it from the TLB.
pub fn unmap(virt: u64) void {
const pml4e = tableAt(kernel_pml4)[(virt >> 39) & 0x1FF];
if (pml4e & present == 0) return;
const pdpte = tableAt(pml4e & addr_mask)[(virt >> 30) & 0x1FF];
if (pdpte & present == 0) return;
const pde = tableAt(pdpte & addr_mask)[(virt >> 21) & 0x1FF];
if (pde & present == 0) return;
tableAt(pde & addr_mask)[(virt >> 12) & 0x1FF] = 0;
invalidate(virt);
}
fn invalidate(virt: u64) void {
// invlpg needs its operand via a register-indirect memory reference that Zig
// inline asm won't form directly, so stage the address in a register first.
asm volatile (
\\mov %[v], %%rax
\\invlpg (%%rax)
:
: [v] "r" (virt),
: .{ .rax = true, .memory = true }
);
}
+46
View File
@@ -0,0 +1,46 @@
//! COM1 serial port (16550 UART) — the kernel's machine-readable output channel.
//! Unlike the framebuffer console, serial text can be captured to a file by QEMU
//! (`-serial file:...`), which is what the test harness asserts on. Each
//! architecture has its own UART; this is the x86 one, driven by port I/O.
const port = 0x3F8; // COM1 base
fn outb(p: u16, value: u8) void {
asm volatile ("outb %[value], %[p]"
:
: [value] "{al}" (value),
[p] "{dx}" (p),
);
}
fn inb(p: u16) u8 {
return asm volatile ("inb %[p], %[value]"
: [value] "={al}" (-> u8),
: [p] "{dx}" (p),
);
}
/// Configure the UART: 38400 baud, 8N1, FIFO on. Safe to call before anything
/// else; it has no dependencies.
pub fn init() void {
outb(port + 1, 0x00); // disable interrupts
outb(port + 3, 0x80); // enable DLAB (set baud divisor)
outb(port + 0, 0x03); // divisor low: 38400 baud
outb(port + 1, 0x00); // divisor high
outb(port + 3, 0x03); // 8 bits, no parity, one stop bit; DLAB off
outb(port + 2, 0xC7); // enable + clear FIFO, 14-byte threshold
outb(port + 4, 0x0B); // RTS/DSR set
}
fn writeByte(c: u8) void {
while (inb(port + 5) & 0x20 == 0) {} // wait until the transmit holding register is empty
outb(port, c);
}
/// Write bytes, translating LF to CRLF so terminals and logs line up.
pub fn write(bytes: []const u8) void {
for (bytes) |c| {
if (c == '\n') writeByte('\r');
writeByte(c);
}
}
+48
View File
@@ -0,0 +1,48 @@
//! Task State Segment and its interrupt stack. In long mode the TSS's main job
//! is the Interrupt Stack Table: an IDT gate can name an IST entry, and the CPU
//! switches to that stack when the exception fires — no matter how broken the
//! interrupted stack was. We use IST1 for the double-fault handler, so a fault
//! that happens *because* the current stack is unusable still lands on solid
//! ground instead of triple-faulting.
const gdt = @import("gdt.zig");
/// x86_64 TSS. `packed` because several 64-bit fields sit at 4-byte-unaligned
/// offsets (rsp0 at byte 4), which a normal struct would pad away.
const Tss = packed struct {
reserved0: u32 = 0,
rsp0: u64 = 0,
rsp1: u64 = 0,
rsp2: u64 = 0,
reserved1: u64 = 0,
ist1: u64 = 0,
ist2: u64 = 0,
ist3: u64 = 0,
ist4: u64 = 0,
ist5: u64 = 0,
ist6: u64 = 0,
ist7: u64 = 0,
reserved2: u64 = 0,
reserved3: u16 = 0,
iomap_base: u16 = 0,
};
/// The IST slot (1-based, as the IDT gate encodes it) used for critical faults.
pub const double_fault_ist = 1;
var tss: Tss align(16) = .{};
/// Dedicated stack for IST1. Static so it needs no allocator and is always valid.
var ist1_stack: [16 * 1024]u8 align(16) = undefined;
/// Loads the task register with the TSS selector. Defined in isr.s.
extern fn load_tr(selector: u16) callconv(.c) void;
/// Point IST1 at its stack, publish the TSS through the GDT, and load it into the
/// task register. Requires the GDT to already be loaded (gdt.init first).
pub fn init() void {
tss.ist1 = @intFromPtr(&ist1_stack) + ist1_stack.len; // stacks grow down
tss.iomap_base = @sizeOf(Tss); // == limit: no I/O permission bitmap
gdt.setTss(@intFromPtr(&tss), @sizeOf(Tss) - 1);
load_tr(gdt.tss_selector);
}