Hardened paging into a real VMM
This commit is contained in:
+16
-4
@@ -4,6 +4,7 @@
|
||||
//! build.zig — no change to the generic code. Keep everything CPU-specific here
|
||||
//! (halt, the descriptor tables, later paging), and nothing generic.
|
||||
|
||||
const danos = @import("danos");
|
||||
const gdt = @import("gdt.zig");
|
||||
const tss = @import("tss.zig");
|
||||
const idt = @import("idt.zig");
|
||||
@@ -35,10 +36,21 @@ pub fn init() void {
|
||||
idt.init();
|
||||
}
|
||||
|
||||
/// Build the kernel's own page tables and switch onto them. Needs a physical
|
||||
/// frame allocator; call once the frame allocator is up.
|
||||
pub fn enablePaging(allocFrame: *const fn () ?u64) void {
|
||||
paging.init(allocFrame);
|
||||
/// Build the kernel's own page tables (with real permissions) and switch onto
|
||||
/// them. Needs the frame allocator and the boot info (for the memory map and the
|
||||
/// kernel's segment layout). Call once the frame allocator is up.
|
||||
pub fn enablePaging(allocFrame: *const fn () ?u64, boot_info: *const danos.BootInfo) void {
|
||||
paging.init(allocFrame, boot_info);
|
||||
}
|
||||
|
||||
/// Map a page into the kernel address space (non-executable). For the heap, etc.
|
||||
pub fn mapPage(virt: u64, phys: u64, writable: bool) void {
|
||||
paging.map(virt, phys, writable);
|
||||
}
|
||||
|
||||
/// Remove a kernel mapping.
|
||||
pub fn unmapPage(virt: u64) void {
|
||||
paging.unmap(virt);
|
||||
}
|
||||
|
||||
/// CR3 holds the physical address of the active top-level page table.
|
||||
|
||||
+124
-41
@@ -1,70 +1,153 @@
|
||||
//! The kernel's own 4-level page tables. Until now we've been running on the
|
||||
//! firmware's page tables, which live in memory we'd like to reclaim and which we
|
||||
//! don't control. This builds our own set, identity-mapping the low 4 GiB, and
|
||||
//! loads CR3 to switch onto them.
|
||||
//! The kernel's page tables and virtual memory manager.
|
||||
//!
|
||||
//! "Identity map" means virtual address == physical address, which keeps
|
||||
//! everything already running — kernel image, stack, framebuffer, the frame
|
||||
//! allocator's bitmap, MMIO — valid across the switch without having to relocate
|
||||
//! anything. 4 GiB comfortably covers all of that (RAM low down, the framebuffer
|
||||
//! at 2 GiB, device MMIO below 4 GiB). Higher-half mapping and per-region
|
||||
//! permissions come later; this is the bootstrap.
|
||||
//! Builds our own 4-level page tables and switches CR3 onto them, replacing the
|
||||
//! firmware's. Unlike the earlier bootstrap this maps with real permissions:
|
||||
//! RAM is identity-mapped read-write + no-execute, the kernel's own segments get
|
||||
//! their ELF permissions (code R+X, rodata R, data R+W+NX), and page 0 is left
|
||||
//! unmapped as a null guard. It also exposes map/unmap for on-demand mapping,
|
||||
//! which the kernel heap will build on.
|
||||
//!
|
||||
//! We use 2 MiB pages, so the whole map is cheap: a PML4, a PDPT, and four page
|
||||
//! directories.
|
||||
//! Everything is 4 KiB pages — precise and simple; the extra table memory is
|
||||
//! negligible against available RAM.
|
||||
|
||||
const KiB = 1024;
|
||||
const MiB = 1024 * KiB;
|
||||
const GiB = 1024 * MiB;
|
||||
const danos = @import("danos");
|
||||
const io = @import("io.zig");
|
||||
|
||||
const page_size = danos.page_size;
|
||||
|
||||
// Page-table entry bits.
|
||||
const present: u64 = 1 << 0;
|
||||
const writable: u64 = 1 << 1;
|
||||
const huge: u64 = 1 << 7; // in a PD entry: this maps a 2 MiB page directly
|
||||
const addr_mask: u64 = 0x000F_FFFF_FFFF_F000; // physical address bits of an entry
|
||||
const no_execute: u64 = 1 << 63;
|
||||
const addr_mask: u64 = 0x000F_FFFF_FFFF_F000;
|
||||
|
||||
// ELF segment flags (p_flags).
|
||||
const pf_x: u32 = 1;
|
||||
const pf_w: u32 = 2;
|
||||
|
||||
// State kept after init so map()/unmap() can serve later callers (e.g. the heap).
|
||||
var kernel_pml4: u64 = 0;
|
||||
var alloc_frame: *const fn () ?u64 = undefined;
|
||||
|
||||
/// A page table is 512 64-bit entries. While building the tables the firmware's
|
||||
/// identity map is still active, so a physical frame address is usable directly.
|
||||
fn tableAt(phys: u64) *[512]u64 {
|
||||
return @ptrFromInt(phys);
|
||||
}
|
||||
|
||||
/// Allocate and zero a fresh page-table frame. Zeroing matters: the frame comes
|
||||
/// from previously-used memory, and any stale non-zero entry would map a bogus
|
||||
/// region.
|
||||
fn allocTable(allocFrame: *const fn () ?u64) u64 {
|
||||
const frame = allocFrame() orelse @panic("paging: out of memory building page tables");
|
||||
fn allocTable() u64 {
|
||||
const frame = alloc_frame() orelse @panic("paging: out of memory building page tables");
|
||||
@memset(tableAt(frame)[0..], 0);
|
||||
return frame;
|
||||
}
|
||||
|
||||
/// Return the table an entry points at, creating it if the entry is empty.
|
||||
fn descend(entry: *u64, allocFrame: *const fn () ?u64) u64 {
|
||||
/// Return the table an entry points at, creating it if empty. Intermediate
|
||||
/// entries are writable and executable so the leaf's bits govern (a page is
|
||||
/// writable only if every level is; non-executable if any level is).
|
||||
fn descend(entry: *u64) u64 {
|
||||
if (entry.* & present != 0) return entry.* & addr_mask;
|
||||
const frame = allocTable(allocFrame);
|
||||
const frame = allocTable();
|
||||
entry.* = frame | present | writable;
|
||||
return frame;
|
||||
}
|
||||
|
||||
/// Identity-map one 2 MiB page: walk PML4 -> PDPT -> PD and write the leaf.
|
||||
fn mapHugePage(pml4: u64, addr: u64, allocFrame: *const fn () ?u64) void {
|
||||
const pml4e = &tableAt(pml4)[(addr >> 39) & 0x1FF];
|
||||
const pdpt = descend(pml4e, allocFrame);
|
||||
const pdpte = &tableAt(pdpt)[(addr >> 30) & 0x1FF];
|
||||
const pd = descend(pdpte, allocFrame);
|
||||
tableAt(pd)[(addr >> 21) & 0x1FF] = addr | present | writable | huge;
|
||||
/// Map one 4 KiB page `virt` -> `phys` with `flags` (present is added).
|
||||
fn mapPage(pml4: u64, virt: u64, phys: u64, flags: u64) void {
|
||||
const pml4e = &tableAt(pml4)[(virt >> 39) & 0x1FF];
|
||||
const pdpt = descend(pml4e);
|
||||
const pdpte = &tableAt(pdpt)[(virt >> 30) & 0x1FF];
|
||||
const pd = descend(pdpte);
|
||||
const pde = &tableAt(pd)[(virt >> 21) & 0x1FF];
|
||||
const pt = descend(pde);
|
||||
tableAt(pt)[(virt >> 12) & 0x1FF] = (phys & addr_mask) | flags | present;
|
||||
}
|
||||
|
||||
/// Build the tables, identity-map the low 4 GiB, and switch CR3 onto them.
|
||||
pub fn init(allocFrame: *const fn () ?u64) void {
|
||||
const pml4 = allocTable(allocFrame);
|
||||
var addr: u64 = 0;
|
||||
while (addr < 4 * GiB) : (addr += 2 * MiB) {
|
||||
mapHugePage(pml4, addr, allocFrame);
|
||||
/// Identity-map [base, base+len) with `flags`, rounded out to whole pages.
|
||||
fn mapRangeIdentity(pml4: u64, base: u64, len: u64, flags: u64) void {
|
||||
var addr = base & ~@as(u64, page_size - 1);
|
||||
const end = base + len;
|
||||
while (addr < end) : (addr += page_size) {
|
||||
if (addr == 0) continue; // leave page 0 unmapped: the null guard
|
||||
mapPage(pml4, addr, addr, flags);
|
||||
}
|
||||
// Loading CR3 switches address spaces and flushes the TLB in one step.
|
||||
}
|
||||
|
||||
fn regions(mm: danos.MemoryMap) []const danos.MemoryRegion {
|
||||
return @as([*]const danos.MemoryRegion, @ptrFromInt(mm.regions))[0..mm.len];
|
||||
}
|
||||
|
||||
/// Enable the NX bit in the page-table format (EFER.NXE). Must happen before we
|
||||
/// load a CR3 whose entries set the NX bit, or those bits are reserved and fault.
|
||||
fn enableNx() void {
|
||||
const efer_msr = 0xC0000080;
|
||||
io.wrmsr(efer_msr, io.rdmsr(efer_msr) | (1 << 11));
|
||||
}
|
||||
|
||||
/// Build the address space and switch onto it.
|
||||
pub fn init(allocFrame: *const fn () ?u64, boot_info: *const danos.BootInfo) void {
|
||||
alloc_frame = allocFrame;
|
||||
enableNx();
|
||||
const pml4 = allocTable();
|
||||
|
||||
// 1. All RAM identity-mapped RW + NX. Non-RAM (MMIO) is skipped and stays
|
||||
// unmapped unless mapped explicitly below.
|
||||
for (regions(boot_info.memory_map)) |r| {
|
||||
if (r.kind == .mmio) continue;
|
||||
mapRangeIdentity(pml4, r.base, r.pages * page_size, present | writable | no_execute);
|
||||
}
|
||||
|
||||
// 2. The framebuffer and the Local APIC (device memory we need), RW + NX.
|
||||
const fb = boot_info.framebuffer;
|
||||
mapRangeIdentity(pml4, fb.base, @as(u64, fb.height) * fb.pitch, present | writable | no_execute);
|
||||
mapPage(pml4, 0xFEE00000, 0xFEE00000, present | writable | no_execute);
|
||||
|
||||
// 3. Overlay the kernel's own segments with their real ELF permissions,
|
||||
// replacing the blanket RW+NX from step 1: code becomes R+X, rodata R,
|
||||
// data R+W+NX. This is the W^X guarantee.
|
||||
for (boot_info.kernel_segments[0..boot_info.kernel_segment_count]) |seg| {
|
||||
var flags: u64 = present;
|
||||
if (seg.flags & pf_w != 0) flags |= writable;
|
||||
if (seg.flags & pf_x == 0) flags |= no_execute;
|
||||
var addr = seg.virt;
|
||||
const end = seg.virt + seg.pages * page_size;
|
||||
while (addr < end) : (addr += page_size) mapPage(pml4, addr, addr, flags);
|
||||
}
|
||||
|
||||
kernel_pml4 = pml4;
|
||||
asm volatile ("mov %[pml4], %%cr3"
|
||||
:
|
||||
: [pml4] "r" (pml4),
|
||||
: .{ .memory = true }
|
||||
);
|
||||
}
|
||||
|
||||
/// Map a page into the kernel address space on demand (for the heap, etc.).
|
||||
/// `writable_page` controls W; pages are always mapped non-executable.
|
||||
pub fn map(virt: u64, phys: u64, writable_page: bool) void {
|
||||
var flags: u64 = present | no_execute;
|
||||
if (writable_page) flags |= writable;
|
||||
mapPage(kernel_pml4, virt, phys, flags);
|
||||
invalidate(virt);
|
||||
}
|
||||
|
||||
/// Remove a mapping and flush it from the TLB.
|
||||
pub fn unmap(virt: u64) void {
|
||||
const pml4e = tableAt(kernel_pml4)[(virt >> 39) & 0x1FF];
|
||||
if (pml4e & present == 0) return;
|
||||
const pdpte = tableAt(pml4e & addr_mask)[(virt >> 30) & 0x1FF];
|
||||
if (pdpte & present == 0) return;
|
||||
const pde = tableAt(pdpte & addr_mask)[(virt >> 21) & 0x1FF];
|
||||
if (pde & present == 0) return;
|
||||
tableAt(pde & addr_mask)[(virt >> 12) & 0x1FF] = 0;
|
||||
invalidate(virt);
|
||||
}
|
||||
|
||||
fn invalidate(virt: u64) void {
|
||||
// invlpg needs its operand via a register-indirect memory reference that Zig
|
||||
// inline asm won't form directly, so stage the address in a register first.
|
||||
asm volatile (
|
||||
\\mov %[v], %%rax
|
||||
\\invlpg (%%rax)
|
||||
:
|
||||
: [v] "r" (virt),
|
||||
: .{ .rax = true, .memory = true }
|
||||
);
|
||||
}
|
||||
|
||||
+20
-5
@@ -36,9 +36,11 @@ fn boot() !noreturn {
|
||||
var boot_info: BootInfo = .{
|
||||
.framebuffer = try queryFramebuffer(bs),
|
||||
.memory_map = undefined, // filled by exitBootServices, just below
|
||||
.kernel_segments = undefined, // filled by loadKernel
|
||||
.kernel_segment_count = 0,
|
||||
};
|
||||
|
||||
const entry = try loadKernel(bs);
|
||||
const entry = try loadKernel(bs, &boot_info);
|
||||
|
||||
log("danos: kernel loaded, exiting boot services\r\n");
|
||||
boot_info.memory_map = try exitBootServices(bs);
|
||||
@@ -152,7 +154,7 @@ fn edidNative(edid: []const u8) ?Resolution {
|
||||
|
||||
/// Open the kernel on the volume we booted from, read it into a pool buffer,
|
||||
/// load its segments, and return the physical entry-point address.
|
||||
fn loadKernel(bs: *uefi.tables.BootServices) !usize {
|
||||
fn loadKernel(bs: *uefi.tables.BootServices, boot_info: *BootInfo) !usize {
|
||||
const loaded = (try bs.handleProtocol(uefi.protocol.LoadedImage, uefi.handle)) orelse
|
||||
return error.NoLoadedImage;
|
||||
const device = loaded.device_handle orelse return error.NoBootDevice;
|
||||
@@ -181,11 +183,13 @@ fn loadKernel(bs: *uefi.tables.BootServices) !usize {
|
||||
read_total += n;
|
||||
}
|
||||
|
||||
return loadElf(bs, image);
|
||||
return loadElf(bs, image, boot_info);
|
||||
}
|
||||
|
||||
/// Validate the ELF and copy every PT_LOAD segment to its physical address.
|
||||
fn loadElf(bs: *uefi.tables.BootServices, image: []u8) !usize {
|
||||
/// Validate the ELF, copy every PT_LOAD segment to its physical address, and
|
||||
/// record each segment's layout so the kernel can re-map itself with the right
|
||||
/// permissions.
|
||||
fn loadElf(bs: *uefi.tables.BootServices, image: []u8, boot_info: *BootInfo) !usize {
|
||||
if (image.len < @sizeOf(elf.Elf64_Ehdr)) return error.NotElf;
|
||||
const ehdr: *const elf.Elf64_Ehdr = @ptrCast(@alignCast(image.ptr));
|
||||
|
||||
@@ -214,6 +218,17 @@ fn loadElf(bs: *uefi.tables.BootServices, image: []u8) !usize {
|
||||
const off: usize = @intCast(phdr.p_offset);
|
||||
@memcpy(bytes[0..file_sz], image[off..][0..file_sz]);
|
||||
@memset(bytes[file_sz..mem_sz], 0);
|
||||
|
||||
// Record it (identity-loaded: virtual == physical) for the kernel's VMM.
|
||||
const n = boot_info.kernel_segment_count;
|
||||
if (n < boot_info.kernel_segments.len) {
|
||||
boot_info.kernel_segments[n] = .{
|
||||
.virt = phdr.p_vaddr,
|
||||
.pages = pages,
|
||||
.flags = phdr.p_flags,
|
||||
};
|
||||
boot_info.kernel_segment_count = n + 1;
|
||||
}
|
||||
}
|
||||
|
||||
return @intCast(ehdr.e_entry);
|
||||
|
||||
+3
-2
@@ -86,10 +86,11 @@ fn kmain(boot_info: *const BootInfo) noreturn {
|
||||
if (f2) |p| pmm.free(p);
|
||||
con.print(" after free : {d} frames free\n", .{pmm.stats().free_frames});
|
||||
|
||||
// Switch off the firmware's page tables onto our own.
|
||||
arch.enablePaging(pmm.alloc);
|
||||
// Switch off the firmware's page tables onto our own (with real permissions).
|
||||
arch.enablePaging(pmm.alloc, boot_info);
|
||||
con.print("\ndanos: paging enabled\n", .{});
|
||||
con.print(" page tables: CR3 = 0x{x:0>16}\n", .{arch.readCr3()});
|
||||
con.print(" kernel segs: {d} (mapped with W^X permissions)\n", .{boot_info.kernel_segment_count});
|
||||
|
||||
// Start the timer and unmask interrupts — the kernel now has a heartbeat.
|
||||
arch.startTimer();
|
||||
|
||||
@@ -76,9 +76,22 @@ pub const MemoryMap = extern struct {
|
||||
len: usize,
|
||||
};
|
||||
|
||||
/// One PT_LOAD segment of the kernel image, so the kernel can re-map itself with
|
||||
/// correct permissions (code R+X, rodata R, data R+W+NX). `flags` are raw ELF
|
||||
/// segment flags: PF_X=1, PF_W=2, PF_R=4.
|
||||
pub const KernelSegment = extern struct {
|
||||
virt: u64,
|
||||
pages: u64,
|
||||
flags: u32,
|
||||
_pad: u32 = 0,
|
||||
};
|
||||
|
||||
/// Handoff structure the bootloader fills in and passes to the kernel's
|
||||
/// `_start` in RDI (the first argument under the SysV AMD64 C ABI).
|
||||
pub const BootInfo = extern struct {
|
||||
framebuffer: Framebuffer,
|
||||
memory_map: MemoryMap,
|
||||
/// The kernel's own PT_LOAD segments (it has three: text, rodata, data).
|
||||
kernel_segments: [8]KernelSegment,
|
||||
kernel_segment_count: u32,
|
||||
};
|
||||
|
||||
+60
-13
@@ -33,17 +33,33 @@ fn check(name: []const u8, ok: bool) void {
|
||||
}
|
||||
}
|
||||
|
||||
/// Emit the overall result line the harness matches, then the done sentinel.
|
||||
fn result() void {
|
||||
log("DANOS-TEST-RESULT: {s} ({d} passed, {d} failed)\n", .{
|
||||
if (failed == 0) "PASS" else "FAIL",
|
||||
passed,
|
||||
failed,
|
||||
});
|
||||
log("DANOS-TEST-DONE\n", .{});
|
||||
}
|
||||
|
||||
pub fn run(case: []const u8, boot_info: *const BootInfo) void {
|
||||
if (eql(case, "smoke")) {
|
||||
smoke(boot_info);
|
||||
} else if (eql(case, "timer")) {
|
||||
timer();
|
||||
} else if (eql(case, "vmm")) {
|
||||
vmm();
|
||||
} else if (eql(case, "fault-ud")) {
|
||||
faultInvalidOpcode();
|
||||
} else if (eql(case, "fault-pf")) {
|
||||
faultPageFault();
|
||||
} else if (eql(case, "fault-df")) {
|
||||
faultDoubleFault();
|
||||
} else if (eql(case, "fault-nx")) {
|
||||
faultNoExecute();
|
||||
} else if (eql(case, "fault-null")) {
|
||||
faultNull();
|
||||
} else {
|
||||
log("DANOS-TEST-RESULT: FAIL (unknown case '{s}')\n", .{case});
|
||||
}
|
||||
@@ -85,12 +101,7 @@ fn smoke(boot_info: *const BootInfo) void {
|
||||
const cr3 = arch.readCr3();
|
||||
check("paging active (CR3 set)", cr3 != 0 and cr3 % danos.page_size == 0);
|
||||
|
||||
log("DANOS-TEST-RESULT: {s} ({d} passed, {d} failed)\n", .{
|
||||
if (failed == 0) "PASS" else "FAIL",
|
||||
passed,
|
||||
failed,
|
||||
});
|
||||
log("DANOS-TEST-DONE\n", .{});
|
||||
result();
|
||||
}
|
||||
|
||||
/// Verify device interrupts fire and return: the timer tick counter must advance
|
||||
@@ -105,12 +116,26 @@ fn timer() void {
|
||||
while (arch.ticks() == start and spins < 5_000_000_000) spins +%= 1;
|
||||
check("timer interrupts advance the tick count", arch.ticks() > start);
|
||||
|
||||
log("DANOS-TEST-RESULT: {s} ({d} passed, {d} failed)\n", .{
|
||||
if (failed == 0) "PASS" else "FAIL",
|
||||
passed,
|
||||
failed,
|
||||
});
|
||||
log("DANOS-TEST-DONE\n", .{});
|
||||
result();
|
||||
}
|
||||
|
||||
/// Verify the on-demand VMM: map a fresh frame at an unused virtual address, and
|
||||
/// check it's writable and reads back.
|
||||
fn vmm() void {
|
||||
log("DANOS-TEST-BEGIN: vmm\n", .{});
|
||||
const frame = pmm.alloc();
|
||||
check("frame available to map", frame != null);
|
||||
if (frame) |phys| {
|
||||
var virt: u64 = 0x0000_4000_0000_0000; // canonical, well clear of everything mapped
|
||||
arch.mapPage(virt, phys, true);
|
||||
const p: *volatile u64 = @ptrFromInt(virt);
|
||||
p.* = 0xdead_c0de_cafe_babe;
|
||||
check("mapped page is writable and reads back", p.* == 0xdead_c0de_cafe_babe);
|
||||
arch.unmapPage(virt);
|
||||
pmm.free(phys);
|
||||
virt += 0;
|
||||
}
|
||||
result();
|
||||
}
|
||||
|
||||
fn faultInvalidOpcode() void {
|
||||
@@ -118,11 +143,33 @@ fn faultInvalidOpcode() void {
|
||||
asm volatile ("ud2");
|
||||
}
|
||||
|
||||
/// Verify NX: fetching an instruction from a data page (mapped no-execute) faults.
|
||||
fn faultNoExecute() void {
|
||||
log("DANOS-TEST-BEGIN: fault-nx\n", .{});
|
||||
var scratch: u64 = 0xC3; // a lone `ret` — harmless if NX somehow let it run
|
||||
const f: *const fn () void = @ptrFromInt(@intFromPtr(&scratch));
|
||||
f(); // instruction fetch from an NX page -> #PF before it executes
|
||||
log("DANOS-TEST-RESULT: FAIL (NX not enforced)\n", .{});
|
||||
}
|
||||
|
||||
/// Verify the null guard: dereferencing address 0 (page 0 left unmapped) faults.
|
||||
fn faultNull() void {
|
||||
log("DANOS-TEST-BEGIN: fault-null\n", .{});
|
||||
// Launder the address through empty asm so the compiler no longer knows it's
|
||||
// 0 (otherwise it folds a null-pointer safety panic instead of doing the real
|
||||
// access). `allowzero` skips the same null check on the cast. The write then
|
||||
// hits the unmapped page 0 and takes a real hardware #PF.
|
||||
var addr: u64 = 0;
|
||||
addr = asm ("" : [ret] "=r" (-> u64) : [in] "0" (addr));
|
||||
const p: *allowzero volatile u64 = @ptrFromInt(addr);
|
||||
p.* = 1;
|
||||
}
|
||||
|
||||
fn faultPageFault() void {
|
||||
log("DANOS-TEST-BEGIN: fault-pf\n", .{});
|
||||
// Runtime address so the backend emits a register store (not a `mov moffs`,
|
||||
// which the self-hosted x86_64 backend can't encode).
|
||||
var addr: u64 = 0xdeadbeef000; // above our identity-mapped 4 GiB
|
||||
var addr: u64 = 0xdeadbeef000; // well above all mapped RAM
|
||||
const p: *volatile u64 = @ptrFromInt(addr);
|
||||
p.* = 1;
|
||||
addr += 0;
|
||||
|
||||
Reference in New Issue
Block a user