M2 step 3: route every physical dereference through the physmap
paging.init now builds the physmap (physToVirt(phys)) alongside the low identity map, so both addressing modes resolve during the transition. tableAt (the page-table walk hinge), the pmm bitmap, the framebuffer, LAPIC/IOAPIC/HPET/PM-timer/SPCR MMIO, the ACPI table walk, AML OperationRegions, the ACPI power registers, the user-ELF frame fills, and the AP trampoline arm/disarm all reach physical memory through the physmap. Hal.mapMmio now maps into the physmap and returns the virtual address, so the device layer never learns the layout. boot_info and its pointees are converted at kmain entry. The kernel still links and runs low; identity is the safety net until it's removed. Suite 27/27. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Fable 5
parent
75bc429aa1
commit
724de7bbd0
+13
-12
@@ -147,7 +147,7 @@ var aml_block_count: usize = 0;
|
|||||||
|
|
||||||
fn addAmlBlock(sdt_phys: u64) void {
|
fn addAmlBlock(sdt_phys: u64) void {
|
||||||
if (aml_block_count >= aml_block_phys.len or sdt_phys == 0) return;
|
if (aml_block_count >= aml_block_phys.len or sdt_phys == 0) return;
|
||||||
const h: *const SystemDescriptorTableHeader = @ptrFromInt(sdt_phys);
|
const h: *const SystemDescriptorTableHeader = @ptrFromInt(danos.physToVirt(sdt_phys));
|
||||||
if (h.length <= @sizeOf(SystemDescriptorTableHeader)) return;
|
if (h.length <= @sizeOf(SystemDescriptorTableHeader)) return;
|
||||||
aml_block_phys[aml_block_count] = sdt_phys + @sizeOf(SystemDescriptorTableHeader);
|
aml_block_phys[aml_block_count] = sdt_phys + @sizeOf(SystemDescriptorTableHeader);
|
||||||
aml_block_len[aml_block_count] = h.length - @sizeOf(SystemDescriptorTableHeader);
|
aml_block_len[aml_block_count] = h.length - @sizeOf(SystemDescriptorTableHeader);
|
||||||
@@ -376,14 +376,14 @@ pub fn discover(rsdp_phys: u64, dt: *DeviceTree, hal: Hal) !void {
|
|||||||
dsdt_phys = 0;
|
dsdt_phys = 0;
|
||||||
aml_block_count = 0;
|
aml_block_count = 0;
|
||||||
|
|
||||||
const rsdp: *const RootSystemDescriptionPointer = @ptrFromInt(rsdp_phys);
|
const rsdp: *const RootSystemDescriptionPointer = @ptrFromInt(danos.physToVirt(rsdp_phys));
|
||||||
if (!std.mem.eql(u8, &rsdp.signature, "RSD PTR ")) return error.BadRsdpSignature;
|
if (!std.mem.eql(u8, &rsdp.signature, "RSD PTR ")) return error.BadRsdpSignature;
|
||||||
// Revision 0 checksums only the first 20 bytes (the v1.0 RSDP).
|
// Revision 0 checksums only the first 20 bytes (the v1.0 RSDP).
|
||||||
if (!checksumOk(@ptrFromInt(rsdp_phys), 20)) return error.BadRsdpChecksum;
|
if (!checksumOk(@ptrFromInt(danos.physToVirt(rsdp_phys)), 20)) return error.BadRsdpChecksum;
|
||||||
|
|
||||||
if (rsdp.revision >= 2) {
|
if (rsdp.revision >= 2) {
|
||||||
const xsdp: *const ExtendedSystemDescriptorPointer = @ptrFromInt(rsdp_phys);
|
const xsdp: *const ExtendedSystemDescriptorPointer = @ptrFromInt(danos.physToVirt(rsdp_phys));
|
||||||
if (!checksumOk(@ptrFromInt(rsdp_phys), xsdp.length)) return error.BadXsdpChecksum;
|
if (!checksumOk(@ptrFromInt(danos.physToVirt(rsdp_phys)), xsdp.length)) return error.BadXsdpChecksum;
|
||||||
try walkRoot(u64, xsdp.extended_system_descriptor_table_address, dt, hal);
|
try walkRoot(u64, xsdp.extended_system_descriptor_table_address, dt, hal);
|
||||||
} else {
|
} else {
|
||||||
try walkRoot(u32, rsdp.root_system_description_table_address, dt, hal);
|
try walkRoot(u32, rsdp.root_system_description_table_address, dt, hal);
|
||||||
@@ -393,7 +393,7 @@ pub fn discover(rsdp_phys: u64, dt: *DeviceTree, hal: Hal) !void {
|
|||||||
// read the sleep types from it.
|
// read the sleep types from it.
|
||||||
var blocks: [aml_block_phys.len][]const u8 = undefined;
|
var blocks: [aml_block_phys.len][]const u8 = undefined;
|
||||||
for (0..aml_block_count) |i| {
|
for (0..aml_block_count) |i| {
|
||||||
blocks[i] = @as([*]const u8, @ptrFromInt(aml_block_phys[i]))[0..aml_block_len[i]];
|
blocks[i] = @as([*]const u8, @ptrFromInt(danos.physToVirt(aml_block_phys[i])))[0..aml_block_len[i]];
|
||||||
}
|
}
|
||||||
const active = blocks[0..aml_block_count];
|
const active = blocks[0..aml_block_count];
|
||||||
if (aml.parse(dt.allocator, active)) |pr| {
|
if (aml.parse(dt.allocator, active)) |pr| {
|
||||||
@@ -412,11 +412,11 @@ pub fn discover(rsdp_phys: u64, dt: *DeviceTree, hal: Hal) !void {
|
|||||||
/// Walk the RSDT (Entry = u32) or XSDT (Entry = u64): validate it, then dispatch
|
/// Walk the RSDT (Entry = u32) or XSDT (Entry = u64): validate it, then dispatch
|
||||||
/// each SDT it points at. A bad individual table is skipped, not fatal.
|
/// each SDT it points at. A bad individual table is skipped, not fatal.
|
||||||
fn walkRoot(comptime Entry: type, root_phys: u64, dt: *DeviceTree, hal: Hal) !void {
|
fn walkRoot(comptime Entry: type, root_phys: u64, dt: *DeviceTree, hal: Hal) !void {
|
||||||
const header: *const SystemDescriptorTableHeader = @ptrFromInt(root_phys);
|
const header: *const SystemDescriptorTableHeader = @ptrFromInt(danos.physToVirt(root_phys));
|
||||||
if (!checksumOk(@ptrFromInt(root_phys), header.length)) return error.BadRootChecksum;
|
if (!checksumOk(@ptrFromInt(danos.physToVirt(root_phys)), header.length)) return error.BadRootChecksum;
|
||||||
|
|
||||||
const count = (header.length - @sizeOf(SystemDescriptorTableHeader)) / @sizeOf(Entry);
|
const count = (header.length - @sizeOf(SystemDescriptorTableHeader)) / @sizeOf(Entry);
|
||||||
const base: [*]const u8 = @ptrFromInt(root_phys);
|
const base: [*]const u8 = @ptrFromInt(danos.physToVirt(root_phys));
|
||||||
const entries: [*]align(1) const Entry = @ptrCast(base + @sizeOf(SystemDescriptorTableHeader));
|
const entries: [*]align(1) const Entry = @ptrCast(base + @sizeOf(SystemDescriptorTableHeader));
|
||||||
|
|
||||||
for (entries[0..count]) |ent| {
|
for (entries[0..count]) |ent| {
|
||||||
@@ -427,7 +427,7 @@ fn walkRoot(comptime Entry: type, root_phys: u64, dt: *DeviceTree, hal: Hal) !vo
|
|||||||
|
|
||||||
/// Dispatch a single SDT on its signature.
|
/// Dispatch a single SDT on its signature.
|
||||||
fn handleTable(dt: *DeviceTree, hal: Hal, sdt_phys: u64) !void {
|
fn handleTable(dt: *DeviceTree, hal: Hal, sdt_phys: u64) !void {
|
||||||
const header: *const SystemDescriptorTableHeader = @ptrFromInt(sdt_phys);
|
const header: *const SystemDescriptorTableHeader = @ptrFromInt(danos.physToVirt(sdt_phys));
|
||||||
const sig = header.signature;
|
const sig = header.signature;
|
||||||
if (std.mem.eql(u8, &sig, &APIC)) {
|
if (std.mem.eql(u8, &sig, &APIC)) {
|
||||||
try parseMadt(dt, header);
|
try parseMadt(dt, header);
|
||||||
@@ -1080,8 +1080,9 @@ fn pciConfigPtr(alloc: McfgAllocation, hal: Hal, bus: u8, dev: u8, func: u8) [*]
|
|||||||
(@as(u64, bus - alloc.start_bus) << 20) +
|
(@as(u64, bus - alloc.start_bus) << 20) +
|
||||||
(@as(u64, dev) << 15) +
|
(@as(u64, dev) << 15) +
|
||||||
(@as(u64, func) << 12);
|
(@as(u64, func) << 12);
|
||||||
hal.mapMmio(phys, phys, true); // identity-map this config page (writable)
|
// Map the config page (writable, for BAR sizing) and use the virtual
|
||||||
return @ptrFromInt(phys);
|
// address the HAL hands back.
|
||||||
|
return @ptrFromInt(hal.mapMmio(phys, danos.page_size, true));
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Read a little-endian integer at `off` from a (possibly unaligned) byte pointer.
|
/// Read a little-endian integer at `off` from a (possibly unaligned) byte pointer.
|
||||||
|
|||||||
@@ -173,7 +173,9 @@ test "parses a nested namespace and finds the sleep package" {
|
|||||||
try std.testing.expectEqual(@as(u8, 0), s5.slp_typ_b);
|
try std.testing.expectEqual(@as(u8, 0), s5.slp_typ_b);
|
||||||
}
|
}
|
||||||
|
|
||||||
fn noMap(_: u64, _: u64, _: bool) void {}
|
fn noMap(phys: u64, _: u64, _: bool) u64 {
|
||||||
|
return phys;
|
||||||
|
}
|
||||||
fn noRead(_: u8, _: u16) u32 {
|
fn noRead(_: u8, _: u16) u32 {
|
||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -21,7 +21,7 @@ const Namespace = nsp.Namespace;
|
|||||||
|
|
||||||
/// Injected hardware access for OperationRegion reads/writes (the arch VMM + pio).
|
/// Injected hardware access for OperationRegion reads/writes (the arch VMM + pio).
|
||||||
pub const Hal = struct {
|
pub const Hal = struct {
|
||||||
mapMmio: *const fn (virt: u64, phys: u64, writable: bool) void,
|
mapMmio: *const fn (phys: u64, len: u64, writable: bool) u64,
|
||||||
pioRead: *const fn (width: u8, port: u16) u32,
|
pioRead: *const fn (width: u8, port: u16) u32,
|
||||||
pioWrite: *const fn (width: u8, port: u16, value: u32) void,
|
pioWrite: *const fn (width: u8, port: u16, value: u32) void,
|
||||||
};
|
};
|
||||||
@@ -665,8 +665,8 @@ pub const Interp = struct {
|
|||||||
fn readRegionByte(self: *Interp, space: u8, addr: u64) Error!u8 {
|
fn readRegionByte(self: *Interp, space: u8, addr: u64) Error!u8 {
|
||||||
switch (space) {
|
switch (space) {
|
||||||
0 => { // SystemMemory
|
0 => { // SystemMemory
|
||||||
self.hal.mapMmio(addr & ~@as(u64, 0xFFF), addr & ~@as(u64, 0xFFF), true);
|
const virt = self.hal.mapMmio(addr & ~@as(u64, 0xFFF), 0x1000, true);
|
||||||
const p: *align(1) const volatile u8 = @ptrFromInt(addr);
|
const p: *align(1) const volatile u8 = @ptrFromInt(virt + (addr & 0xFFF));
|
||||||
return p.*;
|
return p.*;
|
||||||
},
|
},
|
||||||
1 => return @truncate(self.hal.pioRead(1, @intCast(addr & 0xFFFF))), // SystemIO
|
1 => return @truncate(self.hal.pioRead(1, @intCast(addr & 0xFFFF))), // SystemIO
|
||||||
@@ -677,8 +677,8 @@ pub const Interp = struct {
|
|||||||
fn writeRegionByte(self: *Interp, space: u8, addr: u64, value: u8) Error!void {
|
fn writeRegionByte(self: *Interp, space: u8, addr: u64, value: u8) Error!void {
|
||||||
switch (space) {
|
switch (space) {
|
||||||
0 => {
|
0 => {
|
||||||
self.hal.mapMmio(addr & ~@as(u64, 0xFFF), addr & ~@as(u64, 0xFFF), true);
|
const virt = self.hal.mapMmio(addr & ~@as(u64, 0xFFF), 0x1000, true);
|
||||||
const p: *align(1) volatile u8 = @ptrFromInt(addr);
|
const p: *align(1) volatile u8 = @ptrFromInt(virt + (addr & 0xFFF));
|
||||||
p.* = value;
|
p.* = value;
|
||||||
},
|
},
|
||||||
1 => self.hal.pioWrite(1, @intCast(addr & 0xFFFF), value),
|
1 => self.hal.pioWrite(1, @intCast(addr & 0xFFFF), value),
|
||||||
|
|||||||
@@ -18,7 +18,10 @@ const std = @import("std");
|
|||||||
/// layer touches hardware without importing `arch` — the same discipline that lets
|
/// layer touches hardware without importing `arch` — the same discipline that lets
|
||||||
/// it stay firmware-agnostic. `pioRead`/`pioWrite` take a width in bytes (1/2/4).
|
/// it stay firmware-agnostic. `pioRead`/`pioWrite` take a width in bytes (1/2/4).
|
||||||
pub const Hal = struct {
|
pub const Hal = struct {
|
||||||
mapMmio: *const fn (virt: u64, phys: u64, writable: bool) void,
|
/// Map a physical MMIO range and return the virtual address to reach it at.
|
||||||
|
/// The device layer dereferences the returned address and never learns how
|
||||||
|
/// the kernel places it (identity, physmap, a window — the kernel's choice).
|
||||||
|
mapMmio: *const fn (phys: u64, len: u64, writable: bool) u64,
|
||||||
pioRead: *const fn (width: u8, port: u16) u32,
|
pioRead: *const fn (width: u8, port: u16) u32,
|
||||||
pioWrite: *const fn (width: u8, port: u16, value: u32) void,
|
pioWrite: *const fn (width: u8, port: u16, value: u32) void,
|
||||||
};
|
};
|
||||||
|
|||||||
@@ -78,8 +78,7 @@ fn sleepValue(slp_typ: u8) u32 {
|
|||||||
|
|
||||||
fn readReg(hal: Hal, reg: acpi.RegAccess) u32 {
|
fn readReg(hal: Hal, reg: acpi.RegAccess) u32 {
|
||||||
if (reg.mmio) {
|
if (reg.mmio) {
|
||||||
hal.mapMmio(reg.address, reg.address, true);
|
const p: *align(1) volatile u32 = @ptrFromInt(hal.mapMmio(reg.address, 4, true));
|
||||||
const p: *align(1) volatile u32 = @ptrFromInt(reg.address);
|
|
||||||
return p.*;
|
return p.*;
|
||||||
}
|
}
|
||||||
return hal.pioRead(reg.width, @intCast(reg.address));
|
return hal.pioRead(reg.width, @intCast(reg.address));
|
||||||
@@ -87,8 +86,7 @@ fn readReg(hal: Hal, reg: acpi.RegAccess) u32 {
|
|||||||
|
|
||||||
fn writeReg(hal: Hal, reg: acpi.RegAccess, value: u32) void {
|
fn writeReg(hal: Hal, reg: acpi.RegAccess, value: u32) void {
|
||||||
if (reg.mmio) {
|
if (reg.mmio) {
|
||||||
hal.mapMmio(reg.address, reg.address, true);
|
const p: *align(1) volatile u32 = @ptrFromInt(hal.mapMmio(reg.address, 4, true));
|
||||||
const p: *align(1) volatile u32 = @ptrFromInt(reg.address);
|
|
||||||
p.* = value;
|
p.* = value;
|
||||||
} else {
|
} else {
|
||||||
hal.pioWrite(reg.width, @intCast(reg.address), value);
|
hal.pioWrite(reg.width, @intCast(reg.address), value);
|
||||||
|
|||||||
@@ -9,6 +9,7 @@
|
|||||||
//! map). Every interrupt must be acknowledged with an end-of-interrupt write, or
|
//! map). Every interrupt must be acknowledged with an end-of-interrupt write, or
|
||||||
//! the LAPIC won't deliver the next one.
|
//! the LAPIC won't deliver the next one.
|
||||||
|
|
||||||
|
const danos = @import("danos");
|
||||||
const io = @import("io.zig");
|
const io = @import("io.zig");
|
||||||
const paging = @import("paging.zig");
|
const paging = @import("paging.zig");
|
||||||
|
|
||||||
@@ -118,7 +119,8 @@ pub fn init() void {
|
|||||||
if (cfg_pic_present) remapAndMaskPic();
|
if (cfg_pic_present) remapAndMaskPic();
|
||||||
|
|
||||||
const msr = io.rdmsr(ia32_apic_base_msr);
|
const msr = io.rdmsr(ia32_apic_base_msr);
|
||||||
base = @intCast(msr & 0xFFFFF000); // physical base is bits 12+
|
// Reach the LAPIC through the physmap (paging.init maps its page there).
|
||||||
|
base = @intCast(danos.physToVirt(msr & 0xFFFFF000)); // physical base is bits 12+
|
||||||
io.wrmsr(ia32_apic_base_msr, msr | (1 << 11)); // global enable
|
io.wrmsr(ia32_apic_base_msr, msr | (1 << 11)); // global enable
|
||||||
|
|
||||||
write(reg_spurious, 0x100 | spurious_vector); // bit 8 = software enable
|
write(reg_spurious, 0x100 | spurious_vector); // bit 8 = software enable
|
||||||
@@ -299,8 +301,10 @@ fn hpetWrite64(off: usize, value: u64) void {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Map + enable the HPET and return its tick frequency, or null if unusable.
|
/// Map + enable the HPET and return its tick frequency, or null if unusable.
|
||||||
|
/// Maps the HPET into the physmap and switches cfg_hpet_base to that virtual
|
||||||
|
/// address, so the register accessors reach it without the identity map.
|
||||||
fn hpetHz() ?u64 {
|
fn hpetHz() ?u64 {
|
||||||
paging.map(cfg_hpet_base & ~@as(u64, 0xFFF), cfg_hpet_base & ~@as(u64, 0xFFF), true);
|
cfg_hpet_base = paging.mapMmio(cfg_hpet_base, 0x400, true);
|
||||||
const caps = hpetRead64(0x00);
|
const caps = hpetRead64(0x00);
|
||||||
const period_fs = caps >> 32; // femtoseconds per tick
|
const period_fs = caps >> 32; // femtoseconds per tick
|
||||||
if (period_fs == 0) return null;
|
if (period_fs == 0) return null;
|
||||||
@@ -319,7 +323,9 @@ fn readHpet() u64 {
|
|||||||
|
|
||||||
fn readPmTimer() u64 {
|
fn readPmTimer() u64 {
|
||||||
const pt = cfg_pm_timer.?;
|
const pt = cfg_pm_timer.?;
|
||||||
if (pt.mmio) return @as(*volatile u32, @ptrFromInt(pt.address)).*;
|
// MMIO PM timer via the physmap (mapMmio is idempotent); the common case is
|
||||||
|
// a legacy I/O port.
|
||||||
|
if (pt.mmio) return @as(*volatile u32, @ptrFromInt(paging.mapMmio(pt.address, 4, false))).*;
|
||||||
return io.inl(@intCast(pt.address));
|
return io.inl(@intCast(pt.address));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -71,6 +71,13 @@ pub fn mapPage(virt: u64, phys: u64, writable: bool) void {
|
|||||||
paging.map(virt, phys, writable);
|
paging.map(virt, phys, writable);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Map a device MMIO range and return the virtual address to reach it at. This
|
||||||
|
/// is the device layer's `Hal.mapMmio` — it hands back a physmap pointer and
|
||||||
|
/// never exposes how the mapping is placed.
|
||||||
|
pub fn mapMmio(phys: u64, len: u64, writable: bool) u64 {
|
||||||
|
return paging.mapMmio(phys, len, writable);
|
||||||
|
}
|
||||||
|
|
||||||
/// Remove a kernel mapping.
|
/// Remove a kernel mapping.
|
||||||
pub fn unmapPage(virt: u64) void {
|
pub fn unmapPage(virt: u64) void {
|
||||||
paging.unmap(virt);
|
paging.unmap(virt);
|
||||||
|
|||||||
@@ -52,7 +52,9 @@ fn writeEntry(n: u32, low: u32, high: u32) void {
|
|||||||
/// Map the I/O APIC and mask every redirection entry — the safe quiescent state.
|
/// Map the I/O APIC and mask every redirection entry — the safe quiescent state.
|
||||||
pub fn init() void {
|
pub fn init() void {
|
||||||
if (base == 0) return;
|
if (base == 0) return;
|
||||||
paging.map(base & ~@as(u64, 0xFFF), base & ~@as(u64, 0xFFF), true);
|
// Reach the I/O APIC through the physmap; switch `base` to that virtual
|
||||||
|
// address so the register accessors work without the identity map.
|
||||||
|
base = paging.mapMmio(base, 0x1000, true);
|
||||||
max_entries = ((regRead(reg_version) >> 16) & 0xFF) + 1;
|
max_entries = ((regRead(reg_version) >> 16) & 0xFF) + 1;
|
||||||
var n: u32 = 0;
|
var n: u32 = 0;
|
||||||
while (n < max_entries) : (n += 1) writeEntry(n, redir_mask, 0);
|
while (n < max_entries) : (n += 1) writeEntry(n, redir_mask, 0);
|
||||||
|
|||||||
@@ -30,8 +30,14 @@ const pf_w: u32 = 2;
|
|||||||
var kernel_pml4: u64 = 0;
|
var kernel_pml4: u64 = 0;
|
||||||
var alloc_frame: *const fn () ?u64 = undefined;
|
var alloc_frame: *const fn () ?u64 = undefined;
|
||||||
|
|
||||||
|
/// Dereference a page-table frame by its physical address, via the physmap.
|
||||||
|
/// This is the single hinge for the higher-half move: page tables hold physical
|
||||||
|
/// frame addresses (pmm gives out physical frames, and CR3/PTEs must be
|
||||||
|
/// physical), but the kernel reaches them at `physmap_base + phys`. Valid under
|
||||||
|
/// both the loader's bootstrap tables and the kernel's own, which share the
|
||||||
|
/// physmap base.
|
||||||
fn tableAt(phys: u64) *[512]u64 {
|
fn tableAt(phys: u64) *[512]u64 {
|
||||||
return @ptrFromInt(phys);
|
return @ptrFromInt(danos.physToVirt(phys));
|
||||||
}
|
}
|
||||||
|
|
||||||
fn allocTable() u64 {
|
fn allocTable() u64 {
|
||||||
@@ -71,8 +77,19 @@ fn mapRangeIdentity(pml4: u64, base: u64, len: u64, flags: u64) void {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Map [phys_base, phys_base+len) into the physmap (at physToVirt(phys)) with
|
||||||
|
/// `flags`, rounded out to whole pages. This is how the kernel keeps a permanent
|
||||||
|
/// window onto physical memory once the low identity map goes away.
|
||||||
|
fn mapRangePhysmap(pml4: u64, phys_base: u64, len: u64, flags: u64) void {
|
||||||
|
var addr = phys_base & ~@as(u64, page_size - 1);
|
||||||
|
const end = phys_base + len;
|
||||||
|
while (addr < end) : (addr += page_size) {
|
||||||
|
mapPage(pml4, danos.physToVirt(addr), addr, flags);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
fn regions(mm: danos.MemoryMap) []const danos.MemoryRegion {
|
fn regions(mm: danos.MemoryMap) []const danos.MemoryRegion {
|
||||||
return @as([*]const danos.MemoryRegion, @ptrFromInt(mm.regions))[0..mm.len];
|
return @as([*]const danos.MemoryRegion, @ptrFromInt(danos.physToVirt(mm.regions)))[0..mm.len];
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Enable the NX bit in the page-table format (EFER.NXE). Must happen before we
|
/// Enable the NX bit in the page-table format (EFER.NXE). Must happen before we
|
||||||
@@ -88,16 +105,22 @@ pub fn init(allocFrame: *const fn () ?u64, boot_info: *const danos.BootInfo) voi
|
|||||||
enableNx();
|
enableNx();
|
||||||
const pml4 = allocTable();
|
const pml4 = allocTable();
|
||||||
|
|
||||||
// 1. All RAM identity-mapped RW + NX. Non-RAM (MMIO) is skipped and stays
|
// 1. All RAM in the physmap (physToVirt(phys)) RW + NX, plus — during the
|
||||||
// unmapped unless mapped explicitly below.
|
// higher-half transition — a low identity map so any not-yet-converted
|
||||||
|
// physical deref still resolves. Non-RAM (MMIO) is skipped here and
|
||||||
|
// mapped explicitly below. The identity half is removed in a later step.
|
||||||
for (regions(boot_info.memory_map)) |r| {
|
for (regions(boot_info.memory_map)) |r| {
|
||||||
if (r.kind == .mmio) continue;
|
if (r.kind == .mmio) continue;
|
||||||
|
mapRangePhysmap(pml4, r.base, r.pages * page_size, present | writable | no_execute);
|
||||||
mapRangeIdentity(pml4, r.base, r.pages * page_size, present | writable | no_execute);
|
mapRangeIdentity(pml4, r.base, r.pages * page_size, present | writable | no_execute);
|
||||||
}
|
}
|
||||||
|
|
||||||
// 2. The framebuffer and the Local APIC (device memory we need), RW + NX.
|
// 2. The framebuffer and the Local APIC (device memory we need), RW + NX —
|
||||||
|
// in the physmap and (transitionally) identity.
|
||||||
const fb = boot_info.framebuffer;
|
const fb = boot_info.framebuffer;
|
||||||
|
mapRangePhysmap(pml4, fb.base, @as(u64, fb.height) * fb.pitch, present | writable | no_execute);
|
||||||
mapRangeIdentity(pml4, fb.base, @as(u64, fb.height) * fb.pitch, present | writable | no_execute);
|
mapRangeIdentity(pml4, fb.base, @as(u64, fb.height) * fb.pitch, present | writable | no_execute);
|
||||||
|
mapPage(pml4, danos.physToVirt(0xFEE00000), 0xFEE00000, present | writable | no_execute);
|
||||||
mapPage(pml4, 0xFEE00000, 0xFEE00000, present | writable | no_execute);
|
mapPage(pml4, 0xFEE00000, 0xFEE00000, present | writable | no_execute);
|
||||||
|
|
||||||
// 3. Overlay the kernel's own segments with their real ELF permissions,
|
// 3. Overlay the kernel's own segments with their real ELF permissions,
|
||||||
@@ -130,6 +153,25 @@ pub fn map(virt: u64, phys: u64, writable_page: bool) void {
|
|||||||
invalidate(virt);
|
invalidate(virt);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Map a device MMIO range into the physmap and return the virtual address to
|
||||||
|
/// use for it (physToVirt(phys)). The single way the kernel (and the device
|
||||||
|
/// layer, via the HAL) reaches memory-mapped registers once the identity map is
|
||||||
|
/// gone: physmap pages are RW + NX, so a driver never executes device memory.
|
||||||
|
/// Idempotent for already-mapped ranges. `len` 0 maps one page.
|
||||||
|
pub fn mapMmio(phys: u64, len: u64, writable_page: bool) u64 {
|
||||||
|
var flags: u64 = present | no_execute;
|
||||||
|
if (writable_page) flags |= writable;
|
||||||
|
const first = phys & ~@as(u64, page_size - 1);
|
||||||
|
const last = phys + (if (len == 0) 1 else len) - 1;
|
||||||
|
var addr = first;
|
||||||
|
while (addr <= (last & ~@as(u64, page_size - 1))) : (addr += page_size) {
|
||||||
|
const virt = danos.physToVirt(addr);
|
||||||
|
mapPage(kernel_pml4, virt, addr, flags);
|
||||||
|
invalidate(virt);
|
||||||
|
}
|
||||||
|
return danos.physToVirt(phys);
|
||||||
|
}
|
||||||
|
|
||||||
/// Like `descend`, but also sets the U/S bit on the intermediate entry (new or
|
/// Like `descend`, but also sets the U/S bit on the intermediate entry (new or
|
||||||
/// pre-existing): ring-3 access requires U at *every* level, and `descend` leaves
|
/// pre-existing): ring-3 access requires U at *every* level, and `descend` leaves
|
||||||
/// existing entries untouched. Only used under user-exclusive virtual ranges, so
|
/// existing entries untouched. Only used under user-exclusive virtual ranges, so
|
||||||
|
|||||||
@@ -9,6 +9,8 @@
|
|||||||
//! optimistically to COM1 (harmless if absent); the framebuffer console is the
|
//! optimistically to COM1 (harmless if absent); the framebuffer console is the
|
||||||
//! always-present log.
|
//! always-present log.
|
||||||
|
|
||||||
|
const paging = @import("paging.zig");
|
||||||
|
|
||||||
/// How the UART registers are reached: legacy I/O ports or memory-mapped.
|
/// How the UART registers are reached: legacy I/O ports or memory-mapped.
|
||||||
const Access = enum { port, mmio };
|
const Access = enum { port, mmio };
|
||||||
|
|
||||||
@@ -61,7 +63,9 @@ pub fn init() void {
|
|||||||
/// re-run the UART setup there. Called after discovery when an SPCR entry exists.
|
/// re-run the UART setup there. Called after discovery when an SPCR entry exists.
|
||||||
pub fn reconfigure(is_mmio: bool, addr: u64) void {
|
pub fn reconfigure(is_mmio: bool, addr: u64) void {
|
||||||
access = if (is_mmio) .mmio else .port;
|
access = if (is_mmio) .mmio else .port;
|
||||||
base = addr;
|
// An MMIO UART is reached through the physmap; an I/O-port UART keeps its
|
||||||
|
// port number unchanged.
|
||||||
|
base = if (is_mmio) paging.mapMmio(addr, 0x100, true) else addr;
|
||||||
init();
|
init();
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -13,6 +13,7 @@
|
|||||||
//! Once a core has its own descriptor tables, LAPIC, and timer, it calls the generic
|
//! Once a core has its own descriptor tables, LAPIC, and timer, it calls the generic
|
||||||
//! scheduler entry and joins the run loop — mechanism here, policy there.
|
//! scheduler entry and joins the run loop — mechanism here, policy there.
|
||||||
|
|
||||||
|
const danos = @import("danos");
|
||||||
const io = @import("io.zig");
|
const io = @import("io.zig");
|
||||||
const gdt = @import("gdt.zig");
|
const gdt = @import("gdt.zig");
|
||||||
const tss = @import("tss.zig");
|
const tss = @import("tss.zig");
|
||||||
@@ -75,22 +76,27 @@ pub fn trampolinePage() u64 {
|
|||||||
/// Arm the trampoline for a wake: make its page executable (W^X exception for the
|
/// Arm the trampoline for a wake: make its page executable (W^X exception for the
|
||||||
/// duration of the climb) and copy the blob in.
|
/// duration of the climb) and copy the blob in.
|
||||||
fn arm() void {
|
fn arm() void {
|
||||||
|
// The AP executes this page at its physical address (identity) while it
|
||||||
|
// climbs from real to long mode, so it needs a low identity mapping that is
|
||||||
|
// executable — the one deliberate, transient W^X exception. The BSP writes
|
||||||
|
// the blob into the frame through the physmap.
|
||||||
paging.setExecutable(tramp_phys);
|
paging.setExecutable(tramp_phys);
|
||||||
const start = @extern([*]const u8, .{ .name = "ap_trampoline_start" });
|
const start = @extern([*]const u8, .{ .name = "ap_trampoline_start" });
|
||||||
const end = @extern([*]const u8, .{ .name = "ap_trampoline_end" });
|
const end = @extern([*]const u8, .{ .name = "ap_trampoline_end" });
|
||||||
const len = @intFromPtr(end) - @intFromPtr(start);
|
const len = @intFromPtr(end) - @intFromPtr(start);
|
||||||
const dst: [*]u8 = @ptrFromInt(tramp_phys);
|
const dst: [*]u8 = @ptrFromInt(danos.physToVirt(tramp_phys));
|
||||||
@memcpy(dst[0..len], start[0..len]);
|
@memcpy(dst[0..len], start[0..len]);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Disarm after a wake: wipe the page and restore it to inert RW+NX, so no
|
/// Disarm after a wake: wipe the page through the physmap and remove its low
|
||||||
/// executable code (nor any stale bytes) lingers between wakes. Safe to run once the
|
/// identity mapping, so no executable code (nor any stale bytes, nor any
|
||||||
/// woken core has reported in — it's long past the trampoline by then, in the kernel
|
/// low-half mapping) lingers between wakes. Safe once the woken core has
|
||||||
/// image; a core that never answered is dead and can't be mid-climb.
|
/// reported in — it's long past the trampoline by then, in the kernel image; a
|
||||||
|
/// core that never answered is dead and can't be mid-climb.
|
||||||
fn disarm() void {
|
fn disarm() void {
|
||||||
const dst: [*]u8 = @ptrFromInt(tramp_phys);
|
const dst: [*]u8 = @ptrFromInt(danos.physToVirt(tramp_phys));
|
||||||
@memset(dst[0..page_size], 0);
|
@memset(dst[0..page_size], 0);
|
||||||
paging.map(tramp_phys, tramp_phys, true); // RW + NX, like every other RAM frame
|
paging.unmap(tramp_phys); // drop the transient low identity mapping
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Address of a patchable trampoline parameter, by symbol name: the copied blob's
|
/// Address of a patchable trampoline parameter, by symbol name: the copied blob's
|
||||||
@@ -100,7 +106,7 @@ fn disarm() void {
|
|||||||
fn param(comptime name: []const u8) *align(1) volatile u64 {
|
fn param(comptime name: []const u8) *align(1) volatile u64 {
|
||||||
const start = @intFromPtr(@extern([*]const u8, .{ .name = "ap_trampoline_start" }));
|
const start = @intFromPtr(@extern([*]const u8, .{ .name = "ap_trampoline_start" }));
|
||||||
const sym = @intFromPtr(@extern([*]const u8, .{ .name = name }));
|
const sym = @intFromPtr(@extern([*]const u8, .{ .name = name }));
|
||||||
return @ptrFromInt(tramp_phys + (sym - start));
|
return @ptrFromInt(danos.physToVirt(tramp_phys + (sym - start)));
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Wake the core with Local APIC id `apic_id` as dense CPU `index`, hand it
|
/// Wake the core with Local APIC id `apic_id` as dense CPU `index`, hand it
|
||||||
|
|||||||
@@ -67,8 +67,13 @@ pub const Console = struct {
|
|||||||
bg: u32 = 0x0000_0000, // black
|
bg: u32 = 0x0000_0000, // black
|
||||||
|
|
||||||
pub fn init(fb: danos.Framebuffer) Console {
|
pub fn init(fb: danos.Framebuffer) Console {
|
||||||
|
// Reach the framebuffer through the physmap, so the pointer stays valid
|
||||||
|
// once the low identity map is gone. The base is mapped by both the
|
||||||
|
// loader's bootstrap tables and paging.init.
|
||||||
|
var mapped = fb;
|
||||||
|
if (fb.base != 0) mapped.base = danos.physToVirt(fb.base);
|
||||||
return .{
|
return .{
|
||||||
.fb = fb,
|
.fb = mapped,
|
||||||
.cols = fb.width / glyph_w,
|
.cols = fb.width / glyph_w,
|
||||||
.rows = fb.height / glyph_h,
|
.rows = fb.height / glyph_h,
|
||||||
};
|
};
|
||||||
|
|||||||
+8
-4
@@ -41,7 +41,11 @@ var ap_trampoline_page: u64 = 0;
|
|||||||
/// pointer to the handoff data. There is no runtime, no stack unwinding, and no
|
/// pointer to the handoff data. There is no runtime, no stack unwinding, and no
|
||||||
/// caller to return to, so this never returns.
|
/// caller to return to, so this never returns.
|
||||||
export fn _start(boot_info: *const BootInfo) callconv(kernel_abi) noreturn {
|
export fn _start(boot_info: *const BootInfo) callconv(kernel_abi) noreturn {
|
||||||
kmain(boot_info);
|
// The loader passes a low physical pointer (its own stack). Reach it — and
|
||||||
|
// everything it points at — through the physmap, so it stays valid once the
|
||||||
|
// low identity map is gone. The physmap base is the same under the loader's
|
||||||
|
// bootstrap tables and the kernel's own.
|
||||||
|
kmain(@ptrFromInt(danos.physToVirt(@intFromPtr(boot_info))));
|
||||||
}
|
}
|
||||||
|
|
||||||
fn kmain(boot_info: *const BootInfo) noreturn {
|
fn kmain(boot_info: *const BootInfo) noreturn {
|
||||||
@@ -82,7 +86,7 @@ fn kmain(boot_info: *const BootInfo) noreturn {
|
|||||||
|
|
||||||
// Summarise the physical memory the loader handed us. The array is danos's
|
// Summarise the physical memory the loader handed us. The array is danos's
|
||||||
// own MemoryRegion, so this is a plain slice — no firmware layout in sight.
|
// own MemoryRegion, so this is a plain slice — no firmware layout in sight.
|
||||||
const regions = @as([*]const danos.MemoryRegion, @ptrFromInt(boot_info.memory_map.regions))[0..boot_info.memory_map.len];
|
const regions = @as([*]const danos.MemoryRegion, @ptrFromInt(danos.physToVirt(boot_info.memory_map.regions)))[0..boot_info.memory_map.len];
|
||||||
var usable_pages: u64 = 0;
|
var usable_pages: u64 = 0;
|
||||||
var reserved_pages: u64 = 0; // reserved RAM only — MMIO is device space, not RAM
|
var reserved_pages: u64 = 0; // reserved RAM only — MMIO is device space, not RAM
|
||||||
for (regions) |r| {
|
for (regions) |r| {
|
||||||
@@ -141,7 +145,7 @@ fn kmain(boot_info: *const BootInfo) noreturn {
|
|||||||
// mapped) and maps PCIe config space on demand via the VMM. A failure here is
|
// mapped) and maps PCIe config space on demand via the VMM. A failure here is
|
||||||
// not fatal yet — log it and carry on.
|
// not fatal yet — log it and carry on.
|
||||||
const hal = platform.Hal{
|
const hal = platform.Hal{
|
||||||
.mapMmio = arch.mapPage,
|
.mapMmio = arch.mapMmio,
|
||||||
.pioRead = arch.pioRead,
|
.pioRead = arch.pioRead,
|
||||||
.pioWrite = arch.pioWrite,
|
.pioWrite = arch.pioWrite,
|
||||||
};
|
};
|
||||||
@@ -253,7 +257,7 @@ fn kmain(boot_info: *const BootInfo) noreturn {
|
|||||||
// idle). init becomes a real schedulable process in M3.
|
// idle). init becomes a real schedulable process in M3.
|
||||||
if (boot_info.init_len != 0) {
|
if (boot_info.init_len != 0) {
|
||||||
status("starting /sbin/init...\n");
|
status("starting /sbin/init...\n");
|
||||||
const image = @as([*]const u8, @ptrFromInt(boot_info.init_base))[0..boot_info.init_len];
|
const image = @as([*]const u8, @ptrFromInt(danos.physToVirt(boot_info.init_base)))[0..boot_info.init_len];
|
||||||
scheduler.setPreemption(false);
|
scheduler.setPreemption(false);
|
||||||
const code = usermode.runInitElf(image);
|
const code = usermode.runInitElf(image);
|
||||||
scheduler.setPreemption(true);
|
scheduler.setPreemption(true);
|
||||||
|
|||||||
+9
-5
@@ -49,12 +49,16 @@ inline fn setFree(frame: usize) void {
|
|||||||
}
|
}
|
||||||
|
|
||||||
fn regions(map: danos.MemoryMap) []const danos.MemoryRegion {
|
fn regions(map: danos.MemoryMap) []const danos.MemoryRegion {
|
||||||
return @as([*]const danos.MemoryRegion, @ptrFromInt(map.regions))[0..map.len];
|
return @as([*]const danos.MemoryRegion, @ptrFromInt(danos.physToVirt(map.regions)))[0..map.len];
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Build the allocator from the loader's memory map. Relies on the firmware's
|
/// Build the allocator from the loader's memory map. Reaches physical memory
|
||||||
/// identity mapping still being in effect (a physical address is usable directly
|
/// (the region array, the bitmap's own storage) through the physmap, which the
|
||||||
/// as a pointer) — true until the kernel installs its own page tables.
|
/// loader's bootstrap tables already provide — so this works before the kernel
|
||||||
|
/// installs its own tables. Invariant: the bitmap lands in the first usable
|
||||||
|
/// region (lowest address), which must sit under the bootstrap physmap's reach
|
||||||
|
/// (4 GiB); it always does, as both this and the page-table allocator scan from
|
||||||
|
/// low addresses up.
|
||||||
pub fn init(map: danos.MemoryMap) void {
|
pub fn init(map: danos.MemoryMap) void {
|
||||||
const regs = regions(map);
|
const regs = regions(map);
|
||||||
|
|
||||||
@@ -87,7 +91,7 @@ pub fn init(map: danos.MemoryMap) void {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
const bitmap_base = storage orelse @panic("pmm: no region large enough for the frame bitmap");
|
const bitmap_base = storage orelse @panic("pmm: no region large enough for the frame bitmap");
|
||||||
bitmap = @as([*]u8, @ptrFromInt(bitmap_base))[0..bitmap_bytes];
|
bitmap = @as([*]u8, @ptrFromInt(danos.physToVirt(bitmap_base)))[0..bitmap_bytes];
|
||||||
|
|
||||||
// 3. Start with everything marked used, then free the usable regions. Doing
|
// 3. Start with everything marked used, then free the usable regions. Doing
|
||||||
// it this way means every gap, reserved span and MMIO hole is unallocatable
|
// it this way means every gap, reserved span and MMIO hole is unallocatable
|
||||||
|
|||||||
@@ -110,7 +110,7 @@ pub fn run(case: []const u8, boot_info: *const BootInfo) void {
|
|||||||
|
|
||||||
fn platformHal() platform.Hal {
|
fn platformHal() platform.Hal {
|
||||||
return .{
|
return .{
|
||||||
.mapMmio = arch.mapPage,
|
.mapMmio = arch.mapMmio,
|
||||||
.pioRead = arch.pioRead,
|
.pioRead = arch.pioRead,
|
||||||
.pioWrite = arch.pioWrite,
|
.pioWrite = arch.pioWrite,
|
||||||
};
|
};
|
||||||
@@ -552,7 +552,7 @@ fn smpTest() void {
|
|||||||
const tramp = arch.trampolinePage();
|
const tramp = arch.trampolinePage();
|
||||||
check("trampoline frame reserved", tramp != 0);
|
check("trampoline frame reserved", tramp != 0);
|
||||||
if (tramp != 0) {
|
if (tramp != 0) {
|
||||||
const bytes: [*]const u8 = @ptrFromInt(tramp);
|
const bytes: [*]const u8 = @ptrFromInt(danos.physToVirt(tramp));
|
||||||
var zeroed = true;
|
var zeroed = true;
|
||||||
for (0..4096) |b| {
|
for (0..4096) |b| {
|
||||||
if (bytes[b] != 0) zeroed = false;
|
if (bytes[b] != 0) zeroed = false;
|
||||||
@@ -753,7 +753,7 @@ fn initTest(boot_info: *const BootInfo) void {
|
|||||||
result();
|
result();
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
const image = @as([*]const u8, @ptrFromInt(boot_info.init_base))[0..boot_info.init_len];
|
const image = @as([*]const u8, @ptrFromInt(danos.physToVirt(boot_info.init_base)))[0..boot_info.init_len];
|
||||||
sched.setPreemption(false); // see userTest: pins the run to this core's rsp0
|
sched.setPreemption(false); // see userTest: pins the run to this core's rsp0
|
||||||
const code = usermode.runInitElf(image);
|
const code = usermode.runInitElf(image);
|
||||||
sched.setPreemption(true);
|
sched.setPreemption(true);
|
||||||
|
|||||||
@@ -128,10 +128,10 @@ pub fn run(blob: []const u8) RunError!void {
|
|||||||
return error.OutOfMemory;
|
return error.OutOfMemory;
|
||||||
};
|
};
|
||||||
|
|
||||||
// Fill the code frame through its identity mapping (supervisor RW): the
|
// Fill the code frame through the physmap (supervisor RW): the user-facing
|
||||||
// user-facing mapping is read-only, and this also sidesteps CR0.WP/SMAP.
|
// mapping is read-only, and this also sidesteps CR0.WP/SMAP. The tail is
|
||||||
// The tail is padded with int3 so a stray jump traps instead of sliding.
|
// padded with int3 so a stray jump traps instead of sliding.
|
||||||
const code: [*]u8 = @ptrFromInt(code_frame);
|
const code: [*]u8 = @ptrFromInt(danos.physToVirt(code_frame));
|
||||||
@memcpy(code[0..blob.len], blob);
|
@memcpy(code[0..blob.len], blob);
|
||||||
@memset(code[blob.len..page_size], 0xCC);
|
@memset(code[blob.len..page_size], 0xCC);
|
||||||
|
|
||||||
@@ -251,7 +251,7 @@ fn parseSegments(image: []const u8, segs: *[max_segments]Segment) InitError!stru
|
|||||||
/// with the segment's W^X permissions. Records the page for rollback.
|
/// with the segment's W^X permissions. Records the page for rollback.
|
||||||
fn loadPage(image: []const u8, seg: Segment, page_index: u64) InitError!void {
|
fn loadPage(image: []const u8, seg: Segment, page_index: u64) InitError!void {
|
||||||
const frame = pmm.alloc() orelse return error.OutOfMemory;
|
const frame = pmm.alloc() orelse return error.OutOfMemory;
|
||||||
const dst: [*]u8 = @ptrFromInt(frame);
|
const dst: [*]u8 = @ptrFromInt(danos.physToVirt(frame));
|
||||||
@memset(dst[0..page_size], 0);
|
@memset(dst[0..page_size], 0);
|
||||||
const page_off = page_index * page_size;
|
const page_off = page_index * page_size;
|
||||||
if (page_off < seg.filesz) {
|
if (page_off < seg.filesz) {
|
||||||
|
|||||||
Reference in New Issue
Block a user