kernel: user memory is reached only through a checked copy
A new user-memory module owns every kernel touch of a user buffer: copyFromUser, the new copyToUser, and the resolve behind both. The walk accumulates the U/S and writable bits down all four levels with the MMU's own AND rule — folding a 2 MiB leaf in before it resolves and refusing a 1 GiB leaf outright — so a copy honours what ring 3 itself would be allowed, closing the presence-only trust model the IPC layer carried since bring-up. It then confirms the frame is physmap-backed, because that is how the copy reaches it: an mmio_map'd BAR passes the permission walk and would otherwise fault ring 0 on an alias the physmap never mapped, on the IPC path as much as the new one. The nine stragglers that dereferenced user pointers raw now route through it, so a bad pointer returns -EFAULT where it used to fault the kernel. The write direction restructures its callees around kernel bounce buffers: scheduler and devices-broker enumerate from a slot cursor (a task exiting between chunks can neither duplicate nor lose an entry), klog_read drains the ring in chunks, and fs_node stages headers and names contiguously. fs_resolve copies out before installing the endpoint handle, so a faulting copy cannot strand a capability; its out-capacity bound no longer adds an unbounded ring-3 length to the base, which wrapped and trapped the kernel's own overflow check. debug_write reads the caller's message once. Suite 107/107 (new user-memory case: seven bad pointers refused, each paired with a sound call that must still succeed).
This commit is contained in:
@@ -17,18 +17,20 @@
|
||||
//! endpoint's own FIFO (threaded through the otherwise-idle `Task.next`); servers
|
||||
//! waiting for work use a normal WaitQueue.
|
||||
//!
|
||||
//! Trust model (bring-up): copies honour only page presence and a user-half bound,
|
||||
//! not the leaf U/S or R/W bits and not SMAP — a #PF-tolerant, permission-checked
|
||||
//! copy is a later security-track item, matching the existing debug_write gap.
|
||||
//! Trust model: every side of a copy that names a *user* address space goes
|
||||
//! through system/kernel/user-memory.zig — user-half bound, page presence, and
|
||||
//! the leaf permissions ring 3 itself would face (U/S to read, U/S + R/W to
|
||||
//! write). A kernel-side buffer is trusted and translated as-is. An unmapped or
|
||||
//! wrongly-permissioned page fails the operation; it never faults ring 0.
|
||||
|
||||
const std = @import("std");
|
||||
const boot_handoff = @import("boot-handoff");
|
||||
const abi = @import("abi");
|
||||
const architecture = @import("architecture");
|
||||
const scheduler = @import("scheduler.zig");
|
||||
const sync = @import("sync.zig");
|
||||
const heap = @import("heap.zig");
|
||||
const pmm = @import("pmm.zig");
|
||||
const user_memory = @import("user-memory.zig");
|
||||
|
||||
const page_size = abi.page_size;
|
||||
const Task = scheduler.Task;
|
||||
@@ -83,8 +85,9 @@ const PostSlot = struct {
|
||||
bytes: [POST_MAXIMUM]u8 = undefined,
|
||||
};
|
||||
|
||||
/// End of the user (low) canonical half — user buffers must lie below it.
|
||||
const user_half_end: u64 = 0x0000_8000_0000_0000;
|
||||
/// End of the user (low) canonical half — user buffers must lie below it. One
|
||||
/// definition, in the module that owns the user-memory contract.
|
||||
const user_half_end: u64 = user_memory.user_half_end;
|
||||
|
||||
/// A rendezvous endpoint. Allocated from the kernel heap; referenced by handle
|
||||
/// (per process) and/or by a registry slot, counted by `refcount`.
|
||||
@@ -300,18 +303,22 @@ pub fn abandonSenderLocked(t: *Task) void {
|
||||
/// Copy `len` bytes from `source_va` in address space `source_as` to `destination_va` in
|
||||
/// `destination_as`, walking each side's page tables through the physmap (no CR3 switch).
|
||||
/// `*_as == 0` means the kernel address space (for kernel-task endpoints). User
|
||||
/// buffers must lie in the low half. Returns false — never #PFs — if any page is
|
||||
/// unmapped or out of range. Handles page-straddling buffers.
|
||||
/// buffers must lie in the low half and carry the permission ring 3 would need for
|
||||
/// their side of the copy — readable to send from, writable to receive into.
|
||||
/// Returns false — never #PFs — if any page is unmapped, out of range, or
|
||||
/// wrongly permissioned. Handles page-straddling buffers.
|
||||
///
|
||||
/// This is the process↔process case, which `user-memory` deliberately does not
|
||||
/// cover (it knows one user address space at a time); both sides resolve through
|
||||
/// `user_memory.resolve`, so the permission rules are the same ones.
|
||||
fn copyAcross(source_as: u64, source_va: u64, destination_as: u64, destination_va: u64, len: usize) bool {
|
||||
const source_root = if (source_as != 0) source_as else architecture.kernelPageTable();
|
||||
const destination_root = if (destination_as != 0) destination_as else architecture.kernelPageTable();
|
||||
if (source_as != 0 and (source_va >= user_half_end or source_va + len > user_half_end)) return false;
|
||||
if (destination_as != 0 and (destination_va >= user_half_end or destination_va + len > user_half_end)) return false;
|
||||
if (source_as != 0 and !user_memory.userRangeOk(source_va, len)) return false;
|
||||
if (destination_as != 0 and !user_memory.userRangeOk(destination_va, len)) return false;
|
||||
|
||||
var off: usize = 0;
|
||||
while (off < len) {
|
||||
const s = architecture.translate(source_root, source_va + off) orelse return false;
|
||||
const d = architecture.translate(destination_root, destination_va + off) orelse return false;
|
||||
const s = user_memory.resolve(source_as, source_va + off, false) orelse return false;
|
||||
const d = user_memory.resolve(destination_as, destination_va + off, true) orelse return false;
|
||||
const s_left = page_size - ((source_va + off) & (page_size - 1));
|
||||
const d_left = page_size - ((destination_va + off) & (page_size - 1));
|
||||
const n = @min(@min(s_left, d_left), len - off);
|
||||
@@ -323,27 +330,10 @@ fn copyAcross(source_as: u64, source_va: u64, destination_as: u64, destination_v
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Copy `destination.len` bytes from `user_va` in address space `user_as` into the kernel
|
||||
/// buffer `destination`, walking the user page tables through the physmap. Returns false if
|
||||
/// the range escapes the user half or any source page is unmapped — so a bad user
|
||||
/// pointer *fails the system_call* rather than faulting the kernel (danos has no
|
||||
/// fault-recovering copy-in, so a raw dereference of an unmapped user page would halt
|
||||
/// the machine). The correct way to pull a fixed-size struct in from user space, and
|
||||
/// a single fetch: no TOCTOU against a hostile pointer.
|
||||
pub fn copyFromUser(user_as: u64, user_va: u64, destination: []u8) bool {
|
||||
if (user_as == 0) return false; // not a user address space
|
||||
if (user_va >= user_half_end or user_va + destination.len > user_half_end) return false;
|
||||
var off: usize = 0;
|
||||
while (off < destination.len) {
|
||||
const s = architecture.translate(user_as, user_va + off) orelse return false;
|
||||
const s_left = page_size - ((user_va + off) & (page_size - 1));
|
||||
const n = @min(s_left, destination.len - off);
|
||||
const source: [*]const u8 = @ptrFromInt(boot_handoff.physicalToVirtual(s));
|
||||
@memcpy(destination[off..][0..n], source[0..n]);
|
||||
off += n;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
/// The checked copy-in, re-exported from its home in `user-memory` so the many
|
||||
/// `ipc.copyFromUser` call sites keep reading naturally. New code should reach
|
||||
/// for `user-memory` directly — it is where the write direction lives too.
|
||||
pub const copyFromUser = user_memory.copyFromUser;
|
||||
|
||||
// --- the two IPC operations -------------------------------------------------
|
||||
|
||||
|
||||
Reference in New Issue
Block a user