kernel: user memory is reached only through a checked copy

A new user-memory module owns every kernel touch of a user buffer:
copyFromUser, the new copyToUser, and the resolve behind both. The walk
accumulates the U/S and writable bits down all four levels with the MMU's
own AND rule — folding a 2 MiB leaf in before it resolves and refusing a
1 GiB leaf outright — so a copy honours what ring 3 itself would be
allowed, closing the presence-only trust model the IPC layer carried since
bring-up. It then confirms the frame is physmap-backed, because that is how
the copy reaches it: an mmio_map'd BAR passes the permission walk and would
otherwise fault ring 0 on an alias the physmap never mapped, on the IPC path
as much as the new one.

The nine stragglers that dereferenced user pointers raw now route through
it, so a bad pointer returns -EFAULT where it used to fault the kernel.
The write direction restructures its callees around kernel bounce buffers:
scheduler and devices-broker enumerate from a slot cursor (a task exiting
between chunks can neither duplicate nor lose an entry), klog_read drains
the ring in chunks, and fs_node stages headers and names contiguously.
fs_resolve copies out before installing the endpoint handle, so a faulting
copy cannot strand a capability; its out-capacity bound no longer adds an
unbounded ring-3 length to the base, which wrapped and trapped the kernel's
own overflow check. debug_write reads the caller's message once.

Suite 107/107 (new user-memory case: seven bad pointers refused, each
paired with a sound call that must still succeed).
This commit is contained in:
Daniel Samson
2026-07-31 20:57:49 +01:00
parent c4f16a5448
commit 8d4a7cf240
16 changed files with 691 additions and 105 deletions
+145 -41
View File
@@ -31,6 +31,7 @@ const scheduler = @import("scheduler.zig");
const console = @import("console.zig");
const sync = @import("sync.zig");
const ipc = @import("ipc-synchronous.zig");
const user_memory = @import("user-memory.zig");
const devices_broker = @import("devices-broker.zig");
const irq = @import("irq.zig");
const iommu = @import("iommu.zig");
@@ -110,6 +111,14 @@ pub const maximum_arguments = 8;
/// Ceiling on the `system_spawn` extra-arguments blob (argv[1..], NUL-separated).
pub const maximum_argument_bytes = 256;
/// Longest path `fs_resolve` accepts, longest prefix `fs_mount`/`fs_unmount`
/// accept, and longest backend rewrite prefix. Each is also the size of the
/// kernel staging buffer the argument is copied into, which is why they are
/// named here rather than spelled as literals at the check.
pub const maximum_resolve_path = 224;
pub const maximum_mount_prefix = 64;
pub const maximum_mount_rewrite = 32;
/// Auxiliary-vector entry types (System V AMD64 process entry). Only what the
/// kernel emits today; a C runtime scans the vector until the null terminator.
const auxiliary_vector_null: u64 = 0; // AT_NULL — end of the vector
@@ -378,6 +387,12 @@ fn systemIpcSend(state: *architecture.CpuState) void {
/// device_enumerate(buffer, maximum) -> total: snapshot the device table into the caller's
/// buffer (up to `maximum` entries), returning the total device count.
///
/// The broker fills a small kernel chunk which `copyToUser` then places in the
/// caller's buffer: the kernel never stores through a user pointer, so a bad one
/// is -EFAULT instead of a ring-0 page fault. A DeviceDescriptor is a few hundred
/// bytes, so the chunk is deliberately tiny — the 16 KiB kernel stack could not
/// hold a whole user-requested array of them.
fn systemDeviceEnumerate(state: *architecture.CpuState) void {
const buffer_ptr = architecture.systemCallArg(state, 0);
const maximum = architecture.systemCallArg(state, 1);
@@ -385,8 +400,20 @@ fn systemDeviceEnumerate(state: *architecture.CpuState) void {
if (t.address_space == 0 or buffer_ptr >= user_half_end) return fail(state);
const sz = @sizeOf(device_abi.DeviceDescriptor);
const cap = @min(maximum, (user_half_end - buffer_ptr) / sz); // clamp to the user half
const out: [*]device_abi.DeviceDescriptor = @ptrFromInt(buffer_ptr);
architecture.setSystemCallResult(state, devices_broker.enumerate(out[0..@intCast(cap)]));
var chunk: [2]device_abi.DeviceDescriptor = undefined;
var copied: u64 = 0;
var start: usize = 0;
while (copied < cap) {
const filled = devices_broker.enumerateFrom(start, &chunk);
if (filled == 0) break;
start += filled;
const take = @min(@as(u64, filled), cap - copied);
const bytes = std.mem.sliceAsBytes(chunk[0..@intCast(take)]);
if (!user_memory.copyToUser(t.address_space, buffer_ptr + copied * sz, bytes)) return failErr(state, ipc.EFAULT);
copied += take;
}
architecture.setSystemCallResult(state, devices_broker.deviceCount());
}
/// device_claim(id) -> 0/-1: take exclusive ownership of a device for this process.
@@ -969,7 +996,17 @@ fn systemSpawn(state: *architecture.CpuState) void {
const image = ramdisk_image orelse return fail(state);
const rd = initial_ramdisk.Reader.init(image) orelse return fail(state);
const name = @as([*]const u8, @ptrFromInt(ptr))[0..len];
// Both buffers come in through the checked copy layer, once. The lengths are
// already bounded above, so the staging arrays are small and fixed — and
// because the bytes are now the kernel's own, nothing below can be changed
// by another thread of the caller between validation and use.
var name_storage: [scheduler.maximum_task_name]u8 = undefined;
const name = name_storage[0..@intCast(len)];
if (!user_memory.copyFromUser(t.address_space, ptr, name)) return failErr(state, ipc.EFAULT);
var argument_storage: [maximum_argument_bytes]u8 = undefined;
const arguments = argument_storage[0..@intCast(arguments_len)];
if (arguments_len != 0 and !user_memory.copyFromUser(t.address_space, arguments_ptr, arguments)) return failErr(state, ipc.EFAULT);
// Exact path first, basename fallback second; either way argv[0] (and hence
// the task name, and the log ring's attribution) is the stored full path.
const item = rd.find(name) orelse return fail(state); // no bundled binary by that name
@@ -977,8 +1014,7 @@ fn systemSpawn(state: *architecture.CpuState) void {
argv[0] = item.name;
var argc: usize = 1;
if (arguments_len != 0) {
const blob = @as([*]const u8, @ptrFromInt(arguments_ptr))[0..arguments_len];
var pieces = std.mem.tokenizeScalar(u8, blob, 0);
var pieces = std.mem.tokenizeScalar(u8, arguments, 0);
while (pieces.next()) |piece| {
if (argc == maximum_arguments) return fail(state);
argv[argc] = piece;
@@ -1129,8 +1165,27 @@ fn systemProcessEnumerate(state: *architecture.CpuState) void {
if (t.address_space == 0 or buffer_ptr >= user_half_end) return fail(state);
const sz = @sizeOf(abi.ProcessDescriptor);
const cap = @min(maximum, (user_half_end - buffer_ptr) / sz); // clamp to the user half
const out: [*]abi.ProcessDescriptor = @ptrFromInt(buffer_ptr);
architecture.setSystemCallResult(state, scheduler.enumerate(out[0..@intCast(cap)]));
// The scheduler describes a chunk of the table into kernel memory, then
// `copyToUser` places it — the kernel never stores through the user pointer.
// Once the caller's buffer is full the walk continues with an empty chunk,
// because the result is the true live count, not what fitted.
var chunk: [8]abi.ProcessDescriptor = undefined;
var copied: u64 = 0;
var total: u64 = 0;
var cursor: usize = 0;
while (true) {
const room: []abi.ProcessDescriptor = if (copied < cap) chunk[0..@intCast(@min(chunk.len, cap - copied))] else chunk[0..0];
const found = scheduler.enumerateFrom(&cursor, room);
total += found.live;
if (found.filled != 0) {
const bytes = std.mem.sliceAsBytes(chunk[0..found.filled]);
if (!user_memory.copyToUser(t.address_space, buffer_ptr + copied * sz, bytes)) return failErr(state, ipc.EFAULT);
copied += found.filled;
}
if (found.done) break;
}
architecture.setSystemCallResult(state, total);
}
/// process_kill(id) -> 0 / -ESRCH / -EPERM: end the process `id`. Only its
@@ -1682,9 +1737,10 @@ fn systemIrqAck(state: *architecture.CpuState) void {
/// The pointer must lie in the user (low) half, so kernel addresses and
/// non-canonical values fall outside it and the read below can't be steered at
/// kernel data. Length is checked first so the upper-bound add can't overflow.
/// Known gap (fine for trusted user code): a pointer into an *unmapped* hole in
/// the user half passes the check and the read #PFs -> on_fault halts — a
/// self-DoS, not an isolation break. Fault-recovering copy-in is a later item.
/// The message then comes in ONCE through the checked copy layer: an unmapped
/// hole in the user half is -EFAULT rather than a kernel fault, and the bytes the
/// log stamps are the same bytes that were validated (the old code read the user
/// buffer twice — once to stage it, once again inside `log.append`).
///
/// The emit runs under the kernel lock, so a message is atomic on the wire — two
/// processes writing from different cores can interleave *messages*, never bytes.
@@ -1697,7 +1753,6 @@ fn systemDebugWrite(state: *architecture.CpuState) void {
const len = architecture.systemCallArg(state, 1);
const level_raw = architecture.systemCallArg(state, 2);
if (len <= write_buffer.len and ptr < user_half_end and ptr + len <= user_half_end) {
const source: [*]const u8 = @ptrFromInt(ptr);
// Levels above the enum range clamp to raw — old two-arg callers land
// there naturally (garbage in arg 2 stays harmless).
const level: abi.KlogLevel = if (level_raw <= @intFromEnum(abi.KlogLevel.raw))
@@ -1707,14 +1762,16 @@ fn systemDebugWrite(state: *architecture.CpuState) void {
const t = scheduler.current();
const flags = sync.enter();
defer sync.leave(flags);
@memcpy(write_buffer[0..len], source[0..len]); // keep the latest message
// One copy in, under the lock; `write_buffer` (the latest message, which
// the kernel tests assert on) doubles as the staging buffer the log reads.
if (!user_memory.copyFromUser(t.address_space, ptr, write_buffer[0..len])) return failErr(state, ipc.EFAULT);
write_len = len;
write_from_user = architecture.fromUser(state);
write_count += 1;
// The kernel stamps the sender's identity — attribution is structural,
// not a prefix convention the payload could forge (and it is stamped
// per line inside log.append).
log.append(t.id, t.name(), level, source[0..len]);
log.append(t.id, t.name(), level, write_buffer[0..len]);
architecture.setSystemCallResult(state, len);
} else {
fail(state);
@@ -1729,18 +1786,35 @@ fn systemDebugWrite(state: *architecture.CpuState) void {
/// [KlogRecordHeader][name][message] frames out of the byte stream (abi.zig).
///
/// The mirror of `debug_write`: the same overflow-safe user-half bounds check,
/// but the copy runs kernel -> user, under the log lock (inside log.readAt) so
/// the stream can't move underneath the copy. A read-only diagnostic.
/// but the copy runs kernel -> user. The ring is drained a chunk at a time into a
/// kernel staging buffer (each chunk read under the log lock, so the stream can't
/// move underneath it) and each chunk is then placed with `copyToUser` — a
/// reader may ask for a megabyte, and the kernel stack is 16 KiB. A partial
/// result is honest: the reader advances its cursor by what it got. A read-only
/// diagnostic.
fn systemKlogRead(state: *architecture.CpuState) void {
const offset = architecture.systemCallArg(state, 0);
const ptr = architecture.systemCallArg(state, 1);
const len = architecture.systemCallArg(state, 2);
const t = scheduler.current();
// Confine the whole destination span to the user (low) half. `len <=
// user_half_end - ptr` bounds the length without an overflowing add.
if (ptr < user_half_end and len <= user_half_end - ptr) {
const dest: [*]u8 = @ptrFromInt(ptr);
const n = log.readAt(offset, dest[0..len]) orelse return fail(state);
architecture.setSystemCallResult(state, n);
var chunk: [512]u8 = undefined;
var done: u64 = 0;
while (done < len) {
const want = @min(@as(u64, chunk.len), len - done);
const n = log.readAt(offset + done, chunk[0..@intCast(want)]) orelse {
// The cursor fell behind the ring's tail mid-drain. What was
// already placed stands; only a first-chunk miss fails the call.
if (done == 0) return fail(state);
break;
};
if (n == 0) break; // caught up
if (!user_memory.copyToUser(t.address_space, ptr + done, chunk[0..n])) return failErr(state, ipc.EFAULT);
done += n;
}
architecture.setSystemCallResult(state, done);
} else {
fail(state);
}
@@ -1753,10 +1827,10 @@ fn systemKlogRead(state: *architecture.CpuState) void {
fn systemKlogStatus(state: *architecture.CpuState) void {
const ptr = architecture.systemCallArg(state, 0);
const size = @sizeOf(abi.KlogStatus);
const t = scheduler.current();
if (ptr < user_half_end and size <= user_half_end - ptr) {
var status = log.status();
const dest: [*]u8 = @ptrFromInt(ptr);
@memcpy(dest[0..size], std.mem.asBytes(&status)[0..size]);
if (!user_memory.copyValueToUser(t.address_space, ptr, &status)) return failErr(state, ipc.EFAULT);
architecture.setSystemCallResult(state, 0);
} else {
fail(state);
@@ -1775,10 +1849,12 @@ fn systemFsResolve(state: *architecture.CpuState) void {
const flags = architecture.systemCallArg(state, 2);
const out_ptr = architecture.systemCallArg(state, 3);
const out_cap = architecture.systemCallArg(state, 4);
if (path_len == 0 or path_len > 224 or path_ptr >= user_half_end or path_ptr + path_len > user_half_end) return fail(state);
if (out_cap != 0 and (out_ptr >= user_half_end or out_ptr + out_cap > user_half_end)) return fail(state);
const path = @as([*]const u8, @ptrFromInt(path_ptr))[0..path_len];
if (path_len == 0 or path_len > maximum_resolve_path or path_ptr >= user_half_end or path_ptr + path_len > user_half_end) return fail(state);
if (out_cap != 0 and !user_memory.userRangeOk(out_ptr, @intCast(out_cap))) return fail(state);
const t = scheduler.current();
var path_storage: [maximum_resolve_path]u8 = undefined;
const path = path_storage[0..@intCast(path_len)];
if (!user_memory.copyFromUser(t.address_space, path_ptr, path)) return failErr(state, ipc.EFAULT);
const flags_lock = sync.enter();
defer sync.leave(flags_lock);
@@ -1790,14 +1866,15 @@ fn systemFsResolve(state: *architecture.CpuState) void {
.backend => |*backend| {
// The rewritten path goes back in the out buffer behind a u16
// length prefix (a third result register would collide with r8's
// argument role in the userspace stub).
// argument role in the userspace stub). Both halves are placed with
// the checked copy, and *before* the handle is installed, so an
// -EFAULT never strands a capability in the caller's table.
if (backend.path_len + 2 > out_cap) return fail(state);
const prefix = [2]u8{ @intCast(backend.path_len & 0xFF), @intCast(backend.path_len >> 8) };
if (!user_memory.copyToUser(t.address_space, out_ptr, &prefix)) return failErr(state, ipc.EFAULT);
if (!user_memory.copyToUser(t.address_space, out_ptr + 2, backend.path[0..backend.path_len])) return failErr(state, ipc.EFAULT);
const handle = ipc.installHandleDeduped(t, backend.endpoint);
if (handle < 0) return fail(state);
const destination: [*]u8 = @ptrFromInt(out_ptr);
destination[0] = @intCast(backend.path_len & 0xFF);
destination[1] = @intCast(backend.path_len >> 8);
@memcpy(destination[2..][0..backend.path_len], backend.path[0..backend.path_len]);
architecture.setSystemCallResult(state, abi.fs_route_backend);
architecture.setSystemCallResult2(state, @intCast(handle));
},
@@ -1809,6 +1886,12 @@ fn systemFsResolve(state: *architecture.CpuState) void {
/// kernel-backed node. read copies file bytes; status copies a FileAttributes;
/// readdir copies [DirectoryEntryHeader][name] for the `offset`th child. Reads
/// of the immutable initrd never take the kernel lock.
///
/// Every result reaches the caller through `copyToUser`, never a store through
/// the user pointer. `read` stages the file bytes a chunk at a time — a caller
/// may ask for the 64 KiB ceiling, which no kernel stack could hold — so a
/// mid-way -EFAULT is possible; the call fails and the already-placed prefix is
/// meaningless, exactly as a failed read should be.
fn systemFsNode(state: *architecture.CpuState) void {
const operation = architecture.systemCallArg(state, 0);
const node_token = architecture.systemCallArg(state, 1);
@@ -1817,16 +1900,24 @@ fn systemFsNode(state: *architecture.CpuState) void {
const buf_len = architecture.systemCallArg(state, 4);
if (buf_ptr >= user_half_end or buf_len > user_half_end - buf_ptr) return fail(state);
const capped = @min(buf_len, 64 * 1024); // bound any single copy
const destination: [*]u8 = @ptrFromInt(buf_ptr);
const t = scheduler.current();
switch (operation) {
abi.fs_node_read => {
const n = vfs.nodeRead(node_token, offset, destination[0..capped]) orelse return fail(state);
architecture.setSystemCallResult(state, n);
var chunk: [512]u8 = undefined;
var done: u64 = 0;
while (done < capped) {
const want = @min(@as(u64, chunk.len), capped - done);
const n = vfs.nodeRead(node_token, offset + done, chunk[0..@intCast(want)]) orelse return fail(state);
if (n == 0) break; // end of file
if (!user_memory.copyToUser(t.address_space, buf_ptr + done, chunk[0..n])) return failErr(state, ipc.EFAULT);
done += n;
}
architecture.setSystemCallResult(state, done);
},
abi.fs_node_status => {
var attributes = vfs.nodeStatus(node_token) orelse return fail(state);
if (capped < @sizeOf(abi.FileAttributes)) return fail(state);
@memcpy(destination[0..@sizeOf(abi.FileAttributes)], std.mem.asBytes(&attributes));
if (!user_memory.copyValueToUser(t.address_space, buf_ptr, &attributes)) return failErr(state, ipc.EFAULT);
architecture.setSystemCallResult(state, @sizeOf(abi.FileAttributes));
},
abi.fs_node_readdir => {
@@ -1837,10 +1928,13 @@ fn systemFsNode(state: *architecture.CpuState) void {
architecture.setSystemCallResult(state, 0); // past the end
return;
};
var header = result.header;
// Header and name are staged contiguously so one entry is one copy.
var entry: [@sizeOf(abi.DirectoryEntryHeader) + name_buffer.len]u8 = undefined;
const header = result.header;
const total = header_size + @min(result.name_len, capped - header_size);
@memcpy(destination[0..header_size], std.mem.asBytes(&header));
@memcpy(destination[header_size..total], name_buffer[0 .. total - header_size]);
@memcpy(entry[0..header_size], std.mem.asBytes(&header));
@memcpy(entry[header_size..total], name_buffer[0 .. total - header_size]);
if (!user_memory.copyToUser(t.address_space, buf_ptr, entry[0..total])) return failErr(state, ipc.EFAULT);
architecture.setSystemCallResult(state, total);
},
else => fail(state),
@@ -1857,12 +1951,19 @@ fn systemFsMount(state: *architecture.CpuState) void {
const backend_handle = architecture.systemCallArg(state, 2);
const rewrite_ptr = architecture.systemCallArg(state, 3);
const rewrite_len = architecture.systemCallArg(state, 4);
if (prefix_len == 0 or prefix_len > 64 or prefix_ptr >= user_half_end or prefix_ptr + prefix_len > user_half_end) return fail(state);
if (rewrite_len > 32) return fail(state);
if (prefix_len == 0 or prefix_len > maximum_mount_prefix or prefix_ptr >= user_half_end or prefix_ptr + prefix_len > user_half_end) return fail(state);
if (rewrite_len > maximum_mount_rewrite) return fail(state);
if (rewrite_len != 0 and (rewrite_ptr >= user_half_end or rewrite_ptr + rewrite_len > user_half_end)) return fail(state);
const t = scheduler.current();
const prefix = @as([*]const u8, @ptrFromInt(prefix_ptr))[0..prefix_len];
const rewrite = if (rewrite_len == 0) "" else @as([*]const u8, @ptrFromInt(rewrite_ptr))[0..rewrite_len];
// Both strings come in through the checked copy; `mountBackend` copies them
// again into the mount table, so these staging buffers only need to outlive
// this call.
var prefix_storage: [maximum_mount_prefix]u8 = undefined;
const prefix = prefix_storage[0..@intCast(prefix_len)];
if (!user_memory.copyFromUser(t.address_space, prefix_ptr, prefix)) return failErr(state, ipc.EFAULT);
var rewrite_storage: [maximum_mount_rewrite]u8 = undefined;
const rewrite = rewrite_storage[0..@intCast(rewrite_len)];
if (rewrite_len != 0 and !user_memory.copyFromUser(t.address_space, rewrite_ptr, rewrite)) return failErr(state, ipc.EFAULT);
const flags = sync.enter();
defer sync.leave(flags);
@@ -1879,8 +1980,11 @@ fn systemFsMount(state: *architecture.CpuState) void {
fn systemFsUnmount(state: *architecture.CpuState) void {
const prefix_ptr = architecture.systemCallArg(state, 0);
const prefix_len = architecture.systemCallArg(state, 1);
if (prefix_len == 0 or prefix_len > 64 or prefix_ptr >= user_half_end or prefix_ptr + prefix_len > user_half_end) return fail(state);
const prefix = @as([*]const u8, @ptrFromInt(prefix_ptr))[0..prefix_len];
if (prefix_len == 0 or prefix_len > maximum_mount_prefix or prefix_ptr >= user_half_end or prefix_ptr + prefix_len > user_half_end) return fail(state);
const t = scheduler.current();
var prefix_storage: [maximum_mount_prefix]u8 = undefined;
const prefix = prefix_storage[0..@intCast(prefix_len)];
if (!user_memory.copyFromUser(t.address_space, prefix_ptr, prefix)) return failErr(state, ipc.EFAULT);
const flags = sync.enter();
defer sync.leave(flags);
if (!vfs.unmount(prefix)) return fail(state);