Pass argv to processes on a SysV entry stack; grow the user stack to 32 KiB

Processes now start with C-compatible arguments: the kernel builds the
System V AMD64 entry block (argc, argv, empty envp, auxiliary vector)
at the top of the stack, argv[0] is the path or initial-ramdisk name
the process was spawned as, and system_spawn carries an optional
NUL-separated blob that becomes argv[1..]. The runtime parses the block
(runtime.argumentCount/argument) and its spawn wrappers pass arguments
through. The name is also recorded on the task, so a fault report says
which binary died, not just its id.

The user stack grows from one page to eight (32 KiB,
parameters.user_stack_pages), with the page below left unmapped as a
guard so an overflow faults into a clean process kill rather than
corrupting the image. Task.name_buffer is zero-initialised, not
undefined: an undefined default is materialised as a 0xAA fill that
moved the static task pool out of .bss and made the whole kernel ~7x
slower under QEMU TCG (caught by the affinity test).

Proven end to end by the new args test: args-echo respawns itself with
arguments via the syscall blob, burns more stack than one page could
hold, and echoes its argv intact. Full suite: 44/44.
This commit is contained in:
Daniel Samson
2026-07-11 08:33:12 +01:00
parent 6b3ae0c997
commit a5fe63c1dd
14 changed files with 385 additions and 50 deletions
+140 -21
View File
@@ -24,6 +24,7 @@ const elf = std.elf;
const boot_handoff = @import("boot-handoff");
const abi = @import("abi");
const device_abi = @import("device-abi");
const parameters = @import("parameters");
const architecture = @import("architecture");
const pmm = @import("pmm.zig");
const scheduler = @import("scheduler.zig");
@@ -40,9 +41,18 @@ const SystemCall = abi.SystemCall;
/// User virtual addresses. PML4 index 224 — a user-exclusive region, far from
/// the identity map (low indices) and the vmm test address (index 128), so
/// setting the U/S bit on its intermediate tables widens no kernel mapping.
/// An ELF image may occupy [code_virtual, stack_virtual); the stack page sits above.
/// An ELF image may occupy [code_virtual, stack_virtual); the stack sits above.
pub const code_virtual: u64 = 0x0000_7000_0000_0000;
pub const stack_virtual: u64 = 0x0000_7000_0020_0000;
/// The stack region, above the image. The page at `stack_virtual` is **never
/// mapped** — it is the guard page: a process that overflows its stack walks into
/// it and faults (killing only that process) rather than silently corrupting the
/// top of its own image. The stack proper is `parameters.user_stack_pages` pages
/// at [stack_base_virtual, stack_top_virtual), RW + NX, with the System V entry
/// block (argc/argv) at the very top.
pub const stack_virtual: u64 = 0x0000_7000_0020_0000; // guard page (unmapped)
pub const stack_base_virtual: u64 = stack_virtual + page_size;
pub const stack_top_virtual: u64 = stack_base_virtual + parameters.user_stack_pages * page_size;
/// The mmap grant arena: where `mmap` hands out fresh user pages, above the image
/// and stack but still inside PML4[224] (so no kernel mapping is widened). Each
@@ -73,6 +83,19 @@ pub const dma_arena_end: u64 = dma_arena_base + (256 << 20); // 256 MiB per proc
/// chunks, so this bound is generous; it also caps the frame scratch array below.
const maximum_mmap_pages = 256;
/// Ceiling on a process's argv entries, including argv[0]. Arguments are spawn
/// parameters ("you are the driver for device 12"), not bulk data — IPC carries
/// that — so the bound is small and everything fits the single stack page.
pub const maximum_arguments = 8;
/// Ceiling on the `system_spawn` extra-arguments blob (argv[1..], NUL-separated).
pub const maximum_argument_bytes = 256;
/// Auxiliary-vector entry types (System V AMD64 process entry). Only what the
/// kernel emits today; a C runtime scans the vector until the null terminator.
const auxiliary_vector_null: u64 = 0; // AT_NULL — end of the vector
const auxiliary_vector_page_size: u64 = 6; // AT_PAGESZ
// The hand-assembled user program blob (isr.s, .rodata) — the isolation probe.
const pf_start = @extern([*]const u8, .{ .name = "user_pf_start" });
const pf_end = @extern([*]const u8, .{ .name = "user_pf_end" });
@@ -392,28 +415,50 @@ fn systemDeviceRegister(state: *architecture.CpuState) void {
architecture.setSystemCallResult(state, id);
}
/// system_spawn(name_ptr, name_len) -> 0 on success, -1 on failure. Load the binary
/// bundled in the initial-ramdisk under `name` as a fresh ring-3 process. This is the
/// mechanism a user-space supervisor (the device manager) uses to start a driver it
/// matched: discovery and policy stay in user space, the kernel only spawns.
/// system_spawn(name_ptr, name_len, arguments_ptr, arguments_len) -> 0 on success,
/// -1 on failure. Load the binary bundled in the initial-ramdisk under `name` as a
/// fresh ring-3 process. `name` becomes the child's argv[0] (and its task name, so
/// a fault report can say which binary died); `arguments` is an optional
/// NUL-separated blob that becomes argv[1..] — how a supervisor parameterises what
/// it starts ("you are the driver for device 12"). 0/0 means no extra arguments.
/// This is the mechanism a user-space supervisor (the device manager) uses to start
/// a driver it matched: discovery and policy stay in user space, the kernel only
/// spawns.
///
/// Ungated for now — any process may spawn any bundled binary. A capability (only a
/// supervisor holds the right to spawn) belongs here once the model grows one; see
/// docs/driver-model.md. The name is bounds-checked into the user half exactly like
/// `debug_write`, and an unknown name or a load failure returns -1.
/// docs/driver-model.md. Both buffers are bounds-checked into the user half exactly
/// like `debug_write`, and an unknown name or a load failure returns -1.
fn systemSpawn(state: *architecture.CpuState) void {
const ptr = architecture.systemCallArg(state, 0);
const len = architecture.systemCallArg(state, 1);
const arguments_ptr = architecture.systemCallArg(state, 2);
const arguments_len = architecture.systemCallArg(state, 3);
if (len == 0 or len > 64 or ptr >= user_half_end or ptr + len > user_half_end) return fail(state);
if (arguments_len > maximum_argument_bytes) return fail(state);
if (arguments_len != 0 and (arguments_ptr >= user_half_end or arguments_ptr + arguments_len > user_half_end)) return fail(state);
const image = ramdisk_image orelse return fail(state);
const rd = initial_ramdisk.Reader.init(image) orelse return fail(state);
const name = @as([*]const u8, @ptrFromInt(ptr))[0..len];
var argv: [maximum_arguments][]const u8 = undefined;
argv[0] = name;
var argc: usize = 1;
if (arguments_len != 0) {
const blob = @as([*]const u8, @ptrFromInt(arguments_ptr))[0..arguments_len];
var pieces = std.mem.tokenizeScalar(u8, blob, 0);
while (pieces.next()) |piece| {
if (argc == maximum_arguments) return fail(state);
argv[argc] = piece;
argc += 1;
}
}
var i: u32 = 0;
while (i < rd.count) : (i += 1) {
const item = rd.entry(i) orelse continue;
if (!std.mem.eql(u8, item.name, name)) continue;
spawnProcess(item.blob, 4) catch return fail(state);
spawnProcess(item.blob, 4, argv[0..argc]) catch return fail(state);
architecture.setSystemCallResult(state, 0);
return;
}
@@ -641,15 +686,15 @@ pub fn run(blob: []const u8) RunError!void {
@memset(code[blob.len..page_size], 0xCC);
architecture.mapUserPage(code_virtual, code_frame, false, true); // RO + X
architecture.mapUserPage(stack_virtual, stack_frame, true, false); // RW + NX
architecture.mapUserPage(stack_base_virtual, stack_frame, true, false); // RW + NX (one page; the probe barely stacks)
resetRecords();
architecture.enterUser(scheduler.currentCpuIndex(), code_virtual, stack_virtual + page_size);
architecture.enterUser(scheduler.currentCpuIndex(), code_virtual, stack_base_virtual + page_size);
// Back via the exit system_call; the interrupt gate left IF clear.
architecture.enableInterrupts();
architecture.unmapPage(code_virtual);
architecture.unmapPage(stack_virtual);
architecture.unmapPage(stack_base_virtual);
pmm.free(code_frame);
pmm.free(stack_frame);
}
@@ -660,6 +705,7 @@ pub const InitError = error{
BadElf, // malformed/inapplicable image (magic, class, machine, type, bounds)
BadSegment, // PT_LOAD unaligned, out of the user region, W&X, or overlapping
BadEntry, // e_entry not inside an executable segment
BadArguments, // no argv[0], too many entries, or too many bytes for the entry stack
ProgramTooBig, // more pages than the loader's budget
OutOfMemory,
};
@@ -760,13 +806,74 @@ fn loadPageInto(aspace: u64, image: []const u8, seg: Segment, page_index: u64) I
architecture.mapUserPageInto(aspace, seg.vaddr + page_off, frame, seg.writable, seg.executable);
}
/// Build the System V AMD64 process-entry block at the top of a process's stack
/// page and return the initial user stack pointer. At the first user instruction,
/// rsp is 16-byte aligned and points at (addresses growing upward):
///
/// argc, argv[0..argc-1], NULL, NULL (empty envp), auxiliary vector, strings
///
/// — the layout every C runtime's startup code walks, so danos's own runtime and a
/// future libc port read arguments identically (docs/sysv.md). `page` is the kernel
/// (physmap) view of the stack's **top** frame and `page_user_base` that frame's
/// user address (stack_top_virtual - page_size); the pointers written into it are
/// user addresses inside that page. The caller has validated the sizes
/// (`entryStackBytes`), so this cannot overrun.
fn buildEntryStack(page: [*]u8, page_user_base: u64, argv: []const []const u8) u64 {
// The strings live at the very top of the page, packed from the end downward.
var string_offset: usize = page_size;
var pointers: [maximum_arguments]u64 = undefined;
var i: usize = argv.len;
while (i > 0) {
i -= 1;
string_offset -= argv[i].len + 1;
@memcpy(page[string_offset..][0..argv[i].len], argv[i]);
page[string_offset + argv[i].len] = 0; // NUL-terminated, as C expects
pointers[i] = page_user_base + string_offset;
}
// The vector sits below the strings: argc, the argv pointers, the argv
// terminator, an empty envp (terminator only), then the auxiliary vector.
const word_count = 1 + argv.len + 1 + 1 + 4;
const vector_offset = (string_offset - word_count * 8) & ~@as(usize, 15); // entry rsp % 16 == 0
const words: [*]u64 = @ptrCast(@alignCast(page + vector_offset));
var w: usize = 0;
words[w] = argv.len; // argc
w += 1;
for (pointers[0..argv.len]) |pointer| {
words[w] = pointer;
w += 1;
}
words[w] = 0; // argv terminator
words[w + 1] = 0; // envp: no environment yet, just the terminator
words[w + 2] = auxiliary_vector_page_size;
words[w + 3] = page_size;
words[w + 4] = auxiliary_vector_null; // end of the auxiliary vector
words[w + 5] = 0;
return page_user_base + vector_offset;
}
/// Bytes the entry block for `argv` occupies at the top of the stack page:
/// strings (each NUL-terminated), vector words, and the alignment slack.
fn entryStackBytes(argv: []const []const u8) usize {
var string_bytes: usize = 0;
for (argv) |argument| string_bytes += argument.len + 1;
return string_bytes + (1 + argv.len + 1 + 1 + 4) * 8 + 16;
}
/// Load a user ELF image into a fresh address space and spawn it as a scheduled
/// ring-3 process at `priority`. Returns immediately — the process runs
/// preemptively on its own page tables alongside everything else, and its exit
/// is handled by the system_call layer. The whole build (address space + ELF load +
/// task) runs under the kernel lock so it appears atomically and can't race
/// pmm/heap on another core.
pub fn spawnProcess(image: []const u8, priority: u3) InitError!void {
/// ring-3 process at `priority`, entered with `argv` on its stack per the System V
/// convention (`buildEntryStack`). `argv[0]` is required — it names the process:
/// the path or initial-ramdisk name it was spawned as. It is also recorded on the
/// task, so a fault report can say *which* binary died, not just its id.
/// Returns immediately — the process runs preemptively on its own page tables
/// alongside everything else, and its exit is handled by the system_call layer.
/// The whole build (address space + ELF load + task) runs under the kernel lock so
/// it appears atomically and can't race pmm/heap on another core.
pub fn spawnProcess(image: []const u8, priority: u3, argv: []const []const u8) InitError!void {
if (argv.len == 0 or argv.len > maximum_arguments) return error.BadArguments;
// The entry block must leave most of the page as actual stack.
if (entryStackBytes(argv) > page_size / 2) return error.BadArguments;
var segs: [maximum_segments]Segment = undefined;
const parsed = try parseSegments(image, &segs);
@@ -779,10 +886,22 @@ pub fn spawnProcess(image: []const u8, priority: u3) InitError!void {
for (segs[0..parsed.count]) |seg| {
for (0..seg.pages()) |i| try loadPageInto(aspace, image, seg, i);
}
const stack_frame = pmm.alloc() orelse return error.OutOfMemory;
architecture.mapUserPageInto(aspace, stack_virtual, stack_frame, true, false); // RW + NX
if (!scheduler.spawnUserLocked(aspace, parsed.entry, stack_virtual + page_size, priority))
// The stack: `user_stack_pages` zeroed pages below stack_top_virtual, RW + NX.
// The page below them (`stack_virtual`) stays unmapped as the overflow guard.
// The entry block goes at the top of the highest page.
var user_sp: u64 = 0;
for (0..parameters.user_stack_pages) |i| {
const stack_frame = pmm.alloc() orelse return error.OutOfMemory;
const stack_page: [*]u8 = @ptrFromInt(boot_handoff.physicalToVirtual(stack_frame));
@memset(stack_page[0..page_size], 0); // no stale frame contents leak into user space
const page_virtual = stack_base_virtual + i * page_size;
if (i == parameters.user_stack_pages - 1)
user_sp = buildEntryStack(stack_page, page_virtual, argv);
architecture.mapUserPageInto(aspace, page_virtual, stack_frame, true, false); // RW + NX
}
if (!scheduler.spawnUserLocked(aspace, parsed.entry, user_sp, priority, argv[0]))
return error.OutOfMemory;
}