do not edit — generated by btf.
git.druid.rocksindexdruid520kaboomsrc/kernel/exec.nsc

src/kernel/exec.nsc


/*
 * exec: the syscall a shell actually needs to run anything. wraps the
 * kfs_dir_find + kfs_read_file + elf parsing/vmm mapping +
 * elf_call_entry sequence, exposed as syscall 7 so userspace (the
 * shell) can do it too -- those are all kernel-internal, not
 * syscalls, so a userspace program has no other way to load and run
 * something.
 *
 * elf_call_entry's own asm does `call *rax; ret` -- whatever the
 * loaded program leaves in %rax when it ret's becomes elf_call_entry's
 * own return value too (nothing clobbers rax in between), so this
 * genuinely returns the child's "exit code" (a c-style int main(void)
 * return value, since there's no separate exit syscall -- see
 * src/user/crt.s's note on why returning already is exiting).
 *
 * the process table (kaboom_proc_depth and kaboom_proc_pid/
 * kaboom_proc_name/kaboom_proc_pml4, exec.nsh): the honest answer to
 * "what even is a process here" for a kernel with no pcb and no
 * scheduler -- "the process IS the call stack" (elf_call_entry does
 * `call *rax`, sys_exec doesn't return to ITS caller until the child
 * returns). that's not nothing, though: it's a real, if shallow,
 * parent/child tree, exactly the shape a foreground-only shell has on
 * a real os too (the parent is genuinely blocked in a syscall while
 * its one child runs). kmain's own boot-time sys_exec("sh") is the
 * first process (depth 0->1, slot 0).
 *
 * only sh ever calls exec again from there -- but "only sh does" isn't
 * "only once": a running sh can itself exec a NESTED sh (the user
 * typing `sh` at the prompt, or a `sh script` shebang chain), which can
 * in turn exec a command, and so on. the table used to assume depth
 * never passed 2 and reused its one "child" slot for every push past
 * that, which silently lost a nested sh's own identity from /proc/ps
 * the moment IT execed anything (found live: `sh` from an interactive
 * shell, then `cat /proc/ps` from inside it, showed only the outer sh
 * and cat -- the nested sh's own row never appeared).
 *
 * every exec now gets a genuinely separate address space (vmm.nsc):
 * sys_exec builds a fresh pml4 (vmm_create_addrspace), maps+copies
 * every PT_LOAD segment into it one 4kib frame at a time
 * (vmm_map_page over vmm_alloc_frame's pool), switches cr3 into it for
 * the duration of elf_call_entry, then switches cr3 back and tears the
 * whole address space down (vmm_destroy_addrspace) once the child
 * returns -- no shared physical window, no remapping trick, each
 * process's code/data really is its own physical memory. the real
 * per-process frame pool is now big enough (thousands of 4kib frames,
 * not the old shared window's 8) that it's no longer the thing that
 * actually stops nesting too deep -- the process table's own 10 slots
 * (kaboom_proc_pid/kaboom_proc_name/kaboom_proc_pml4, exec.nsh) are:
 * sys_exec checks kaboom_proc_depth against that real limit itself
 * before ever pushing (see the body below), since proc_push has no
 * bounds check of its own and a 10-process-deep stack really is this
 * table's whole real capacity, not a soft guess. /proc/self and
 * /proc/ps (virtfs.nsc) are what read this.
 *
 * shebang scripts: if the exec'd file isn't a valid elf but starts
 * with "#!", sys_exec parses the interpreter path off that first line
 * and re-execs IT instead, with argv shifted the same way a real
 * unix execve() does (interpreter, then the script's own path, then
 * whatever args the original call had past its own argv[0]) -- see
 * shebang_parse_interp/the sys_exec body below. kaboom's only real
 * interpreter is sh itself, and a shebang script naming sh is exec'd
 * from WITHIN an already-running sh -- that recursive sys_exec call
 * gets its own real address space exactly like any other process, no
 * special case needed anymore. no real cycle detection, but a hard
 * nesting limit: at most 8 shebang re-execs can be live on the call
 * stack at once (exec_shebang_depth below), and the 9th fails the exec
 * cleanly. an earlier version had no limit at all, on the theory that
 * a chain pointing back to itself would just "recurse until the stack
 * runs out" -- which is exactly what it did, for real: a script whose
 * own shebang line named itself ran the shared 64kib kernel stack
 * (boot.s) straight down into the live page tables below it and
 * triple-faulted the machine. 8 is far deeper than any real script
 * chain nests.
 */
 
include "../fs/kfs.nsh";
include "vmm.nsh";
include "exec.nsh";
include "alloc.nsh";
 
global struct arena fs_colosseum;
global ptr exec_filebuf;
 
i32 elf_validate(ptr buf);
u16 elf_read_u16(ptr buf, u64 off);
u32 elf_read_u32(ptr buf, u64 off);
u64 elf_read_u64(ptr buf, u64 off);
i32 elf_segments_ok(ptr buf, u64 sz, u64 e_entry, u64 e_phoff, u16 e_phentsize, u16 e_phnum);
void load_cr3(u64 pml4_phys);
u32 fd_open_mask(void);
void fd_close_unless(u32 keep);
i64 elf_call_entry(ptr entry, i64 argc, ptr argv);
 
u8
ptr_byte_at(ptr base, u64 index)
{
	ptr p;
	u64 word;
	p = base + index;
	word = (u64)*p;
	return (u8)(word & (u64)0xff);
}
 
/* copies up to 23 bytes of path into dst (a 24-byte buffer, room for
 * a null terminator) -- a read-modify-write per byte, same technique
 * fd_read_stdin/virtfs.nsc's own put_byte already use, since dst here
 * is a bare ptr, not a typed array nscc could index directly. longer
 * names are silently truncated: this is a display label for ps, not a
 * real identifier anything looks back up. */
void
proc_copy_name(ptr dst, ptr path, u64 path_len)
{
	u64 i;
	u64 n;
	u64 wordoff;
	u64 byteoff;
	u64 word;
	u8 ch;
 
	n = path_len;
	if(n > (u64)23)
	{
		n = (u64)23;
	}
 
	i = 0;
	while(i <= n)
	{
		if(i < n)
		{
			ch = ptr_byte_at(path, i);
		}
		else
		{
			ch = 0; /* null terminator */
		}
		wordoff = i & ~(u64)7;
		byteoff = i & (u64)7;
		word = (u64)*(dst + wordoff);
		word = word & ~((u64)0xff << (byteoff * (u64)8));
		word = word | ((u64)ch << (byteoff * (u64)8));
		*(dst + wordoff) = (i64)word;
		i = i + (u64)1;
	}
}
 
/* reads the i-th slot of a raw argv array (8 bytes per slot, plain
 * ptr-sized values) -- userspace has lib.nsc's own argv_get for
 * exactly this; this is the kernel-side duplicate, same reasoning as
 * every other tiny helper this file/kfs.nsc/virtfs.nsc separately
 * keep their own copy of. */
ptr
exec_argv_get(ptr argv, i32 i)
{
	return (ptr)(u64)*(argv + (u64)i * (u64)8);
}
 
/* parses "#!<interpreter path>" off the first line of a shebang
 * script (buf[0..1] already confirmed to be '#'/'!' by the caller),
 * skipping leading whitespace after the "#!" and stopping at the
 * first whitespace/newline or eof -- no shebang-line argument support
 * (e.g. "#!/bin/sh -x"), matching "super simple, just commands".
 * writes into out (room for 64 bytes) and returns the length, or 0 if
 * there was no real interpreter name to find (an empty or all-
 * whitespace shebang line). */
u64
shebang_parse_interp(ptr buf, u64 sz, ptr out)
{
	u64 i;
	u64 j;
	u64 wordoff;
	u64 byteoff;
	u64 word;
	u8 ch;
 
	i = 2; /* skip "#!" */
	while(i < sz && (ptr_byte_at(buf, i) == (u8)32 || ptr_byte_at(buf, i) == (u8)9))
	{
		i = i + (u64)1;
	}
 
	j = 0;
	while(i < sz && j < (u64)63)
	{
		ch = ptr_byte_at(buf, i);
		if(ch == (u8)10 || ch == (u8)32 || ch == (u8)9 || ch == (u8)13)
		{
			i = sz; /* stop the outer while too -- no break in this codebase's own style */
		}
		else
		{
			wordoff = j & ~(u64)7;
			byteoff = j & (u64)7;
			word = (u64)*(out + wordoff);
			word = word & ~((u64)0xff << (byteoff * (u64)8));
			word = word | ((u64)ch << (byteoff * (u64)8));
			*(out + wordoff) = (i64)word;
			j = j + (u64)1;
			i = i + (u64)1;
		}
	}
	wordoff = j & ~(u64)7;
	byteoff = j & (u64)7;
	word = (u64)*(out + wordoff);
	word = word & ~((u64)0xff << (byteoff * (u64)8));
	*(out + wordoff) = (i64)word; /* null terminator */
 
	return j;
}
 
/* how many shebang re-execs are live on the call stack right now --
 * see this file's own top note. a plain counter, not per-process
 * state: exec nesting is strictly stack-shaped (the process IS the
 * call stack), so incrementing before the recursive sys_exec and
 * decrementing after it returns is always exact. */
global i32 exec_shebang_depth;
 
global void
proc_init(void)
{
	kaboom_next_pid = 1;
	kaboom_proc_depth = 0;
	exec_shebang_depth = 0;
}
 
/* whichever pid is currently deepest on the process stack -- the
 * same "depth>=2 means a running command, else sh itself" logic
 * virtfs.nsc's own /proc/self used to duplicate inline. idt.nsc's
 * sys_doerror (syscall_dispatch) uses this too: whatever's making a
 * syscall right now IS the deepest pid, by construction (a syscall
 * traps from that process's own code, nothing else runs concurrently
 * to have pushed anything deeper in between). */
global u64
kaboom_current_pid(void)
{
	return kaboom_proc_pid[(u64)kaboom_proc_depth - (u64)1];
}
 
/* pushes a new process onto the (at most 10-deep, see this file's own
 * top note on why that's the real, provable ceiling) process stack,
 * returns the depth this call pushed FROM so the matching proc_pop can
 * restore it exactly once the child returns. */
i32
proc_push(ptr path, u64 path_len)
{
	i32 from;
	u64 slot;
 
	from = kaboom_proc_depth;
	slot = (u64)kaboom_proc_depth;
	kaboom_proc_pid[slot] = kaboom_next_pid;
	kaboom_next_pid = kaboom_next_pid + (u64)1;
	proc_copy_name(&kaboom_proc_name[slot * (u64)24], path, path_len);
	kaboom_proc_depth = kaboom_proc_depth + 1;
	return from;
}
 
void
proc_pop(i32 to_depth)
{
	kaboom_proc_depth = to_depth;
}
 
/* a ptr into fs_colosseum, not a stack local or a fixed global array:
 * the buffer stays alive (as far as the stack is concerned) for as
 * long as the child it loads is running, since elf_call_entry's `call
 * *rax` doesn't return until the child does -- exec is not a leaf
 * call, and kaboom's single shared 64kib stack (see boot.s) has no
 * room to spend on a buffer that isn't actually part of any live
 * call frame's own working set. a single, permanent 65536-byte arena
 * (see alloc.nsc/alloc_init), not grown per exec the way this used to
 * be - see this file's own sys_exec for the real reasoning and the
 * rare-oversized-file fallback. */
/* exec_filebuf itself (global ptr, alloc.nsh) is one single,
 * permanent 65536-byte arena, carved once at alloc_init time -- not
 * grown per exec the way it used to be. safe to share across a
 * shebang chain's recursive sys_exec calls for the same reason it
 * always was: the outer call has already parsed everything it needs
 * out of the buffer (interp_path, a stack local) before recursing,
 * and never touches exec_filebuf again after the inner call returns.
 * a file bigger than 65536 bytes (kfs files can be up to 8450048; the
 * dynamite bootloader blob is the one documented real case that big)
 * falls through to a direct, uncounted arena_alloc against
 * fs_colosseum instead, at the call site below. */
 
/* looks up vaddr's existing mapping in pml4_phys's private region,
 * without allocating anything -- returns the already-mapped physical
 * frame, or (u64)0 if vaddr has no mapping yet. mirrors vmm_map_page's
 * own pml4->pdpt->pd->pt walk exactly (same fixed entry-0 shortcut:
 * see vmm_create_addrspace, and the same out-of-range check), but
 * read-only.
 *
 * needed because this project's own userspace link script (user.ld)
 * doesn't page-align its PT_LOAD segments -- deliberately so, back when
 * elf.nsc's old elf_load did a flat memcpy with no paging involved at
 * all (see user.ld's own comment).
 * now that loading goes through real per-4kib-page frames, two
 * adjacent segments (the common case: R+X text/rodata, then R+W bss,
 * sharing the one page their boundary falls inside) would otherwise
 * each separately allocate+zero a FRESH frame for that same shared
 * page -- the later segment's zeroing silently wiping out the earlier
 * segment's already-copied bytes, and orphaning (leaking) the first
 * frame entirely, since the pt entry no longer points at it. checking
 * for an already-mapped page before allocating a new one, and
 * reusing it instead (zeroed once, by whichever segment got there
 * first), is what keeps two segments sharing a page additive instead
 * of destructive. */
u64
exec_page_lookup(u64 pml4_phys, u64 vaddr)
{
	ptr pd;
	u64 pd_index;
	ptr pt_entry_in_pd;
	ptr pt;
	u64 pt_index;
	u64 pte;
 
	if(vaddr < (u64)0x400000 || vaddr >= (u64)0x2000000)
	{
		return (u64)0; /* outside the private region entirely -- same
		                 * bounds vmm_map_page itself enforces */
	}
 
	pd = (ptr)((u64)*(ptr)((u64)*(ptr)pml4_phys & ~(u64)0xfff) & ~(u64)0xfff);
	pd_index = vaddr / (u64)0x200000;
	pt_entry_in_pd = pd + pd_index * (u64)8;
	if(((u64)*pt_entry_in_pd & (u64)1) == (u64)0)
	{
		return (u64)0;
	}
	pt = (ptr)((u64)*pt_entry_in_pd & ~(u64)0xfff);
	pt_index = (vaddr / (u64)4096) % (u64)512;
	pte = (u64)*(pt + pt_index * (u64)8);
	if((pte & (u64)1) == (u64)0)
	{
		return (u64)0;
	}
	return pte & ~(u64)0xfff;
}
 
global i64
sys_exec(ptr path, u64 path_len, i64 argc, ptr argv)
{
	i32 n;
	u64 sz;
	u64 entry;
	i32 from_depth;
	i64 result;
	i8 interp_path[64];
	u64 interp_len;
	ptr new_argv[16];
	i64 new_argc;
	i32 ai;
	i8 resolved_path[80];
	ptr use_path;
	u64 use_path_len;
	i32 joined_len;
	u64 need;
	u64 zi;
	ptr nb;
	u32 fd_keep;
	u64 new_pml4;
	u64 saved_cr3;
	u64 seg_i;
	u64 ph_off_local;
	u64 e_phoff;
	u16 e_phentsize;
	u16 e_phnum;
	u32 p_type;
	u64 p_offset;
	u64 p_vaddr;
	u64 p_filesz;
	u64 p_memsz;
	u64 page_va;
	u64 page_pa;
	u64 copy_off;
	u64 word;
	u64 bi;
	u64 k;
 
	/* kfs_resolve, not kfs_dir_find: a bare name still resolves
	 * against cwd exactly as before, but this also handles a real
	 * absolute or relative multi-component path ("/bin/sh", "a/b") --
	 * needed for real, conventional shebang lines ("#!/bin/sh"), which
	 * a bare-name-only lookup could never have found. /bin is still
	 * the fallback $PATH once cwd/the given path itself doesn't
	 * resolve, kfs_bin_dir_lba resolved once at mount time as always. */
	use_path = path;
	use_path_len = path_len;
	n = kfs_resolve(path, path_len);
	if(n < 0)
	{
		n = kfs_dir_find_in(kfs_bin_dir_lba, path, path_len);
		if(n >= 0)
		{
			/* found only via the $PATH fallback -- the bare name that
			 * worked here is NOT independently resolvable by anything
			 * downstream that doesn't ALSO know about kfs_bin_dir_lba
			 * (fd_open/kfs_resolve, unlike this function, never
			 * consult it). that matters specifically for a shebang: a
			 * script found this way passes its own path on as the
			 * interpreter's argv[1], and the interpreter (sh,
			 * run_script) just calls plain sys_open on it -- which
			 * silently failed to find a name that only ever resolved
			 * via this fallback in the first place. building a real,
			 * independently-resolvable "/bin/<name>" here (kaboom's
			 * $PATH is always exactly /bin, never anything else) fixes
			 * that, and doubles as a more accurate process name for
			 * ps/proc_push below. kfs_join_path (kfs.nsc) does the
			 * actual byte-building -- this used to hand-spell the 5
			 * ascii bytes of "/bin/" out itself, the one place in this
			 * whole kernel that hardcoded a path as raw byte constants
			 * instead of a normal string literal. */
			joined_len = kfs_join_path(&resolved_path[0], (u64)80, "/bin", (u64)4, path, path_len);
			if(joined_len >= 0)
			{
				use_path = &resolved_path[0];
				use_path_len = (u64)joined_len;
			}
		}
	}
	if(n < 0)
	{
		return -1;
	}
	if(kfs_check_perm(n, (u64)4) == 0)
	{
		return -1; /* no execute permission -- applies to a shebang
		            * interpreter re-exec too, same as the original
		            * script/binary: both go through this exact check. */
	}
 
	need = kfs_file_size(n);
	if(need > (u64)8450048)
	{
		return -1; /* past kfs's own max file size -- a corrupt inode */
	}
	need = (need + (u64)7) & ~(u64)7;
	if(need < (u64)63488)
	{
		need = (u64)63488;
	}
	if(need <= (u64)65536)
	{
		nb = exec_filebuf; /* the single, permanent, shared buffer --
		                     * no allocation call needed at all. */
	}
	else
	{
		nb = arena_alloc(&fs_colosseum, need); /* rare oversized case,
		                     * direct and uncounted, same as this
		                     * region's own behavior before colosseums
		                     * existed - never reclaimed. */
		if(nb == (ptr)0)
		{
			return -1; /* fs_colosseum can't fit this file -- fail, never overflow */
		}
	}
 
	zi = 0;
	while(zi < need)
	{
		*(nb + zi) = 0;
		zi = zi + (u64)8;
	}
	sz = kfs_read_file(n, nb);
	if(sz == (u64)0)
	{
		return -1;
	}
 
	if(sz >= (u64)2 && ptr_byte_at(nb, (u64)0) == (u8)35 && ptr_byte_at(nb, (u64)1) == (u8)33)
	{
		/* "#!" -- a shebang script, not an elf. parse the interpreter
		 * path into interp_path (a real stack local, not nb itself)
		 * BEFORE doing anything else with nb, since the recursive
		 * sys_exec call below allocates its OWN nb (the same shared
		 * exec_filebuf, in the common case) for the interpreter's own
		 * content -- see this file's own top note for why re-execing
		 * an interpreter (in practice, always sh) this way is safe
		 * here. */
		interp_len = shebang_parse_interp(nb, sz, &interp_path[0]);
		if(interp_len == (u64)0)
		{
			return -1;
		}
 
		/* argv for the interpreter: itself, then the script's own
		 * path, then whatever args the ORIGINAL call had past its own
		 * argv[0] (the script's own invocation name) -- the same
		 * shift a real unix shebang does. new_argv[16] reserves slots
		 * 0/1 for the interpreter+script path, leaving 14 slots for the
		 * original argv -- anything past the 14th original arg is
		 * silently dropped, same undocumented-until-now truncation
		 * convention as sh.nsc's own words[16]/proc_copy_name caps
		 * elsewhere in this kernel. */
		new_argv[0] = &interp_path[0];
		new_argv[1] = use_path;
		new_argc = 2;
		ai = 1;
		while(ai < (i32)argc && new_argc < (i64)16)
		{
			new_argv[new_argc] = exec_argv_get(argv, ai);
			new_argc = new_argc + (i64)1;
			ai = ai + 1;
		}
 
		/* the interpreter here is, in every real case, sh -- exec'd
		 * recursively below, which now gets its own real address space
		 * like any other process, no special case. */
		if(exec_shebang_depth >= 8)
		{
			return -1; /* nested too deep (a self-referential shebang
			            * chain, in practice) -- fail this exec cleanly
			            * long before the shared kernel stack could
			            * overflow, see the top note. */
		}
		exec_shebang_depth = exec_shebang_depth + 1;
		result = sys_exec(&interp_path[0], interp_len, new_argc, (ptr)&new_argv[0]);
		exec_shebang_depth = exec_shebang_depth - 1;
		return result;
	}
 
	/* everything from here down replaces the old window/shwin logic
	 * entirely: a real per-process address space (vmm.nsc), built and
	 * torn down around the call instead of a shared physical window
	 * remap. real per-process isolation -- separate physical frames,
	 * separate page tables per process -- comes from this, not from
	 * any virtual-address sub-window check (elf_segments_ok's own
	 * bounds check is now just one shared 0x400000-0x2000000 region). */
	if(elf_validate(nb) == 0 || sz < (u64)64)
	{
		return -1;
	}
	e_phoff = elf_read_u64(nb, (u64)32);
	e_phentsize = elf_read_u16(nb, (u64)54);
	e_phnum = elf_read_u16(nb, (u64)56);
	if(elf_segments_ok(nb, sz, elf_read_u64(nb, (u64)24), e_phoff, e_phentsize, e_phnum) == 0)
	{
		return -1;
	}
 
	/* the process table (kaboom_proc_pid/kaboom_proc_name/
	 * kaboom_proc_pml4, exec.nsh) has exactly 10 slots, 0..9 -- a real,
	 * provable ceiling this file's own top note already explains (up
	 * to 10 processes genuinely alive on the call stack at once), not
	 * a soft guideline: proc_push (called once, further down, once
	 * this exec is actually committed to running) indexes it by
	 * kaboom_proc_depth with no bounds check of its own, so this guard
	 * is what has to catch depth reaching 10 (every slot already in
	 * use) BEFORE that call -- pushing anyway would index slot 10,
	 * past every one of those arrays' real storage, corrupting
	 * whatever global happens to sit next in memory. that could never
	 * actually happen under the OLD shared-window design (its own much
	 * smaller frame pool, 8 frames, always exhausted and cleanly
	 * failed the exec first); the real per-process frame pool this
	 * task adds is easily big enough to nest straight past 10 without
	 * ever exhausting it, so this depth check is what now has to catch
	 * it instead -- confirmed for real: without this check, nesting
	 * `sh` past depth 10 corrupted memory silently (no panic, the vm
	 * just stopped responding a couple of levels further in) rather
	 * than failing the exec. */
	if(kaboom_proc_depth >= 10)
	{
		return -1;
	}
 
	new_pml4 = vmm_create_addrspace();
	if(new_pml4 == (u64)0)
	{
		return -1;
	}
 
	/* map + copy every PT_LOAD segment, one 4kib page at a time,
	 * rolling back the whole address space on any allocation failure
	 * partway through -- every fresh frame is zeroed BEFORE any file
	 * bytes are copied into it, since a frame vmm_alloc_frame hands
	 * back may still hold a previous, now-destroyed process's data. */
	seg_i = 0;
	while(seg_i < (u64)e_phnum)
	{
		ph_off_local = e_phoff + seg_i * (u64)e_phentsize;
		p_type = elf_read_u32(nb, ph_off_local);
		if(p_type == (u32)1)
		{
			p_offset = elf_read_u64(nb, ph_off_local + (u64)8);
			p_vaddr = elf_read_u64(nb, ph_off_local + (u64)16);
			p_filesz = elf_read_u64(nb, ph_off_local + (u64)32);
			p_memsz = elf_read_u64(nb, ph_off_local + (u64)40);
 
			/* a zero-memsz segment (the netbsd-empty-.bss case
			 * elf_segments_ok's own comment documents -- p_vaddr/
			 * p_filesz/p_memsz all 0, or at least p_memsz 0) is
			 * inert: nothing is ever copied or zeroed for it, and
			 * elf_segments_ok deliberately skips validating p_vaddr's
			 * range in exactly this case (nothing ever reads it, for
			 * a REAL elf). this loop's own bound is in terms of
			 * p_vaddr/p_memsz regardless, so without this explicit
			 * guard a zero-memsz segment with a garbage, unvalidated,
			 * non-page-aligned p_vaddr could still drive one iteration
			 * into exec_page_lookup/vmm_map_page with an address
			 * nothing ever checked -- skip the whole segment instead. */
			if(p_memsz != (u64)0)
			{
				page_va = p_vaddr & ~(u64)0xfff;
				while(page_va < p_vaddr + p_memsz)
				{
					/* a page an EARLIER segment already mapped (segments
					 * sharing a page at their boundary, see
					 * exec_page_lookup's own note) is reused as-is --
					 * only a genuinely fresh page gets a new frame
					 * allocated and zeroed. */
					page_pa = exec_page_lookup(new_pml4, page_va);
					if(page_pa == (u64)0)
					{
						page_pa = vmm_alloc_frame();
						if(page_pa == (u64)0)
						{
							vmm_destroy_addrspace(new_pml4);
							return -1;
						}
						if(vmm_map_page(new_pml4, page_va, page_pa) == 0)
						{
							/* vmm_map_page can fail AFTER vmm_alloc_frame
							 * already handed out page_pa -- its own
							 * internal per-process pt frame alloc (a
							 * SEPARATE vmm_alloc_frame call, vmm.nsc) is
							 * what actually failed, not this one. page_pa
							 * itself never got linked into any pte in
							 * that case, so vmm_destroy_addrspace's own
							 * pd/pt walk can never find and free it --
							 * free it explicitly first, or it leaks for
							 * good. */
							vmm_free_frame(page_pa);
							vmm_destroy_addrspace(new_pml4);
							return -1;
						}
						/* zero the whole frame first (a fresh frame may
						 * still hold a previous, now-destroyed process's
						 * data), then copy whatever file bytes this page
						 * actually covers */
						bi = 0;
						while(bi < (u64)4096)
						{
							*((ptr)page_pa + bi) = 0;
							bi = bi + (u64)8;
						}
					}
					/* built byte-by-byte, same technique the old (now
					 * deleted) elf_load used for its own segment-relative
					 * copy loop (git log -p -- src/kernel/elf.nsc), adapted
					 * to this loop's page-relative addressing: a plain
					 * elf_read_u64 here would always pull a full 8-byte
					 * word, which is wrong at BOTH ends of a segment that
					 * isn't 8-aligned. at the tail, p_filesz itself isn't
					 * necessarily a multiple of 8, so the last word's read
					 * would run up to 7 bytes past the segment's real file
					 * data (garbage from whatever follows it in the file,
					 * landing in memory that's supposed to be zero-filled
					 * .bss). at the head, if p_vaddr itself isn't 8-aligned,
					 * requiring page_va+copy_off >= p_vaddr (an 8-aligned
					 * boundary check) would skip the whole word containing
					 * p_vaddr, silently leaving the segment's first 1-7
					 * file bytes as zero instead of their real content.
					 * masking per BYTE against [p_vaddr, p_vaddr+p_filesz)
					 * fixes both: any byte outside that range is left out
					 * of the OR entirely, so it keeps whatever the zeroing
					 * loop above already set (0) rather than pulling in an
					 * out-of-segment file byte or dropping an in-segment
					 * one. */
					copy_off = 0;
					while(copy_off < (u64)4096)
					{
						if(page_va + copy_off + (u64)8 > p_vaddr && page_va + copy_off < p_vaddr + p_filesz)
						{
							word = 0;
							k = 0;
							while(k < (u64)8)
							{
								if(page_va + copy_off + k >= p_vaddr && page_va + copy_off + k < p_vaddr + p_filesz)
								{
									word = word | ((u64)ptr_byte_at(nb, p_offset + (page_va + copy_off + k - p_vaddr)) << (k * (u64)8));
								}
								k = k + (u64)1;
							}
							*((ptr)page_pa + copy_off) = (i64)word;
						}
						copy_off = copy_off + (u64)8;
					}
					page_va = page_va + (u64)4096;
				}
			}
		}
		seg_i = seg_i + (u64)1;
	}
 
	entry = elf_read_u64(nb, (u64)24);
 
	/* pushed only once the exec is actually about to run (found, read,
	 * every PT_LOAD segment mapped+copied) -- a failed exec above never
	 * pollutes ps with something that didn't really start. */
	from_depth = proc_push(use_path, use_path_len);
	kaboom_proc_pml4[from_depth] = new_pml4;
 
	if(from_depth == 0)
	{
		saved_cr3 = kernel_cr3;
	}
	else
	{
		saved_cr3 = kaboom_proc_pml4[from_depth - 1];
	}
	load_cr3(new_pml4);
 
	/* fds the caller already had open before the child started are
	 * the caller's; anything else still open once the child returns
	 * was opened by the child (or something it exec'd in turn) and
	 * never closed -- close those now, at the child's own exit. there
	 * is no fd inheritance or passing in this kernel, so nothing a
	 * child opens can ever legitimately outlive it, and a caller that
	 * stays resident across the exec (sh) keeps every fd it had.
	 * without this, a program that forgot to close its files leaked
	 * those slots until reboot -- only 5 exist. */
	/* argv itself, and every string it points to, are dereferenced by
	 * elf_call_entry AFTER load_cr3(new_pml4) above -- the child reads
	 * them through its OWN page tables, not the caller's. that's silently
	 * fine today ONLY because every current caller keeps its argv
	 * strings in memory that's genuinely SHARED across every address
	 * space, not private per-process: sh's interactive line/words
	 * buffers live on the one shared kernel stack, script lines come out
	 * of sys_alloc (the shared, never-reclaimed heap_colosseum), and the
	 * shebang path just above builds new_argv/interp_path/resolved_path
	 * as kernel-stack locals too. nothing in this kernel currently passes
	 * an argv string out of a process's own PRIVATE memory (its own
	 * per-process .bss or heap) -- if something ever did, the child
	 * would silently read whatever happens to live at that same virtual
	 * address in ITS OWN address space instead, with no error at all.
	 * copying argv/its strings into shared memory before the switch
	 * would close that gap, but there's no real caller to fix it for
	 * yet -- left as future work for whoever builds real per-thread
	 * execution. */
	fd_keep = fd_open_mask();
	result = elf_call_entry((ptr)entry, argc, argv);
	fd_close_unless(fd_keep);
 
	load_cr3(saved_cr3);
	vmm_destroy_addrspace(new_pml4);
	proc_pop(from_depth);
	return result;
}
powered by btf.