do not edit — generated by btf.
git.druid.rocksindexdruid520kaboomsrc/kernel/elf.nsc

src/kernel/elf.nsc


/*
 * minimal elf64 header/program-header parsing: validates the header
 * (elf_validate), reads arbitrary fields out of the raw file bytes
 * (elf_read_u16/u32/u64), and checks every program header's bounds
 * before anything is loaded anywhere (elf_segments_ok). the actual
 * per-page mapping and file-byte copying now lives in sys_exec
 * (exec.nsc), one real address space per process via vmm.nsc
 * (vmm_create_addrspace/vmm_map_page) -- this file used to also own
 * that copy itself (elf_load, into one of two shared fixed virtual
 * windows), back when every process shared the kernel's one address
 * space; real per-process page tables replaced that, so only the
 * format-parsing helpers remain here. no relocations, no dynamic
 * linking, no sections -- a plain static non-pie executable only,
 * which is all elf_call_entry (elf_asm.s) can actually run anyway:
 * there's no ring3/tss/user segments yet, so a "loaded" program runs
 * in ring0 as an ordinary called function (call *entry, not a real
 * process replacement) and is expected to ret back when done, same
 * simplification as everywhere else in this kernel that doesn't have
 * a scheduler yet.
 *
 * elf's byte layout is dictated by the format spec, not kaboom's own
 * invention (unlike kfs), so this needs the same byte-extraction
 * technique as gdt/idt parsing rather than kfs's word-aligned shortcut.
 */
 
u8
ptr_byte_at(ptr base, u64 index)
{
	ptr p;
	u64 word;
	p = base + index;
	word = (u64)*p;
	return (u8)(word & (u64)0xff);
}
 
global u16
elf_read_u16(ptr buf, u64 off)
{
	u64 b0;
	u64 b1;
	b0 = (u64)ptr_byte_at(buf, off);
	b1 = (u64)ptr_byte_at(buf, off + (u64)1);
	return (u16)(b0 | (b1 << (u64)8));
}
 
global u32
elf_read_u32(ptr buf, u64 off)
{
	u64 v;
	u64 i;
	v = 0;
	i = 0;
	while(i < (u64)4)
	{
		v = v | ((u64)ptr_byte_at(buf, off + i) << (i * (u64)8));
		i = i + (u64)1;
	}
	return (u32)v;
}
 
global u64
elf_read_u64(ptr buf, u64 off)
{
	u64 v;
	u64 i;
	v = 0;
	i = 0;
	while(i < (u64)8)
	{
		v = v | ((u64)ptr_byte_at(buf, off + i) << (i * (u64)8));
		i = i + (u64)1;
	}
	return v;
}
 
/* validates the elf64/x86-64 magic+class+machine fields. segment
 * bounds are checked separately, by elf_segments_ok. */
global i32
elf_validate(ptr buf)
{
	if(ptr_byte_at(buf, (u64)0) != (u8)0x7f) { return 0; }
	if(ptr_byte_at(buf, (u64)1) != (u8)69) { return 0; }  /* 'E' */
	if(ptr_byte_at(buf, (u64)2) != (u8)76) { return 0; }  /* 'L' */
	if(ptr_byte_at(buf, (u64)3) != (u8)70) { return 0; }  /* 'F' */
	if(ptr_byte_at(buf, (u64)4) != (u8)2) { return 0; }   /* ELFCLASS64 */
	if(elf_read_u16(buf, (u64)18) != (u16)62) { return 0; } /* EM_X86_64 */
	return 1;
}
 
/* checks every program header against the file actually read (sz
 * bytes) and every PT_LOAD segment against the one shared private
 * region (0x400000-0x2000000, see vmm_map_page's own identical bounds
 * check, vmm.nsc) every process's address space reserves for it,
 * BEFORE anything is mapped or copied anywhere -- returns 1 if the
 * whole file is safe to load, 0 otherwise. a p_vaddr/p_memsz pair is
 * otherwise trusted blindly, and p_memsz costs nothing on disk, so an
 * oversized .bss alone could drive sys_exec's own per-page loop
 * (exec.nsc) into mapping straight past the private region's own
 * ceiling. the end checked is p_vaddr + p_memsz rounded up to 8, since
 * a packed 8-byte store (sys_exec's own per-page copy loop now, same
 * as elf_load's used to be) can run up to 7 bytes past p_memsz.
 * nothing nscc plus this project's own link scripts produce gets
 * anywhere near any of these limits -- this exists to reject a
 * malformed file cleanly rather than trust it.
 *
 * real per-process isolation (separate physical frames, separate page
 * tables per process, see vmm.nsc) is what actually keeps one
 * process's load from touching another's memory now -- this is only
 * a single shared region's own bounds check, not a per-program
 * sub-window the way it used to be back when every process shared one
 * flat address space (collapsing the old two-window check to this one
 * region is exactly what real per-process address spaces make safe:
 * see exec.nsc's own note on this). */
global i32
elf_segments_ok(ptr buf, u64 sz, u64 e_entry, u64 e_phoff, u16 e_phentsize, u16 e_phnum)
{
	u64 lo;
	u64 hi;
	u64 i;
	u64 ph_off;
	u64 p_offset;
	u64 p_vaddr;
	u64 p_filesz;
	u64 p_memsz;
 
	lo = (u64)0x400000;
	hi = (u64)0x2000000;
	if(e_entry < lo || e_entry >= hi)
	{
		return 0;
	}
 
	/* the program header table itself has to be inside the file, with
	 * room for every field read below (56 bytes: a real elf64 phdr). */
	if(e_phentsize < (u16)56 || e_phoff > sz)
	{
		return 0;
	}
	if((u64)e_phnum * (u64)e_phentsize > sz - e_phoff)
	{
		return 0;
	}
 
	i = 0;
	while(i < (u64)e_phnum)
	{
		ph_off = e_phoff + i * (u64)e_phentsize;
		if(elf_read_u32(buf, ph_off) == (u32)1) /* PT_LOAD */
		{
			p_offset = elf_read_u64(buf, ph_off + (u64)8);
			p_vaddr = elf_read_u64(buf, ph_off + (u64)16);
			p_filesz = elf_read_u64(buf, ph_off + (u64)32);
			p_memsz = elf_read_u64(buf, ph_off + (u64)40);
 
			if(p_filesz > p_memsz) { return 0; }
			if(p_offset > sz || p_filesz > sz - p_offset) { return 0; }
			/* a zero-memsz PT_LOAD segment copies and zeroes nothing
			 * at all (sys_exec's own per-page loop never even enters
			 * its while(page_va < p_vaddr + p_memsz) for one -- see
			 * its own explicit p_memsz!=0 guard, exec.nsc), so its own
			 * p_vaddr is never actually read from or written to --
			 * skip the region check for it rather than reject it.
			 * real, live-hit case: gnu ld omits an empty .bss segment
			 * outright, but not every linker does -- netbsd's own ld
			 * still emits one, with p_vaddr/p_filesz/p_memsz all 0,
			 * which used to fail the check below (0 is below the
			 * region's own start) and made every binary that linker
			 * produced fail to exec at all, "command not found"
			 * (sys_exec's own ambiguous failure message) with no
			 * other symptom -- found from a real built-on-netbsd
			 * binary, not guessed at. */
			if(p_memsz > (u64)0)
			{
				if(p_vaddr < lo || p_vaddr >= hi) { return 0; }
				/* p_memsz <= hi - p_vaddr first, so the round-up below
				 * can't overflow on a garbage 64-bit p_memsz. */
				if(p_memsz > hi - p_vaddr) { return 0; }
				if(((p_memsz + (u64)7) & ~(u64)7) > hi - p_vaddr) { return 0; }
			}
		}
		i = i + (u64)1;
	}
	return 1;
}
 
powered by btf.