| git.druid.rocks | index | druid520 | kaboom | src/ | kernel/ | elf.nsc |
src/kernel/elf.nsc
/*
* minimal elf64 header/program-header parsing: validates the header
* (elf_validate), reads arbitrary fields out of the raw file bytes
* (elf_read_u16/u32/u64), and checks every program header's bounds
* before anything is loaded anywhere (elf_segments_ok). the actual
* per-page mapping and file-byte copying now lives in sys_exec
* (exec.nsc), one real address space per process via vmm.nsc
* (vmm_create_addrspace/vmm_map_page) -- this file used to also own
* that copy itself (elf_load, into one of two shared fixed virtual
* windows), back when every process shared the kernel's one address
* space; real per-process page tables replaced that, so only the
* format-parsing helpers remain here. no relocations, no dynamic
* linking, no sections -- a plain static non-pie executable only,
* which is all elf_call_entry (elf_asm.s) can actually run anyway:
* there's no ring3/tss/user segments yet, so a "loaded" program runs
* in ring0 as an ordinary called function (call *entry, not a real
* process replacement) and is expected to ret back when done, same
* simplification as everywhere else in this kernel that doesn't have
* a scheduler yet.
*
* elf's byte layout is dictated by the format spec, not kaboom's own
* invention (unlike kfs), so this needs the same byte-extraction
* technique as gdt/idt parsing rather than kfs's word-aligned shortcut.
*/
u8
ptr_byte_at(ptr base, u64 index)
{
ptr p;
u64 word;
p = base + index;
word = (u64)*p;
return (u8)(word & (u64)0xff);
}
global u16
elf_read_u16(ptr buf, u64 off)
{
u64 b0;
u64 b1;
b0 = (u64)ptr_byte_at(buf, off);
b1 = (u64)ptr_byte_at(buf, off + (u64)1);
return (u16)(b0 | (b1 << (u64)8));
}
global u32
elf_read_u32(ptr buf, u64 off)
{
u64 v;
u64 i;
v = 0;
i = 0;
while(i < (u64)4)
{
v = v | ((u64)ptr_byte_at(buf, off + i) << (i * (u64)8));
i = i + (u64)1;
}
return (u32)v;
}
global u64
elf_read_u64(ptr buf, u64 off)
{
u64 v;
u64 i;
v = 0;
i = 0;
while(i < (u64)8)
{
v = v | ((u64)ptr_byte_at(buf, off + i) << (i * (u64)8));
i = i + (u64)1;
}
return v;
}
/* validates the elf64/x86-64 magic+class+machine fields. segment
* bounds are checked separately, by elf_segments_ok. */
global i32
elf_validate(ptr buf)
{
if(ptr_byte_at(buf, (u64)0) != (u8)0x7f) { return 0; }
if(ptr_byte_at(buf, (u64)1) != (u8)69) { return 0; } /* 'E' */
if(ptr_byte_at(buf, (u64)2) != (u8)76) { return 0; } /* 'L' */
if(ptr_byte_at(buf, (u64)3) != (u8)70) { return 0; } /* 'F' */
if(ptr_byte_at(buf, (u64)4) != (u8)2) { return 0; } /* ELFCLASS64 */
if(elf_read_u16(buf, (u64)18) != (u16)62) { return 0; } /* EM_X86_64 */
return 1;
}
/* checks every program header against the file actually read (sz
* bytes) and every PT_LOAD segment against the one shared private
* region (0x400000-0x2000000, see vmm_map_page's own identical bounds
* check, vmm.nsc) every process's address space reserves for it,
* BEFORE anything is mapped or copied anywhere -- returns 1 if the
* whole file is safe to load, 0 otherwise. a p_vaddr/p_memsz pair is
* otherwise trusted blindly, and p_memsz costs nothing on disk, so an
* oversized .bss alone could drive sys_exec's own per-page loop
* (exec.nsc) into mapping straight past the private region's own
* ceiling. the end checked is p_vaddr + p_memsz rounded up to 8, since
* a packed 8-byte store (sys_exec's own per-page copy loop now, same
* as elf_load's used to be) can run up to 7 bytes past p_memsz.
* nothing nscc plus this project's own link scripts produce gets
* anywhere near any of these limits -- this exists to reject a
* malformed file cleanly rather than trust it.
*
* real per-process isolation (separate physical frames, separate page
* tables per process, see vmm.nsc) is what actually keeps one
* process's load from touching another's memory now -- this is only
* a single shared region's own bounds check, not a per-program
* sub-window the way it used to be back when every process shared one
* flat address space (collapsing the old two-window check to this one
* region is exactly what real per-process address spaces make safe:
* see exec.nsc's own note on this). */
global i32
elf_segments_ok(ptr buf, u64 sz, u64 e_entry, u64 e_phoff, u16 e_phentsize, u16 e_phnum)
{
u64 lo;
u64 hi;
u64 i;
u64 ph_off;
u64 p_offset;
u64 p_vaddr;
u64 p_filesz;
u64 p_memsz;
lo = (u64)0x400000;
hi = (u64)0x2000000;
if(e_entry < lo || e_entry >= hi)
{
return 0;
}
/* the program header table itself has to be inside the file, with
* room for every field read below (56 bytes: a real elf64 phdr). */
if(e_phentsize < (u16)56 || e_phoff > sz)
{
return 0;
}
if((u64)e_phnum * (u64)e_phentsize > sz - e_phoff)
{
return 0;
}
i = 0;
while(i < (u64)e_phnum)
{
ph_off = e_phoff + i * (u64)e_phentsize;
if(elf_read_u32(buf, ph_off) == (u32)1) /* PT_LOAD */
{
p_offset = elf_read_u64(buf, ph_off + (u64)8);
p_vaddr = elf_read_u64(buf, ph_off + (u64)16);
p_filesz = elf_read_u64(buf, ph_off + (u64)32);
p_memsz = elf_read_u64(buf, ph_off + (u64)40);
if(p_filesz > p_memsz) { return 0; }
if(p_offset > sz || p_filesz > sz - p_offset) { return 0; }
/* a zero-memsz PT_LOAD segment copies and zeroes nothing
* at all (sys_exec's own per-page loop never even enters
* its while(page_va < p_vaddr + p_memsz) for one -- see
* its own explicit p_memsz!=0 guard, exec.nsc), so its own
* p_vaddr is never actually read from or written to --
* skip the region check for it rather than reject it.
* real, live-hit case: gnu ld omits an empty .bss segment
* outright, but not every linker does -- netbsd's own ld
* still emits one, with p_vaddr/p_filesz/p_memsz all 0,
* which used to fail the check below (0 is below the
* region's own start) and made every binary that linker
* produced fail to exec at all, "command not found"
* (sys_exec's own ambiguous failure message) with no
* other symptom -- found from a real built-on-netbsd
* binary, not guessed at. */
if(p_memsz > (u64)0)
{
if(p_vaddr < lo || p_vaddr >= hi) { return 0; }
/* p_memsz <= hi - p_vaddr first, so the round-up below
* can't overflow on a garbage 64-bit p_memsz. */
if(p_memsz > hi - p_vaddr) { return 0; }
if(((p_memsz + (u64)7) & ~(u64)7) > hi - p_vaddr) { return 0; }
}
}
i = i + (u64)1;
}
return 1;
}