do not edit — generated by btf.
git.druid.rocksindexdruid520kaboomsrc/kernel/vmm.nsc

src/kernel/vmm.nsc


/*
 * vmm: physical frame allocation and per-process page tables. this
 * file's own first job is turning whatever the boot path handed us
 * (pvh's hvm_start_info, or dynamite's own real int 15h e820 call --
 * see vmm_e820_from_pvh/Task 13) into one common, flat array of
 * (addr, size, type) entries -- 24 bytes each, matching the real xen
 * hvm_memmap_table_entry layout exactly (empirically verified against
 * a real qemu boot, see the note on Task 1), so the pvh path is a
 * straight per-entry copy and the bios path just fills the same
 * fields directly.
 */
 
include "alloc.nsh";
include "klog.nsh";
void halt(void);
 
u32 get_start_info_ptr32(void);
 
u8
vmm_byte_at(ptr base, u64 index)
{
	ptr p;
	u64 word;
	p = base + index;
	word = (u64)*p;
	return (u8)(word & (u64)0xff);
}
 
u64
vmm_read_u64(ptr buf, u64 off)
{
	u64 v;
	u64 i;
	v = 0;
	i = 0;
	while(i < (u64)8)
	{
		v = v | ((u64)vmm_byte_at(buf, off + i) << (i * (u64)8));
		i = i + (u64)1;
	}
	return v;
}
 
u32
vmm_read_u32(ptr buf, u64 off)
{
	u64 v;
	u64 i;
	v = 0;
	i = 0;
	while(i < (u64)4)
	{
		v = v | ((u64)vmm_byte_at(buf, off + i) << (i * (u64)8));
		i = i + (u64)1;
	}
	return (u32)v;
}
 
global i8 vmm_e820_buf[6144];
global u64 vmm_e820_count;
 
global void
vmm_e820_add(u64 addr, u64 size, u32 type)
{
	ptr slot;
 
	if(vmm_e820_count >= (u64)256)
	{
		return; /* silently drop past the fixed ceiling, same
		         * documented-truncation convention exec.nsc's argv
		         * cap already uses */
	}
	slot = &vmm_e820_buf[0] + vmm_e820_count * (u64)24;
	*slot = (i64)addr;
	*(slot + (u64)8) = (i64)size;
	*(slot + (u64)16) = (i64)(u64)type;
	vmm_e820_count = vmm_e820_count + (u64)1;
}
 
global void
vmm_e820_from_pvh(void)
{
	ptr sip;
	u64 memmap_paddr;
	u32 memmap_entries;
	u32 version;
	u64 i;
	ptr entry;
	u64 addr;
	u64 size;
	u32 type;
 
	vmm_e820_count = 0;
	sip = (ptr)(u64)get_start_info_ptr32();
	version = vmm_read_u32(sip, (u64)4);
	if(version < (u32)1)
	{
		return; /* no memmap field at all on a version-0 struct --
		         * leaves vmm_e820_count at 0, vmm_init (Task 3) halts
		         * cleanly on that same as any other detection failure */
	}
	memmap_paddr = vmm_read_u64(sip, (u64)40);
	memmap_entries = (u32)vmm_read_u32(sip, (u64)48);
 
	i = 0;
	while(i < (u64)memmap_entries)
	{
		entry = (ptr)(memmap_paddr + i * (u64)24);
		addr = vmm_read_u64(entry, (u64)0);
		size = vmm_read_u64(entry, (u64)8);
		type = (u32)vmm_read_u32(entry, (u64)16);
		vmm_e820_add(addr, size, type);
		i = i + (u64)1;
	}
}
 
global void
vmm_e820_from_bios(u64 buf_addr, u64 count)
{
	ptr buf;
	u64 i;
	ptr entry;
	u64 addr;
	u64 size;
	u32 type;
 
	/* dynamite's real int 15h/e820h loop (Task 13) already writes the
	 * exact same 24-byte (addr, size, type, reserved) shape
	 * vmm_e820_add expects straight into its own scratch buffer, so
	 * unlike vmm_e820_from_pvh there's no hvm_start_info indirection to
	 * walk first -- just copy count entries starting at buf_addr. */
	vmm_e820_count = 0;
	buf = (ptr)buf_addr;
	i = 0;
	while(i < count)
	{
		entry = buf + i * (u64)24;
		addr = vmm_read_u64(entry, (u64)0);
		size = vmm_read_u64(entry, (u64)8);
		type = (u32)vmm_read_u32(entry, (u64)16);
		vmm_e820_add(addr, size, type);
		i = i + (u64)1;
	}
}
 
/* one bit per 4KiB frame, up to 4GiB of real RAM (2^32/4096/8 =
 * 131072 bytes) -- a real, generous ceiling in the same spirit as the
 * 10-slot process table, not an arbitrary guess. usable RAM is
 * further capped at the first 1GiB regardless of what e820 reports
 * (see this project's own spec -- growing the identity map past 1GiB
 * is a separate, later project), so in practice only the first 32768
 * bytes of this bitmap (1GiB/4KiB/8) are ever touched. */
global i8 vmm_frame_bitmap[131072];
global u64 vmm_frame_hint; /* next index to start scanning from --
                             * pure optimization, never load-bearing:
                             * every alloc still verifies the bit is
                             * actually clear before taking it. */
 
u64
vmm_bit_get(u64 frame_idx)
{
	u64 byte;
	u64 bit;
	byte = (u64)vmm_byte_at(&vmm_frame_bitmap[0], frame_idx / (u64)8);
	bit = frame_idx & (u64)7;
	return (byte >> bit) & (u64)1;
}
 
void
vmm_bit_set(u64 frame_idx, u64 val)
{
	ptr slot;
	u64 byteoff;
	u64 bytebit;
	u64 wordoff;
	u64 byteinword;
	u64 byteval;
	u64 word;
 
	/* *ptr is always a full 8-byte store here (no byte-granular write
	 * exists) -- same read-modify-write-the-whole-word technique
	 * proc_copy_name (exec.nsc:144-147) already uses: read the
	 * containing 8-byte word, clear/set only this one byte within it
	 * via shift+mask, write the whole word back. */
	slot = &vmm_frame_bitmap[0];
	byteoff = frame_idx / (u64)8;
	bytebit = frame_idx & (u64)7;
	wordoff = byteoff & ~(u64)7;
	byteinword = byteoff & (u64)7;
 
	word = (u64)*(slot + wordoff);
	byteval = (word >> (byteinword * (u64)8)) & (u64)0xff;
	if(val == (u64)1)
	{
		byteval = byteval | ((u64)1 << bytebit);
	}
	else
	{
		byteval = byteval & ~((u64)1 << bytebit);
	}
	word = word & ~((u64)0xff << (byteinword * (u64)8));
	word = word | (byteval << (byteinword * (u64)8));
	*(slot + wordoff) = (i64)word;
}
 
global u64
vmm_alloc_frame(void)
{
	u64 total_frames;
	u64 i;
	u64 idx;
 
	total_frames = (u64)0x40000000 / (u64)4096; /* capped at 1gib */
	i = 0;
	while(i < total_frames)
	{
		idx = (vmm_frame_hint + i) % total_frames;
		if(vmm_bit_get(idx) == (u64)0)
		{
			vmm_bit_set(idx, (u64)1);
			vmm_frame_hint = idx + (u64)1;
			return idx * (u64)4096;
		}
		i = i + (u64)1;
	}
	return (u64)0;
}
 
global void
vmm_free_frame(u64 addr)
{
	vmm_bit_set(addr / (u64)4096, (u64)0);
}
 
global u64
vmm_frames_free(void)
{
	u64 total_frames;
	u64 i;
	u64 n;
 
	total_frames = (u64)0x40000000 / (u64)4096;
	i = 0;
	n = 0;
	while(i < total_frames)
	{
		if(vmm_bit_get(i) == (u64)0)
		{
			n = n + (u64)1;
		}
		i = i + (u64)1;
	}
	return n;
}
 
global void
vmm_init(void)
{
	u64 i;
	ptr entry;
	u64 addr;
	u64 size;
	u32 type;
	u64 start_frame;
	u64 end_frame;
	u64 f;
	u64 reserved_end;
 
	i = 0;
	while(i < (u64)131072)
	{
		vmm_frame_bitmap[i] = (i8)0xff; /* start with everything marked
		                                  * used -- e820 usable ranges
		                                  * below clear only what's
		                                  * genuinely free */
		i = i + (u64)1;
	}
	vmm_frame_hint = 0;
 
	if(vmm_e820_count == (u64)0)
	{
		klog_write("kaboom: no usable memory map -- halting\n");
		while(1)
		{
			halt();
		}
	}
 
	reserved_end = heap_colosseum_limit(); /* 0x3000000 -- everything
	                                         * at or below this is
	                                         * already spoken for
	                                         * (kernel image, boot's own
	                                         * tables, the stack,
	                                         * heap_colosseum) and must
	                                         * never be handed out as a
	                                         * general frame */
 
	i = 0;
	while(i < vmm_e820_count)
	{
		entry = &vmm_e820_buf[0] + i * (u64)24;
		addr = vmm_read_u64(entry, (u64)0);
		size = vmm_read_u64(entry, (u64)8);
		type = (u32)vmm_read_u32(entry, (u64)16);
 
		if(type == (u32)1) /* usable RAM */
		{
			start_frame = (addr + (u64)4095) / (u64)4096;
			end_frame = (addr + size) / (u64)4096; /* exclusive */
			if(addr < reserved_end)
			{
				start_frame = reserved_end / (u64)4096;
			}
			if(end_frame > (u64)0x40000000 / (u64)4096)
			{
				end_frame = (u64)0x40000000 / (u64)4096; /* cap at 1gib */
			}
			f = start_frame;
			while(f < end_frame)
			{
				vmm_bit_set(f, (u64)0);
				f = f + (u64)1;
			}
		}
		i = i + (u64)1;
	}
 
	/* an e820 map that exists but describes too little usable RAM
	 * (every usable range entirely below reserved_end, or simply not
	 * enough total memory) can leave EVERY frame still marked used here
	 * -- vmm_e820_count == 0 above only catches a map that's missing
	 * outright, not one that's present but useless. without this check,
	 * that case fell through silently and only surfaced much later as a
	 * confusing, misattributed failure (confirmed live: 40mib of RAM
	 * produced "kaboom: exec sh failed -- halting" instead of any real
	 * memory-detection message) -- halt cleanly here instead, at the
	 * point the real cause is actually known. */
	if(vmm_frames_free() == (u64)0)
	{
		klog_write("kaboom: not enough usable memory -- halting\n");
		while(1)
		{
			halt();
		}
	}
}
 
u64 get_cr3(void);
void load_cr3(u64 pml4_phys);
ptr get_pd_table(void);
 
global u64
vmm_create_addrspace(void)
{
	u64 pml4_phys;
	u64 pdpt_phys;
	u64 pd_phys;
	ptr boot_pd;
	ptr new_pd;
	u64 i;
	u64 word;
 
	pml4_phys = vmm_alloc_frame();
	pdpt_phys = vmm_alloc_frame();
	pd_phys = vmm_alloc_frame();
	if(pml4_phys == (u64)0 || pdpt_phys == (u64)0 || pd_phys == (u64)0)
	{
		if(pml4_phys != (u64)0) { vmm_free_frame(pml4_phys); }
		if(pdpt_phys != (u64)0) { vmm_free_frame(pdpt_phys); }
		if(pd_phys != (u64)0) { vmm_free_frame(pd_phys); }
		return (u64)0;
	}
 
	/* zero all three fresh frames -- every pml4/pdpt/pd entry starts
	 * not-present (bit 0 clear) until explicitly set below */
	i = 0;
	while(i < (u64)512)
	{
		*((ptr)pml4_phys + i * (u64)8) = 0;
		*((ptr)pdpt_phys + i * (u64)8) = 0;
		*((ptr)pd_phys + i * (u64)8) = 0;
		i = i + (u64)1;
	}
 
	*(ptr)pml4_phys = (i64)(pdpt_phys | (u64)0x03); /* present+writable */
	*(ptr)pdpt_phys = (i64)(pd_phys | (u64)0x03);
 
	/* copy the shared indices (0-1: kernel low memory/stack; 16-511:
	 * heap_colosseum plus the entire general frame pool above it) BY
	 * VALUE from the boot-time pd -- these are still 2mib huge-page
	 * entries, identical physical==virtual mapping every process
	 * shares. indices 2-15 (the private region) stay zero/not-present,
	 * filled in lazily by vmm_map_page.
	 *
	 * 16-511, not just 16-23 (heap_colosseum alone): every frame
	 * vmm_alloc_frame ever hands out -- whether for this call's own
	 * pml4/pdpt/pd/pt scratch, or for a PT_LOAD segment's backing page
	 * -- comes from the general pool at/above heap_colosseum_limit()
	 * (physical 48mib up), and the kernel code building/filling a
	 * fresh address space (this function, vmm_map_page, sys_exec's own
	 * per-page copy loop) always dereferences those frames as bare
	 * physical==virtual pointers, regardless of which process's own
	 * cr3 happens to be active at the time (a nested sys_exec runs
	 * under its CALLER's address space, not the boot one). without the
	 * general pool ALSO being identity-mapped here, any such pointer
	 * write/read made while a non-boot address space is active faults
	 * (found for real: the first command run from a freshly-booted
	 * shell page-faulted building its own address space, every frame
	 * it touched sitting outside the narrower 16-23 range this used to
	 * copy). copying all the way to 511 costs nothing beyond 512 more
	 * quad writes -- entries past whatever's real backing RAM are
	 * already present-but-unbacked in boot_pd itself (boot.s's own
	 * fill_pd flatly fills all 512 entries, independent of how much
	 * real memory is actually installed), so this changes nothing
	 * about what boot_pd itself already guarantees. */
	boot_pd = get_pd_table();
	new_pd = (ptr)pd_phys;
	i = 0;
	while(i < (u64)2)
	{
		word = (u64)*(boot_pd + i * (u64)8);
		*(new_pd + i * (u64)8) = (i64)word;
		i = i + (u64)1;
	}
	i = 16;
	while(i < (u64)512)
	{
		word = (u64)*(boot_pd + i * (u64)8);
		*(new_pd + i * (u64)8) = (i64)word;
		i = i + (u64)1;
	}
 
	return pml4_phys;
}
 
global i32
vmm_map_page(u64 pml4_phys, u64 vaddr, u64 paddr)
{
	ptr pd;
	u64 pd_index;
	ptr pt_entry_in_pd;
	u64 pt_phys;
	ptr pt;
	u64 pt_index;
	u64 i;
 
	if(vaddr < (u64)0x400000 || vaddr >= (u64)0x2000000)
	{
		return 0; /* outside the private region entirely */
	}
 
	/* pml4_phys's own entry 0 always points at this addrspace's one
	 * pdpt, whose entry 0 always points at this addrspace's one pd --
	 * both fixed by vmm_create_addrspace, never touched again, so
	 * walking straight to the pd needs no further pml4/pdpt indexing. */
	pd = (ptr)((u64)*(ptr)((u64)*(ptr)pml4_phys & ~(u64)0xfff) & ~(u64)0xfff);
 
	pd_index = vaddr / (u64)0x200000;
	pt_entry_in_pd = pd + pd_index * (u64)8;
 
	if(((u64)*pt_entry_in_pd & (u64)1) == (u64)0)
	{
		/* this 2mib slice has never been touched by this process
		 * before -- allocate a fresh per-process pt for it */
		pt_phys = vmm_alloc_frame();
		if(pt_phys == (u64)0)
		{
			return 0;
		}
		i = 0;
		while(i < (u64)512)
		{
			*((ptr)pt_phys + i * (u64)8) = 0;
			i = i + (u64)1;
		}
		*pt_entry_in_pd = (i64)(pt_phys | (u64)0x03);
	}
 
	pt = (ptr)((u64)*pt_entry_in_pd & ~(u64)0xfff);
	pt_index = (vaddr / (u64)4096) % (u64)512;
	*(pt + pt_index * (u64)8) = (i64)(paddr | (u64)0x03);
	return 1;
}
 
global void
vmm_destroy_addrspace(u64 pml4_phys)
{
	ptr pd;
	u64 pdpt_phys;
	u64 i;
	u64 pde;
	u64 pt_phys;
	ptr pt;
	u64 j;
	u64 pte;
 
	pdpt_phys = (u64)*(ptr)pml4_phys & ~(u64)0xfff;
	pd = (ptr)((u64)*(ptr)pdpt_phys & ~(u64)0xfff);
 
	/* free only the private indices (2-15) and whatever per-process
	 * pt/frames they point at -- indices 0-1/16-511 are the shared
	 * kernel mapping (vmm_create_addrspace's own comment has the full
	 * reasoning for why that range is 16-511, not just 16-23), never
	 * owned by this process, never freed here */
	i = 2;
	while(i < (u64)16)
	{
		pde = (u64)*(pd + i * (u64)8);
		if((pde & (u64)1) == (u64)1)
		{
			pt_phys = pde & ~(u64)0xfff;
			pt = (ptr)pt_phys;
			j = 0;
			while(j < (u64)512)
			{
				pte = (u64)*(pt + j * (u64)8);
				if((pte & (u64)1) == (u64)1)
				{
					vmm_free_frame(pte & ~(u64)0xfff);
				}
				j = j + (u64)1;
			}
			vmm_free_frame(pt_phys);
		}
		i = i + (u64)1;
	}
 
	vmm_free_frame((u64)pd);
	vmm_free_frame(pdpt_phys);
	vmm_free_frame(pml4_phys);
}
powered by btf.