| git.druid.rocks | index | druid520 | kaboom | src/ | kernel/ | vmm.nsc |
src/kernel/vmm.nsc
/*
* vmm: physical frame allocation and per-process page tables. this
* file's own first job is turning whatever the boot path handed us
* (pvh's hvm_start_info, or dynamite's own real int 15h e820 call --
* see vmm_e820_from_pvh/Task 13) into one common, flat array of
* (addr, size, type) entries -- 24 bytes each, matching the real xen
* hvm_memmap_table_entry layout exactly (empirically verified against
* a real qemu boot, see the note on Task 1), so the pvh path is a
* straight per-entry copy and the bios path just fills the same
* fields directly.
*/
include "alloc.nsh";
include "klog.nsh";
void halt(void);
u32 get_start_info_ptr32(void);
u8
vmm_byte_at(ptr base, u64 index)
{
ptr p;
u64 word;
p = base + index;
word = (u64)*p;
return (u8)(word & (u64)0xff);
}
u64
vmm_read_u64(ptr buf, u64 off)
{
u64 v;
u64 i;
v = 0;
i = 0;
while(i < (u64)8)
{
v = v | ((u64)vmm_byte_at(buf, off + i) << (i * (u64)8));
i = i + (u64)1;
}
return v;
}
u32
vmm_read_u32(ptr buf, u64 off)
{
u64 v;
u64 i;
v = 0;
i = 0;
while(i < (u64)4)
{
v = v | ((u64)vmm_byte_at(buf, off + i) << (i * (u64)8));
i = i + (u64)1;
}
return (u32)v;
}
global i8 vmm_e820_buf[6144];
global u64 vmm_e820_count;
global void
vmm_e820_add(u64 addr, u64 size, u32 type)
{
ptr slot;
if(vmm_e820_count >= (u64)256)
{
return; /* silently drop past the fixed ceiling, same
* documented-truncation convention exec.nsc's argv
* cap already uses */
}
slot = &vmm_e820_buf[0] + vmm_e820_count * (u64)24;
*slot = (i64)addr;
*(slot + (u64)8) = (i64)size;
*(slot + (u64)16) = (i64)(u64)type;
vmm_e820_count = vmm_e820_count + (u64)1;
}
global void
vmm_e820_from_pvh(void)
{
ptr sip;
u64 memmap_paddr;
u32 memmap_entries;
u32 version;
u64 i;
ptr entry;
u64 addr;
u64 size;
u32 type;
vmm_e820_count = 0;
sip = (ptr)(u64)get_start_info_ptr32();
version = vmm_read_u32(sip, (u64)4);
if(version < (u32)1)
{
return; /* no memmap field at all on a version-0 struct --
* leaves vmm_e820_count at 0, vmm_init (Task 3) halts
* cleanly on that same as any other detection failure */
}
memmap_paddr = vmm_read_u64(sip, (u64)40);
memmap_entries = (u32)vmm_read_u32(sip, (u64)48);
i = 0;
while(i < (u64)memmap_entries)
{
entry = (ptr)(memmap_paddr + i * (u64)24);
addr = vmm_read_u64(entry, (u64)0);
size = vmm_read_u64(entry, (u64)8);
type = (u32)vmm_read_u32(entry, (u64)16);
vmm_e820_add(addr, size, type);
i = i + (u64)1;
}
}
global void
vmm_e820_from_bios(u64 buf_addr, u64 count)
{
ptr buf;
u64 i;
ptr entry;
u64 addr;
u64 size;
u32 type;
/* dynamite's real int 15h/e820h loop (Task 13) already writes the
* exact same 24-byte (addr, size, type, reserved) shape
* vmm_e820_add expects straight into its own scratch buffer, so
* unlike vmm_e820_from_pvh there's no hvm_start_info indirection to
* walk first -- just copy count entries starting at buf_addr. */
vmm_e820_count = 0;
buf = (ptr)buf_addr;
i = 0;
while(i < count)
{
entry = buf + i * (u64)24;
addr = vmm_read_u64(entry, (u64)0);
size = vmm_read_u64(entry, (u64)8);
type = (u32)vmm_read_u32(entry, (u64)16);
vmm_e820_add(addr, size, type);
i = i + (u64)1;
}
}
/* one bit per 4KiB frame, up to 4GiB of real RAM (2^32/4096/8 =
* 131072 bytes) -- a real, generous ceiling in the same spirit as the
* 10-slot process table, not an arbitrary guess. usable RAM is
* further capped at the first 1GiB regardless of what e820 reports
* (see this project's own spec -- growing the identity map past 1GiB
* is a separate, later project), so in practice only the first 32768
* bytes of this bitmap (1GiB/4KiB/8) are ever touched. */
global i8 vmm_frame_bitmap[131072];
global u64 vmm_frame_hint; /* next index to start scanning from --
* pure optimization, never load-bearing:
* every alloc still verifies the bit is
* actually clear before taking it. */
u64
vmm_bit_get(u64 frame_idx)
{
u64 byte;
u64 bit;
byte = (u64)vmm_byte_at(&vmm_frame_bitmap[0], frame_idx / (u64)8);
bit = frame_idx & (u64)7;
return (byte >> bit) & (u64)1;
}
void
vmm_bit_set(u64 frame_idx, u64 val)
{
ptr slot;
u64 byteoff;
u64 bytebit;
u64 wordoff;
u64 byteinword;
u64 byteval;
u64 word;
/* *ptr is always a full 8-byte store here (no byte-granular write
* exists) -- same read-modify-write-the-whole-word technique
* proc_copy_name (exec.nsc:144-147) already uses: read the
* containing 8-byte word, clear/set only this one byte within it
* via shift+mask, write the whole word back. */
slot = &vmm_frame_bitmap[0];
byteoff = frame_idx / (u64)8;
bytebit = frame_idx & (u64)7;
wordoff = byteoff & ~(u64)7;
byteinword = byteoff & (u64)7;
word = (u64)*(slot + wordoff);
byteval = (word >> (byteinword * (u64)8)) & (u64)0xff;
if(val == (u64)1)
{
byteval = byteval | ((u64)1 << bytebit);
}
else
{
byteval = byteval & ~((u64)1 << bytebit);
}
word = word & ~((u64)0xff << (byteinword * (u64)8));
word = word | (byteval << (byteinword * (u64)8));
*(slot + wordoff) = (i64)word;
}
global u64
vmm_alloc_frame(void)
{
u64 total_frames;
u64 i;
u64 idx;
total_frames = (u64)0x40000000 / (u64)4096; /* capped at 1gib */
i = 0;
while(i < total_frames)
{
idx = (vmm_frame_hint + i) % total_frames;
if(vmm_bit_get(idx) == (u64)0)
{
vmm_bit_set(idx, (u64)1);
vmm_frame_hint = idx + (u64)1;
return idx * (u64)4096;
}
i = i + (u64)1;
}
return (u64)0;
}
global void
vmm_free_frame(u64 addr)
{
vmm_bit_set(addr / (u64)4096, (u64)0);
}
global u64
vmm_frames_free(void)
{
u64 total_frames;
u64 i;
u64 n;
total_frames = (u64)0x40000000 / (u64)4096;
i = 0;
n = 0;
while(i < total_frames)
{
if(vmm_bit_get(i) == (u64)0)
{
n = n + (u64)1;
}
i = i + (u64)1;
}
return n;
}
global void
vmm_init(void)
{
u64 i;
ptr entry;
u64 addr;
u64 size;
u32 type;
u64 start_frame;
u64 end_frame;
u64 f;
u64 reserved_end;
i = 0;
while(i < (u64)131072)
{
vmm_frame_bitmap[i] = (i8)0xff; /* start with everything marked
* used -- e820 usable ranges
* below clear only what's
* genuinely free */
i = i + (u64)1;
}
vmm_frame_hint = 0;
if(vmm_e820_count == (u64)0)
{
klog_write("kaboom: no usable memory map -- halting\n");
while(1)
{
halt();
}
}
reserved_end = heap_colosseum_limit(); /* 0x3000000 -- everything
* at or below this is
* already spoken for
* (kernel image, boot's own
* tables, the stack,
* heap_colosseum) and must
* never be handed out as a
* general frame */
i = 0;
while(i < vmm_e820_count)
{
entry = &vmm_e820_buf[0] + i * (u64)24;
addr = vmm_read_u64(entry, (u64)0);
size = vmm_read_u64(entry, (u64)8);
type = (u32)vmm_read_u32(entry, (u64)16);
if(type == (u32)1) /* usable RAM */
{
start_frame = (addr + (u64)4095) / (u64)4096;
end_frame = (addr + size) / (u64)4096; /* exclusive */
if(addr < reserved_end)
{
start_frame = reserved_end / (u64)4096;
}
if(end_frame > (u64)0x40000000 / (u64)4096)
{
end_frame = (u64)0x40000000 / (u64)4096; /* cap at 1gib */
}
f = start_frame;
while(f < end_frame)
{
vmm_bit_set(f, (u64)0);
f = f + (u64)1;
}
}
i = i + (u64)1;
}
/* an e820 map that exists but describes too little usable RAM
* (every usable range entirely below reserved_end, or simply not
* enough total memory) can leave EVERY frame still marked used here
* -- vmm_e820_count == 0 above only catches a map that's missing
* outright, not one that's present but useless. without this check,
* that case fell through silently and only surfaced much later as a
* confusing, misattributed failure (confirmed live: 40mib of RAM
* produced "kaboom: exec sh failed -- halting" instead of any real
* memory-detection message) -- halt cleanly here instead, at the
* point the real cause is actually known. */
if(vmm_frames_free() == (u64)0)
{
klog_write("kaboom: not enough usable memory -- halting\n");
while(1)
{
halt();
}
}
}
u64 get_cr3(void);
void load_cr3(u64 pml4_phys);
ptr get_pd_table(void);
global u64
vmm_create_addrspace(void)
{
u64 pml4_phys;
u64 pdpt_phys;
u64 pd_phys;
ptr boot_pd;
ptr new_pd;
u64 i;
u64 word;
pml4_phys = vmm_alloc_frame();
pdpt_phys = vmm_alloc_frame();
pd_phys = vmm_alloc_frame();
if(pml4_phys == (u64)0 || pdpt_phys == (u64)0 || pd_phys == (u64)0)
{
if(pml4_phys != (u64)0) { vmm_free_frame(pml4_phys); }
if(pdpt_phys != (u64)0) { vmm_free_frame(pdpt_phys); }
if(pd_phys != (u64)0) { vmm_free_frame(pd_phys); }
return (u64)0;
}
/* zero all three fresh frames -- every pml4/pdpt/pd entry starts
* not-present (bit 0 clear) until explicitly set below */
i = 0;
while(i < (u64)512)
{
*((ptr)pml4_phys + i * (u64)8) = 0;
*((ptr)pdpt_phys + i * (u64)8) = 0;
*((ptr)pd_phys + i * (u64)8) = 0;
i = i + (u64)1;
}
*(ptr)pml4_phys = (i64)(pdpt_phys | (u64)0x03); /* present+writable */
*(ptr)pdpt_phys = (i64)(pd_phys | (u64)0x03);
/* copy the shared indices (0-1: kernel low memory/stack; 16-511:
* heap_colosseum plus the entire general frame pool above it) BY
* VALUE from the boot-time pd -- these are still 2mib huge-page
* entries, identical physical==virtual mapping every process
* shares. indices 2-15 (the private region) stay zero/not-present,
* filled in lazily by vmm_map_page.
*
* 16-511, not just 16-23 (heap_colosseum alone): every frame
* vmm_alloc_frame ever hands out -- whether for this call's own
* pml4/pdpt/pd/pt scratch, or for a PT_LOAD segment's backing page
* -- comes from the general pool at/above heap_colosseum_limit()
* (physical 48mib up), and the kernel code building/filling a
* fresh address space (this function, vmm_map_page, sys_exec's own
* per-page copy loop) always dereferences those frames as bare
* physical==virtual pointers, regardless of which process's own
* cr3 happens to be active at the time (a nested sys_exec runs
* under its CALLER's address space, not the boot one). without the
* general pool ALSO being identity-mapped here, any such pointer
* write/read made while a non-boot address space is active faults
* (found for real: the first command run from a freshly-booted
* shell page-faulted building its own address space, every frame
* it touched sitting outside the narrower 16-23 range this used to
* copy). copying all the way to 511 costs nothing beyond 512 more
* quad writes -- entries past whatever's real backing RAM are
* already present-but-unbacked in boot_pd itself (boot.s's own
* fill_pd flatly fills all 512 entries, independent of how much
* real memory is actually installed), so this changes nothing
* about what boot_pd itself already guarantees. */
boot_pd = get_pd_table();
new_pd = (ptr)pd_phys;
i = 0;
while(i < (u64)2)
{
word = (u64)*(boot_pd + i * (u64)8);
*(new_pd + i * (u64)8) = (i64)word;
i = i + (u64)1;
}
i = 16;
while(i < (u64)512)
{
word = (u64)*(boot_pd + i * (u64)8);
*(new_pd + i * (u64)8) = (i64)word;
i = i + (u64)1;
}
return pml4_phys;
}
global i32
vmm_map_page(u64 pml4_phys, u64 vaddr, u64 paddr)
{
ptr pd;
u64 pd_index;
ptr pt_entry_in_pd;
u64 pt_phys;
ptr pt;
u64 pt_index;
u64 i;
if(vaddr < (u64)0x400000 || vaddr >= (u64)0x2000000)
{
return 0; /* outside the private region entirely */
}
/* pml4_phys's own entry 0 always points at this addrspace's one
* pdpt, whose entry 0 always points at this addrspace's one pd --
* both fixed by vmm_create_addrspace, never touched again, so
* walking straight to the pd needs no further pml4/pdpt indexing. */
pd = (ptr)((u64)*(ptr)((u64)*(ptr)pml4_phys & ~(u64)0xfff) & ~(u64)0xfff);
pd_index = vaddr / (u64)0x200000;
pt_entry_in_pd = pd + pd_index * (u64)8;
if(((u64)*pt_entry_in_pd & (u64)1) == (u64)0)
{
/* this 2mib slice has never been touched by this process
* before -- allocate a fresh per-process pt for it */
pt_phys = vmm_alloc_frame();
if(pt_phys == (u64)0)
{
return 0;
}
i = 0;
while(i < (u64)512)
{
*((ptr)pt_phys + i * (u64)8) = 0;
i = i + (u64)1;
}
*pt_entry_in_pd = (i64)(pt_phys | (u64)0x03);
}
pt = (ptr)((u64)*pt_entry_in_pd & ~(u64)0xfff);
pt_index = (vaddr / (u64)4096) % (u64)512;
*(pt + pt_index * (u64)8) = (i64)(paddr | (u64)0x03);
return 1;
}
global void
vmm_destroy_addrspace(u64 pml4_phys)
{
ptr pd;
u64 pdpt_phys;
u64 i;
u64 pde;
u64 pt_phys;
ptr pt;
u64 j;
u64 pte;
pdpt_phys = (u64)*(ptr)pml4_phys & ~(u64)0xfff;
pd = (ptr)((u64)*(ptr)pdpt_phys & ~(u64)0xfff);
/* free only the private indices (2-15) and whatever per-process
* pt/frames they point at -- indices 0-1/16-511 are the shared
* kernel mapping (vmm_create_addrspace's own comment has the full
* reasoning for why that range is 16-511, not just 16-23), never
* owned by this process, never freed here */
i = 2;
while(i < (u64)16)
{
pde = (u64)*(pd + i * (u64)8);
if((pde & (u64)1) == (u64)1)
{
pt_phys = pde & ~(u64)0xfff;
pt = (ptr)pt_phys;
j = 0;
while(j < (u64)512)
{
pte = (u64)*(pt + j * (u64)8);
if((pte & (u64)1) == (u64)1)
{
vmm_free_frame(pte & ~(u64)0xfff);
}
j = j + (u64)1;
}
vmm_free_frame(pt_phys);
}
i = i + (u64)1;
}
vmm_free_frame((u64)pd);
vmm_free_frame(pdpt_phys);
vmm_free_frame(pml4_phys);
}