do not edit — generated by btf.
git.druid.rocksindexdruid520kaboomsrc/boot/boot.s

src/boot/boot.s


/*
 * multiboot1 entry + 32-bit protected mode -> long mode transition.
 *
 * multiboot1 only guarantees 32-bit protected mode with flat segments.
 * nscc emits pure 64-bit (long mode) code, so nothing nsc-compiled can
 * run until we've built page tables, turned on PAE+LME+paging, and
 * far-jumped into a 64-bit code segment. get any of this wrong and the
 * cpu triple-faults silently (instant reboot, no error) -- this is the
 * single highest-risk piece of the whole kernel and gets tested before
 * anything else is built on top of it.
 */
 
/*
 * qemu's -kernel loader hard-refuses any 64-bit-class ELF through its
 * multiboot1 path (confirmed empirically: "Cannot load x86-64 image,
 * give a 32bit one") regardless of what's actually inside the load
 * segments -- and nscc only emits 64-bit object files, so the whole
 * kernel has to link as ELF64. a multiboot1 header was tried
 * alongside the PVH note below, but qemu detects multiboot1 first and
 * that path's ELFCLASS32 check runs unconditionally before PVH is
 * ever considered -- having both present doesn't work, it's one or
 * the other. PVH wins here because `qemu -kernel` working today is a
 * hard requirement; a real multiboot1 header for GRUB/real hardware
 * would need a genuinely separate 32-bit build, not attempted yet.
 *
 * PVH (xen/hvm direct-boot protocol, also implemented by qemu) is
 * fine with an ELF64 container -- it just needs a note pointing at a
 * 32-bit physical entry address. confirmed working empirically before
 * wiring it in for real.
 */
.section .note.pvh, "a", @note
.align 4
	.long 4                       /* namesz: "Xen\0" */
	.long 4                       /* descsz: 4-byte phys32 entry addr */
	.long 0x12                    /* type: XEN_ELFNOTE_PHYS32_ENTRY */
	.ascii "Xen\0"
	.long _start
 
.section .bss
.align 4096
.global pd
pml4:
	.skip 4096
pdpt:
	.skip 4096
pd:
	.skip 4096
.align 4
.global start_info_ptr32
start_info_ptr32:
	.skip 4
/* dynamite's own real-bios handoff (Task 13) -- see dynamite.s's own
 * comment just before its `jmp KERNEL_ENTRY_ADDR` for why these three
 * registers specifically (ebp/esi/edx) and why an explicit magic
 * rather than an inferred one. unused/zero on a genuine pvh boot,
 * since qemu's -kernel/pvh loader jumps straight to _start without
 * ever running dynamite.s at all. */
.align 4
.global bios_boot_flag
bios_boot_flag:
	.skip 4
.global bios_e820_ptr32
bios_e820_ptr32:
	.skip 4
.global bios_e820_count32
bios_e820_count32:
	.skip 4
/* stack guard gap: 512kib of never-used .bss between the live page
 * tables above and the kernel stack below. there is exactly one stack
 * in this kernel -- every "process" (exec.nsc) runs on it as an
 * ordinary nested call -- and it grows DOWN toward stack_bottom, so
 * with nothing in between, an overflow past its 64kib landed straight
 * in pd/pdpt/pml4: found for real, a shebang script naming itself as
 * its own interpreter recursed until it overwrote them and the
 * machine triple-faulted (full reboot). exec.nsc now caps that
 * recursion directly; this gap is defense-in-depth for any OTHER deep
 * stack use, not a structural guarantee -- an overflow of more than
 * 64kib + 512kib would still reach the page tables. the real fix is an
 * actual unmapped guard page below stack_bottom (a not-present 4kib
 * page, so an overflow faults immediately instead of corrupting
 * anything), which means splitting fill_pd's first 2mib huge page
 * into a real 4kib page table -- a genuine future improvement, not
 * attempted here. costs nothing in the binary (.bss is nobits, and
 * this whole range is already inside the 1gib identity map); only
 * the zeroing loop above touches it, once, at boot. */
	.skip 524288
.align 16
stack_bottom:
	.skip 65536
stack_top:
 
.section .rodata
gdt64:
	.quad 0                                  /* null descriptor */
	.quad 0x00af9a000000ffff                 /* 0x08: 64-bit code, ring0 */
	.quad 0x00af92000000ffff                 /* 0x10: 64-bit data, ring0 */
gdt64_end:
gdt64_ptr:
	.word gdt64_end - gdt64 - 1
	.quad gdt64
 
.section .text
.code32
.global _start
.type _start, @function
_start:
	cli
	movl $stack_top, %esp
 
	/* zero .bss unconditionally. dynamite's raw objcopy blob (real
	 * bios path) doesn't include .bss at all -- objcopy -O binary only
	 * emits PROGBITS content -- and real hardware ram starts as
	 * garbage, unlike qemu's guest ram, which happens to start zeroed.
	 * without this, boot.s's own page tables and any zero-initialized
	 * kernel globals would start from garbage on real hardware.
	 * redundant but harmless on the existing -kernel/pvh path, whose
	 * elf loader already zeroes .bss per the program header.
	 *
	 * %ebx (the PVH hvm_start_info pointer, per the entry note above)
	 * is deliberately still untouched here and saved to start_info_ptr32
	 * (itself a .bss slot) only after this loop, not before: this loop
	 * zeroes the entire .bss section, so storing into a .bss slot any
	 * earlier would just get wiped out by the same loop before anything
	 * ever reads it back -- confirmed the hard way empirically (the
	 * pre-fix ordering read back a zeroed slot, i.e. a null pointer,
	 * whose dereference produced IVT-segment-shaped garbage instead of
	 * the real hvm_start_info magic). none of edi/ecx/eax used below
	 * touch ebx, so it survives this loop in the register untouched. */
	movl $_bss_start, %edi
	movl $_bss_end, %ecx
	subl %edi, %ecx
	shrl $2, %ecx
	xorl %eax, %eax
	cld                         /* pvh and bios alike leave df unspecified */
	rep stosl
 
	movl %ebx, start_info_ptr32
 
	/* dynamite's real-bios handoff (Task 13), saved at this exact same
	 * point for the exact same reason as %ebx just above: none of
	 * ebp/esi/edx are touched by the zeroing loop or anything before
	 * it, so they still hold whatever dynamite.s (or, on a real pvh
	 * boot, nothing at all) last put there. */
	movl %ebp, bios_boot_flag
	movl %esi, bios_e820_ptr32
	movl %edx, bios_e820_count32
 
	/* identity-map the first 1GiB with 2MiB pages: pml4[0] -> pdpt,
	 * pdpt[0] -> pd, pd[0..511] -> 2MiB pages covering 0..1GiB. this
	 * is more than the kernel needs right now but avoids revisiting
	 * this table for a good while. */
	movl $pdpt, %eax
	orl $0x03, %eax
	movl %eax, pml4
 
	movl $pd, %eax
	orl $0x03, %eax
	movl %eax, pdpt
 
	movl $0, %ecx
fill_pd:
	movl %ecx, %eax
	shll $21, %eax
	orl $0x83, %eax             /* present, writable, huge (2MiB) */
	movl %eax, pd(,%ecx,8)
	movl $0, pd+4(,%ecx,8)
	incl %ecx
	cmpl $512, %ecx
	jne fill_pd
 
	/* enable PAE (CR4.PAE) */
	movl %cr4, %eax
	orl $0x20, %eax
	movl %eax, %cr4
 
	/* load CR3 with the PML4 base */
	movl $pml4, %eax
	movl %eax, %cr3
 
	/* set EFER.LME (long mode enable) via MSR 0xC0000080 */
	movl $0xC0000080, %ecx
	rdmsr
	orl $0x100, %eax
	wrmsr
 
	/* enable paging (CR0.PG) -- now in compatibility mode */
	movl %cr0, %eax
	orl $0x80000000, %eax
	movl %eax, %cr0
 
	lgdt gdt64_ptr
	ljmp $0x08, $long_mode_entry
 
.code64
long_mode_entry:
	movw $0x10, %ax
	movw %ax, %ds
	movw %ax, %es
	movw %ax, %ss
	movw %ax, %fs
	movw %ax, %gs
 
	movq $stack_top, %rsp
	movq $0, %rbp
 
	/* raw pre-nsc sanity beacon: write 'B' to COM1 (0x3f8) directly,
	 * independent of any nsc code or the io.s helpers being correct.
	 * if this byte never shows up in the serial log, the fault is in
	 * this file, not in kmain or anything it calls. */
	movw $0x3f8, %dx
	movb $0x42, %al
	outb %al, %dx
 
	call kmain
 
	/* kmain should never return; halt forever if it somehow does */
halt_forever:
	cli
	hlt
	jmp halt_forever
powered by btf.