| git.druid.rocks | index | druid520 | kaboom | src/ | boot/ | boot.s |
src/boot/boot.s
/*
* multiboot1 entry + 32-bit protected mode -> long mode transition.
*
* multiboot1 only guarantees 32-bit protected mode with flat segments.
* nscc emits pure 64-bit (long mode) code, so nothing nsc-compiled can
* run until we've built page tables, turned on PAE+LME+paging, and
* far-jumped into a 64-bit code segment. get any of this wrong and the
* cpu triple-faults silently (instant reboot, no error) -- this is the
* single highest-risk piece of the whole kernel and gets tested before
* anything else is built on top of it.
*/
/*
* qemu's -kernel loader hard-refuses any 64-bit-class ELF through its
* multiboot1 path (confirmed empirically: "Cannot load x86-64 image,
* give a 32bit one") regardless of what's actually inside the load
* segments -- and nscc only emits 64-bit object files, so the whole
* kernel has to link as ELF64. a multiboot1 header was tried
* alongside the PVH note below, but qemu detects multiboot1 first and
* that path's ELFCLASS32 check runs unconditionally before PVH is
* ever considered -- having both present doesn't work, it's one or
* the other. PVH wins here because `qemu -kernel` working today is a
* hard requirement; a real multiboot1 header for GRUB/real hardware
* would need a genuinely separate 32-bit build, not attempted yet.
*
* PVH (xen/hvm direct-boot protocol, also implemented by qemu) is
* fine with an ELF64 container -- it just needs a note pointing at a
* 32-bit physical entry address. confirmed working empirically before
* wiring it in for real.
*/
.section .note.pvh, "a", @note
.align 4
.long 4 /* namesz: "Xen\0" */
.long 4 /* descsz: 4-byte phys32 entry addr */
.long 0x12 /* type: XEN_ELFNOTE_PHYS32_ENTRY */
.ascii "Xen\0"
.long _start
.section .bss
.align 4096
.global pd
pml4:
.skip 4096
pdpt:
.skip 4096
pd:
.skip 4096
.align 4
.global start_info_ptr32
start_info_ptr32:
.skip 4
/* dynamite's own real-bios handoff (Task 13) -- see dynamite.s's own
* comment just before its `jmp KERNEL_ENTRY_ADDR` for why these three
* registers specifically (ebp/esi/edx) and why an explicit magic
* rather than an inferred one. unused/zero on a genuine pvh boot,
* since qemu's -kernel/pvh loader jumps straight to _start without
* ever running dynamite.s at all. */
.align 4
.global bios_boot_flag
bios_boot_flag:
.skip 4
.global bios_e820_ptr32
bios_e820_ptr32:
.skip 4
.global bios_e820_count32
bios_e820_count32:
.skip 4
/* stack guard gap: 512kib of never-used .bss between the live page
* tables above and the kernel stack below. there is exactly one stack
* in this kernel -- every "process" (exec.nsc) runs on it as an
* ordinary nested call -- and it grows DOWN toward stack_bottom, so
* with nothing in between, an overflow past its 64kib landed straight
* in pd/pdpt/pml4: found for real, a shebang script naming itself as
* its own interpreter recursed until it overwrote them and the
* machine triple-faulted (full reboot). exec.nsc now caps that
* recursion directly; this gap is defense-in-depth for any OTHER deep
* stack use, not a structural guarantee -- an overflow of more than
* 64kib + 512kib would still reach the page tables. the real fix is an
* actual unmapped guard page below stack_bottom (a not-present 4kib
* page, so an overflow faults immediately instead of corrupting
* anything), which means splitting fill_pd's first 2mib huge page
* into a real 4kib page table -- a genuine future improvement, not
* attempted here. costs nothing in the binary (.bss is nobits, and
* this whole range is already inside the 1gib identity map); only
* the zeroing loop above touches it, once, at boot. */
.skip 524288
.align 16
stack_bottom:
.skip 65536
stack_top:
.section .rodata
gdt64:
.quad 0 /* null descriptor */
.quad 0x00af9a000000ffff /* 0x08: 64-bit code, ring0 */
.quad 0x00af92000000ffff /* 0x10: 64-bit data, ring0 */
gdt64_end:
gdt64_ptr:
.word gdt64_end - gdt64 - 1
.quad gdt64
.section .text
.code32
.global _start
.type _start, @function
_start:
cli
movl $stack_top, %esp
/* zero .bss unconditionally. dynamite's raw objcopy blob (real
* bios path) doesn't include .bss at all -- objcopy -O binary only
* emits PROGBITS content -- and real hardware ram starts as
* garbage, unlike qemu's guest ram, which happens to start zeroed.
* without this, boot.s's own page tables and any zero-initialized
* kernel globals would start from garbage on real hardware.
* redundant but harmless on the existing -kernel/pvh path, whose
* elf loader already zeroes .bss per the program header.
*
* %ebx (the PVH hvm_start_info pointer, per the entry note above)
* is deliberately still untouched here and saved to start_info_ptr32
* (itself a .bss slot) only after this loop, not before: this loop
* zeroes the entire .bss section, so storing into a .bss slot any
* earlier would just get wiped out by the same loop before anything
* ever reads it back -- confirmed the hard way empirically (the
* pre-fix ordering read back a zeroed slot, i.e. a null pointer,
* whose dereference produced IVT-segment-shaped garbage instead of
* the real hvm_start_info magic). none of edi/ecx/eax used below
* touch ebx, so it survives this loop in the register untouched. */
movl $_bss_start, %edi
movl $_bss_end, %ecx
subl %edi, %ecx
shrl $2, %ecx
xorl %eax, %eax
cld /* pvh and bios alike leave df unspecified */
rep stosl
movl %ebx, start_info_ptr32
/* dynamite's real-bios handoff (Task 13), saved at this exact same
* point for the exact same reason as %ebx just above: none of
* ebp/esi/edx are touched by the zeroing loop or anything before
* it, so they still hold whatever dynamite.s (or, on a real pvh
* boot, nothing at all) last put there. */
movl %ebp, bios_boot_flag
movl %esi, bios_e820_ptr32
movl %edx, bios_e820_count32
/* identity-map the first 1GiB with 2MiB pages: pml4[0] -> pdpt,
* pdpt[0] -> pd, pd[0..511] -> 2MiB pages covering 0..1GiB. this
* is more than the kernel needs right now but avoids revisiting
* this table for a good while. */
movl $pdpt, %eax
orl $0x03, %eax
movl %eax, pml4
movl $pd, %eax
orl $0x03, %eax
movl %eax, pdpt
movl $0, %ecx
fill_pd:
movl %ecx, %eax
shll $21, %eax
orl $0x83, %eax /* present, writable, huge (2MiB) */
movl %eax, pd(,%ecx,8)
movl $0, pd+4(,%ecx,8)
incl %ecx
cmpl $512, %ecx
jne fill_pd
/* enable PAE (CR4.PAE) */
movl %cr4, %eax
orl $0x20, %eax
movl %eax, %cr4
/* load CR3 with the PML4 base */
movl $pml4, %eax
movl %eax, %cr3
/* set EFER.LME (long mode enable) via MSR 0xC0000080 */
movl $0xC0000080, %ecx
rdmsr
orl $0x100, %eax
wrmsr
/* enable paging (CR0.PG) -- now in compatibility mode */
movl %cr0, %eax
orl $0x80000000, %eax
movl %eax, %cr0
lgdt gdt64_ptr
ljmp $0x08, $long_mode_entry
.code64
long_mode_entry:
movw $0x10, %ax
movw %ax, %ds
movw %ax, %es
movw %ax, %ss
movw %ax, %fs
movw %ax, %gs
movq $stack_top, %rsp
movq $0, %rbp
/* raw pre-nsc sanity beacon: write 'B' to COM1 (0x3f8) directly,
* independent of any nsc code or the io.s helpers being correct.
* if this byte never shows up in the serial log, the fault is in
* this file, not in kmain or anything it calls. */
movw $0x3f8, %dx
movb $0x42, %al
outb %al, %dx
call kmain
/* kmain should never return; halt forever if it somehow does */
halt_forever:
cli
hlt
jmp halt_forever