do not edit — generated by btf.
git.druid.rocksindexdruid520kaboomsrc/kernel/fd.nsc

src/kernel/fd.nsc


/*
 * file descriptors: 0/1/2 are the fixed stdin/stdout/stderr (keyboard
 * and vga+serial console -- stderr is just stdout, no separate error
 * stream exists), 3+ are open files. each open file's whole content
 * is cached in its fd slot at open() time -- simplest thing that works
 * at this scale. each slot's buffer is sized to the real file it's
 * holding (see fd_buf_need), not a fixed 63488 bytes: kfs files can
 * now be up to 8450048 bytes (double indirection, kfs.nsc), and
 * kfs_read_file writes the whole file into whatever buffer it's given
 * with no bounds of its own. a real seek/range-read against the disk
 * on every read() call, instead of caching the whole file, is still a
 * fine follow-up once caching whole files stops being good enough.
 */
 
include "../fs/kfs.nsh";
include "virtfs.nsh";
include "alloc.nsh";
 
global struct arena fs_colosseum;
global ptr fd_cache_vec;
 
i32 kbd_getchar(void);
void vga_putc(u8 ch);
void vga_backspace(void);
void serial_putc(u8 ch);
void sti(void);
void halt(void);
 
/* fixed-size open-file table, 5 slots (fd 3..7), backed by
 * fd_cache_vec (alloc.nsh) -- one real colosseum vector, replacing
 * what used to be 5 hand-duplicated fd_inuseN/fd_dataN/fd_capN
 * globals dispatched by a plain if-chain (nsc has no arrays-of-
 * structs, so the vector's own bookkeeping is a raw carved region,
 * not a native array -- see alloc.nsc). fd_size/fd_cursor stay real
 * nsc scalar arrays (that much the language does support), indexed
 * the same way. a slot's "in use" state lives on the arena itself
 * (issued/done, see alloc.nsc) -- there's no separate inuse flag to
 * keep in sync by hand anymore.
 *
 * fd_dataptr[idx], 0 in the common case, holds the one real exception:
 * a file bigger than fd_cache_vec's own fixed 65536-byte slot size
 * (kfs files can be up to 8450048 bytes) gets a direct, uncounted
 * arena_alloc against fs_colosseum instead of a vector slot's own
 * buffer -- the vector slot is still checked out (for its fd-number
 * bookkeeping and issued/done lifecycle), it just isn't where this
 * particular fd's real content lives. never reclaimed on close, same
 * as this rare case has always behaved. */
global u64 fd_size[5];
global u64 fd_cursor[5];
global ptr fd_dataptr[5];
 
/* the real data buffer for slot IDX/SLOT right now: fd_dataptr[idx]
 * if set (the oversized case), else the vector slot's own fixed
 * region. */
ptr
fd_data_for(u64 idx, ptr slot)
{
	if(fd_dataptr[idx] != (ptr)0)
	{
		return fd_dataptr[idx];
	}
	return (ptr)slot->start;
}
 
/* SLOT for fd (3..7), or 0 if fd is out of that range -- the one
 * place the fd-number <-> vector-index arithmetic happens. */
ptr
fd_slot(i32 fd)
{
	if(fd < 3 || fd > 7)
	{
		return (ptr)0;
	}
	return (ptr)((u64)fd_cache_vec->children + (u64)(fd - 3) * (u64)sizeof(struct arena));
}
 
/* true only once a slot has actually been opened and not yet closed --
 * issued alone isn't enough (it stays 1 forever after a slot's first
 * ever use, by design, see alloc.nsc), and done alone isn't enough
 * either (a never-issued fresh slot also has done == 0). */
i32
fd_slot_open(ptr slot)
{
	if(slot->issued == (u64)1 && slot->done == (u64)0)
	{
		return 1;
	}
	return 0;
}
 
/* how big a slot's buffer has to be to safely hold inode's content:
 * its real kfs_file_size rounded up to a multiple of 8 (kfs_read_file's
 * own last word-sized store can run up to 7 bytes past the logical
 * end, see there), floored at 63488 -- the fixed size every slot used
 * to be, so a small file or a /proc//int virtual file (whose generated
 * content has nothing to do with its on-disk placeholder's size) gets
 * exactly the buffer it always did, and the common case never
 * allocates anything new. sized to the ACTUAL file, never a blanket
 * 8450048-byte worst case -- fs_colosseum is only 4mib, and eagerly
 * reserving the max for every slot would exhaust it on the first
 * open. returns 0 for a size past kfs's own 8450048-byte max file
 * size (a corrupt inode -- nothing kfs_write_file wrote can be that
 * big), which fd_open treats as a failed open. */
u64
fd_buf_need(i32 inode)
{
	u64 size;
 
	size = kfs_file_size(inode);
	if(size > (u64)8450048)
	{
		return (u64)0;
	}
	size = (size + (u64)7) & ~(u64)7;
	if(size < (u64)63488)
	{
		size = (u64)63488;
	}
	return size;
}
 
/* fills data (a slot's buffer, at least fd_buf_need bytes) with a
 * file's actual content, and returns how many bytes are really
 * there -- virtfs_read (virtfs.nsc) gets first refusal whenever the
 * path resolved to a name whose PARENT is a known virtual directory
 * (/proc, /int), so an open() of one of those always returns live,
 * freshly-generated content instead of the empty placeholder bytes
 * disk.pl put on disk for it; -1 from virtfs_read (not a name it
 * recognizes, or have_parent itself failed -- a bare root-level open,
 * which is never a virtual-file candidate) falls straight through to
 * the real, disk-backed kfs_read_file, unchanged from before virtfs
 * existed. */
u64
fd_load_content(i32 inode, i32 have_parent, u64 parent_lba, ptr base_name, u64 base_len, ptr data)
{
	i32 vsz;
 
	if(have_parent == 1)
	{
		vsz = virtfs_read(parent_lba, base_name, base_len, data);
		if(vsz >= 0)
		{
			return (u64)vsz;
		}
	}
	return kfs_read_file(inode, data);
}
 
global i32
fd_open(ptr path, u64 path_len)
{
	i32 inode;
	u64 sz;
	i32 have_parent;
	u64 parent_lba;
	i32 parent_inode_unused;
	u64 base_off;
	ptr base_name;
	u64 base_len;
	u64 need;
	ptr nb;
	ptr slot;
	u64 idx;
 
	/* kfs_resolve, not kfs_dir_find: a bare name (no '/') resolves
	 * against the cwd exactly like kfs_dir_find always did, but this
	 * also walks a real multi-component or absolute path ("doc/x",
	 * "/doc/x") one directory at a time -- see kfs_resolve's own
	 * comment for why cat/open needed that and kfs_dir_find alone
	 * never could. */
	inode = kfs_resolve(path, path_len);
	if(inode < 0)
	{
		return -1;
	}
	/* plain files only (KFS_TYPE_FILE). a directory used to "open"
	 * fine, with its zero size read back as an empty file -- so
	 * `mv dir file`/`cp dir file` read 0 bytes and kfs_save truncated
	 * the real destination to nothing. refused here, once, for every
	 * caller (cat, cp, mv, ed, sh's script loader, ...). */
	if(kfs_inode_type(inode) != (i64)1)
	{
		return -1;
	}
	if(kfs_check_perm(inode, (u64)1) == 0)
	{
		return -1; /* no read permission */
	}
 
	/* also resolve the parent directory -- fd_load_content uses it to
	 * recognize a virtual-file candidate (see its own comment). doing
	 * this unconditionally, even for the overwhelming majority of
	 * opens that are ordinary disk files, costs one extra directory
	 * walk per open() -- cheap, and far simpler than teaching
	 * kfs_resolve itself to also hand back the parent it walked
	 * through on its way to the final inode. parent_inode_unused: this
	 * caller only wants the parent's DATA lba, not its own permission
	 * bits (there's nothing here that creates or removes a dirent, the
	 * two operations that actually need the parent directory's write
	 * bit -- see kfs_create/kfs_mkdir/kfs_rm/kfs_rmdir instead). */
	have_parent = kfs_resolve_parent(path, path_len, &parent_lba, &parent_inode_unused, &base_off);
	base_name = path + base_off;
	base_len = path_len - base_off;
 
	need = fd_buf_need(inode);
	if(need == (u64)0)
	{
		return -1;
	}
 
	slot = arena_child(fd_cache_vec);
	if(slot == (ptr)0)
	{
		return -1; /* every slot is genuinely in use -- no free fd */
	}
	idx = ((u64)slot - (u64)fd_cache_vec->children) / (u64)sizeof(struct arena);
 
	if(need <= (u64)65536)
	{
		fd_dataptr[idx] = (ptr)0; /* common case -- the vector slot's
		                           * own fixed region holds this file */
	}
	else
	{
		/* rare: bigger than fd_cache_vec's own fixed slot size. a
		 * direct, uncounted arena_alloc against fs_colosseum instead --
		 * never reclaimed, same as this case has always behaved. the
		 * vector slot itself is still what's checked out (it owns this
		 * fd's issued/done lifecycle and its number), it just isn't
		 * where the real content lives. */
		nb = arena_alloc(&fs_colosseum, need);
		if(nb == (ptr)0)
		{
			arena_mark_done(slot); /* give the slot straight back --
			                        * never really used for anything */
			return -1;
		}
		fd_dataptr[idx] = nb;
	}
 
	sz = fd_load_content(inode, have_parent, parent_lba, base_name, base_len, fd_data_for(idx, slot));
	fd_size[idx] = sz;
	fd_cursor[idx] = (u64)0;
	return (i32)(idx + (u64)3);
}
 
/* the size of an open file fd's content -- exactly what fd_read will
 * hand back from the start (virtfs's live-generated content for /proc
 * and /int, kfs_read_file's for everything else, see fd_load_content),
 * so userspace can size one buffer to the real file up front instead
 * of guessing a fixed size, the same way fd_buf_need sizes the slot
 * itself kernel-side. -1 for anything that isn't an open file slot. */
global i64
fd_fsize(i32 fd)
{
	ptr slot;
 
	slot = fd_slot(fd);
	if(slot == (ptr)0 || fd_slot_open(slot) == 0)
	{
		return -1;
	}
	return (i64)fd_size[(u64)(fd - 3)];
}
 
/* -1 for anything that isn't a currently-open file slot, including a
 * valid slot number that was never opened or is already closed -- an
 * earlier version returned 0 (success) for those too. */
global i32
fd_close(i32 fd)
{
	ptr slot;
 
	slot = fd_slot(fd);
	if(slot == (ptr)0 || fd_slot_open(slot) == 0)
	{
		return -1;
	}
	arena_mark_done(slot);
	return 0;
}
 
/* which file slots are open right now, one bit per slot (bit 0 = fd 3
 * .. bit 4 = fd 7) -- sys_exec (exec.nsc) snapshots this before
 * running a child and hands it back to fd_close_unless once the child
 * returns, so anything the child opened and never closed gets closed
 * at its own exit, while everything the caller already had open
 * stays open. */
global u32
fd_open_mask(void)
{
	u32 m;
 
	m = (u32)0;
	if(fd_slot_open(fd_slot(3)) == 1) { m = m | (u32)1; }
	if(fd_slot_open(fd_slot(4)) == 1) { m = m | (u32)2; }
	if(fd_slot_open(fd_slot(5)) == 1) { m = m | (u32)4; }
	if(fd_slot_open(fd_slot(6)) == 1) { m = m | (u32)8; }
	if(fd_slot_open(fd_slot(7)) == 1) { m = m | (u32)16; }
	return m;
}
 
/* closes every open file slot whose bit is NOT set in keep (see
 * fd_open_mask). unconditional mark_done, same as this always did
 * (the original set fd_inuseN = 0 regardless of prior state too) --
 * arena_mark_done on an already-done or never-issued slot is harmless. */
global void
fd_close_unless(u32 keep)
{
	if((keep & (u32)1) == (u32)0) { arena_mark_done(fd_slot(3)); }
	if((keep & (u32)2) == (u32)0) { arena_mark_done(fd_slot(4)); }
	if((keep & (u32)4) == (u32)0) { arena_mark_done(fd_slot(5)); }
	if((keep & (u32)8) == (u32)0) { arena_mark_done(fd_slot(6)); }
	if((keep & (u32)16) == (u32)0) { arena_mark_done(fd_slot(7)); }
}
 
/* reads a full line (up to and including the newline) into buf, up to
 * maxlen bytes. blocks for every character, not just the first --
 * real keystrokes arrive with human-typing-speed gaps between them
 * (tens to hundreds of ms), so a design that only blocks for the
 * first byte and then grabs "whatever else happens to already be
 * queued" would return after a single character almost every time,
 * which is exactly the bug the first version of this function had
 * (caught by actually typing a multi-character line at a real qemu
 * instance with real gaps between keystrokes, not by reasoning about
 * it -- it looked fine on paper). `int 0x80` is an interrupt gate,
 * which auto-clears IF on entry, so hlt would never wake up without
 * re-enabling interrupts first: a halted cpu with IF=0 can never
 * receive the very keyboard irq it's waiting for. sti() here doesn't
 * leak past this handler -- iretq restores the original caller's
 * rflags (saved at interrupt entry), not whatever sti() set
 * mid-handler. buf must be padded to a multiple of 8 bytes, same as
 * every other packed-word destination in this kernel: *ptr is always
 * a full 8-byte load/store, so filling it byte-by-byte is a
 * read-modify-write of the containing word, which can touch up to 7
 * bytes past the last byte actually written.
 *
 * backspace (ascii 8) is handled here, not at the keyboard-driver
 * level: kbd_handle_irq echoes every OTHER character the moment it's
 * typed, before any reader ever sees it, so erasing a character
 * (both the visible glyph and the byte already written into buf)
 * has to happen wherever the line's current length is actually
 * known -- kbd.nsc itself has no idea how many characters this
 * particular read has accepted so far. a backspace with nothing to
 * erase (n==0) is simply dropped. */
i64
fd_read_stdin(ptr buf, u64 maxlen)
{
	u64 n;
	i32 ch;
	ptr dst;
	u64 word;
	u64 wordoff;
	u64 byteoff;
 
	if(maxlen == (u64)0)
	{
		return (i64)0;
	}
 
	dst = buf;
	n = 0;
	while(n < maxlen)
	{
		ch = kbd_getchar();
		if(ch < 0)
		{
			sti();
			while(ch < 0)
			{
				halt();
				ch = kbd_getchar();
			}
		}
 
		if(ch == 8)
		{
			if(n > (u64)0)
			{
				n = n - (u64)1;
				vga_backspace();
				/* the standard backspace-space-backspace idiom: a
				 * bare 8 just moves a real terminal's cursor back
				 * without erasing anything already displayed, so the
				 * next character typed would overwrite in place
				 * rather than the line visibly shrinking. this is
				 * the ONLY thing a real interactive session over
				 * -nographic (serial and the console sharing one
				 * stream, no separate vga window at all) actually
				 * sees -- found by testing backspace specifically
				 * and noticing the erased character was still
				 * sitting there, not by reading the code. */
				serial_putc((u8)8);
				serial_putc((u8)32);
				serial_putc((u8)8);
			}
		}
		else
		{
			wordoff = n & ~(u64)7;
			byteoff = n & (u64)7;
			word = (u64)*(dst + wordoff);
			word = word & ~((u64)0xff << (byteoff * (u64)8));
			word = word | ((u64)(u8)ch << (byteoff * (u64)8));
			*(dst + wordoff) = (i64)word;
 
			n = n + (u64)1;
			if(ch == 10)
			{
				return (i64)n;
			}
		}
	}
	return (i64)n;
}
 
u8
ptr_byte_at(ptr base, u64 index)
{
	ptr p;
	u64 word;
	p = base + index;
	word = (u64)*p;
	return (u8)(word & (u64)0xff);
}
 
i64
fd_read_from_cache(ptr data, u64 size, ptr cursor, ptr buf, u64 maxlen)
{
	u64 pos;
	u64 remaining;
	u64 n;
	u64 i;
	u64 word;
	u64 wordoff;
	u64 byteoff;
	ptr dst;
	u8 b;
 
	pos = (u64)*cursor;
	if(pos >= size)
	{
		return (i64)0; /* eof */
	}
	remaining = size - pos;
	n = maxlen;
	if(n > remaining)
	{
		n = remaining;
	}
 
	dst = buf;
	i = 0;
	while(i < n)
	{
		b = ptr_byte_at(data, pos + i);
		wordoff = i & ~(u64)7;
		byteoff = i & (u64)7;
		word = (u64)*(dst + wordoff);
		word = word & ~((u64)0xff << (byteoff * (u64)8));
		word = word | ((u64)b << (byteoff * (u64)8));
		*(dst + wordoff) = (i64)word;
		i = i + (u64)1;
	}
	*cursor = (i64)(pos + n);
	return (i64)n;
}
 
global i64
fd_read(i32 fd, ptr buf, u64 maxlen)
{
	ptr slot;
	u64 idx;
 
	if(fd == 0)
	{
		return fd_read_stdin(buf, maxlen);
	}
	/* only an open slot: a closed (or never-opened) one still has the
	 * last file's cached content sitting there, and an earlier version
	 * happily handed that back. */
	slot = fd_slot(fd);
	if(slot == (ptr)0 || fd_slot_open(slot) == 0)
	{
		return -1;
	}
	idx = (u64)(fd - 3);
	return fd_read_from_cache(fd_data_for(idx, slot), fd_size[idx], &fd_cursor[idx], buf, maxlen);
}
 
/* console only -- files aren't writable through this v1 fd layer
 * (nothing needs it yet: cat/ls/echo are all read/list/stdout-only). */
global i64
fd_write(i32 fd, ptr buf, u64 len)
{
	u64 i;
	u8 ch;
 
	if(fd == 1 || fd == 2)
	{
		i = 0;
		while(i < len)
		{
			ch = ptr_byte_at(buf, i);
			vga_putc(ch);
			serial_putc(ch);
			i = i + (u64)1;
		}
		return (i64)len;
	}
	return -1;
}
powered by btf.