| git.druid.rocks | index | druid520 | kaboom | src/ | kernel/ | fd.nsc |
src/kernel/fd.nsc
/*
* file descriptors: 0/1/2 are the fixed stdin/stdout/stderr (keyboard
* and vga+serial console -- stderr is just stdout, no separate error
* stream exists), 3+ are open files. each open file's whole content
* is cached in its fd slot at open() time -- simplest thing that works
* at this scale. each slot's buffer is sized to the real file it's
* holding (see fd_buf_need), not a fixed 63488 bytes: kfs files can
* now be up to 8450048 bytes (double indirection, kfs.nsc), and
* kfs_read_file writes the whole file into whatever buffer it's given
* with no bounds of its own. a real seek/range-read against the disk
* on every read() call, instead of caching the whole file, is still a
* fine follow-up once caching whole files stops being good enough.
*/
include "../fs/kfs.nsh";
include "virtfs.nsh";
include "alloc.nsh";
global struct arena fs_colosseum;
global ptr fd_cache_vec;
i32 kbd_getchar(void);
void vga_putc(u8 ch);
void vga_backspace(void);
void serial_putc(u8 ch);
void sti(void);
void halt(void);
/* fixed-size open-file table, 5 slots (fd 3..7), backed by
* fd_cache_vec (alloc.nsh) -- one real colosseum vector, replacing
* what used to be 5 hand-duplicated fd_inuseN/fd_dataN/fd_capN
* globals dispatched by a plain if-chain (nsc has no arrays-of-
* structs, so the vector's own bookkeeping is a raw carved region,
* not a native array -- see alloc.nsc). fd_size/fd_cursor stay real
* nsc scalar arrays (that much the language does support), indexed
* the same way. a slot's "in use" state lives on the arena itself
* (issued/done, see alloc.nsc) -- there's no separate inuse flag to
* keep in sync by hand anymore.
*
* fd_dataptr[idx], 0 in the common case, holds the one real exception:
* a file bigger than fd_cache_vec's own fixed 65536-byte slot size
* (kfs files can be up to 8450048 bytes) gets a direct, uncounted
* arena_alloc against fs_colosseum instead of a vector slot's own
* buffer -- the vector slot is still checked out (for its fd-number
* bookkeeping and issued/done lifecycle), it just isn't where this
* particular fd's real content lives. never reclaimed on close, same
* as this rare case has always behaved. */
global u64 fd_size[5];
global u64 fd_cursor[5];
global ptr fd_dataptr[5];
/* the real data buffer for slot IDX/SLOT right now: fd_dataptr[idx]
* if set (the oversized case), else the vector slot's own fixed
* region. */
ptr
fd_data_for(u64 idx, ptr slot)
{
if(fd_dataptr[idx] != (ptr)0)
{
return fd_dataptr[idx];
}
return (ptr)slot->start;
}
/* SLOT for fd (3..7), or 0 if fd is out of that range -- the one
* place the fd-number <-> vector-index arithmetic happens. */
ptr
fd_slot(i32 fd)
{
if(fd < 3 || fd > 7)
{
return (ptr)0;
}
return (ptr)((u64)fd_cache_vec->children + (u64)(fd - 3) * (u64)sizeof(struct arena));
}
/* true only once a slot has actually been opened and not yet closed --
* issued alone isn't enough (it stays 1 forever after a slot's first
* ever use, by design, see alloc.nsc), and done alone isn't enough
* either (a never-issued fresh slot also has done == 0). */
i32
fd_slot_open(ptr slot)
{
if(slot->issued == (u64)1 && slot->done == (u64)0)
{
return 1;
}
return 0;
}
/* how big a slot's buffer has to be to safely hold inode's content:
* its real kfs_file_size rounded up to a multiple of 8 (kfs_read_file's
* own last word-sized store can run up to 7 bytes past the logical
* end, see there), floored at 63488 -- the fixed size every slot used
* to be, so a small file or a /proc//int virtual file (whose generated
* content has nothing to do with its on-disk placeholder's size) gets
* exactly the buffer it always did, and the common case never
* allocates anything new. sized to the ACTUAL file, never a blanket
* 8450048-byte worst case -- fs_colosseum is only 4mib, and eagerly
* reserving the max for every slot would exhaust it on the first
* open. returns 0 for a size past kfs's own 8450048-byte max file
* size (a corrupt inode -- nothing kfs_write_file wrote can be that
* big), which fd_open treats as a failed open. */
u64
fd_buf_need(i32 inode)
{
u64 size;
size = kfs_file_size(inode);
if(size > (u64)8450048)
{
return (u64)0;
}
size = (size + (u64)7) & ~(u64)7;
if(size < (u64)63488)
{
size = (u64)63488;
}
return size;
}
/* fills data (a slot's buffer, at least fd_buf_need bytes) with a
* file's actual content, and returns how many bytes are really
* there -- virtfs_read (virtfs.nsc) gets first refusal whenever the
* path resolved to a name whose PARENT is a known virtual directory
* (/proc, /int), so an open() of one of those always returns live,
* freshly-generated content instead of the empty placeholder bytes
* disk.pl put on disk for it; -1 from virtfs_read (not a name it
* recognizes, or have_parent itself failed -- a bare root-level open,
* which is never a virtual-file candidate) falls straight through to
* the real, disk-backed kfs_read_file, unchanged from before virtfs
* existed. */
u64
fd_load_content(i32 inode, i32 have_parent, u64 parent_lba, ptr base_name, u64 base_len, ptr data)
{
i32 vsz;
if(have_parent == 1)
{
vsz = virtfs_read(parent_lba, base_name, base_len, data);
if(vsz >= 0)
{
return (u64)vsz;
}
}
return kfs_read_file(inode, data);
}
global i32
fd_open(ptr path, u64 path_len)
{
i32 inode;
u64 sz;
i32 have_parent;
u64 parent_lba;
i32 parent_inode_unused;
u64 base_off;
ptr base_name;
u64 base_len;
u64 need;
ptr nb;
ptr slot;
u64 idx;
/* kfs_resolve, not kfs_dir_find: a bare name (no '/') resolves
* against the cwd exactly like kfs_dir_find always did, but this
* also walks a real multi-component or absolute path ("doc/x",
* "/doc/x") one directory at a time -- see kfs_resolve's own
* comment for why cat/open needed that and kfs_dir_find alone
* never could. */
inode = kfs_resolve(path, path_len);
if(inode < 0)
{
return -1;
}
/* plain files only (KFS_TYPE_FILE). a directory used to "open"
* fine, with its zero size read back as an empty file -- so
* `mv dir file`/`cp dir file` read 0 bytes and kfs_save truncated
* the real destination to nothing. refused here, once, for every
* caller (cat, cp, mv, ed, sh's script loader, ...). */
if(kfs_inode_type(inode) != (i64)1)
{
return -1;
}
if(kfs_check_perm(inode, (u64)1) == 0)
{
return -1; /* no read permission */
}
/* also resolve the parent directory -- fd_load_content uses it to
* recognize a virtual-file candidate (see its own comment). doing
* this unconditionally, even for the overwhelming majority of
* opens that are ordinary disk files, costs one extra directory
* walk per open() -- cheap, and far simpler than teaching
* kfs_resolve itself to also hand back the parent it walked
* through on its way to the final inode. parent_inode_unused: this
* caller only wants the parent's DATA lba, not its own permission
* bits (there's nothing here that creates or removes a dirent, the
* two operations that actually need the parent directory's write
* bit -- see kfs_create/kfs_mkdir/kfs_rm/kfs_rmdir instead). */
have_parent = kfs_resolve_parent(path, path_len, &parent_lba, &parent_inode_unused, &base_off);
base_name = path + base_off;
base_len = path_len - base_off;
need = fd_buf_need(inode);
if(need == (u64)0)
{
return -1;
}
slot = arena_child(fd_cache_vec);
if(slot == (ptr)0)
{
return -1; /* every slot is genuinely in use -- no free fd */
}
idx = ((u64)slot - (u64)fd_cache_vec->children) / (u64)sizeof(struct arena);
if(need <= (u64)65536)
{
fd_dataptr[idx] = (ptr)0; /* common case -- the vector slot's
* own fixed region holds this file */
}
else
{
/* rare: bigger than fd_cache_vec's own fixed slot size. a
* direct, uncounted arena_alloc against fs_colosseum instead --
* never reclaimed, same as this case has always behaved. the
* vector slot itself is still what's checked out (it owns this
* fd's issued/done lifecycle and its number), it just isn't
* where the real content lives. */
nb = arena_alloc(&fs_colosseum, need);
if(nb == (ptr)0)
{
arena_mark_done(slot); /* give the slot straight back --
* never really used for anything */
return -1;
}
fd_dataptr[idx] = nb;
}
sz = fd_load_content(inode, have_parent, parent_lba, base_name, base_len, fd_data_for(idx, slot));
fd_size[idx] = sz;
fd_cursor[idx] = (u64)0;
return (i32)(idx + (u64)3);
}
/* the size of an open file fd's content -- exactly what fd_read will
* hand back from the start (virtfs's live-generated content for /proc
* and /int, kfs_read_file's for everything else, see fd_load_content),
* so userspace can size one buffer to the real file up front instead
* of guessing a fixed size, the same way fd_buf_need sizes the slot
* itself kernel-side. -1 for anything that isn't an open file slot. */
global i64
fd_fsize(i32 fd)
{
ptr slot;
slot = fd_slot(fd);
if(slot == (ptr)0 || fd_slot_open(slot) == 0)
{
return -1;
}
return (i64)fd_size[(u64)(fd - 3)];
}
/* -1 for anything that isn't a currently-open file slot, including a
* valid slot number that was never opened or is already closed -- an
* earlier version returned 0 (success) for those too. */
global i32
fd_close(i32 fd)
{
ptr slot;
slot = fd_slot(fd);
if(slot == (ptr)0 || fd_slot_open(slot) == 0)
{
return -1;
}
arena_mark_done(slot);
return 0;
}
/* which file slots are open right now, one bit per slot (bit 0 = fd 3
* .. bit 4 = fd 7) -- sys_exec (exec.nsc) snapshots this before
* running a child and hands it back to fd_close_unless once the child
* returns, so anything the child opened and never closed gets closed
* at its own exit, while everything the caller already had open
* stays open. */
global u32
fd_open_mask(void)
{
u32 m;
m = (u32)0;
if(fd_slot_open(fd_slot(3)) == 1) { m = m | (u32)1; }
if(fd_slot_open(fd_slot(4)) == 1) { m = m | (u32)2; }
if(fd_slot_open(fd_slot(5)) == 1) { m = m | (u32)4; }
if(fd_slot_open(fd_slot(6)) == 1) { m = m | (u32)8; }
if(fd_slot_open(fd_slot(7)) == 1) { m = m | (u32)16; }
return m;
}
/* closes every open file slot whose bit is NOT set in keep (see
* fd_open_mask). unconditional mark_done, same as this always did
* (the original set fd_inuseN = 0 regardless of prior state too) --
* arena_mark_done on an already-done or never-issued slot is harmless. */
global void
fd_close_unless(u32 keep)
{
if((keep & (u32)1) == (u32)0) { arena_mark_done(fd_slot(3)); }
if((keep & (u32)2) == (u32)0) { arena_mark_done(fd_slot(4)); }
if((keep & (u32)4) == (u32)0) { arena_mark_done(fd_slot(5)); }
if((keep & (u32)8) == (u32)0) { arena_mark_done(fd_slot(6)); }
if((keep & (u32)16) == (u32)0) { arena_mark_done(fd_slot(7)); }
}
/* reads a full line (up to and including the newline) into buf, up to
* maxlen bytes. blocks for every character, not just the first --
* real keystrokes arrive with human-typing-speed gaps between them
* (tens to hundreds of ms), so a design that only blocks for the
* first byte and then grabs "whatever else happens to already be
* queued" would return after a single character almost every time,
* which is exactly the bug the first version of this function had
* (caught by actually typing a multi-character line at a real qemu
* instance with real gaps between keystrokes, not by reasoning about
* it -- it looked fine on paper). `int 0x80` is an interrupt gate,
* which auto-clears IF on entry, so hlt would never wake up without
* re-enabling interrupts first: a halted cpu with IF=0 can never
* receive the very keyboard irq it's waiting for. sti() here doesn't
* leak past this handler -- iretq restores the original caller's
* rflags (saved at interrupt entry), not whatever sti() set
* mid-handler. buf must be padded to a multiple of 8 bytes, same as
* every other packed-word destination in this kernel: *ptr is always
* a full 8-byte load/store, so filling it byte-by-byte is a
* read-modify-write of the containing word, which can touch up to 7
* bytes past the last byte actually written.
*
* backspace (ascii 8) is handled here, not at the keyboard-driver
* level: kbd_handle_irq echoes every OTHER character the moment it's
* typed, before any reader ever sees it, so erasing a character
* (both the visible glyph and the byte already written into buf)
* has to happen wherever the line's current length is actually
* known -- kbd.nsc itself has no idea how many characters this
* particular read has accepted so far. a backspace with nothing to
* erase (n==0) is simply dropped. */
i64
fd_read_stdin(ptr buf, u64 maxlen)
{
u64 n;
i32 ch;
ptr dst;
u64 word;
u64 wordoff;
u64 byteoff;
if(maxlen == (u64)0)
{
return (i64)0;
}
dst = buf;
n = 0;
while(n < maxlen)
{
ch = kbd_getchar();
if(ch < 0)
{
sti();
while(ch < 0)
{
halt();
ch = kbd_getchar();
}
}
if(ch == 8)
{
if(n > (u64)0)
{
n = n - (u64)1;
vga_backspace();
/* the standard backspace-space-backspace idiom: a
* bare 8 just moves a real terminal's cursor back
* without erasing anything already displayed, so the
* next character typed would overwrite in place
* rather than the line visibly shrinking. this is
* the ONLY thing a real interactive session over
* -nographic (serial and the console sharing one
* stream, no separate vga window at all) actually
* sees -- found by testing backspace specifically
* and noticing the erased character was still
* sitting there, not by reading the code. */
serial_putc((u8)8);
serial_putc((u8)32);
serial_putc((u8)8);
}
}
else
{
wordoff = n & ~(u64)7;
byteoff = n & (u64)7;
word = (u64)*(dst + wordoff);
word = word & ~((u64)0xff << (byteoff * (u64)8));
word = word | ((u64)(u8)ch << (byteoff * (u64)8));
*(dst + wordoff) = (i64)word;
n = n + (u64)1;
if(ch == 10)
{
return (i64)n;
}
}
}
return (i64)n;
}
u8
ptr_byte_at(ptr base, u64 index)
{
ptr p;
u64 word;
p = base + index;
word = (u64)*p;
return (u8)(word & (u64)0xff);
}
i64
fd_read_from_cache(ptr data, u64 size, ptr cursor, ptr buf, u64 maxlen)
{
u64 pos;
u64 remaining;
u64 n;
u64 i;
u64 word;
u64 wordoff;
u64 byteoff;
ptr dst;
u8 b;
pos = (u64)*cursor;
if(pos >= size)
{
return (i64)0; /* eof */
}
remaining = size - pos;
n = maxlen;
if(n > remaining)
{
n = remaining;
}
dst = buf;
i = 0;
while(i < n)
{
b = ptr_byte_at(data, pos + i);
wordoff = i & ~(u64)7;
byteoff = i & (u64)7;
word = (u64)*(dst + wordoff);
word = word & ~((u64)0xff << (byteoff * (u64)8));
word = word | ((u64)b << (byteoff * (u64)8));
*(dst + wordoff) = (i64)word;
i = i + (u64)1;
}
*cursor = (i64)(pos + n);
return (i64)n;
}
global i64
fd_read(i32 fd, ptr buf, u64 maxlen)
{
ptr slot;
u64 idx;
if(fd == 0)
{
return fd_read_stdin(buf, maxlen);
}
/* only an open slot: a closed (or never-opened) one still has the
* last file's cached content sitting there, and an earlier version
* happily handed that back. */
slot = fd_slot(fd);
if(slot == (ptr)0 || fd_slot_open(slot) == 0)
{
return -1;
}
idx = (u64)(fd - 3);
return fd_read_from_cache(fd_data_for(idx, slot), fd_size[idx], &fd_cursor[idx], buf, maxlen);
}
/* console only -- files aren't writable through this v1 fd layer
* (nothing needs it yet: cat/ls/echo are all read/list/stdout-only). */
global i64
fd_write(i32 fd, ptr buf, u64 len)
{
u64 i;
u8 ch;
if(fd == 1 || fd == 2)
{
i = 0;
while(i < len)
{
ch = ptr_byte_at(buf, i);
vga_putc(ch);
serial_putc(ch);
i = i + (u64)1;
}
return (i64)len;
}
return -1;
}