| git.druid.rocks | index | druid520 | mp | src/ | sandbox.c |
src/sandbox.c
#define _POSIX_C_SOURCE 200809L
#define _NETBSD_SOURCE 1
/* included unconditionally: on a non-Linux build everything below this
* point is compiled out, and -pedantic-errors rejects a translation
* unit left completely empty (which is exactly what this file is
* without the Linux sandbox). */
#include "mpx.h"
#ifdef __linux__
#include <stdio.h>
#include <string.h>
#include <errno.h>
/* unshare()/CLONE_NEWNET/CLONE_NEWNS (<sched.h>) and mount()/its MS_
* mount flags (<sys/mount.h>) are gated behind _GNU_SOURCE on glibc --
* this tree uses no GNU extensions or GNU feature-test macros ever, so
* both are declared here directly instead: the CLONE_ and MS_ values
* are Linux kernel uapi constants (include/uapi/linux/sched.h,
* include/uapi/linux/mount.h), architecture-independent and stable
* across kernel versions, and unshare()/mount()'s signatures are the
* same on every libc that provides them (glibc, musl, ...). */
extern int unshare(int flags);
extern int mount(const char* source, const char* target,
const char* filesystemtype, unsigned long mountflags, const void* data);
#define MP_MS_RDONLY 1UL
#define MP_MS_REMOUNT 32UL
#define MP_MS_BIND 4096UL
#define MP_MS_REC 16384UL
#define MP_MS_PRIVATE 262144UL
#define MP_CLONE_NEWNS 0x00020000
#define MP_CLONE_NEWNET 0x40000000
/* a mount-heavy host/container (overlay-per-container storage drivers,
* some systemd-managed setups) can easily exceed a couple hundred real
* mount points -- once nm hits this cap, listmounts stops scanning
* (not just stops recording) and every mount beyond it silently falls
* back to the single top-level recursive remount this function's own
* comment already explains is NOT reliable for pre-existing submounts.
* generously sized (static storage, not stack, so the cost is a few MB
* of BSS, not a real constraint) rather than exactly right-sized, since
* getting this wrong fails silently with no diagnostic either way. */
#define MAXMOUNTS 4096
#define MAXMOUNTINFO 65536
static long listmounts(char out[MAXMOUNTS][MAXPATH]);
/* a single "mount(NULL, path, NULL, MS_REMOUNT|MS_BIND|MS_RDONLY|MS_REC,
* NULL)" does not reliably cascade read-only to submounts that already
* existed before the remount (a well-known kernel limitation, the same
* reason bwrap/docker walk mountinfo themselves rather than trusting
* MS_REC here) -- a separate filesystem bind-mounted under "/" (/tmp as
* tmpfs is the common case) stays writable otherwise. so every submount
* gets its own explicit remount call. returns the number of mount point
* paths found (each newline-terminated in out, most MAXMOUNTS kept). */
static long
listmounts(char out[MAXMOUNTS][MAXPATH])
{
static char buf[MAXMOUNTINFO];
long n;
long i;
long nm;
nm = 0;
n = readfile("/proc/self/mountinfo", buf, (long)sizeof(buf) - 1);
if(n < 0)
{
return 0;
}
i = 0;
while(i < n && nm < MAXMOUNTS)
{
long linestart = i;
long lineend = i;
long field = 0;
long fieldstart = i;
long mpstart = -1;
long mpend = -1;
while(lineend < n && buf[lineend] != '\n')
{
lineend = lineend + 1;
}
fieldstart = linestart;
field = 0;
{
long j = linestart;
while(j <= lineend)
{
if(j == lineend || buf[j] == ' ')
{
field = field + 1;
if(field == 5)
{
mpstart = fieldstart;
mpend = j;
}
fieldstart = j + 1;
}
j = j + 1;
}
}
if(mpstart >= 0 && mpend > mpstart)
{
/* the kernel escapes space/tab/backslash/newline in this
* field as "\NNN" (octal) -- a mount path containing any of
* those (a space is common: autofs home dirs, removable
* media, some overlay setups) would otherwise copy through
* as the literal 4-byte escape sequence, and the later
* mount(2) remount call on that garbled path then silently
* fails (every call here is fire-and-forget), quietly
* leaving that one submount out of the "every submount goes
* read-only" guarantee this whole function exists for. */
long src = mpstart;
long dst = 0;
while(src < mpend && dst < MAXPATH - 1)
{
if(buf[src] == '\\' && src + 3 < mpend
&& buf[src + 1] >= '0' && buf[src + 1] <= '7'
&& buf[src + 2] >= '0' && buf[src + 2] <= '7'
&& buf[src + 3] >= '0' && buf[src + 3] <= '7')
{
out[nm][dst] = (char)((buf[src + 1] - '0') * 64
+ (buf[src + 2] - '0') * 8 + (buf[src + 3] - '0'));
dst = dst + 1;
src = src + 4;
}
else
{
out[nm][dst] = buf[src];
dst = dst + 1;
src = src + 1;
}
}
if(src >= mpend)
{
out[nm][dst] = '\0';
nm = nm + 1;
}
}
i = lineend + 1;
}
return nm;
}
/* network + mount namespace isolation for a build/check/install/remove
* phase (never fetch, which needs real network): unshare gives a fresh
* network namespace with only loopback (so anything but localhost is
* simply unreachable, no firewall rules needed) and a fresh mount
* namespace whose whole tree (every submount, individually) gets
* remounted read-only, with stagedir and prefix bind-remounted back to
* read-write on top. still no separate rootfs, no copying -- a bounded
* list walk plus a few dozen syscalls at most, not a portage-style
* LD_PRELOAD syscall interceptor on every access. any setup failure
* warns and falls back to running unsandboxed rather than failing the
* build outright -- this is defense in depth, not a hard boundary. */
void
setup_sandbox(const char* stagedir, const char* prefix)
{
static char mounts[MAXMOUNTS][MAXPATH];
long nm;
long i;
if(unshare(MP_CLONE_NEWNET | MP_CLONE_NEWNS) != 0)
{
fprintf(stderr, "warn: sandbox unshare failed: %s, continuing unsandboxed.\n", strerror(errno));
return;
}
/* isolate mount propagation from the host FIRST -- otherwise a mount
* change below could still propagate back to the real system. */
if(mount(NULL, "/", NULL, MP_MS_REC | MP_MS_PRIVATE, NULL) != 0)
{
fprintf(stderr, "warn: sandbox mount isolation failed: %s, continuing unsandboxed.\n", strerror(errno));
return;
}
/* turn "/" into a (recursive) bind mount of itself, so every mount
* under it (this one included) can be remounted read-only below. */
if(mount("/", "/", NULL, MP_MS_BIND | MP_MS_REC, NULL) != 0)
{
fprintf(stderr, "warn: sandbox bind mount failed: %s, continuing unsandboxed.\n", strerror(errno));
return;
}
nm = listmounts(mounts);
for(i = 0; i < nm; i = i + 1)
{
/* best-effort per mount point: one unusual/virtual submount
* refusing to go read-only shouldn't abort the whole sandbox. */
mount(NULL, mounts[i], NULL, MP_MS_REMOUNT | MP_MS_BIND | MP_MS_RDONLY, NULL);
}
/* belt and braces: the top-level recursive remount too, in case
* listmounts missed something mountinfo didn't report. */
mount(NULL, "/", NULL, MP_MS_REMOUNT | MP_MS_BIND | MP_MS_RDONLY | MP_MS_REC, NULL);
/* a bind mount created over an already-read-only path inherits that
* read-only state (binding doesn't grant new permissions, only a new
* reference to the same vfsmount data) -- the explicit MS_REMOUNT
* without MS_RDONLY is what actually restores write access. */
if(stagedir != NULL && stagedir[0] != '\0')
{
mount(stagedir, stagedir, NULL, MP_MS_BIND, NULL);
mount(NULL, stagedir, NULL, MP_MS_REMOUNT | MP_MS_BIND, NULL);
}
if(prefix != NULL && prefix[0] != '\0')
{
mount(prefix, prefix, NULL, MP_MS_BIND, NULL);
mount(NULL, prefix, NULL, MP_MS_REMOUNT | MP_MS_BIND, NULL);
}
}
#endif