do not edit — generated by btf.
git.druid.rocksindexdruid520mpsrc/sandbox.c

src/sandbox.c


#define _POSIX_C_SOURCE 200809L
#define _NETBSD_SOURCE 1
 
/* included unconditionally: on a non-Linux build everything below this
 * point is compiled out, and -pedantic-errors rejects a translation
 * unit left completely empty (which is exactly what this file is
 * without the Linux sandbox). */
#include "mpx.h"
 
#ifdef __linux__
 
#include <stdio.h>
#include <string.h>
#include <errno.h>
 
/* unshare()/CLONE_NEWNET/CLONE_NEWNS (<sched.h>) and mount()/its MS_
 * mount flags (<sys/mount.h>) are gated behind _GNU_SOURCE on glibc --
 * this tree uses no GNU extensions or GNU feature-test macros ever, so
 * both are declared here directly instead: the CLONE_ and MS_ values
 * are Linux kernel uapi constants (include/uapi/linux/sched.h,
 * include/uapi/linux/mount.h), architecture-independent and stable
 * across kernel versions, and unshare()/mount()'s signatures are the
 * same on every libc that provides them (glibc, musl, ...). */
extern int unshare(int flags);
extern int mount(const char* source, const char* target,
    const char* filesystemtype, unsigned long mountflags, const void* data);
#define MP_MS_RDONLY  1UL
#define MP_MS_REMOUNT 32UL
#define MP_MS_BIND    4096UL
#define MP_MS_REC     16384UL
#define MP_MS_PRIVATE 262144UL
#define MP_CLONE_NEWNS  0x00020000
#define MP_CLONE_NEWNET 0x40000000
 
/* a mount-heavy host/container (overlay-per-container storage drivers,
 * some systemd-managed setups) can easily exceed a couple hundred real
 * mount points -- once nm hits this cap, listmounts stops scanning
 * (not just stops recording) and every mount beyond it silently falls
 * back to the single top-level recursive remount this function's own
 * comment already explains is NOT reliable for pre-existing submounts.
 * generously sized (static storage, not stack, so the cost is a few MB
 * of BSS, not a real constraint) rather than exactly right-sized, since
 * getting this wrong fails silently with no diagnostic either way. */
#define MAXMOUNTS 4096
#define MAXMOUNTINFO 65536
 
static long listmounts(char out[MAXMOUNTS][MAXPATH]);
 
/* a single "mount(NULL, path, NULL, MS_REMOUNT|MS_BIND|MS_RDONLY|MS_REC,
 * NULL)" does not reliably cascade read-only to submounts that already
 * existed before the remount (a well-known kernel limitation, the same
 * reason bwrap/docker walk mountinfo themselves rather than trusting
 * MS_REC here) -- a separate filesystem bind-mounted under "/" (/tmp as
 * tmpfs is the common case) stays writable otherwise. so every submount
 * gets its own explicit remount call. returns the number of mount point
 * paths found (each newline-terminated in out, most MAXMOUNTS kept). */
static long
listmounts(char out[MAXMOUNTS][MAXPATH])
{
	static char buf[MAXMOUNTINFO];
	long n;
	long i;
	long nm;
 
	nm = 0;
	n = readfile("/proc/self/mountinfo", buf, (long)sizeof(buf) - 1);
	if(n < 0)
	{
		return 0;
	}
	i = 0;
	while(i < n && nm < MAXMOUNTS)
	{
		long linestart = i;
		long lineend = i;
		long field = 0;
		long fieldstart = i;
		long mpstart = -1;
		long mpend = -1;
 
		while(lineend < n && buf[lineend] != '\n')
		{
			lineend = lineend + 1;
		}
		fieldstart = linestart;
		field = 0;
		{
			long j = linestart;
			while(j <= lineend)
			{
				if(j == lineend || buf[j] == ' ')
				{
					field = field + 1;
					if(field == 5)
					{
						mpstart = fieldstart;
						mpend = j;
					}
					fieldstart = j + 1;
				}
				j = j + 1;
			}
		}
		if(mpstart >= 0 && mpend > mpstart)
		{
			/* the kernel escapes space/tab/backslash/newline in this
			 * field as "\NNN" (octal) -- a mount path containing any of
			 * those (a space is common: autofs home dirs, removable
			 * media, some overlay setups) would otherwise copy through
			 * as the literal 4-byte escape sequence, and the later
			 * mount(2) remount call on that garbled path then silently
			 * fails (every call here is fire-and-forget), quietly
			 * leaving that one submount out of the "every submount goes
			 * read-only" guarantee this whole function exists for. */
			long src = mpstart;
			long dst = 0;
			while(src < mpend && dst < MAXPATH - 1)
			{
				if(buf[src] == '\\' && src + 3 < mpend
				    && buf[src + 1] >= '0' && buf[src + 1] <= '7'
				    && buf[src + 2] >= '0' && buf[src + 2] <= '7'
				    && buf[src + 3] >= '0' && buf[src + 3] <= '7')
				{
					out[nm][dst] = (char)((buf[src + 1] - '0') * 64
					    + (buf[src + 2] - '0') * 8 + (buf[src + 3] - '0'));
					dst = dst + 1;
					src = src + 4;
				}
				else
				{
					out[nm][dst] = buf[src];
					dst = dst + 1;
					src = src + 1;
				}
			}
			if(src >= mpend)
			{
				out[nm][dst] = '\0';
				nm = nm + 1;
			}
		}
		i = lineend + 1;
	}
	return nm;
}
 
/* network + mount namespace isolation for a build/check/install/remove
 * phase (never fetch, which needs real network): unshare gives a fresh
 * network namespace with only loopback (so anything but localhost is
 * simply unreachable, no firewall rules needed) and a fresh mount
 * namespace whose whole tree (every submount, individually) gets
 * remounted read-only, with stagedir and prefix bind-remounted back to
 * read-write on top. still no separate rootfs, no copying -- a bounded
 * list walk plus a few dozen syscalls at most, not a portage-style
 * LD_PRELOAD syscall interceptor on every access. any setup failure
 * warns and falls back to running unsandboxed rather than failing the
 * build outright -- this is defense in depth, not a hard boundary. */
void
setup_sandbox(const char* stagedir, const char* prefix)
{
	static char mounts[MAXMOUNTS][MAXPATH];
	long nm;
	long i;
 
	if(unshare(MP_CLONE_NEWNET | MP_CLONE_NEWNS) != 0)
	{
		fprintf(stderr, "warn: sandbox unshare failed: %s, continuing unsandboxed.\n", strerror(errno));
		return;
	}
	/* isolate mount propagation from the host FIRST -- otherwise a mount
	 * change below could still propagate back to the real system. */
	if(mount(NULL, "/", NULL, MP_MS_REC | MP_MS_PRIVATE, NULL) != 0)
	{
		fprintf(stderr, "warn: sandbox mount isolation failed: %s, continuing unsandboxed.\n", strerror(errno));
		return;
	}
	/* turn "/" into a (recursive) bind mount of itself, so every mount
	 * under it (this one included) can be remounted read-only below. */
	if(mount("/", "/", NULL, MP_MS_BIND | MP_MS_REC, NULL) != 0)
	{
		fprintf(stderr, "warn: sandbox bind mount failed: %s, continuing unsandboxed.\n", strerror(errno));
		return;
	}
	nm = listmounts(mounts);
	for(i = 0; i < nm; i = i + 1)
	{
		/* best-effort per mount point: one unusual/virtual submount
		 * refusing to go read-only shouldn't abort the whole sandbox. */
		mount(NULL, mounts[i], NULL, MP_MS_REMOUNT | MP_MS_BIND | MP_MS_RDONLY, NULL);
	}
	/* belt and braces: the top-level recursive remount too, in case
	 * listmounts missed something mountinfo didn't report. */
	mount(NULL, "/", NULL, MP_MS_REMOUNT | MP_MS_BIND | MP_MS_RDONLY | MP_MS_REC, NULL);
	/* a bind mount created over an already-read-only path inherits that
	 * read-only state (binding doesn't grant new permissions, only a new
	 * reference to the same vfsmount data) -- the explicit MS_REMOUNT
	 * without MS_RDONLY is what actually restores write access. */
	if(stagedir != NULL && stagedir[0] != '\0')
	{
		mount(stagedir, stagedir, NULL, MP_MS_BIND, NULL);
		mount(NULL, stagedir, NULL, MP_MS_REMOUNT | MP_MS_BIND, NULL);
	}
	if(prefix != NULL && prefix[0] != '\0')
	{
		mount(prefix, prefix, NULL, MP_MS_BIND, NULL);
		mount(NULL, prefix, NULL, MP_MS_REMOUNT | MP_MS_BIND, NULL);
	}
}
 
#endif
 
powered by btf.