| git.druid.rocks | index | druid520 | ports | ports/ | coreutils/ | sbase/ | patches/ | 0003-eregcomp-gnu-escapes.patch |
ports/coreutils/sbase/patches/0003-eregcomp-gnu-escapes.patch
# real posix regcomp() (musl's included, confirmed for real) gives
# backslash NO special meaning at all inside a bracket expression --
# "[^ \t]" is genuinely "not space, not backslash, not the letter t",
# three literal characters, not "not space, not tab". gnu sed/grep
# both treat \t \n \r \a \f \v as escapes unconditionally regardless
# of context, and real-world scripts widely assume that (this tree's
# own toybox port among them -- confirmed for real, this exact gap
# silently truncated a generated identifier by excluding the letter
# "t" from a match). translates those specific escapes to their
# literal byte form in the pattern string BEFORE ever calling
# regcomp() -- not a libc change, this stays entirely inside sbase's
# own eregcomp/enregcomp wrapper (shared by every caller: expr, grep,
# nl, sed). every other backslash sequence (backreferences, \(, \|,
# \+, anchors, ...) passes through completely unchanged for regcomp()
# to interpret on its own.
--- a/libutil/eregcomp.c 2026-09-11 21:58:42.855930624 +0000
+++ b/libutil/eregcomp.c 2026-09-11 21:58:42.856930615 +0000
@@ -5,13 +5,59 @@
#include "../util.h"
+/* translate the common gnu-regex backslash escapes (\t \n \r \a \f \v) into
+ * their literal byte form before ever handing the pattern to regcomp() --
+ * real posix regcomp() (musl's included) gives backslash NO special meaning
+ * at all inside a bracket expression ("[^ \t]" is genuinely "not space, not
+ * backslash, not t", three literal characters -- confirmed for real: it
+ * silently truncated a real identifier by excluding the letter "t") and
+ * doesn't define these escapes outside brackets either. gnu sed/grep both
+ * treat them as escapes unconditionally regardless of context, and enough
+ * real-world scripts already assume that (this tree's own toybox build
+ * among them) to make it worth closing here, once, for every eregcomp/
+ * enregcomp caller (expr, grep, nl, sed) instead of per-caller. every
+ * OTHER backslash sequence (backreferences, \(, \|, \+, anchors, ...) is
+ * passed through completely unchanged for regcomp() to interpret on its
+ * own -- only these specific extra letters are unknown to it otherwise.
+ * a fixed, static buffer (same BUFSIZ convention this file's own errbuf
+ * already uses below) rather than a dynamic allocation: regcomp() copies
+ * whatever it needs out of the pattern immediately, nothing here ever
+ * holds onto the returned pointer past that one call.
+ */
+static char *
+unescape_regex(const char *regex)
+{
+ static char buf[BUFSIZ];
+ size_t i, o;
+
+ for (i = 0, o = 0; regex[i] && o < sizeof(buf) - 1; i++) {
+ if (regex[i] == '\\' && regex[i + 1]) {
+ switch (regex[i + 1]) {
+ case 't': buf[o++] = '\t'; i++; continue;
+ case 'n': buf[o++] = '\n'; i++; continue;
+ case 'r': buf[o++] = '\r'; i++; continue;
+ case 'a': buf[o++] = '\a'; i++; continue;
+ case 'f': buf[o++] = '\f'; i++; continue;
+ case 'v': buf[o++] = '\v'; i++; continue;
+ }
+ buf[o++] = regex[i++];
+ if (o < sizeof(buf) - 1)
+ buf[o++] = regex[i];
+ continue;
+ }
+ buf[o++] = regex[i];
+ }
+ buf[o] = '\0';
+ return buf;
+}
+
int
enregcomp(int status, regex_t *preg, const char *regex, int cflags)
{
char errbuf[BUFSIZ] = "";
int r;
- if ((r = regcomp(preg, regex, cflags)) == 0)
+ if ((r = regcomp(preg, unescape_regex(regex), cflags)) == 0)
return r;
regerror(r, preg, errbuf, sizeof(errbuf));