Files
regexx/regexx.c
T
retoorandClaude Sonnet 5 b0991814cb Expand BINARY/UTF8/ASCII edge-case testing to 3,252 cases, fix a real LOCALE bug
Three new categories added to tests/cases.py (TEST_PLAN.md records the
detail): Category H exercises BINARY mode with genuinely arbitrary raw
bytes (embedded NUL, high bytes, non-UTF-8 sequences) via real Python
`bytes` subjects, not just UTF-8-encoded text; Category I exercises
UTF8 mode edge cases (4-byte/astral code points, combining marks,
Arabic, Hebrew, CJK, offset correctness across multi-byte characters);
Category J is a fixed-seed (reproducible, not flaky) random generator
combining the existing atom/quantifier/grouping vocabulary across all
three modes.

This found and fixed a real bug, not just a test-generation one:
LOCALE, in BINARY/ASCII mode, treated bytes 0x80-0xFF as word
characters, based on an unverified assumption about what the "C"
locale does. Checked directly against both the C standard's own
guarantee for isalnum() under "C" and a real CPython interpreter with
re.LOCALE and the "C" locale explicitly set, neither treats anything
above 0x7f as a word character. Fixed in regexx.c's cls_is_word;
LOCALE is now documented as an accepted no-op in non-UTF8 mode,
matching verified reality instead of a prior assumption (concept.md
13.4, docs/API.md, README.md "Known deviations").

Two more findings were test-generation bugs, not regexx bugs: gen.py's
own ASCII-mode ground truth used Python str + re.ASCII (code-point
space) instead of a bytes pattern against a bytes subject (what
regexx's byte-oriented ASCII mode actually is), and LOCALE combined
with the (now removed as redundant) auto-added re.ASCII flag raised
ValueError in Python for being an incompatible combination. Both
fixed in gen.py.

Two further findings were concrete instances of an already-documented
category (glibc's wctype.h Unicode tables not matching CPython's own
exactly): U+00A0 and fullwidth digits U+FF10-FF19 are recognized by
CPython's \s/\d but not by glibc's iswspace()/iswdigit() under C.utf8.
Recorded in README.md, not patched, for the reason already given for
the first such instance (NBSP) in the previous commit.

After these fixes: all 3,252 committed cases pass, clean under
AddressSanitizer/UndefinedBehaviorSanitizer. The Category J generator
was additionally run against 5 more seeds at 3,000 iterations each
(26,760 further checks) as exploratory validation, all passing; not
committed, to keep the regular suite's size proportionate.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01EjuMk8kY9SDus1wWe2K9xY
2026-09-14 09:25:50 +00:00

1929 lines
81 KiB
C

/* regexx.c - a single-file C regex interpreter with Python `re` semantics.
*
* See concept.md for the full design rationale. This file is the v1
* implementation: it covers the public API and pattern syntax listed in
* README.md "Implementation status" in full, executed by a single
* recursive backtracking engine (concept.md Section 7.3) over a fully
* materialized copy of the input. It does not yet implement the
* streaming, bounded-memory regular engine of concept.md Section 7.2;
* that remains future work, tracked in README.md.
*
* That backtracking engine is memoized (run_memo) and, for quantifiers
* over a single character/class/./ (OP_REPEAT1), backed by precomputed
* run-length and skip-ahead tables (compute_maxrun,
* compute_next_prevmatch), so that Pattern_search/finditer/split/sub
* run in linear rather than quadratic time for the common,
* backreference-free case; see README.md "Implementation status" for
* the measured numbers, the precise remaining limitations, and
* citations to the published techniques this matches.
*
* Naming convention (concept.md Section 9.0): every identifier with a
* direct Python `re` counterpart uses that counterpart's exact spelling.
*
* Not reentrant: re_compile() uses one process-wide parser scratch pool
* and the module level functions share one process-wide pattern cache
* (concept.md 13.6); do not call regexx functions from more than one
* thread without external synchronization.
*/
#define _DEFAULT_SOURCE
#include "regexx.h"
#include <stdlib.h>
#include <string.h>
#include <stdio.h>
#include <stdarg.h>
#include <ctype.h>
#include <wctype.h>
#include <wchar.h>
#include <locale.h>
#include <errno.h>
/* ================================================================
* 0. Small utilities: dynamic array, allocation helpers
* ================================================================ */
typedef struct { void *data; size_t len, cap, elemsize; } DArr;
static void da_init(DArr *a, size_t elemsize) {
a->data = NULL; a->len = 0; a->cap = 0; a->elemsize = elemsize;
}
static void *da_push(DArr *a) {
if (a->len == a->cap) {
size_t nc = a->cap ? a->cap * 2 : 8;
a->data = realloc(a->data, nc * a->elemsize);
a->cap = nc;
}
void *p = (char *)a->data + a->len * a->elemsize;
a->len++;
return p;
}
static void da_free(DArr *a) { free(a->data); a->data = NULL; a->len = a->cap = 0; }
static char *xstrndup(const char *s, size_t n) {
char *r = malloc(n + 1);
memcpy(r, s, n);
r[n] = 0;
return r;
}
static void ensure_unicode_locale(void) {
static int done = 0;
if (!done) { setlocale(LC_CTYPE, "C.utf8"); done = 1; }
}
/* ================================================================
* 1. UTF-8 decoding (concept.md Section 7.2's continuation handling is
* not needed here since v1 materializes the whole input up front,
* see README.md "Implementation status")
* ================================================================ */
static int utf8_decode(const uint8_t *buf, size_t len, size_t i, uint32_t *cp) {
uint8_t b0 = buf[i];
if (b0 < 0x80) { *cp = b0; return 1; }
if ((b0 & 0xE0) == 0xC0) {
if (i + 1 >= len || (buf[i+1] & 0xC0) != 0x80) return 0;
uint32_t v = (uint32_t)((b0 & 0x1F) << 6) | (buf[i+1] & 0x3F);
if (v < 0x80) return 0;
*cp = v; return 2;
}
if ((b0 & 0xF0) == 0xE0) {
if (i + 2 >= len || (buf[i+1] & 0xC0) != 0x80 || (buf[i+2] & 0xC0) != 0x80) return 0;
uint32_t v = ((uint32_t)(b0 & 0x0F) << 12) | ((uint32_t)(buf[i+1] & 0x3F) << 6) | (buf[i+2] & 0x3F);
if (v < 0x800 || (v >= 0xD800 && v <= 0xDFFF)) return 0;
*cp = v; return 3;
}
if ((b0 & 0xF8) == 0xF0) {
if (i + 3 >= len || (buf[i+1] & 0xC0) != 0x80 || (buf[i+2] & 0xC0) != 0x80 || (buf[i+3] & 0xC0) != 0x80) return 0;
uint32_t v = ((uint32_t)(b0 & 0x07) << 18) | ((uint32_t)(buf[i+1] & 0x3F) << 12) | ((uint32_t)(buf[i+2] & 0x3F) << 6) | (buf[i+3] & 0x3F);
if (v < 0x10000 || v > 0x10FFFF) return 0;
*cp = v; return 4;
}
return 0;
}
/* ================================================================
* 2. Character classification predicates (mode aware, Section 2.2/14.4)
* ================================================================ */
#define MODE_BINARY 0
#define MODE_ASCII 1
#define MODE_UTF8 2
static int is_ascii_word(uint32_t cp) {
return (cp >= 'a' && cp <= 'z') || (cp >= 'A' && cp <= 'Z') || (cp >= '0' && cp <= '9') || cp == '_';
}
static int is_ascii_space(uint32_t cp) {
return cp == ' ' || cp == '\t' || cp == '\n' || cp == '\r' || cp == '\f' || cp == '\v';
}
static int is_ascii_digit(uint32_t cp) { return cp >= '0' && cp <= '9'; }
static int cls_is_digit(uint32_t cp, int mode, int flags) {
if (mode == MODE_UTF8 && !(flags & ASCII)) { ensure_unicode_locale(); return iswdigit((wint_t)cp) ? 1 : 0; }
return is_ascii_digit(cp);
}
static int cls_is_space(uint32_t cp, int mode, int flags) {
if (mode == MODE_UTF8 && !(flags & ASCII)) { ensure_unicode_locale(); return iswspace((wint_t)cp) ? 1 : 0; }
return is_ascii_space(cp);
}
static int cls_is_word(uint32_t cp, int mode, int flags) {
if (mode == MODE_UTF8 && !(flags & ASCII)) {
ensure_unicode_locale();
return (iswalnum((wint_t)cp) || cp == '_') ? 1 : 0;
}
return is_ascii_word(cp); /* LOCALE has no additional effect here, see below */
/* In BINARY/ASCII mode, LOCALE (Section 2.6) is intentionally a
* no-op relative to plain ASCII classification: the C standard
* guarantees the "C" locale's isalnum()/isalpha() classify only
* the ASCII letters/digits, nothing in 0x80-0xFF, and CPython's own
* re.LOCALE, checked directly under an explicit "C" locale, agrees
* (rb'\w+' against b"abc\x80\x90def" still stops at "abc"). An
* earlier revision of this function instead treated every byte in
* 0x80-0xFF as a word character under LOCALE, based on an
* unverified assumption about what "the C locale" does; the large
* combinatorial test expansion (tests/cases.py Category H) caught
* the resulting mismatch against a real CPython interpreter, which
* is what prompted checking the premise directly (see above) and
* finding it false. README.md "Known deviations" records this. */
}
static uint32_t cls_swapcase(uint32_t cp, int mode, int flags) {
if (mode == MODE_UTF8 && !(flags & ASCII)) {
ensure_unicode_locale();
wint_t lo = towlower((wint_t)cp), up = towupper((wint_t)cp);
if ((uint32_t)lo != cp) return (uint32_t)lo;
if ((uint32_t)up != cp) return (uint32_t)up;
return cp;
}
if (cp >= 'a' && cp <= 'z') return cp - 32;
if (cp >= 'A' && cp <= 'Z') return cp + 32;
return cp;
}
static uint32_t cls_fold(uint32_t cp, int mode, int flags) {
if (mode == MODE_UTF8 && !(flags & ASCII)) { ensure_unicode_locale(); return (uint32_t)towlower((wint_t)cp); }
if (cp >= 'A' && cp <= 'Z') return cp + 32;
return cp;
}
/* ================================================================
* 3. AST
* ================================================================ */
enum {
N_CHAR, N_ANY, N_CLASS, N_CONCAT, N_ALT, N_REPEAT, N_GROUP,
N_BACKREF, N_ANCHOR, N_LOOKAROUND, N_ATOMIC, N_EMPTY
};
enum { A_BOL, A_EOL, A_BOS, A_EOS, A_WB, A_NWB };
enum { LK_AHEAD, LK_BEHIND };
enum { CI_RANGE, CI_D, CI_W, CI_S };
typedef struct { int kind; uint32_t lo, hi; int neg; } ClassItem;
typedef struct Node {
int kind;
struct Node *a, *b;
int32_t lo, hi;
int greedy;
uint32_t ch;
DArr items;
int class_negate;
int group_index;
char *group_name;
int backref_index;
char *backref_name;
int anchor_kind;
int lookaround_kind;
int lookaround_negate;
int width;
} Node;
typedef struct { Node **data; size_t len, cap; } NodePool;
static NodePool g_pool;
static Node *node_new(int kind) {
if (g_pool.len == g_pool.cap) {
g_pool.cap = g_pool.cap ? g_pool.cap * 2 : 64;
g_pool.data = realloc(g_pool.data, g_pool.cap * sizeof(Node *));
}
Node *n = calloc(1, sizeof(Node));
n->kind = kind; n->group_index = -1; n->backref_index = -1; n->greedy = 1; n->width = -1;
da_init(&n->items, sizeof(ClassItem));
g_pool.data[g_pool.len++] = n;
return n;
}
static void nodepool_free_all(NodePool *pool) {
for (size_t i = 0; i < pool->len; i++) {
Node *n = pool->data[i];
da_free(&n->items);
free(n->group_name);
free(n->backref_name);
free(n);
}
free(pool->data);
pool->data = NULL; pool->len = pool->cap = 0;
}
static int node_nullable(Node *n) {
if (!n) return 1;
switch (n->kind) {
case N_CHAR: case N_ANY: case N_CLASS: case N_BACKREF: return 0;
case N_CONCAT: return node_nullable(n->a) && node_nullable(n->b);
case N_ALT: return node_nullable(n->a) || node_nullable(n->b);
case N_REPEAT: return n->lo == 0 || node_nullable(n->a);
case N_GROUP: return node_nullable(n->a);
case N_ANCHOR: case N_LOOKAROUND: return 1;
case N_ATOMIC: return node_nullable(n->a);
default: return 1;
}
}
static int node_width(Node *n) {
if (!n) return 0;
switch (n->kind) {
case N_CHAR: case N_ANY: case N_CLASS: return 1;
case N_CONCAT: { int wa = node_width(n->a), wb = node_width(n->b); return (wa < 0 || wb < 0) ? -1 : wa + wb; }
case N_ALT: { int wa = node_width(n->a), wb = node_width(n->b); return (wa < 0 || wb < 0 || wa != wb) ? -1 : wa; }
case N_REPEAT: { if (n->lo != n->hi) return -1; int w = node_width(n->a); return w < 0 ? -1 : w * n->lo; }
case N_GROUP: return node_width(n->a);
case N_BACKREF: return -1;
case N_ANCHOR: case N_LOOKAROUND: return 0;
case N_ATOMIC: return node_width(n->a);
default: return 0;
}
}
static int is_simple_atom(Node *n) { return n->kind == N_CHAR || n->kind == N_ANY || n->kind == N_CLASS; }
/* ================================================================
* 4. Parser
* ================================================================ */
typedef struct { char *name; int index; } GroupName;
typedef struct {
const char *s;
size_t len, i;
int flags;
int mode;
int ngroups;
DArr backrefs; /* Node* */
DArr groupnames; /* GroupName */
char errmsg[256];
int64_t errpos;
int failed;
} Parser;
static void perr(Parser *p, int64_t pos, const char *fmt, ...) {
if (p->failed) return;
p->failed = 1; p->errpos = pos;
va_list ap; va_start(ap, fmt);
vsnprintf(p->errmsg, sizeof p->errmsg, fmt, ap);
va_end(ap);
}
static int peek(Parser *p) { return p->i < p->len ? (unsigned char)p->s[p->i] : -1; }
static int peek2(Parser *p) { return p->i + 1 < p->len ? (unsigned char)p->s[p->i + 1] : -1; }
static int at_end(Parser *p) { return p->i >= p->len; }
static int hexval(int c) {
if (c >= '0' && c <= '9') return c - '0';
if (c >= 'a' && c <= 'f') return c - 'a' + 10;
if (c >= 'A' && c <= 'F') return c - 'A' + 10;
return -1;
}
static Node *parse_alt(Parser *p);
static Node *parse_escape(Parser *p, int in_class, uint32_t *literal_cp, int *is_literal) {
*is_literal = 0;
if (at_end(p)) { perr(p, (int64_t)p->i, "bad escape (end of pattern)"); return NULL; }
int c = peek(p); p->i++;
switch (c) {
case 'n': *literal_cp = '\n'; *is_literal = 1; return NULL;
case 'r': *literal_cp = '\r'; *is_literal = 1; return NULL;
case 't': *literal_cp = '\t'; *is_literal = 1; return NULL;
case 'f': *literal_cp = '\f'; *is_literal = 1; return NULL;
case 'v': *literal_cp = '\v'; *is_literal = 1; return NULL;
case 'a': *literal_cp = '\a'; *is_literal = 1; return NULL;
case '0': {
uint32_t v = 0; int n = 0;
while (n < 2 && peek(p) >= '0' && peek(p) <= '7') { v = v * 8 + (uint32_t)(peek(p) - '0'); p->i++; n++; }
*literal_cp = v; *is_literal = 1; return NULL;
}
case 'x': {
int h1 = hexval(peek(p));
if (h1 < 0) { perr(p, (int64_t)p->i, "incomplete escape \\x"); return NULL; }
p->i++;
int h2 = hexval(peek(p));
if (h2 < 0) { perr(p, (int64_t)p->i, "incomplete escape \\x"); return NULL; }
p->i++;
*literal_cp = (uint32_t)(h1 * 16 + h2); *is_literal = 1; return NULL;
}
case 'u': case 'U': {
int ndig = (c == 'u') ? 4 : 8;
uint32_t v = 0;
for (int k = 0; k < ndig; k++) {
int h = hexval(peek(p));
if (h < 0) { perr(p, (int64_t)p->i, "incomplete escape \\%c", c); return NULL; }
v = v * 16 + (uint32_t)h; p->i++;
}
*literal_cp = v; *is_literal = 1; return NULL;
}
case 'N':
perr(p, (int64_t)p->i, "\\N{...} named code points are not implemented in this build");
return NULL;
case 'd': case 'D': case 'w': case 'W': case 's': case 'S': {
Node *n = node_new(N_CLASS);
ClassItem *it = da_push(&n->items);
it->kind = (c == 'd' || c == 'D') ? CI_D : (c == 'w' || c == 'W') ? CI_W : CI_S;
it->neg = (c == 'D' || c == 'W' || c == 'S') ? 1 : 0;
it->lo = it->hi = 0;
return n;
}
case 'b':
if (in_class) { *literal_cp = '\b'; *is_literal = 1; return NULL; }
{ Node *n = node_new(N_ANCHOR); n->anchor_kind = A_WB; return n; }
case 'B': { Node *n = node_new(N_ANCHOR); n->anchor_kind = A_NWB; return n; }
case 'A': { Node *n = node_new(N_ANCHOR); n->anchor_kind = A_BOS; return n; }
case 'Z': { Node *n = node_new(N_ANCHOR); n->anchor_kind = A_EOS; return n; }
case 'g': {
if (in_class) { *literal_cp = 'g'; *is_literal = 1; return NULL; }
if (peek(p) != '<') { perr(p, (int64_t)p->i, "missing < in \\g<...>"); return NULL; }
p->i++;
size_t start = p->i;
while (!at_end(p) && peek(p) != '>') p->i++;
if (at_end(p)) { perr(p, (int64_t)p->i, "missing > in \\g<...>"); return NULL; }
char *ref = xstrndup(p->s + start, p->i - start);
p->i++;
Node *n = node_new(N_BACKREF);
size_t rl = strlen(ref);
int all_digit = rl > 0;
for (size_t k = 0; k < rl; k++) if (!isdigit((unsigned char)ref[k])) all_digit = 0;
if (all_digit) { n->backref_index = atoi(ref); free(ref); }
else n->backref_name = ref;
Node **slot = da_push(&p->backrefs); *slot = n;
return n;
}
case '1': case '2': case '3': case '4': case '5': case '6': case '7': case '8': case '9': {
if (in_class) { *literal_cp = (uint32_t)c; *is_literal = 1; return NULL; }
uint32_t v = (uint32_t)(c - '0');
while (peek(p) >= '0' && peek(p) <= '9') { v = v * 10 + (uint32_t)(peek(p) - '0'); p->i++; }
Node *n = node_new(N_BACKREF);
n->backref_index = (int)v;
Node **slot = da_push(&p->backrefs); *slot = n;
return n;
}
default:
*literal_cp = (uint32_t)c; *is_literal = 1; return NULL;
}
}
static uint32_t read_raw_char(Parser *p) {
if (p->mode == MODE_UTF8) {
uint32_t cp;
int n = utf8_decode((const uint8_t *)p->s, p->len, p->i, &cp);
if (n == 0) { perr(p, (int64_t)p->i, "invalid UTF-8 in pattern"); p->i++; return 0; }
p->i += (size_t)n;
return cp;
}
return (unsigned char)p->s[p->i++];
}
static Node *parse_class(Parser *p) {
Node *n = node_new(N_CLASS);
p->i++;
if (peek(p) == '^') { n->class_negate = 1; p->i++; }
int first = 1;
while (1) {
if (at_end(p)) { perr(p, (int64_t)p->i, "unterminated character set"); return n; }
if (peek(p) == ']' && !first) { p->i++; break; }
first = 0;
uint32_t lo;
if (peek(p) == '\\') {
p->i++;
uint32_t litcp; int is_lit;
Node *sub = parse_escape(p, 1, &litcp, &is_lit);
if (p->failed) return n;
if (!is_lit && sub) {
ClassItem *it = da_push(&n->items);
*it = ((ClassItem *)sub->items.data)[0];
continue;
}
lo = litcp;
} else if (peek(p) == '[' && peek2(p) == ':') {
perr(p, (int64_t)p->i, "POSIX named classes ([:alpha:] etc) are not part of Python re, see concept.md 14.3");
return n;
} else {
lo = read_raw_char(p);
if (p->failed) return n;
}
uint32_t hi = lo;
if (peek(p) == '-' && peek2(p) != ']' && peek2(p) != -1) {
p->i++;
if (peek(p) == '\\') {
p->i++;
uint32_t litcp; int is_lit;
parse_escape(p, 1, &litcp, &is_lit);
if (p->failed) return n;
if (!is_lit) { perr(p, (int64_t)p->i, "bad character range"); return n; }
hi = litcp;
} else {
hi = read_raw_char(p);
if (p->failed) return n;
}
if (hi < lo) { perr(p, (int64_t)p->i, "bad character range"); return n; }
}
ClassItem *it = da_push(&n->items);
it->kind = CI_RANGE; it->lo = lo; it->hi = hi; it->neg = 0;
}
return n;
}
static Node *make_concat(Node *a, Node *b) {
if (!a) return b;
if (!b) return a;
Node *n = node_new(N_CONCAT);
n->a = a; n->b = b;
return n;
}
static Node *parse_repeat_suffix(Parser *p, Node *atom) {
int lo = -1, hi = -1;
if (peek(p) == '*') { lo = 0; hi = -1; p->i++; }
else if (peek(p) == '+') { lo = 1; hi = -1; p->i++; }
else if (peek(p) == '?') { lo = 0; hi = 1; p->i++; }
else if (peek(p) == '{') {
size_t save = p->i;
p->i++;
int have_lo = 0, have_hi = 0; long vlo = 0, vhi = 0;
while (isdigit(peek(p))) { vlo = vlo * 10 + (peek(p) - '0'); p->i++; have_lo = 1; }
int has_comma = 0;
if (peek(p) == ',') {
has_comma = 1; p->i++;
while (isdigit(peek(p))) { vhi = vhi * 10 + (peek(p) - '0'); p->i++; have_hi = 1; }
}
if (peek(p) != '}' || (!have_lo && !has_comma)) { p->i = save; return atom; }
p->i++;
lo = have_lo ? (int)vlo : 0;
hi = has_comma ? (have_hi ? (int)vhi : -1) : lo;
if (hi != -1 && hi < lo) { perr(p, (int64_t)save, "min repeat greater than max repeat"); return atom; }
} else {
return atom;
}
Node *rep = node_new(N_REPEAT);
rep->a = atom; rep->lo = lo; rep->hi = hi; rep->greedy = 1;
if (peek(p) == '?') { rep->greedy = 0; p->i++; return rep; }
if (peek(p) == '+') {
p->i++;
Node *at = node_new(N_ATOMIC);
at->a = rep;
return at;
}
return rep;
}
static Node *parse_group_or_special(Parser *p) {
size_t open_pos = p->i;
p->i++;
if (peek(p) == '?') {
p->i++;
int c = peek(p);
if (c == ':') {
p->i++;
Node *body = parse_alt(p);
if (peek(p) != ')') { perr(p, (int64_t)p->i, "missing ), unterminated subpattern"); return body; }
p->i++;
Node *g = node_new(N_GROUP); g->a = body; g->group_index = -1;
return g;
}
if (c == '#') {
p->i++;
while (!at_end(p) && peek(p) != ')') p->i++;
if (at_end(p)) { perr(p, (int64_t)p->i, "missing ), unterminated comment"); return node_new(N_EMPTY); }
p->i++;
return node_new(N_EMPTY);
}
if (c == '=' || c == '!') {
p->i++;
Node *body = parse_alt(p);
if (peek(p) != ')') { perr(p, (int64_t)p->i, "missing ), unterminated subpattern"); return body; }
p->i++;
Node *n = node_new(N_LOOKAROUND);
n->a = body; n->lookaround_kind = LK_AHEAD; n->lookaround_negate = (c == '!');
return n;
}
if (c == '<' && (peek2(p) == '=' || peek2(p) == '!')) {
int neg = (peek2(p) == '!');
p->i += 2;
Node *body = parse_alt(p);
if (peek(p) != ')') { perr(p, (int64_t)p->i, "missing ), unterminated subpattern"); return body; }
p->i++;
int w = node_width(body);
if (w < 0) { perr(p, (int64_t)open_pos, "look-behind requires fixed-width pattern"); return node_new(N_EMPTY); }
Node *n = node_new(N_LOOKAROUND);
n->a = body; n->lookaround_kind = LK_BEHIND; n->lookaround_negate = neg; n->width = w;
return n;
}
if (c == '>') {
p->i++;
Node *body = parse_alt(p);
if (peek(p) != ')') { perr(p, (int64_t)p->i, "missing ), unterminated subpattern"); return body; }
p->i++;
Node *n = node_new(N_ATOMIC); n->a = body;
return n;
}
if (c == 'P') {
p->i++;
if (peek(p) == '<') {
p->i++;
size_t start = p->i;
while (!at_end(p) && peek(p) != '>') p->i++;
if (at_end(p)) { perr(p, (int64_t)p->i, "missing >, unterminated name"); return node_new(N_EMPTY); }
char *name = xstrndup(p->s + start, p->i - start);
p->i++;
int idx = ++p->ngroups;
Node *body = parse_alt(p);
if (peek(p) != ')') { perr(p, (int64_t)p->i, "missing ), unterminated subpattern"); free(name); return body; }
p->i++;
Node *g = node_new(N_GROUP); g->a = body; g->group_index = idx; g->group_name = name;
GroupName *gn = da_push(&p->groupnames);
gn->name = xstrndup(name, strlen(name)); gn->index = idx;
return g;
}
if (peek(p) == '=') {
p->i++;
size_t start = p->i;
while (!at_end(p) && peek(p) != ')') p->i++;
if (at_end(p)) { perr(p, (int64_t)p->i, "missing ), unterminated name"); return node_new(N_EMPTY); }
char *name = xstrndup(p->s + start, p->i - start);
p->i++;
Node *n = node_new(N_BACKREF); n->backref_name = name;
Node **slot = da_push(&p->backrefs); *slot = n;
return n;
}
perr(p, (int64_t)p->i, "unknown extension ?P");
return node_new(N_EMPTY);
}
if (c == '(') {
perr(p, (int64_t)open_pos, "conditional groups (?(id)yes|no) are not implemented in this build");
return node_new(N_EMPTY);
}
{
size_t save = p->i;
int newflags = 0, any = 0;
while (!at_end(p) && strchr("aiLmsux", peek(p))) {
switch (peek(p)) {
case 'i': newflags |= IGNORECASE; break;
case 'm': newflags |= MULTILINE; break;
case 's': newflags |= DOTALL; break;
case 'x': newflags |= VERBOSE; break;
case 'a': newflags |= ASCII; break;
case 'L': newflags |= LOCALE; break;
default: break;
}
p->i++; any = 1;
}
if (any && peek(p) == ')') {
p->i++;
if (open_pos != 0) { perr(p, (int64_t)open_pos, "global flags not at the start of the expression"); return node_new(N_EMPTY); }
p->flags |= newflags;
return node_new(N_EMPTY);
}
p->i = save;
perr(p, (int64_t)open_pos, "scoped inline flags (?flags:...) are not implemented in this build");
return node_new(N_EMPTY);
}
}
int idx = ++p->ngroups;
Node *body = parse_alt(p);
if (peek(p) != ')') { perr(p, (int64_t)p->i, "missing ), unterminated subpattern"); return body; }
p->i++;
Node *g = node_new(N_GROUP); g->a = body; g->group_index = idx;
return g;
}
static Node *parse_atom(Parser *p) {
int c = peek(p);
if (c == '(') return parse_group_or_special(p);
if (c == '[') return parse_class(p);
if (c == '.') { p->i++; return node_new(N_ANY); }
if (c == '^') { p->i++; Node *n = node_new(N_ANCHOR); n->anchor_kind = A_BOL; return n; }
if (c == '$') { p->i++; Node *n = node_new(N_ANCHOR); n->anchor_kind = A_EOL; return n; }
if (c == '\\') {
p->i++;
uint32_t litcp; int is_lit;
Node *n = parse_escape(p, 0, &litcp, &is_lit);
if (p->failed) return node_new(N_EMPTY);
if (is_lit) { Node *cn = node_new(N_CHAR); cn->ch = litcp; return cn; }
return n;
}
if (c == '*' || c == '+' || c == '?') {
perr(p, (int64_t)p->i, "nothing to repeat");
p->i++;
return node_new(N_EMPTY);
}
uint32_t cp = read_raw_char(p);
if (p->failed) return node_new(N_EMPTY);
Node *cn = node_new(N_CHAR); cn->ch = cp;
return cn;
}
static Node *parse_concat(Parser *p) {
Node *result = NULL;
while (!at_end(p) && peek(p) != '|' && peek(p) != ')' && !p->failed) {
Node *atom = parse_atom(p);
if (p->failed) break;
atom = parse_repeat_suffix(p, atom);
result = make_concat(result, atom);
}
if (!result) result = node_new(N_EMPTY);
return result;
}
static Node *parse_alt(Parser *p) {
Node *first = parse_concat(p);
if (peek(p) == '|') {
p->i++;
Node *rest = parse_alt(p);
Node *n = node_new(N_ALT);
n->a = first; n->b = rest;
return n;
}
return first;
}
/* Strip VERBOSE-mode whitespace and #-comments, respecting [...] classes
* and backslash escapes. Only called once the VERBOSE flag is known. */
static char *strip_verbose(const char *s, size_t len, size_t *outlen) {
char *out = malloc(len + 1);
size_t o = 0, i = 0;
int in_class = 0;
while (i < len) {
char c = s[i];
if (c == '\\' && i + 1 < len) { out[o++] = s[i++]; out[o++] = s[i++]; continue; }
if (c == '[' && !in_class) { in_class = 1; out[o++] = s[i++]; continue; }
if (c == ']' && in_class) { in_class = 0; out[o++] = s[i++]; continue; }
if (!in_class && isspace((unsigned char)c)) { i++; continue; }
if (!in_class && c == '#') { while (i < len && s[i] != '\n') i++; continue; }
out[o++] = s[i++];
}
out[o] = 0;
*outlen = o;
return out;
}
/* ================================================================
* 5. Bytecode
* ================================================================ */
enum {
OP_CHAR, OP_CLASS, OP_ANY, OP_SPLIT, OP_JMP, OP_SAVE, OP_MATCH,
OP_ASSERT, OP_BACKREF, OP_LOOKAHEAD, OP_LOOKBEHIND, OP_ATOMIC,
OP_RETURN, OP_REPEAT1
};
typedef struct {
int op;
int32_t x, y;
uint32_t data;
int width;
int neg;
int is_loop;
int loop_close;
int32_t lo, hi;
int greedy;
int atomkind; /* OP_REPEAT1: 0=CHAR 1=CLASS 2=ANY */
} Inst;
typedef struct {
Inst *insts;
int n, cap;
DArr classnodes; /* Node* */
int ngroups;
int mode;
int flags;
int has_backref; /* computed once after compiling; gates memoized search, see run_memo */
} Prog;
static int emit(Prog *pr, int op) {
if (pr->n == pr->cap) {
pr->cap = pr->cap ? pr->cap * 2 : 64;
pr->insts = realloc(pr->insts, (size_t)pr->cap * sizeof(Inst));
}
Inst *i = &pr->insts[pr->n];
memset(i, 0, sizeof(*i));
i->op = op; i->x = i->y = -1;
return pr->n++;
}
static void compile_node(Prog *pr, Node *n);
static void compile_repeat(Prog *pr, Node *n) {
if (is_simple_atom(n->a)) {
int pc = emit(pr, OP_REPEAT1);
Inst *in = &pr->insts[pc];
in->lo = n->lo; in->hi = n->hi; in->greedy = n->greedy;
if (n->a->kind == N_CHAR) { in->atomkind = 0; in->data = n->a->ch; }
else if (n->a->kind == N_ANY) { in->atomkind = 2; }
else {
in->atomkind = 1;
int idx = (int)pr->classnodes.len;
Node **slot = da_push(&pr->classnodes); *slot = n->a;
in->data = (uint32_t)idx;
}
return;
}
int lo = n->lo, hi = n->hi;
for (int k = 0; k < lo; k++) compile_node(pr, n->a);
if (hi == -1) {
int nullable = node_nullable(n->a);
int split_pc = emit(pr, OP_SPLIT);
pr->insts[split_pc].is_loop = nullable;
int body_start = pr->n;
compile_node(pr, n->a);
int jmp_pc = emit(pr, OP_JMP);
pr->insts[jmp_pc].x = split_pc;
pr->insts[jmp_pc].loop_close = nullable;
int end = pr->n;
if (n->greedy) { pr->insts[split_pc].x = body_start; pr->insts[split_pc].y = end; }
else { pr->insts[split_pc].x = end; pr->insts[split_pc].y = body_start; }
} else if (hi > lo) {
DArr splits; da_init(&splits, sizeof(int));
for (int k = 0; k < hi - lo; k++) {
int sp = emit(pr, OP_SPLIT);
int *slot = da_push(&splits); *slot = sp;
int body_start = pr->n;
compile_node(pr, n->a);
if (n->greedy) pr->insts[sp].x = body_start; else pr->insts[sp].y = body_start;
}
int end = pr->n;
for (size_t k = 0; k < splits.len; k++) {
int sp = ((int *)splits.data)[k];
if (n->greedy) pr->insts[sp].y = end; else pr->insts[sp].x = end;
}
da_free(&splits);
}
}
static void compile_node(Prog *pr, Node *n) {
switch (n->kind) {
case N_EMPTY: return;
case N_CHAR: { int pc = emit(pr, OP_CHAR); pr->insts[pc].data = n->ch; return; }
case N_ANY: emit(pr, OP_ANY); return;
case N_CLASS: {
int idx = (int)pr->classnodes.len;
Node **slot = da_push(&pr->classnodes); *slot = n;
int pc = emit(pr, OP_CLASS); pr->insts[pc].data = (uint32_t)idx;
return;
}
case N_CONCAT: compile_node(pr, n->a); compile_node(pr, n->b); return;
case N_ALT: {
int split_pc = emit(pr, OP_SPLIT);
int x0 = pr->n;
compile_node(pr, n->a);
int jmp_pc = emit(pr, OP_JMP);
int y0 = pr->n;
compile_node(pr, n->b);
int end = pr->n;
pr->insts[split_pc].x = x0; pr->insts[split_pc].y = y0;
pr->insts[jmp_pc].x = end;
return;
}
case N_REPEAT: compile_repeat(pr, n); return;
case N_GROUP: {
if (n->group_index >= 0) { int pc = emit(pr, OP_SAVE); pr->insts[pc].data = (uint32_t)(2 * n->group_index); }
compile_node(pr, n->a);
if (n->group_index >= 0) { int pc = emit(pr, OP_SAVE); pr->insts[pc].data = (uint32_t)(2 * n->group_index + 1); }
return;
}
case N_BACKREF: { int pc = emit(pr, OP_BACKREF); pr->insts[pc].data = (uint32_t)n->backref_index; return; }
case N_ANCHOR: { int pc = emit(pr, OP_ASSERT); pr->insts[pc].data = (uint32_t)n->anchor_kind; return; }
case N_LOOKAROUND: {
int op = (n->lookaround_kind == LK_AHEAD) ? OP_LOOKAHEAD : OP_LOOKBEHIND;
int pc = emit(pr, op);
pr->insts[pc].neg = n->lookaround_negate;
pr->insts[pc].width = n->width;
int sub_start = pr->n;
compile_node(pr, n->a);
emit(pr, OP_RETURN);
int after = pr->n;
pr->insts[pc].x = sub_start; pr->insts[pc].y = after;
return;
}
case N_ATOMIC: {
int pc = emit(pr, OP_ATOMIC);
int sub_start = pr->n;
compile_node(pr, n->a);
emit(pr, OP_RETURN);
int after = pr->n;
pr->insts[pc].x = sub_start; pr->insts[pc].y = after;
return;
}
}
}
/* ================================================================
* 6. Matcher (recursive backtracking VM, concept.md Section 7.3/7.1)
* ================================================================ */
#define MAX_DEPTH 60000
typedef struct {
Prog *pr;
const uint32_t *text;
int64_t len;
int64_t *caps; /* 2*(ngroups+1) */
int64_t require_end; /* -1 unconstrained, -2 "sub-program, report on OP_RETURN", >=0 exact end required */
int64_t sub_end; /* scratch: end sp reported by a sub-program's OP_RETURN */
int depth_exceeded;
uint8_t *memo; /* NULL, or a (ninsts * (len+1))-bit "known to fail" cache, see run_memo */
int32_t **repeat_maxrun; /* NULL, or per-OP_REPEAT1-instruction run-length tables, see compute_maxrun */
int32_t **repeat_nextpm; /* NULL, or per-OP_REPEAT1-instruction "rightmost position <= X where
* the following atom matches" tables, see compute_next_prevmatch */
int forbid_empty; /* 1: OP_MATCH must not end at the attempt's own start, see
* Pattern_finditer/Pattern_split's same-position retry */
} MCtx;
/* Root-cause fix for the search-family quadratic behavior documented in
* README.md "Implementation status": a plain backtracking matcher has
* no memory that "starting at instruction pc, text position sp is
* hopeless", so re-trying nearby start positions (do_one's loop) or
* nearby alternative partitions of the same run of characters
* ((a+)+b-style catastrophic backtracking) re-derives the same failure
* over and over. For a pattern with no backreference, whether
* execution starting at (pc, sp) can ever reach OP_MATCH is a pure
* function of (pc, sp) alone whenever it is checked outside any active
* nullable-loop guard (guard_pc == -1) and outside any nested
* sub-program's constrained context (require_end == -1); this is
* exactly the "memoized backtracking" technique, provably the same
* complexity class as simulating the compiled program as an NFA
* (concept.md Section 7.2's Pike VM), just expressed recursively with
* a memo table instead of iteratively over an explicit thread list.
* Memoization never caches a *success*, only a proven failure, so it
* cannot change which match (or which captures) is found, only skip
* re-deriving failures already known. This is why it is always safe to
* enable, and it is enabled automatically whenever the compiled
* program has no OP_BACKREF (Prog.has_backref), regardless of which
* Pattern_ function is used. */
static int memo_get(MCtx *c, int pc, int64_t sp) {
if (!c->memo) return 0;
int64_t bit = (int64_t)pc * (c->len + 1) + sp;
return (c->memo[bit >> 3] >> (bit & 7)) & 1;
}
static void memo_set(MCtx *c, int pc, int64_t sp) {
if (!c->memo) return;
int64_t bit = (int64_t)pc * (c->len + 1) + sp;
c->memo[bit >> 3] |= (uint8_t)(1u << (bit & 7));
}
static int class_match(Node *cls, uint32_t cp, int mode, int flags) {
int hit = 0;
for (size_t k = 0; k < cls->items.len && !hit; k++) {
ClassItem *it = &((ClassItem *)cls->items.data)[k];
int m;
switch (it->kind) {
case CI_RANGE:
m = (cp >= it->lo && cp <= it->hi);
if (!m && (flags & IGNORECASE)) {
uint32_t alt = cls_swapcase(cp, mode, flags);
m = (alt >= it->lo && alt <= it->hi);
}
break;
case CI_D: m = cls_is_digit(cp, mode, flags); break;
case CI_W: m = cls_is_word(cp, mode, flags); break;
case CI_S: m = cls_is_space(cp, mode, flags); break;
default: m = 0;
}
if (it->kind != CI_RANGE && it->neg) m = !m;
if (m) hit = 1;
}
if (cls->class_negate) hit = !hit;
return hit;
}
static int char_eq(uint32_t a, uint32_t b, int mode, int flags) {
if (a == b) return 1;
if (flags & IGNORECASE) return cls_fold(a, mode, flags) == cls_fold(b, mode, flags);
return 0;
}
static int is_word_at(MCtx *c, int64_t pos) {
if (pos < 0 || pos >= c->len) return 0;
return cls_is_word(c->text[pos], c->pr->mode, c->pr->flags);
}
static int run_memo(MCtx *c, int pc, int64_t sp, int64_t guard_sp, int32_t guard_pc, int depth);
static int run(MCtx *c, int pc, int64_t sp, int64_t guard_sp, int32_t guard_pc, int depth) {
if (depth > MAX_DEPTH) { c->depth_exceeded = 1; return 0; }
for (;;) {
Inst *in = &c->pr->insts[pc];
switch (in->op) {
case OP_CHAR:
if (sp >= c->len || !char_eq(c->text[sp], in->data, c->pr->mode, c->pr->flags)) return 0;
pc++; sp++; continue;
case OP_ANY:
if (sp >= c->len) return 0;
if (c->text[sp] == '\n' && !(c->pr->flags & DOTALL)) return 0;
pc++; sp++; continue;
case OP_CLASS: {
if (sp >= c->len) return 0;
Node *cls = ((Node **)c->pr->classnodes.data)[in->data];
if (!class_match(cls, c->text[sp], c->pr->mode, c->pr->flags)) return 0;
pc++; sp++; continue;
}
case OP_REPEAT1: {
int64_t maxc = (in->hi == -1) ? (c->len - sp) : in->hi;
if (maxc > c->len - sp) maxc = c->len - sp;
int64_t count;
if (c->repeat_maxrun && c->repeat_maxrun[pc]) {
/* O(1): see compute_maxrun. Without this, counting
* how many atoms match from sp is an O(remaining
* length) scan repeated at every start position
* do_one/finditer/split try, which is exactly the
* quadratic-time defect documented in README.md
* "Implementation status"; this is its fix. */
count = c->repeat_maxrun[pc][sp];
} else {
count = 0;
Node *cls = (in->atomkind == 1) ? ((Node **)c->pr->classnodes.data)[in->data] : NULL;
while (count < maxc) {
uint32_t ch = c->text[sp + count];
int m;
if (in->atomkind == 0) m = char_eq(ch, in->data, c->pr->mode, c->pr->flags);
else if (in->atomkind == 2) m = !(ch == '\n' && !(c->pr->flags & DOTALL));
else m = class_match(cls, ch, c->pr->mode, c->pr->flags);
if (!m) break;
count++;
}
}
if (count > maxc) count = maxc;
if (count < in->lo) return 0;
if (in->greedy && c->repeat_nextpm && c->repeat_nextpm[pc]) {
/* O(candidates actually worth trying), not O(count):
* see compute_next_prevmatch. Trying every k from
* count down to lo one at a time costs O(count)
* loop iterations even though each individual
* OP_CHAR/OP_CLASS/OP_ANY check at pc+1 is O(1),
* because the *number* of iterations, not the cost
* of any one of them, is what stayed quadratic
* across do_one/finditer/split's outer position
* loop; jumping straight to positions where pc+1
* can actually succeed is what fixes that. */
int32_t *pm = c->repeat_nextpm[pc];
int64_t hi = sp + count; if (hi > c->len - 1) hi = c->len - 1;
int64_t lo_bound = sp + in->lo;
int64_t p = (hi >= lo_bound && hi >= 0) ? pm[hi] : -1;
while (p >= lo_bound) {
if (run_memo(c, pc + 1, p, guard_sp, guard_pc, depth + 1)) return 1;
p = (p > 0) ? pm[p - 1] : -1;
}
return 0;
}
if (in->greedy) {
for (int64_t k = count; k >= in->lo; k--)
if (run_memo(c, pc + 1, sp + k, guard_sp, guard_pc, depth + 1)) return 1;
} else {
for (int64_t k = in->lo; k <= count; k++)
if (run_memo(c, pc + 1, sp + k, guard_sp, guard_pc, depth + 1)) return 1;
}
return 0;
}
case OP_ASSERT: {
int ok;
switch (in->data) {
case A_BOL: ok = (sp == 0) || ((c->pr->flags & MULTILINE) && sp > 0 && c->text[sp - 1] == '\n'); break;
case A_EOL: ok = (sp == c->len) || (c->text[sp] == '\n' && ((c->pr->flags & MULTILINE) || sp == c->len - 1)); break;
case A_BOS: ok = (sp == 0); break;
case A_EOS: ok = (sp == c->len); break;
case A_WB: ok = is_word_at(c, sp - 1) != is_word_at(c, sp); break;
case A_NWB: ok = is_word_at(c, sp - 1) == is_word_at(c, sp); break;
default: ok = 0;
}
if (!ok) return 0;
pc++; continue;
}
case OP_SAVE: {
int64_t old = c->caps[in->data];
c->caps[in->data] = sp;
if (run_memo(c, pc + 1, sp, guard_sp, guard_pc, depth + 1)) return 1;
c->caps[in->data] = old;
return 0;
}
case OP_SPLIT:
if (in->is_loop) {
if (run_memo(c, in->x, sp, sp, pc, depth + 1)) return 1;
pc = in->y; continue;
} else {
if (run_memo(c, in->x, sp, guard_sp, guard_pc, depth + 1)) return 1;
pc = in->y; continue;
}
case OP_JMP:
if (in->loop_close && in->x == guard_pc && sp == guard_sp) return 0;
pc = in->x; continue;
case OP_BACKREF: {
int gi = (int)in->data;
if (gi < 0 || gi > c->pr->ngroups) return 0;
int64_t s = c->caps[2 * gi], e = c->caps[2 * gi + 1];
if (s < 0 || e < 0) return 0;
int64_t rl = e - s;
if (sp + rl > c->len) return 0;
for (int64_t k = 0; k < rl; k++)
if (!char_eq(c->text[sp + k], c->text[s + k], c->pr->mode, c->pr->flags)) return 0;
pc++; sp += rl; continue;
}
case OP_LOOKAHEAD: {
size_t ncaps = 2 * (size_t)(c->pr->ngroups + 1);
int64_t *snap = malloc(ncaps * sizeof(int64_t));
memcpy(snap, c->caps, ncaps * sizeof(int64_t));
int64_t save_req = c->require_end; c->require_end = -2;
int ok = run_memo(c, in->x, sp, -1, -1, depth + 1);
c->require_end = save_req;
int accept = in->neg ? !ok : ok;
if (!accept) { memcpy(c->caps, snap, ncaps * sizeof(int64_t)); free(snap); return 0; }
free(snap);
pc = in->y; continue;
}
case OP_LOOKBEHIND: {
size_t ncaps = 2 * (size_t)(c->pr->ngroups + 1);
int64_t *snap = malloc(ncaps * sizeof(int64_t));
memcpy(snap, c->caps, ncaps * sizeof(int64_t));
int64_t start = sp - in->width;
int ok = 0;
if (start >= 0) {
int64_t save_req = c->require_end; c->require_end = sp;
ok = run_memo(c, in->x, start, -1, -1, depth + 1);
c->require_end = save_req;
}
int accept = in->neg ? !ok : ok;
if (!accept) { memcpy(c->caps, snap, ncaps * sizeof(int64_t)); free(snap); return 0; }
free(snap);
pc = in->y; continue;
}
case OP_ATOMIC: {
int64_t save_req = c->require_end; c->require_end = -2;
int ok = run_memo(c, in->x, sp, -1, -1, depth + 1);
c->require_end = save_req;
if (!ok) return 0;
sp = c->sub_end;
pc = in->y; continue;
}
case OP_RETURN:
if (c->require_end == -2) { c->sub_end = sp; return 1; }
if (c->require_end >= 0) return sp == c->require_end;
return 1;
case OP_MATCH:
if (c->require_end >= 0 && sp != c->require_end) return 0;
if (c->forbid_empty && sp == c->caps[0]) return 0;
c->caps[1] = sp;
return 1;
default:
return 0;
}
}
}
/* The only call site allowed to consult/populate the memo (see the
* comment on MCtx.memo above): every recursive call inside run(), and
* every external entry point, goes through here instead of run()
* directly. Caching is restricted to guard_pc == -1 (not inside an
* active nullable-loop guard) and c->require_end == -1 (not inside a
* fullmatch's exact-end constraint, nor inside a nested lookaround/
* atomic sub-program, which always sets require_end elsewhere first);
* outside that combination, whether (pc, sp) succeeds can depend on
* context beyond (pc, sp) itself, so it is never cached or consulted
* there, only computed directly by run(), exactly as before this
* optimization existed. */
static int run_memo(MCtx *c, int pc, int64_t sp, int64_t guard_sp, int32_t guard_pc, int depth) {
int cacheable = c->memo && guard_pc == -1 && c->require_end == -1 && !c->forbid_empty;
if (cacheable && memo_get(c, pc, sp)) return 0;
int result = run(c, pc, sp, guard_sp, guard_pc, depth);
if (cacheable && !result && !c->depth_exceeded) memo_set(c, pc, sp);
return result;
}
/* ================================================================
* 7. Pattern / Match / Input structures
* ================================================================ */
typedef struct { char **names; int *idx; int n; } GroupIndex;
typedef struct {
Prog prog;
GroupIndex gidx;
NodePool pool;
char *pattern_copy;
} PatternImpl;
struct Input {
uint8_t *buf;
size_t len;
int owns_buf;
};
typedef struct {
int64_t *pos; /* 2*(ngroups+1) */
int ngroups;
int64_t *cp_to_byte; /* NULL unless UTF8 mode; length ngroups-independent, sized to text length+1 */
} MatchSlots;
static int groupindex_lookup(GroupIndex *gi, const char *name) {
for (int k = 0; k < gi->n; k++) if (strcmp(gi->names[k], name) == 0) return gi->idx[k];
return -1;
}
static int resolve_mode(int flags) {
if (flags & BINARY) return MODE_BINARY;
if (flags & UTF8) return MODE_UTF8;
return MODE_ASCII;
}
static void pattern_error_fill(PatternError *err, const char *pattern, const char *msg, int64_t pos) {
if (!err) return;
err->msg = strdup(msg ? msg : "");
err->pattern = pattern ? strdup(pattern) : NULL;
err->pos = pos;
err->lineno = 1; err->colno = 1;
if (pattern) {
for (int64_t k = 0; k < pos && pattern[k]; k++) {
if (pattern[k] == '\n') { err->lineno++; err->colno = 1; } else err->colno++;
}
}
}
void PatternError_free(PatternError *err) {
if (!err) return;
free((void *)err->msg); free((void *)err->pattern);
err->msg = NULL; err->pattern = NULL;
}
/* ================================================================
* 8. Compile driver
* ================================================================ */
Pattern *re_compile(const char *pattern, size_t len, int flags, PatternError *err) {
/* Section 9.3: BINARY/UTF8 select their mode; otherwise ASCII. */
int mode = resolve_mode(flags);
/* Step 1: leading global inline flags (?aiLmsux), Section 2.5/2.6 */
int extra_flags = 0;
size_t body_off = 0;
if (len >= 3 && pattern[0] == '(' && pattern[1] == '?') {
size_t k = 2; int any = 0, ok = 1;
while (k < len && strchr("aiLmsux", (unsigned char)pattern[k])) {
switch (pattern[k]) {
case 'i': extra_flags |= IGNORECASE; break;
case 'm': extra_flags |= MULTILINE; break;
case 's': extra_flags |= DOTALL; break;
case 'x': extra_flags |= VERBOSE; break;
case 'a': extra_flags |= ASCII; break;
case 'L': extra_flags |= LOCALE; break;
default: break;
}
k++; any = 1;
}
if (any && k < len && pattern[k] == ')') { body_off = k + 1; }
else { extra_flags = 0; ok = 0; (void)ok; }
}
int all_flags = flags | extra_flags;
const char *body = pattern + body_off;
size_t bodylen = len - body_off;
char *stripped = NULL;
if (all_flags & VERBOSE) {
size_t nl;
stripped = strip_verbose(body, bodylen, &nl);
body = stripped; bodylen = nl;
}
Parser p; memset(&p, 0, sizeof p);
p.s = body; p.len = bodylen; p.i = 0; p.flags = all_flags; p.mode = mode;
da_init(&p.backrefs, sizeof(Node *));
da_init(&p.groupnames, sizeof(GroupName));
g_pool.data = NULL; g_pool.len = 0; g_pool.cap = 0;
Node *ast = parse_alt(&p);
if (!p.failed && !at_end(&p)) {
if (peek(&p) == ')') perr(&p, (int64_t)p.i, "unbalanced parenthesis");
else perr(&p, (int64_t)p.i, "unexpected character");
}
if (!p.failed) {
for (size_t k = 0; k < p.backrefs.len; k++) {
Node *b = ((Node **)p.backrefs.data)[k];
if (b->backref_name) {
int found = -1;
for (size_t j = 0; j < p.groupnames.len; j++) {
GroupName *gn = &((GroupName *)p.groupnames.data)[j];
if (strcmp(gn->name, b->backref_name) == 0) { found = gn->index; break; }
}
if (found < 0) { perr(&p, 0, "unknown group name '%s'", b->backref_name); break; }
b->backref_index = found;
} else if (b->backref_index < 1 || b->backref_index > p.ngroups) {
perr(&p, 0, "invalid group reference %d", b->backref_index);
break;
}
}
}
if (p.failed) {
pattern_error_fill(err, pattern, p.errmsg, p.errpos + (int64_t)body_off);
nodepool_free_all(&g_pool);
da_free(&p.backrefs);
for (size_t k = 0; k < p.groupnames.len; k++) free(((GroupName *)p.groupnames.data)[k].name);
da_free(&p.groupnames);
free(stripped);
return NULL;
}
Prog prog; memset(&prog, 0, sizeof prog);
prog.ngroups = p.ngroups; prog.mode = mode; prog.flags = all_flags;
da_init(&prog.classnodes, sizeof(Node *));
compile_node(&prog, ast);
emit(&prog, OP_MATCH);
for (int k = 0; k < prog.n; k++) if (prog.insts[k].op == OP_BACKREF) { prog.has_backref = 1; break; }
PatternImpl *impl = calloc(1, sizeof(PatternImpl));
impl->prog = prog;
impl->pool = g_pool;
impl->pattern_copy = xstrndup(pattern, len);
impl->gidx.n = (int)p.groupnames.len;
impl->gidx.names = malloc(sizeof(char *) * (size_t)(impl->gidx.n ? impl->gidx.n : 1));
impl->gidx.idx = malloc(sizeof(int) * (size_t)(impl->gidx.n ? impl->gidx.n : 1));
for (int k = 0; k < impl->gidx.n; k++) {
GroupName *gn = &((GroupName *)p.groupnames.data)[k];
impl->gidx.names[k] = gn->name; /* transfer ownership */
impl->gidx.idx[k] = gn->index;
}
da_free(&p.groupnames);
da_free(&p.backrefs);
free(stripped);
Pattern *pat = calloc(1, sizeof(Pattern));
pat->pattern = impl->pattern_copy;
pat->flags = all_flags;
pat->groups = prog.ngroups;
pat->groupindex = &impl->gidx;
pat->program = impl;
return pat;
}
void Pattern_free(Pattern *self) {
if (!self) return;
PatternImpl *impl = (PatternImpl *)self->program;
if (impl) {
free(impl->prog.insts);
da_free(&impl->prog.classnodes);
nodepool_free_all(&impl->pool);
for (int k = 0; k < impl->gidx.n; k++) free(impl->gidx.names[k]);
free(impl->gidx.names);
free(impl->gidx.idx);
free(impl->pattern_copy);
free(impl);
}
free(self);
}
/* ================================================================
* 9. Input
* ================================================================ */
Input *Input_from_buffer(const uint8_t *buf, size_t len) {
Input *in = calloc(1, sizeof(Input));
in->buf = (uint8_t *)buf; in->len = len; in->owns_buf = 0;
return in;
}
Input *Input_from_file(const char *path, PatternError *err) {
FILE *f = fopen(path, "rb");
if (!f) { pattern_error_fill(err, NULL, strerror(errno), 0); return NULL; }
if (fseek(f, 0, SEEK_END) == 0) {
long sz = ftell(f);
if (sz >= 0) {
rewind(f);
uint8_t *buf = malloc((size_t)sz > 0 ? (size_t)sz : 1);
size_t got = fread(buf, 1, (size_t)sz, f);
fclose(f);
Input *in = calloc(1, sizeof(Input));
in->buf = buf; in->len = got; in->owns_buf = 1;
return in;
}
}
/* Not seekable (a pipe, a FIFO, process substitution, stdin): the
* size cannot be known upfront, so read incrementally into a
* growable buffer instead. Still materializes the whole input in
* memory, per this file's "Implementation status" note. */
clearerr(f);
size_t cap = 1 << 16, len = 0;
uint8_t *buf = malloc(cap);
size_t n;
while ((n = fread(buf + len, 1, cap - len, f)) > 0) {
len += n;
if (len == cap) { cap *= 2; buf = realloc(buf, cap); }
}
fclose(f);
Input *in = calloc(1, sizeof(Input));
in->buf = buf; in->len = len; in->owns_buf = 1;
return in;
}
void Input_free(Input *in) {
if (!in) return;
if (in->owns_buf) free(in->buf);
free(in);
}
/* ================================================================
* 10. Match buffer construction (UTF-8 pre-decode, Section 8.4/9.3)
* ================================================================ */
typedef struct {
uint32_t *text;
int64_t len;
int64_t *cp_to_byte; /* len+1 entries, NULL for non-UTF8 */
} MatBuf;
static int build_matbuf(Pattern *pat, Input *in, MatBuf *mb, PatternError *err) {
int mode = resolve_mode(pat->flags);
if (mode == MODE_UTF8) {
DArr cps; da_init(&cps, sizeof(uint32_t));
DArr offs; da_init(&offs, sizeof(int64_t));
size_t i = 0;
while (i < in->len) {
uint32_t cp;
int n = utf8_decode(in->buf, in->len, i, &cp);
if (n == 0) {
pattern_error_fill(err, NULL, "invalid UTF-8 in subject", (int64_t)i);
da_free(&cps); da_free(&offs);
return 0;
}
uint32_t *cpp = da_push(&cps); *cpp = cp;
int64_t *op = da_push(&offs); *op = (int64_t)i;
i += (size_t)n;
}
int64_t *sentinel = da_push(&offs); *sentinel = (int64_t)in->len;
mb->text = (uint32_t *)cps.data;
mb->cp_to_byte = (int64_t *)offs.data;
mb->len = (int64_t)cps.len;
} else {
uint32_t *text = malloc((in->len ? in->len : 1) * sizeof(uint32_t));
for (size_t i = 0; i < in->len; i++) text[i] = in->buf[i];
mb->text = text; mb->len = (int64_t)in->len; mb->cp_to_byte = NULL;
}
return 1;
}
static void free_matbuf(MatBuf *mb) { free(mb->text); free(mb->cp_to_byte); }
/* ================================================================
* 11. Matching driver and Pattern_ methods
* ================================================================ */
static void alloc_caps(int64_t **caps, int ngroups) {
*caps = malloc(2 * (size_t)(ngroups + 1) * sizeof(int64_t));
}
static void reset_caps(int64_t *caps, int ngroups) {
for (int k = 0; k < 2 * (ngroups + 1); k++) caps[k] = -1;
}
static void fill_match_from_caps(Match *out, Pattern *self, Input *string, int64_t pos, int64_t endpos,
int64_t *caps, int ngroups, MatBuf *mb) {
MatchSlots *ms = calloc(1, sizeof(MatchSlots));
ms->pos = caps; ms->ngroups = ngroups;
if (mb->cp_to_byte) {
ms->cp_to_byte = malloc((size_t)(mb->len + 1) * sizeof(int64_t));
memcpy(ms->cp_to_byte, mb->cp_to_byte, (size_t)(mb->len + 1) * sizeof(int64_t));
}
out->re = self; out->string = string; out->pos = pos; out->endpos = endpos;
out->slots = ms; out->lastindex = -1; out->lastgroup = NULL;
PatternImpl *impl = (PatternImpl *)self->program;
for (int g = ngroups; g >= 1; g--) {
if (caps[2 * g] >= 0) {
out->lastindex = g;
for (int k = 0; k < impl->gidx.n; k++)
if (impl->gidx.idx[k] == g) { out->lastgroup = impl->gidx.names[k]; break; }
break;
}
}
}
static int64_t clamp(int64_t v, int64_t lo, int64_t hi) { return v < lo ? lo : (v > hi ? hi : v); }
/* Allocated once per top-level Pattern_/re_ call (not once per start
* position tried, and, for Pattern_finditer/Pattern_split, not once
* per match found either): see the comment on MCtx.memo. NULL, with no
* behavior change beyond the missing optimization, whenever the
* pattern contains a backreference anywhere. */
static uint8_t *alloc_memo(Prog *pr, int64_t textlen) {
if (pr->has_backref) return NULL;
int64_t bits = (int64_t)pr->n * (textlen + 1);
int64_t bytes = (bits + 7) / 8;
return calloc((size_t)(bytes > 0 ? bytes : 1), 1);
}
/* mr[i] = how many consecutive positions starting at text[i] satisfy
* in's atom predicate (0 if text[i] itself does not), computed in one
* backward O(textlen) pass instead of redone forward, from scratch, at
* every position a caller asks about it. This is what makes
* OP_REPEAT1 O(1) per position instead of O(remaining run length),
* which otherwise stays quadratic across do_one/finditer/split's outer
* position loop even with run_memo's (pc, sp) memoization, because the
* counting scan is an internal C loop, not expressed as (pc, sp)
* recursive calls at all. int32_t bounds a single repeat's run length
* to ~2 billion, far past any input this build's fully-materializing
* Input can hold in memory in the first place. */
static int32_t *compute_maxrun(Prog *pr, Inst *in, const uint32_t *text, int64_t len) {
int32_t *mr = malloc((size_t)(len + 1) * sizeof(int32_t));
if (!mr) return NULL;
mr[len] = 0;
Node *cls = (in->atomkind == 1) ? ((Node **)pr->classnodes.data)[in->data] : NULL;
for (int64_t i = len - 1; i >= 0; i--) {
uint32_t ch = text[i];
int m;
if (in->atomkind == 0) m = char_eq(ch, in->data, pr->mode, pr->flags);
else if (in->atomkind == 2) m = !(ch == '\n' && !(pr->flags & DOTALL));
else m = class_match(cls, ch, pr->mode, pr->flags);
int64_t next = mr[i + 1];
mr[i] = m ? (int32_t)(next < INT32_MAX ? next + 1 : INT32_MAX) : 0;
}
return mr;
}
/* One table per OP_REPEAT1 instruction actually present in the
* program, indexed by instruction number; only worth the O(text
* length) memory per instruction for the multi-position operations
* (search/finditer/split) that would otherwise redo the scan at every
* position, so match/fullmatch (a single attempt) do not allocate it. */
static int32_t **alloc_repeat_maxrun(Prog *pr, const uint32_t *text, int64_t len) {
int32_t **arr = calloc((size_t)pr->n, sizeof(int32_t *));
if (!arr) return NULL;
for (int i = 0; i < pr->n; i++)
if (pr->insts[i].op == OP_REPEAT1)
arr[i] = compute_maxrun(pr, &pr->insts[i], text, len);
return arr;
}
static void free_repeat_maxrun(Prog *pr, int32_t **arr) {
if (!arr) return;
for (int i = 0; i < pr->n; i++) free(arr[i]);
free(arr);
}
static int inst_atom_match(Prog *pr, Inst *in, uint32_t ch) {
switch (in->op) {
case OP_CHAR: return char_eq(ch, in->data, pr->mode, pr->flags);
case OP_ANY: return !(ch == '\n' && !(pr->flags & DOTALL));
case OP_CLASS: return class_match(((Node **)pr->classnodes.data)[in->data], ch, pr->mode, pr->flags);
default: return 0;
}
}
/* pm[i] = the largest position <= i where the instruction right after
* an OP_REPEAT1 matches, or -1 if none exists in [0, i]. Only computed
* when that next instruction is itself a single simple atom (OP_CHAR/
* OP_CLASS/OP_ANY); this is what lets OP_REPEAT1's greedy backtrack
* jump straight from "the largest k worth trying" to "the next
* smaller one worth trying" instead of visiting every k in between
* (see the comment at its one call site). Built in one forward
* O(textlen) pass instead of walked freshly, backward, from every
* position a caller asks about it. */
static int32_t *compute_next_prevmatch(Prog *pr, Inst *next, const uint32_t *text, int64_t len) {
if (len <= 0) return NULL;
int32_t *pm = malloc((size_t)len * sizeof(int32_t));
if (!pm) return NULL;
int32_t last = -1;
for (int64_t i = 0; i < len; i++) {
if (inst_atom_match(pr, next, text[i])) last = (int32_t)i;
pm[i] = last;
}
return pm;
}
static int32_t **alloc_repeat_nextpm(Prog *pr, const uint32_t *text, int64_t len) {
int32_t **arr = calloc((size_t)pr->n, sizeof(int32_t *));
if (!arr) return NULL;
for (int i = 0; i < pr->n; i++) {
if (pr->insts[i].op != OP_REPEAT1 || !pr->insts[i].greedy) continue;
if (i + 1 >= pr->n) continue;
int nextop = pr->insts[i + 1].op;
if (nextop == OP_CHAR || nextop == OP_ANY || nextop == OP_CLASS)
arr[i] = compute_next_prevmatch(pr, &pr->insts[i + 1], text, len);
}
return arr;
}
static void free_repeat_nextpm(Prog *pr, int32_t **arr) {
if (!arr) return;
for (int i = 0; i < pr->n; i++) free(arr[i]);
free(arr);
}
static int do_one(Pattern *self, Input *string, int64_t pos, int64_t endpos, int anchored, int fullmatch, Match *out) {
MatBuf mb;
if (!build_matbuf(self, string, &mb, NULL)) return -1;
int64_t ep = (endpos < 0) ? mb.len : clamp(endpos, 0, mb.len);
int64_t p0 = clamp(pos, 0, mb.len);
PatternImpl *impl = (PatternImpl *)self->program;
int ng = impl->prog.ngroups;
int64_t *caps; alloc_caps(&caps, ng);
uint8_t *memo = alloc_memo(&impl->prog, ep);
/* Only search (anchored == 0) tries more than one position, so
* only search pays for the run-length precompute (see
* compute_maxrun); match/fullmatch's single attempt gets no
* benefit from it and skips the O(text length) memory. */
int32_t **maxrun = anchored ? NULL : alloc_repeat_maxrun(&impl->prog, mb.text, ep);
int32_t **nextpm = anchored ? NULL : alloc_repeat_nextpm(&impl->prog, mb.text, ep);
int found = 0;
int64_t last_start = anchored ? p0 : ep;
for (int64_t start = p0; start <= last_start && !found; start++) {
reset_caps(caps, ng);
caps[0] = start;
MCtx c; c.pr = &impl->prog; c.text = mb.text; c.len = ep; c.caps = caps;
c.depth_exceeded = 0; c.require_end = fullmatch ? ep : -1; c.sub_end = -1; c.memo = memo;
c.repeat_maxrun = maxrun; c.repeat_nextpm = nextpm; c.forbid_empty = 0;
found = run_memo(&c, 0, start, -1, -1, 0);
if (!found && c.depth_exceeded) { free(caps); free(memo); free_repeat_maxrun(&impl->prog, maxrun); free_repeat_nextpm(&impl->prog, nextpm); free_matbuf(&mb); return -1; }
}
free(memo);
free_repeat_maxrun(&impl->prog, maxrun);
free_repeat_nextpm(&impl->prog, nextpm);
if (!found) { free(caps); free_matbuf(&mb); return 0; }
fill_match_from_caps(out, self, string, p0, ep, caps, ng, &mb);
free_matbuf(&mb);
return 1;
}
int Pattern_match(Pattern *self, Input *string, int64_t pos, int64_t endpos, Match *out) {
return do_one(self, string, pos, endpos, 1, 0, out);
}
int Pattern_fullmatch(Pattern *self, Input *string, int64_t pos, int64_t endpos, Match *out) {
return do_one(self, string, pos, endpos, 1, 1, out);
}
int Pattern_search(Pattern *self, Input *string, int64_t pos, int64_t endpos, Match *out) {
return do_one(self, string, pos, endpos, 0, 0, out);
}
int Pattern_finditer(Pattern *self, Input *string, int64_t pos, int64_t endpos, MatchIterCb cb, void *ctx) {
MatBuf mb;
if (!build_matbuf(self, string, &mb, NULL)) return -1;
int64_t ep = (endpos < 0) ? mb.len : clamp(endpos, 0, mb.len);
int64_t start = clamp(pos, 0, mb.len);
PatternImpl *impl = (PatternImpl *)self->program;
int ng = impl->prog.ngroups;
/* One memo table for the whole scan: every position tried, across
* every match found, shares it (see the comment on MCtx.memo). A
* fact it records ("(pc, sp) cannot reach OP_MATCH") never becomes
* false later in the same scan, so nothing here ever needs to
* invalidate or reset it between matches. */
uint8_t *memo = alloc_memo(&impl->prog, ep);
int32_t **maxrun = alloc_repeat_maxrun(&impl->prog, mb.text, ep); /* see compute_maxrun */
int32_t **nextpm = alloc_repeat_nextpm(&impl->prog, mb.text, ep); /* see compute_next_prevmatch */
int count = 0;
while (start <= ep) {
int64_t *caps; alloc_caps(&caps, ng);
int found = 0;
int64_t s;
for (s = start; s <= ep && !found; s++) {
reset_caps(caps, ng);
caps[0] = s;
MCtx c; c.pr = &impl->prog; c.text = mb.text; c.len = ep; c.caps = caps;
c.depth_exceeded = 0; c.require_end = -1; c.sub_end = -1; c.memo = memo;
c.repeat_maxrun = maxrun; c.repeat_nextpm = nextpm; c.forbid_empty = 0;
found = run_memo(&c, 0, s, -1, -1, 0);
if (!found && c.depth_exceeded) { free(caps); free(memo); free_repeat_maxrun(&impl->prog, maxrun); free_repeat_nextpm(&impl->prog, nextpm); free_matbuf(&mb); return -1; }
}
if (!found) { free(caps); break; }
Match m; memset(&m, 0, sizeof m);
MatBuf shallow = mb; shallow.cp_to_byte = mb.cp_to_byte; /* share for lookup, copied inside fill */
fill_match_from_caps(&m, self, string, start, ep, caps, ng, &shallow);
cb(ctx, &m);
count++;
int64_t mstart = caps[0], mend = caps[1];
Match_free(&m);
/* CPython's finditer, reverse engineered against a real
* interpreter (not documented): if the match just reported was
* empty, it additionally looks for a second, non-empty match at
* that exact same start position before moving on, and reports
* that one too if it exists (README.md/docs/API.md carry the
* probe cases this was derived from). OP_MATCH's forbid_empty
* check forces exactly that search: the same attempt, with the
* empty solution excluded, so ordinary backtracking finds the
* next (longer) alternative on its own if one exists. */
if (mend == mstart) {
int64_t *caps2; alloc_caps(&caps2, ng);
reset_caps(caps2, ng);
caps2[0] = mstart;
MCtx c2; c2.pr = &impl->prog; c2.text = mb.text; c2.len = ep; c2.caps = caps2;
c2.depth_exceeded = 0; c2.require_end = -1; c2.sub_end = -1; c2.memo = memo;
c2.repeat_maxrun = maxrun; c2.repeat_nextpm = nextpm; c2.forbid_empty = 1;
int found2 = run_memo(&c2, 0, mstart, -1, -1, 0);
if (!found2 && c2.depth_exceeded) { free(caps2); free(memo); free_repeat_maxrun(&impl->prog, maxrun); free_repeat_nextpm(&impl->prog, nextpm); free_matbuf(&mb); return -1; }
if (found2) {
Match m2; memset(&m2, 0, sizeof m2);
fill_match_from_caps(&m2, self, string, mstart, ep, caps2, ng, &shallow);
cb(ctx, &m2);
count++;
mend = caps2[1];
Match_free(&m2);
} else {
free(caps2);
}
}
start = (mend > mstart) ? mend : mstart + 1;
}
free(memo);
free_repeat_maxrun(&impl->prog, maxrun);
free_repeat_nextpm(&impl->prog, nextpm);
free_matbuf(&mb);
return count;
}
int Pattern_findall(Pattern *self, Input *string, int64_t pos, int64_t endpos, MatchIterCb cb, void *ctx) {
return Pattern_finditer(self, string, pos, endpos, cb, ctx);
}
/* Pattern_split's contract: cb is invoked once per element of the list
* Python's re.split() would return, in order, each time as a Match
* whose group 0 (via Match_group(m, NULL, 0, ...)) is that element's
* text; a "None" element (an unparticipated capturing group between
* two matches) reports via Match_group returning 0, matching how an
* unparticipated group reports on any other Match. */
static void emit_split_item(MatchIterCb cb, void *ctx, Pattern *self, Input *string, MatBuf *mb, int64_t s, int64_t e) {
int64_t *caps = malloc(2 * sizeof(int64_t));
caps[0] = s; caps[1] = e;
Match m; memset(&m, 0, sizeof m);
fill_match_from_caps(&m, self, string, 0, mb->len, caps, 0, mb);
cb(ctx, &m);
Match_free(&m);
}
int Pattern_split(Pattern *self, Input *string, int maxsplit, MatchIterCb cb, void *ctx) {
MatBuf mb;
if (!build_matbuf(self, string, &mb, NULL)) return -1;
int64_t ep = mb.len;
int64_t start = 0, seg_start = 0;
PatternImpl *impl = (PatternImpl *)self->program;
int ng = impl->prog.ngroups;
uint8_t *memo = alloc_memo(&impl->prog, ep); /* one table for the whole scan, see MCtx.memo */
int32_t **maxrun = alloc_repeat_maxrun(&impl->prog, mb.text, ep); /* see compute_maxrun */
int32_t **nextpm = alloc_repeat_nextpm(&impl->prog, mb.text, ep); /* see compute_next_prevmatch */
int n = 0;
while (start <= ep) {
if (maxsplit > 0 && n >= maxsplit) break;
int64_t *caps; alloc_caps(&caps, ng);
int found = 0; int64_t s;
for (s = start; s <= ep && !found; s++) {
reset_caps(caps, ng);
caps[0] = s;
MCtx c; c.pr = &impl->prog; c.text = mb.text; c.len = ep; c.caps = caps;
c.depth_exceeded = 0; c.require_end = -1; c.sub_end = -1; c.memo = memo;
c.repeat_maxrun = maxrun; c.repeat_nextpm = nextpm; c.forbid_empty = 0;
found = run_memo(&c, 0, s, -1, -1, 0);
if (!found && c.depth_exceeded) { free(caps); free(memo); free_repeat_maxrun(&impl->prog, maxrun); free_repeat_nextpm(&impl->prog, nextpm); free_matbuf(&mb); return -1; }
}
if (!found) { free(caps); break; }
emit_split_item(cb, ctx, self, string, &mb, seg_start, caps[0]);
for (int g = 1; g <= ng; g++) emit_split_item(cb, ctx, self, string, &mb, caps[2 * g], caps[2 * g + 1]);
n++;
seg_start = caps[1];
int64_t mstart = caps[0], mend = caps[1];
free(caps);
/* Same same-position retry as Pattern_finditer; see its comment. */
if (mend == mstart) {
if (maxsplit <= 0 || n < maxsplit) {
int64_t *caps2; alloc_caps(&caps2, ng);
reset_caps(caps2, ng);
caps2[0] = mstart;
MCtx c2; c2.pr = &impl->prog; c2.text = mb.text; c2.len = ep; c2.caps = caps2;
c2.depth_exceeded = 0; c2.require_end = -1; c2.sub_end = -1; c2.memo = memo;
c2.repeat_maxrun = maxrun; c2.repeat_nextpm = nextpm; c2.forbid_empty = 1;
int found2 = run_memo(&c2, 0, mstart, -1, -1, 0);
if (!found2 && c2.depth_exceeded) { free(caps2); free(memo); free_repeat_maxrun(&impl->prog, maxrun); free_repeat_nextpm(&impl->prog, nextpm); free_matbuf(&mb); return -1; }
if (found2) {
emit_split_item(cb, ctx, self, string, &mb, mstart, caps2[0]);
for (int g = 1; g <= ng; g++) emit_split_item(cb, ctx, self, string, &mb, caps2[2 * g], caps2[2 * g + 1]);
n++;
seg_start = caps2[1];
mend = caps2[1];
}
free(caps2);
}
}
start = (mend > mstart) ? mend : mstart + 1;
}
free(memo);
free_repeat_maxrun(&impl->prog, maxrun);
free_repeat_nextpm(&impl->prog, nextpm);
emit_split_item(cb, ctx, self, string, &mb, seg_start, ep);
free_matbuf(&mb);
return n;
}
static char *expand_template(const char *repl, Match *m, size_t *outlen) {
size_t cap = 64, len = 0;
char *out = malloc(cap);
size_t rl = strlen(repl);
for (size_t i = 0; i < rl; i++) {
char c = repl[i];
const char *piece = NULL; size_t piecelen = 0;
int consumed_one_char = 1;
if (c == '\\' && i + 1 < rl) {
char nc = repl[i + 1];
if (nc == 'g' && i + 2 < rl && repl[i + 2] == '<') {
size_t j = i + 3, start = j;
while (j < rl && repl[j] != '>') j++;
char *name = xstrndup(repl + start, j - start);
int idx = -1;
if (strspn(name, "0123456789") == strlen(name) && name[0]) idx = atoi(name);
else idx = groupindex_lookup((GroupIndex *)m->re->groupindex, name);
free(name);
Match_group(m, NULL, idx, &piece, &piecelen);
i = j;
} else if (isdigit((unsigned char)nc)) {
size_t j = i + 1, start = j;
while (j < rl && isdigit((unsigned char)repl[j]) && j - start < 2) j++;
char *ns = xstrndup(repl + start, j - start);
int idx = atoi(ns); free(ns);
Match_group(m, NULL, idx, &piece, &piecelen);
i = j - 1;
} else if (nc == 'n') { char cc = '\n'; if (len+1>=cap){cap*=2;out=realloc(out,cap);} out[len++]=cc; i++; continue; }
else if (nc == 't') { char cc = '\t'; if (len+1>=cap){cap*=2;out=realloc(out,cap);} out[len++]=cc; i++; continue; }
else if (nc == '\\') { char cc = '\\'; if (len+1>=cap){cap*=2;out=realloc(out,cap);} out[len++]=cc; i++; continue; }
else { piece = repl + i + 1; piecelen = 1; i++; }
} else {
if (len + 1 >= cap) { cap *= 2; out = realloc(out, cap); }
out[len++] = c;
consumed_one_char = 1; (void)consumed_one_char;
continue;
}
if (piece && piecelen) {
while (len + piecelen >= cap) { cap *= 2; out = realloc(out, cap); }
memcpy(out + len, piece, piecelen);
len += piecelen;
}
}
out[len] = 0;
*outlen = len;
return out;
}
typedef struct { char *buf; size_t len, cap; int64_t last_end; Input *string; const char *repl; MatchSubCb cb; void *cbctx; int n; int count_limit; } SubCtx;
static void subctx_append(SubCtx *sc, const void *data, size_t len) {
while (sc->len + len + 1 > sc->cap) { sc->cap = sc->cap ? sc->cap * 2 : 256; sc->buf = realloc(sc->buf, sc->cap); }
memcpy(sc->buf + sc->len, data, len);
sc->len += len;
}
static void sub_cb(void *ctx, const Match *m_const) {
SubCtx *sc = ctx;
Match *m = (Match *)m_const;
if (sc->count_limit > 0 && sc->n >= sc->count_limit) return;
int64_t mstart_byte = Match_start_byte(m, 0);
subctx_append(sc, sc->string->buf + sc->last_end, (size_t)(mstart_byte - sc->last_end));
if (sc->cb) {
char *piece = NULL; size_t plen = 0;
sc->cb(sc->cbctx, m, &piece, &plen);
if (piece) { subctx_append(sc, piece, plen); free(piece); }
} else {
size_t plen; char *piece = expand_template(sc->repl, m, &plen);
subctx_append(sc, piece, plen);
free(piece);
}
sc->last_end = Match_end_byte(m, 0);
sc->n++;
}
int Pattern_subn(Pattern *self, Input *string, const char *repl, MatchSubCb cb, void *ctx, int count, char **out, size_t *outlen, int *n) {
SubCtx sc; memset(&sc, 0, sizeof sc);
sc.string = string; sc.repl = repl; sc.cb = cb; sc.cbctx = ctx; sc.count_limit = count; sc.last_end = 0;
int total = Pattern_finditer(self, string, 0, -1, sub_cb, &sc);
if (total < 0) { free(sc.buf); return -1; }
subctx_append(&sc, string->buf + sc.last_end, (size_t)((int64_t)string->len - sc.last_end));
if (!sc.buf) sc.buf = malloc(1);
sc.buf[sc.len] = 0;
*out = sc.buf; *outlen = sc.len;
if (n) *n = sc.n;
return 1;
}
int Pattern_sub(Pattern *self, Input *string, const char *repl, MatchSubCb cb, void *ctx, int count, char **out, size_t *outlen) {
int n;
return Pattern_subn(self, string, repl, cb, ctx, count, out, outlen, &n);
}
/* ================================================================
* 12. Match accessors
* ================================================================ */
int Pattern_groupindex_lookup(Pattern *self, const char *name) {
PatternImpl *impl = (PatternImpl *)self->program;
return groupindex_lookup(&impl->gidx, name);
}
int Match_group(Match *self, const char *name_or_null, int index, const char **out, size_t *outlen) {
MatchSlots *ms = self->slots;
int g = index;
if (name_or_null) {
PatternImpl *impl = (PatternImpl *)self->re->program;
g = groupindex_lookup(&impl->gidx, name_or_null);
if (g < 0) return -1;
}
if (g < 0 || g > ms->ngroups) return -1;
int64_t s = ms->pos[2 * g], e = ms->pos[2 * g + 1];
if (s < 0 || e < 0) { *out = NULL; *outlen = 0; return 0; }
int64_t bs = ms->cp_to_byte ? ms->cp_to_byte[s] : s;
int64_t be = ms->cp_to_byte ? ms->cp_to_byte[e] : e;
*out = (const char *)(self->string->buf + bs);
*outlen = (size_t)(be - bs);
return 1;
}
int64_t Match_start(Match *self, int group) {
MatchSlots *ms = self->slots;
if (group < 0 || group > ms->ngroups) return -1;
return ms->pos[2 * group];
}
int64_t Match_end(Match *self, int group) {
MatchSlots *ms = self->slots;
if (group < 0 || group > ms->ngroups) return -1;
return ms->pos[2 * group + 1];
}
void Match_span(Match *self, int group, int64_t *start, int64_t *end) {
*start = Match_start(self, group); *end = Match_end(self, group);
}
int64_t Match_start_byte(Match *self, int group) {
MatchSlots *ms = self->slots;
if (group < 0 || group > ms->ngroups) return -1;
int64_t v = ms->pos[2 * group];
if (v < 0) return -1;
return ms->cp_to_byte ? ms->cp_to_byte[v] : v;
}
int64_t Match_end_byte(Match *self, int group) {
MatchSlots *ms = self->slots;
if (group < 0 || group > ms->ngroups) return -1;
int64_t v = ms->pos[2 * group + 1];
if (v < 0) return -1;
return ms->cp_to_byte ? ms->cp_to_byte[v] : v;
}
void Match_span_byte(Match *self, int group, int64_t *start, int64_t *end) {
*start = Match_start_byte(self, group); *end = Match_end_byte(self, group);
}
void Match_free(Match *self) {
if (!self || !self->slots) return;
MatchSlots *ms = self->slots;
free(ms->pos); free(ms->cp_to_byte); free(ms);
self->slots = NULL;
}
/* ================================================================
* 13. Module level functions, cache, escape (Section 9.1)
* ================================================================ */
#define CACHE_CAP 512
typedef struct { char *key; int keylen; int flags; Pattern *pat; } CacheEnt;
static CacheEnt g_cache[CACHE_CAP];
static int g_cache_n = 0;
void re_purge(void) {
for (int i = 0; i < g_cache_n; i++) { free(g_cache[i].key); Pattern_free(g_cache[i].pat); }
g_cache_n = 0;
}
static Pattern *cache_compile(const char *pattern, size_t len, int flags, PatternError *err) {
for (int i = 0; i < g_cache_n; i++)
if (g_cache[i].flags == flags && (size_t)g_cache[i].keylen == len && memcmp(g_cache[i].key, pattern, len) == 0)
return g_cache[i].pat;
Pattern *pat = re_compile(pattern, len, flags, err);
if (!pat) return NULL;
if (g_cache_n >= CACHE_CAP) re_purge();
g_cache[g_cache_n].key = xstrndup(pattern, len);
g_cache[g_cache_n].keylen = (int)len;
g_cache[g_cache_n].flags = flags;
g_cache[g_cache_n].pat = pat;
g_cache_n++;
return pat;
}
int re_match(const char *pattern, size_t len, int flags, Input *string, Match *out) {
Pattern *pat = cache_compile(pattern, len, flags, NULL); if (!pat) return -1;
return Pattern_match(pat, string, 0, -1, out);
}
int re_fullmatch(const char *pattern, size_t len, int flags, Input *string, Match *out) {
Pattern *pat = cache_compile(pattern, len, flags, NULL); if (!pat) return -1;
return Pattern_fullmatch(pat, string, 0, -1, out);
}
int re_search(const char *pattern, size_t len, int flags, Input *string, Match *out) {
Pattern *pat = cache_compile(pattern, len, flags, NULL); if (!pat) return -1;
return Pattern_search(pat, string, 0, -1, out);
}
int re_finditer(const char *pattern, size_t len, int flags, Input *string, MatchIterCb cb, void *ctx) {
Pattern *pat = cache_compile(pattern, len, flags, NULL); if (!pat) return -1;
return Pattern_finditer(pat, string, 0, -1, cb, ctx);
}
int re_findall(const char *pattern, size_t len, int flags, Input *string, MatchIterCb cb, void *ctx) {
return re_finditer(pattern, len, flags, string, cb, ctx);
}
int re_split(const char *pattern, size_t len, int flags, Input *string, int maxsplit, MatchIterCb cb, void *ctx) {
Pattern *pat = cache_compile(pattern, len, flags, NULL); if (!pat) return -1;
return Pattern_split(pat, string, maxsplit, cb, ctx);
}
int re_sub(const char *pattern, size_t len, int flags, Input *string, const char *repl, MatchSubCb cb, void *ctx, int count, char **out, size_t *outlen) {
Pattern *pat = cache_compile(pattern, len, flags, NULL); if (!pat) return -1;
return Pattern_sub(pat, string, repl, cb, ctx, count, out, outlen);
}
int re_subn(const char *pattern, size_t len, int flags, Input *string, const char *repl, MatchSubCb cb, void *ctx, int count, char **out, size_t *outlen, int *n) {
Pattern *pat = cache_compile(pattern, len, flags, NULL); if (!pat) return -1;
return Pattern_subn(pat, string, repl, cb, ctx, count, out, outlen, n);
}
void re_escape(const char *in, size_t len, char **out, size_t *outlen) {
char *buf = malloc(len * 2 + 1);
size_t o = 0;
for (size_t i = 0; i < len; i++) {
unsigned char ch = (unsigned char)in[i];
int is_word = (ch >= 'a' && ch <= 'z') || (ch >= 'A' && ch <= 'Z') || (ch >= '0' && ch <= '9') || ch == '_';
if (!is_word && ch < 0x80) buf[o++] = '\\';
buf[o++] = (char)ch;
}
buf[o] = 0;
*out = buf; *outlen = o;
}