Closed two of the three gaps the previous commit's honest self-review left open (the third, full streaming/bounded-memory input, remains out of scope for this pass and is still documented as such). Literal prefilter (regexx.c, pike_find): when the NFA-only Prog's first instruction is a mandatory OP_CHAR or OP_CLASS, a pattern beginning with a required literal or class rather than a nullable loop or a leading assertion, injecting a fresh unanchored start thread at a position that instruction would reject is certain to die on the very next pike_step call regardless; checking that identical condition before injecting rather than after changes nothing about which threads ever exist, only how much wasted work is done finding out. This targeted exactly examples/bench_vs_posix.c's worst regression: the literal-search scenario went from roughly 70x slower than POSIX <regex.h> (up from roughly 22x before the Pike VM existed) down to roughly 11x-13x, better than the original pre-Pike-VM number; number extraction (starts with a class) improved more modestly; a*b (starts with a nullable loop, structurally unhelped) is unchanged, as expected. Verified with the full 3,252-case suite, three clean AddressSanitizer/UndefinedBehaviorSanitizer passes, and a rerun of the whitebox dual-engine cross-check (24,000 match/fullmatch/search plus ~2,700 finditer comparisons between the two engines on the same compiled patterns, zero mismatches). Memory profiling (concept.md 7.9): Valgrind/Massif on the same adversarial, prefilter-proof pattern shape (a*b, nullable leading loop) used for the backtracking engine's own worst case, for a fair comparison. Peak heap was almost entirely the 10MB input buffer itself; the Pike VM's own contribution was roughly 12KB, confirming the design's O(instruction count x group count), input-length- independent memory bound actually holds for the v1 implementation, not only on paper. concept.md Section 10's table, which had only a "not yet profiled" caveat for this row before, is updated with the measured result. README.md, docs/API.md, USAGE.md, and bench_vs_posix.c's own printed summary are updated throughout with the corrected numbers, rather than left describing the pre-prefilter regression as current. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01EjuMk8kY9SDus1wWe2K9xY
2424 lines
106 KiB
C
2424 lines
106 KiB
C
/* regexx.c - a single-file C regex interpreter with Python `re` semantics.
|
|
*
|
|
* See concept.md for the full design rationale. This file is the v1
|
|
* implementation: it covers the public API and pattern syntax listed in
|
|
* README.md "Implementation status" in full, executed by a single
|
|
* recursive backtracking engine (concept.md Section 7.3) over a fully
|
|
* materialized copy of the input. It does not yet implement the
|
|
* streaming, bounded-memory regular engine of concept.md Section 7.2;
|
|
* that remains future work, tracked in README.md.
|
|
*
|
|
* That backtracking engine is memoized (run_memo) and, for quantifiers
|
|
* over a single character/class/./ (OP_REPEAT1), backed by precomputed
|
|
* run-length and skip-ahead tables (compute_maxrun,
|
|
* compute_next_prevmatch), so that Pattern_search/finditer/split/sub
|
|
* run in linear rather than quadratic time for the common,
|
|
* backreference-free case; see README.md "Implementation status" for
|
|
* the measured numbers, the precise remaining limitations, and
|
|
* citations to the published techniques this matches.
|
|
*
|
|
* Naming convention (concept.md Section 9.0): every identifier with a
|
|
* direct Python `re` counterpart uses that counterpart's exact spelling.
|
|
*
|
|
* Not reentrant: re_compile() uses one process-wide parser scratch pool
|
|
* and the module level functions share one process-wide pattern cache
|
|
* (concept.md 13.6); do not call regexx functions from more than one
|
|
* thread without external synchronization.
|
|
*/
|
|
#define _DEFAULT_SOURCE
|
|
#include "regexx.h"
|
|
#include <stdlib.h>
|
|
#include <string.h>
|
|
#include <stdio.h>
|
|
#include <stdarg.h>
|
|
#include <ctype.h>
|
|
#include <wctype.h>
|
|
#include <wchar.h>
|
|
#include <locale.h>
|
|
#include <errno.h>
|
|
#include <sys/mman.h>
|
|
#include <sys/stat.h>
|
|
#include <fcntl.h>
|
|
#include <unistd.h>
|
|
|
|
/* ================================================================
|
|
* 0. Small utilities: dynamic array, allocation helpers
|
|
* ================================================================ */
|
|
|
|
typedef struct { void *data; size_t len, cap, elemsize; } DArr;
|
|
|
|
static void da_init(DArr *a, size_t elemsize) {
|
|
a->data = NULL; a->len = 0; a->cap = 0; a->elemsize = elemsize;
|
|
}
|
|
/* Returns NULL on allocation failure, leaving `a` exactly as it was
|
|
* (never loses or corrupts the existing elements): most callers in
|
|
* this file grow bounded-by-pattern-size structures (the AST pool,
|
|
* class items, the instruction array), where a failure here is
|
|
* already effectively unreachable in practice and is not separately
|
|
* checked; the two callers that grow a structure sized to the
|
|
* *input* (build_matbuf's UTF-8 decode arrays, the only other
|
|
* uses that can plausibly reach real-world allocation limits) do
|
|
* check it. */
|
|
static void *da_push(DArr *a) {
|
|
if (a->len == a->cap) {
|
|
size_t nc = a->cap ? a->cap * 2 : 8;
|
|
void *nd = realloc(a->data, nc * a->elemsize);
|
|
if (!nd) return NULL;
|
|
a->data = nd; a->cap = nc;
|
|
}
|
|
void *p = (char *)a->data + a->len * a->elemsize;
|
|
a->len++;
|
|
return p;
|
|
}
|
|
static void da_free(DArr *a) { free(a->data); a->data = NULL; a->len = a->cap = 0; }
|
|
|
|
static char *xstrndup(const char *s, size_t n) {
|
|
char *r = malloc(n + 1);
|
|
memcpy(r, s, n);
|
|
r[n] = 0;
|
|
return r;
|
|
}
|
|
|
|
static void ensure_unicode_locale(void) {
|
|
static int done = 0;
|
|
if (!done) { setlocale(LC_CTYPE, "C.utf8"); done = 1; }
|
|
}
|
|
|
|
/* ================================================================
|
|
* 1. UTF-8 decoding (concept.md Section 7.2's continuation handling is
|
|
* not needed here since v1 materializes the whole input up front,
|
|
* see README.md "Implementation status")
|
|
* ================================================================ */
|
|
|
|
static int utf8_decode(const uint8_t *buf, size_t len, size_t i, uint32_t *cp) {
|
|
uint8_t b0 = buf[i];
|
|
if (b0 < 0x80) { *cp = b0; return 1; }
|
|
if ((b0 & 0xE0) == 0xC0) {
|
|
if (i + 1 >= len || (buf[i+1] & 0xC0) != 0x80) return 0;
|
|
uint32_t v = (uint32_t)((b0 & 0x1F) << 6) | (buf[i+1] & 0x3F);
|
|
if (v < 0x80) return 0;
|
|
*cp = v; return 2;
|
|
}
|
|
if ((b0 & 0xF0) == 0xE0) {
|
|
if (i + 2 >= len || (buf[i+1] & 0xC0) != 0x80 || (buf[i+2] & 0xC0) != 0x80) return 0;
|
|
uint32_t v = ((uint32_t)(b0 & 0x0F) << 12) | ((uint32_t)(buf[i+1] & 0x3F) << 6) | (buf[i+2] & 0x3F);
|
|
if (v < 0x800 || (v >= 0xD800 && v <= 0xDFFF)) return 0;
|
|
*cp = v; return 3;
|
|
}
|
|
if ((b0 & 0xF8) == 0xF0) {
|
|
if (i + 3 >= len || (buf[i+1] & 0xC0) != 0x80 || (buf[i+2] & 0xC0) != 0x80 || (buf[i+3] & 0xC0) != 0x80) return 0;
|
|
uint32_t v = ((uint32_t)(b0 & 0x07) << 18) | ((uint32_t)(buf[i+1] & 0x3F) << 12) | ((uint32_t)(buf[i+2] & 0x3F) << 6) | (buf[i+3] & 0x3F);
|
|
if (v < 0x10000 || v > 0x10FFFF) return 0;
|
|
*cp = v; return 4;
|
|
}
|
|
return 0;
|
|
}
|
|
|
|
/* ================================================================
|
|
* 2. Character classification predicates (mode aware, Section 2.2/14.4)
|
|
* ================================================================ */
|
|
|
|
#define MODE_BINARY 0
|
|
#define MODE_ASCII 1
|
|
#define MODE_UTF8 2
|
|
|
|
static int is_ascii_word(uint32_t cp) {
|
|
return (cp >= 'a' && cp <= 'z') || (cp >= 'A' && cp <= 'Z') || (cp >= '0' && cp <= '9') || cp == '_';
|
|
}
|
|
static int is_ascii_space(uint32_t cp) {
|
|
return cp == ' ' || cp == '\t' || cp == '\n' || cp == '\r' || cp == '\f' || cp == '\v';
|
|
}
|
|
static int is_ascii_digit(uint32_t cp) { return cp >= '0' && cp <= '9'; }
|
|
|
|
static int cls_is_digit(uint32_t cp, int mode, int flags) {
|
|
if (mode == MODE_UTF8 && !(flags & ASCII)) { ensure_unicode_locale(); return iswdigit((wint_t)cp) ? 1 : 0; }
|
|
return is_ascii_digit(cp);
|
|
}
|
|
static int cls_is_space(uint32_t cp, int mode, int flags) {
|
|
if (mode == MODE_UTF8 && !(flags & ASCII)) { ensure_unicode_locale(); return iswspace((wint_t)cp) ? 1 : 0; }
|
|
return is_ascii_space(cp);
|
|
}
|
|
static int cls_is_word(uint32_t cp, int mode, int flags) {
|
|
if (mode == MODE_UTF8 && !(flags & ASCII)) {
|
|
ensure_unicode_locale();
|
|
return (iswalnum((wint_t)cp) || cp == '_') ? 1 : 0;
|
|
}
|
|
return is_ascii_word(cp); /* LOCALE has no additional effect here, see below */
|
|
/* In BINARY/ASCII mode, LOCALE (Section 2.6) is intentionally a
|
|
* no-op relative to plain ASCII classification: the C standard
|
|
* guarantees the "C" locale's isalnum()/isalpha() classify only
|
|
* the ASCII letters/digits, nothing in 0x80-0xFF, and CPython's own
|
|
* re.LOCALE, checked directly under an explicit "C" locale, agrees
|
|
* (rb'\w+' against b"abc\x80\x90def" still stops at "abc"). An
|
|
* earlier revision of this function instead treated every byte in
|
|
* 0x80-0xFF as a word character under LOCALE, based on an
|
|
* unverified assumption about what "the C locale" does; the large
|
|
* combinatorial test expansion (tests/cases.py Category H) caught
|
|
* the resulting mismatch against a real CPython interpreter, which
|
|
* is what prompted checking the premise directly (see above) and
|
|
* finding it false. README.md "Known deviations" records this. */
|
|
}
|
|
static uint32_t cls_swapcase(uint32_t cp, int mode, int flags) {
|
|
if (mode == MODE_UTF8 && !(flags & ASCII)) {
|
|
ensure_unicode_locale();
|
|
wint_t lo = towlower((wint_t)cp), up = towupper((wint_t)cp);
|
|
if ((uint32_t)lo != cp) return (uint32_t)lo;
|
|
if ((uint32_t)up != cp) return (uint32_t)up;
|
|
return cp;
|
|
}
|
|
if (cp >= 'a' && cp <= 'z') return cp - 32;
|
|
if (cp >= 'A' && cp <= 'Z') return cp + 32;
|
|
return cp;
|
|
}
|
|
static uint32_t cls_fold(uint32_t cp, int mode, int flags) {
|
|
if (mode == MODE_UTF8 && !(flags & ASCII)) { ensure_unicode_locale(); return (uint32_t)towlower((wint_t)cp); }
|
|
if (cp >= 'A' && cp <= 'Z') return cp + 32;
|
|
return cp;
|
|
}
|
|
|
|
/* ================================================================
|
|
* 3. AST
|
|
* ================================================================ */
|
|
|
|
enum {
|
|
N_CHAR, N_ANY, N_CLASS, N_CONCAT, N_ALT, N_REPEAT, N_GROUP,
|
|
N_BACKREF, N_ANCHOR, N_LOOKAROUND, N_ATOMIC, N_EMPTY
|
|
};
|
|
enum { A_BOL, A_EOL, A_BOS, A_EOS, A_WB, A_NWB };
|
|
enum { LK_AHEAD, LK_BEHIND };
|
|
enum { CI_RANGE, CI_D, CI_W, CI_S };
|
|
|
|
typedef struct { int kind; uint32_t lo, hi; int neg; } ClassItem;
|
|
|
|
typedef struct Node {
|
|
int kind;
|
|
struct Node *a, *b;
|
|
int32_t lo, hi;
|
|
int greedy;
|
|
uint32_t ch;
|
|
DArr items;
|
|
int class_negate;
|
|
int group_index;
|
|
char *group_name;
|
|
int backref_index;
|
|
char *backref_name;
|
|
int anchor_kind;
|
|
int lookaround_kind;
|
|
int lookaround_negate;
|
|
int width;
|
|
} Node;
|
|
|
|
typedef struct { Node **data; size_t len, cap; } NodePool;
|
|
static NodePool g_pool;
|
|
|
|
static Node *node_new(int kind) {
|
|
if (g_pool.len == g_pool.cap) {
|
|
g_pool.cap = g_pool.cap ? g_pool.cap * 2 : 64;
|
|
g_pool.data = realloc(g_pool.data, g_pool.cap * sizeof(Node *));
|
|
}
|
|
Node *n = calloc(1, sizeof(Node));
|
|
n->kind = kind; n->group_index = -1; n->backref_index = -1; n->greedy = 1; n->width = -1;
|
|
da_init(&n->items, sizeof(ClassItem));
|
|
g_pool.data[g_pool.len++] = n;
|
|
return n;
|
|
}
|
|
static void nodepool_free_all(NodePool *pool) {
|
|
for (size_t i = 0; i < pool->len; i++) {
|
|
Node *n = pool->data[i];
|
|
da_free(&n->items);
|
|
free(n->group_name);
|
|
free(n->backref_name);
|
|
free(n);
|
|
}
|
|
free(pool->data);
|
|
pool->data = NULL; pool->len = pool->cap = 0;
|
|
}
|
|
|
|
static int node_nullable(Node *n) {
|
|
if (!n) return 1;
|
|
switch (n->kind) {
|
|
case N_CHAR: case N_ANY: case N_CLASS: case N_BACKREF: return 0;
|
|
case N_CONCAT: return node_nullable(n->a) && node_nullable(n->b);
|
|
case N_ALT: return node_nullable(n->a) || node_nullable(n->b);
|
|
case N_REPEAT: return n->lo == 0 || node_nullable(n->a);
|
|
case N_GROUP: return node_nullable(n->a);
|
|
case N_ANCHOR: case N_LOOKAROUND: return 1;
|
|
case N_ATOMIC: return node_nullable(n->a);
|
|
default: return 1;
|
|
}
|
|
}
|
|
static int node_width(Node *n) {
|
|
if (!n) return 0;
|
|
switch (n->kind) {
|
|
case N_CHAR: case N_ANY: case N_CLASS: return 1;
|
|
case N_CONCAT: { int wa = node_width(n->a), wb = node_width(n->b); return (wa < 0 || wb < 0) ? -1 : wa + wb; }
|
|
case N_ALT: { int wa = node_width(n->a), wb = node_width(n->b); return (wa < 0 || wb < 0 || wa != wb) ? -1 : wa; }
|
|
case N_REPEAT: { if (n->lo != n->hi) return -1; int w = node_width(n->a); return w < 0 ? -1 : w * n->lo; }
|
|
case N_GROUP: return node_width(n->a);
|
|
case N_BACKREF: return -1;
|
|
case N_ANCHOR: case N_LOOKAROUND: return 0;
|
|
case N_ATOMIC: return node_width(n->a);
|
|
default: return 0;
|
|
}
|
|
}
|
|
static int is_simple_atom(Node *n) { return n->kind == N_CHAR || n->kind == N_ANY || n->kind == N_CLASS; }
|
|
|
|
/* ================================================================
|
|
* 4. Parser
|
|
* ================================================================ */
|
|
|
|
typedef struct { char *name; int index; } GroupName;
|
|
|
|
typedef struct {
|
|
const char *s;
|
|
size_t len, i;
|
|
int flags;
|
|
int mode;
|
|
int ngroups;
|
|
DArr backrefs; /* Node* */
|
|
DArr groupnames; /* GroupName */
|
|
char errmsg[256];
|
|
int64_t errpos;
|
|
int failed;
|
|
} Parser;
|
|
|
|
static void perr(Parser *p, int64_t pos, const char *fmt, ...) {
|
|
if (p->failed) return;
|
|
p->failed = 1; p->errpos = pos;
|
|
va_list ap; va_start(ap, fmt);
|
|
vsnprintf(p->errmsg, sizeof p->errmsg, fmt, ap);
|
|
va_end(ap);
|
|
}
|
|
static int peek(Parser *p) { return p->i < p->len ? (unsigned char)p->s[p->i] : -1; }
|
|
static int peek2(Parser *p) { return p->i + 1 < p->len ? (unsigned char)p->s[p->i + 1] : -1; }
|
|
static int at_end(Parser *p) { return p->i >= p->len; }
|
|
static int hexval(int c) {
|
|
if (c >= '0' && c <= '9') return c - '0';
|
|
if (c >= 'a' && c <= 'f') return c - 'a' + 10;
|
|
if (c >= 'A' && c <= 'F') return c - 'A' + 10;
|
|
return -1;
|
|
}
|
|
|
|
static Node *parse_alt(Parser *p);
|
|
|
|
static Node *parse_escape(Parser *p, int in_class, uint32_t *literal_cp, int *is_literal) {
|
|
*is_literal = 0;
|
|
if (at_end(p)) { perr(p, (int64_t)p->i, "bad escape (end of pattern)"); return NULL; }
|
|
int c = peek(p); p->i++;
|
|
switch (c) {
|
|
case 'n': *literal_cp = '\n'; *is_literal = 1; return NULL;
|
|
case 'r': *literal_cp = '\r'; *is_literal = 1; return NULL;
|
|
case 't': *literal_cp = '\t'; *is_literal = 1; return NULL;
|
|
case 'f': *literal_cp = '\f'; *is_literal = 1; return NULL;
|
|
case 'v': *literal_cp = '\v'; *is_literal = 1; return NULL;
|
|
case 'a': *literal_cp = '\a'; *is_literal = 1; return NULL;
|
|
case '0': {
|
|
uint32_t v = 0; int n = 0;
|
|
while (n < 2 && peek(p) >= '0' && peek(p) <= '7') { v = v * 8 + (uint32_t)(peek(p) - '0'); p->i++; n++; }
|
|
*literal_cp = v; *is_literal = 1; return NULL;
|
|
}
|
|
case 'x': {
|
|
int h1 = hexval(peek(p));
|
|
if (h1 < 0) { perr(p, (int64_t)p->i, "incomplete escape \\x"); return NULL; }
|
|
p->i++;
|
|
int h2 = hexval(peek(p));
|
|
if (h2 < 0) { perr(p, (int64_t)p->i, "incomplete escape \\x"); return NULL; }
|
|
p->i++;
|
|
*literal_cp = (uint32_t)(h1 * 16 + h2); *is_literal = 1; return NULL;
|
|
}
|
|
case 'u': case 'U': {
|
|
int ndig = (c == 'u') ? 4 : 8;
|
|
uint32_t v = 0;
|
|
for (int k = 0; k < ndig; k++) {
|
|
int h = hexval(peek(p));
|
|
if (h < 0) { perr(p, (int64_t)p->i, "incomplete escape \\%c", c); return NULL; }
|
|
v = v * 16 + (uint32_t)h; p->i++;
|
|
}
|
|
*literal_cp = v; *is_literal = 1; return NULL;
|
|
}
|
|
case 'N':
|
|
perr(p, (int64_t)p->i, "\\N{...} named code points are not implemented in this build");
|
|
return NULL;
|
|
case 'd': case 'D': case 'w': case 'W': case 's': case 'S': {
|
|
Node *n = node_new(N_CLASS);
|
|
ClassItem *it = da_push(&n->items);
|
|
it->kind = (c == 'd' || c == 'D') ? CI_D : (c == 'w' || c == 'W') ? CI_W : CI_S;
|
|
it->neg = (c == 'D' || c == 'W' || c == 'S') ? 1 : 0;
|
|
it->lo = it->hi = 0;
|
|
return n;
|
|
}
|
|
case 'b':
|
|
if (in_class) { *literal_cp = '\b'; *is_literal = 1; return NULL; }
|
|
{ Node *n = node_new(N_ANCHOR); n->anchor_kind = A_WB; return n; }
|
|
case 'B': { Node *n = node_new(N_ANCHOR); n->anchor_kind = A_NWB; return n; }
|
|
case 'A': { Node *n = node_new(N_ANCHOR); n->anchor_kind = A_BOS; return n; }
|
|
case 'Z': { Node *n = node_new(N_ANCHOR); n->anchor_kind = A_EOS; return n; }
|
|
case 'g': {
|
|
if (in_class) { *literal_cp = 'g'; *is_literal = 1; return NULL; }
|
|
if (peek(p) != '<') { perr(p, (int64_t)p->i, "missing < in \\g<...>"); return NULL; }
|
|
p->i++;
|
|
size_t start = p->i;
|
|
while (!at_end(p) && peek(p) != '>') p->i++;
|
|
if (at_end(p)) { perr(p, (int64_t)p->i, "missing > in \\g<...>"); return NULL; }
|
|
char *ref = xstrndup(p->s + start, p->i - start);
|
|
p->i++;
|
|
Node *n = node_new(N_BACKREF);
|
|
size_t rl = strlen(ref);
|
|
int all_digit = rl > 0;
|
|
for (size_t k = 0; k < rl; k++) if (!isdigit((unsigned char)ref[k])) all_digit = 0;
|
|
if (all_digit) { n->backref_index = atoi(ref); free(ref); }
|
|
else n->backref_name = ref;
|
|
Node **slot = da_push(&p->backrefs); *slot = n;
|
|
return n;
|
|
}
|
|
case '1': case '2': case '3': case '4': case '5': case '6': case '7': case '8': case '9': {
|
|
if (in_class) { *literal_cp = (uint32_t)c; *is_literal = 1; return NULL; }
|
|
uint32_t v = (uint32_t)(c - '0');
|
|
while (peek(p) >= '0' && peek(p) <= '9') { v = v * 10 + (uint32_t)(peek(p) - '0'); p->i++; }
|
|
Node *n = node_new(N_BACKREF);
|
|
n->backref_index = (int)v;
|
|
Node **slot = da_push(&p->backrefs); *slot = n;
|
|
return n;
|
|
}
|
|
default:
|
|
*literal_cp = (uint32_t)c; *is_literal = 1; return NULL;
|
|
}
|
|
}
|
|
|
|
static uint32_t read_raw_char(Parser *p) {
|
|
if (p->mode == MODE_UTF8) {
|
|
uint32_t cp;
|
|
int n = utf8_decode((const uint8_t *)p->s, p->len, p->i, &cp);
|
|
if (n == 0) { perr(p, (int64_t)p->i, "invalid UTF-8 in pattern"); p->i++; return 0; }
|
|
p->i += (size_t)n;
|
|
return cp;
|
|
}
|
|
return (unsigned char)p->s[p->i++];
|
|
}
|
|
|
|
static Node *parse_class(Parser *p) {
|
|
Node *n = node_new(N_CLASS);
|
|
p->i++;
|
|
if (peek(p) == '^') { n->class_negate = 1; p->i++; }
|
|
int first = 1;
|
|
while (1) {
|
|
if (at_end(p)) { perr(p, (int64_t)p->i, "unterminated character set"); return n; }
|
|
if (peek(p) == ']' && !first) { p->i++; break; }
|
|
first = 0;
|
|
uint32_t lo;
|
|
if (peek(p) == '\\') {
|
|
p->i++;
|
|
uint32_t litcp; int is_lit;
|
|
Node *sub = parse_escape(p, 1, &litcp, &is_lit);
|
|
if (p->failed) return n;
|
|
if (!is_lit && sub) {
|
|
ClassItem *it = da_push(&n->items);
|
|
*it = ((ClassItem *)sub->items.data)[0];
|
|
continue;
|
|
}
|
|
lo = litcp;
|
|
} else if (peek(p) == '[' && peek2(p) == ':') {
|
|
perr(p, (int64_t)p->i, "POSIX named classes ([:alpha:] etc) are not part of Python re, see concept.md 14.3");
|
|
return n;
|
|
} else {
|
|
lo = read_raw_char(p);
|
|
if (p->failed) return n;
|
|
}
|
|
uint32_t hi = lo;
|
|
if (peek(p) == '-' && peek2(p) != ']' && peek2(p) != -1) {
|
|
p->i++;
|
|
if (peek(p) == '\\') {
|
|
p->i++;
|
|
uint32_t litcp; int is_lit;
|
|
parse_escape(p, 1, &litcp, &is_lit);
|
|
if (p->failed) return n;
|
|
if (!is_lit) { perr(p, (int64_t)p->i, "bad character range"); return n; }
|
|
hi = litcp;
|
|
} else {
|
|
hi = read_raw_char(p);
|
|
if (p->failed) return n;
|
|
}
|
|
if (hi < lo) { perr(p, (int64_t)p->i, "bad character range"); return n; }
|
|
}
|
|
ClassItem *it = da_push(&n->items);
|
|
it->kind = CI_RANGE; it->lo = lo; it->hi = hi; it->neg = 0;
|
|
}
|
|
return n;
|
|
}
|
|
|
|
static Node *make_concat(Node *a, Node *b) {
|
|
if (!a) return b;
|
|
if (!b) return a;
|
|
Node *n = node_new(N_CONCAT);
|
|
n->a = a; n->b = b;
|
|
return n;
|
|
}
|
|
|
|
static Node *parse_repeat_suffix(Parser *p, Node *atom) {
|
|
int lo = -1, hi = -1;
|
|
if (peek(p) == '*') { lo = 0; hi = -1; p->i++; }
|
|
else if (peek(p) == '+') { lo = 1; hi = -1; p->i++; }
|
|
else if (peek(p) == '?') { lo = 0; hi = 1; p->i++; }
|
|
else if (peek(p) == '{') {
|
|
size_t save = p->i;
|
|
p->i++;
|
|
int have_lo = 0, have_hi = 0; long vlo = 0, vhi = 0;
|
|
while (isdigit(peek(p))) { vlo = vlo * 10 + (peek(p) - '0'); p->i++; have_lo = 1; }
|
|
int has_comma = 0;
|
|
if (peek(p) == ',') {
|
|
has_comma = 1; p->i++;
|
|
while (isdigit(peek(p))) { vhi = vhi * 10 + (peek(p) - '0'); p->i++; have_hi = 1; }
|
|
}
|
|
if (peek(p) != '}' || (!have_lo && !has_comma)) { p->i = save; return atom; }
|
|
p->i++;
|
|
lo = have_lo ? (int)vlo : 0;
|
|
hi = has_comma ? (have_hi ? (int)vhi : -1) : lo;
|
|
if (hi != -1 && hi < lo) { perr(p, (int64_t)save, "min repeat greater than max repeat"); return atom; }
|
|
} else {
|
|
return atom;
|
|
}
|
|
Node *rep = node_new(N_REPEAT);
|
|
rep->a = atom; rep->lo = lo; rep->hi = hi; rep->greedy = 1;
|
|
if (peek(p) == '?') { rep->greedy = 0; p->i++; return rep; }
|
|
if (peek(p) == '+') {
|
|
p->i++;
|
|
Node *at = node_new(N_ATOMIC);
|
|
at->a = rep;
|
|
return at;
|
|
}
|
|
return rep;
|
|
}
|
|
|
|
static Node *parse_group_or_special(Parser *p) {
|
|
size_t open_pos = p->i;
|
|
p->i++;
|
|
if (peek(p) == '?') {
|
|
p->i++;
|
|
int c = peek(p);
|
|
if (c == ':') {
|
|
p->i++;
|
|
Node *body = parse_alt(p);
|
|
if (peek(p) != ')') { perr(p, (int64_t)p->i, "missing ), unterminated subpattern"); return body; }
|
|
p->i++;
|
|
Node *g = node_new(N_GROUP); g->a = body; g->group_index = -1;
|
|
return g;
|
|
}
|
|
if (c == '#') {
|
|
p->i++;
|
|
while (!at_end(p) && peek(p) != ')') p->i++;
|
|
if (at_end(p)) { perr(p, (int64_t)p->i, "missing ), unterminated comment"); return node_new(N_EMPTY); }
|
|
p->i++;
|
|
return node_new(N_EMPTY);
|
|
}
|
|
if (c == '=' || c == '!') {
|
|
p->i++;
|
|
Node *body = parse_alt(p);
|
|
if (peek(p) != ')') { perr(p, (int64_t)p->i, "missing ), unterminated subpattern"); return body; }
|
|
p->i++;
|
|
Node *n = node_new(N_LOOKAROUND);
|
|
n->a = body; n->lookaround_kind = LK_AHEAD; n->lookaround_negate = (c == '!');
|
|
return n;
|
|
}
|
|
if (c == '<' && (peek2(p) == '=' || peek2(p) == '!')) {
|
|
int neg = (peek2(p) == '!');
|
|
p->i += 2;
|
|
Node *body = parse_alt(p);
|
|
if (peek(p) != ')') { perr(p, (int64_t)p->i, "missing ), unterminated subpattern"); return body; }
|
|
p->i++;
|
|
int w = node_width(body);
|
|
if (w < 0) { perr(p, (int64_t)open_pos, "look-behind requires fixed-width pattern"); return node_new(N_EMPTY); }
|
|
Node *n = node_new(N_LOOKAROUND);
|
|
n->a = body; n->lookaround_kind = LK_BEHIND; n->lookaround_negate = neg; n->width = w;
|
|
return n;
|
|
}
|
|
if (c == '>') {
|
|
p->i++;
|
|
Node *body = parse_alt(p);
|
|
if (peek(p) != ')') { perr(p, (int64_t)p->i, "missing ), unterminated subpattern"); return body; }
|
|
p->i++;
|
|
Node *n = node_new(N_ATOMIC); n->a = body;
|
|
return n;
|
|
}
|
|
if (c == 'P') {
|
|
p->i++;
|
|
if (peek(p) == '<') {
|
|
p->i++;
|
|
size_t start = p->i;
|
|
while (!at_end(p) && peek(p) != '>') p->i++;
|
|
if (at_end(p)) { perr(p, (int64_t)p->i, "missing >, unterminated name"); return node_new(N_EMPTY); }
|
|
char *name = xstrndup(p->s + start, p->i - start);
|
|
p->i++;
|
|
int idx = ++p->ngroups;
|
|
Node *body = parse_alt(p);
|
|
if (peek(p) != ')') { perr(p, (int64_t)p->i, "missing ), unterminated subpattern"); free(name); return body; }
|
|
p->i++;
|
|
Node *g = node_new(N_GROUP); g->a = body; g->group_index = idx; g->group_name = name;
|
|
GroupName *gn = da_push(&p->groupnames);
|
|
gn->name = xstrndup(name, strlen(name)); gn->index = idx;
|
|
return g;
|
|
}
|
|
if (peek(p) == '=') {
|
|
p->i++;
|
|
size_t start = p->i;
|
|
while (!at_end(p) && peek(p) != ')') p->i++;
|
|
if (at_end(p)) { perr(p, (int64_t)p->i, "missing ), unterminated name"); return node_new(N_EMPTY); }
|
|
char *name = xstrndup(p->s + start, p->i - start);
|
|
p->i++;
|
|
Node *n = node_new(N_BACKREF); n->backref_name = name;
|
|
Node **slot = da_push(&p->backrefs); *slot = n;
|
|
return n;
|
|
}
|
|
perr(p, (int64_t)p->i, "unknown extension ?P");
|
|
return node_new(N_EMPTY);
|
|
}
|
|
if (c == '(') {
|
|
perr(p, (int64_t)open_pos, "conditional groups (?(id)yes|no) are not implemented in this build");
|
|
return node_new(N_EMPTY);
|
|
}
|
|
{
|
|
size_t save = p->i;
|
|
int newflags = 0, any = 0;
|
|
while (!at_end(p) && strchr("aiLmsux", peek(p))) {
|
|
switch (peek(p)) {
|
|
case 'i': newflags |= IGNORECASE; break;
|
|
case 'm': newflags |= MULTILINE; break;
|
|
case 's': newflags |= DOTALL; break;
|
|
case 'x': newflags |= VERBOSE; break;
|
|
case 'a': newflags |= ASCII; break;
|
|
case 'L': newflags |= LOCALE; break;
|
|
default: break;
|
|
}
|
|
p->i++; any = 1;
|
|
}
|
|
if (any && peek(p) == ')') {
|
|
p->i++;
|
|
if (open_pos != 0) { perr(p, (int64_t)open_pos, "global flags not at the start of the expression"); return node_new(N_EMPTY); }
|
|
p->flags |= newflags;
|
|
return node_new(N_EMPTY);
|
|
}
|
|
p->i = save;
|
|
perr(p, (int64_t)open_pos, "scoped inline flags (?flags:...) are not implemented in this build");
|
|
return node_new(N_EMPTY);
|
|
}
|
|
}
|
|
int idx = ++p->ngroups;
|
|
Node *body = parse_alt(p);
|
|
if (peek(p) != ')') { perr(p, (int64_t)p->i, "missing ), unterminated subpattern"); return body; }
|
|
p->i++;
|
|
Node *g = node_new(N_GROUP); g->a = body; g->group_index = idx;
|
|
return g;
|
|
}
|
|
|
|
static Node *parse_atom(Parser *p) {
|
|
int c = peek(p);
|
|
if (c == '(') return parse_group_or_special(p);
|
|
if (c == '[') return parse_class(p);
|
|
if (c == '.') { p->i++; return node_new(N_ANY); }
|
|
if (c == '^') { p->i++; Node *n = node_new(N_ANCHOR); n->anchor_kind = A_BOL; return n; }
|
|
if (c == '$') { p->i++; Node *n = node_new(N_ANCHOR); n->anchor_kind = A_EOL; return n; }
|
|
if (c == '\\') {
|
|
p->i++;
|
|
uint32_t litcp; int is_lit;
|
|
Node *n = parse_escape(p, 0, &litcp, &is_lit);
|
|
if (p->failed) return node_new(N_EMPTY);
|
|
if (is_lit) { Node *cn = node_new(N_CHAR); cn->ch = litcp; return cn; }
|
|
return n;
|
|
}
|
|
if (c == '*' || c == '+' || c == '?') {
|
|
perr(p, (int64_t)p->i, "nothing to repeat");
|
|
p->i++;
|
|
return node_new(N_EMPTY);
|
|
}
|
|
uint32_t cp = read_raw_char(p);
|
|
if (p->failed) return node_new(N_EMPTY);
|
|
Node *cn = node_new(N_CHAR); cn->ch = cp;
|
|
return cn;
|
|
}
|
|
|
|
static Node *parse_concat(Parser *p) {
|
|
Node *result = NULL;
|
|
while (!at_end(p) && peek(p) != '|' && peek(p) != ')' && !p->failed) {
|
|
Node *atom = parse_atom(p);
|
|
if (p->failed) break;
|
|
atom = parse_repeat_suffix(p, atom);
|
|
result = make_concat(result, atom);
|
|
}
|
|
if (!result) result = node_new(N_EMPTY);
|
|
return result;
|
|
}
|
|
|
|
static Node *parse_alt(Parser *p) {
|
|
Node *first = parse_concat(p);
|
|
if (peek(p) == '|') {
|
|
p->i++;
|
|
Node *rest = parse_alt(p);
|
|
Node *n = node_new(N_ALT);
|
|
n->a = first; n->b = rest;
|
|
return n;
|
|
}
|
|
return first;
|
|
}
|
|
|
|
/* Strip VERBOSE-mode whitespace and #-comments, respecting [...] classes
|
|
* and backslash escapes. Only called once the VERBOSE flag is known. */
|
|
static char *strip_verbose(const char *s, size_t len, size_t *outlen) {
|
|
char *out = malloc(len + 1);
|
|
size_t o = 0, i = 0;
|
|
int in_class = 0;
|
|
while (i < len) {
|
|
char c = s[i];
|
|
if (c == '\\' && i + 1 < len) { out[o++] = s[i++]; out[o++] = s[i++]; continue; }
|
|
if (c == '[' && !in_class) { in_class = 1; out[o++] = s[i++]; continue; }
|
|
if (c == ']' && in_class) { in_class = 0; out[o++] = s[i++]; continue; }
|
|
if (!in_class && isspace((unsigned char)c)) { i++; continue; }
|
|
if (!in_class && c == '#') { while (i < len && s[i] != '\n') i++; continue; }
|
|
out[o++] = s[i++];
|
|
}
|
|
out[o] = 0;
|
|
*outlen = o;
|
|
return out;
|
|
}
|
|
|
|
/* ================================================================
|
|
* 5. Bytecode
|
|
* ================================================================ */
|
|
|
|
enum {
|
|
OP_CHAR, OP_CLASS, OP_ANY, OP_SPLIT, OP_JMP, OP_SAVE, OP_MATCH,
|
|
OP_ASSERT, OP_BACKREF, OP_LOOKAHEAD, OP_LOOKBEHIND, OP_ATOMIC,
|
|
OP_RETURN, OP_REPEAT1
|
|
};
|
|
|
|
typedef struct {
|
|
int op;
|
|
int32_t x, y;
|
|
uint32_t data;
|
|
int width;
|
|
int neg;
|
|
int is_loop;
|
|
int loop_close;
|
|
int32_t lo, hi;
|
|
int greedy;
|
|
int atomkind; /* OP_REPEAT1: 0=CHAR 1=CLASS 2=ANY */
|
|
} Inst;
|
|
|
|
typedef struct {
|
|
Inst *insts;
|
|
int n, cap;
|
|
DArr classnodes; /* Node* */
|
|
int ngroups;
|
|
int mode;
|
|
int flags;
|
|
int has_backref; /* computed once after compiling; gates memoized search, see run_memo */
|
|
int no_repeat1; /* compile time only: when set, compile_repeat never emits the
|
|
* OP_REPEAT1 fast path (Section 6), even for a simple-atom
|
|
* repeat, so the result contains only instructions the Pike
|
|
* VM understands (concept.md 7.7). Never set on the ordinary
|
|
* backtracking Prog every Pattern already compiles; set only
|
|
* on the second, NFA-only Prog compiled from the same AST for
|
|
* an eligible pattern, see PatternImpl.nfa_prog. */
|
|
} Prog;
|
|
|
|
static int emit(Prog *pr, int op) {
|
|
if (pr->n == pr->cap) {
|
|
pr->cap = pr->cap ? pr->cap * 2 : 64;
|
|
pr->insts = realloc(pr->insts, (size_t)pr->cap * sizeof(Inst));
|
|
}
|
|
Inst *i = &pr->insts[pr->n];
|
|
memset(i, 0, sizeof(*i));
|
|
i->op = op; i->x = i->y = -1;
|
|
return pr->n++;
|
|
}
|
|
|
|
static void compile_node(Prog *pr, Node *n);
|
|
|
|
static void compile_repeat(Prog *pr, Node *n) {
|
|
if (!pr->no_repeat1 && is_simple_atom(n->a)) {
|
|
int pc = emit(pr, OP_REPEAT1);
|
|
Inst *in = &pr->insts[pc];
|
|
in->lo = n->lo; in->hi = n->hi; in->greedy = n->greedy;
|
|
if (n->a->kind == N_CHAR) { in->atomkind = 0; in->data = n->a->ch; }
|
|
else if (n->a->kind == N_ANY) { in->atomkind = 2; }
|
|
else {
|
|
in->atomkind = 1;
|
|
int idx = (int)pr->classnodes.len;
|
|
Node **slot = da_push(&pr->classnodes); *slot = n->a;
|
|
in->data = (uint32_t)idx;
|
|
}
|
|
return;
|
|
}
|
|
int lo = n->lo, hi = n->hi;
|
|
for (int k = 0; k < lo; k++) compile_node(pr, n->a);
|
|
if (hi == -1) {
|
|
int nullable = node_nullable(n->a);
|
|
int split_pc = emit(pr, OP_SPLIT);
|
|
pr->insts[split_pc].is_loop = nullable;
|
|
int body_start = pr->n;
|
|
compile_node(pr, n->a);
|
|
int jmp_pc = emit(pr, OP_JMP);
|
|
pr->insts[jmp_pc].x = split_pc;
|
|
pr->insts[jmp_pc].loop_close = nullable;
|
|
int end = pr->n;
|
|
if (n->greedy) { pr->insts[split_pc].x = body_start; pr->insts[split_pc].y = end; }
|
|
else { pr->insts[split_pc].x = end; pr->insts[split_pc].y = body_start; }
|
|
} else if (hi > lo) {
|
|
DArr splits; da_init(&splits, sizeof(int));
|
|
for (int k = 0; k < hi - lo; k++) {
|
|
int sp = emit(pr, OP_SPLIT);
|
|
int *slot = da_push(&splits); *slot = sp;
|
|
int body_start = pr->n;
|
|
compile_node(pr, n->a);
|
|
if (n->greedy) pr->insts[sp].x = body_start; else pr->insts[sp].y = body_start;
|
|
}
|
|
int end = pr->n;
|
|
for (size_t k = 0; k < splits.len; k++) {
|
|
int sp = ((int *)splits.data)[k];
|
|
if (n->greedy) pr->insts[sp].y = end; else pr->insts[sp].x = end;
|
|
}
|
|
da_free(&splits);
|
|
}
|
|
}
|
|
|
|
static void compile_node(Prog *pr, Node *n) {
|
|
switch (n->kind) {
|
|
case N_EMPTY: return;
|
|
case N_CHAR: { int pc = emit(pr, OP_CHAR); pr->insts[pc].data = n->ch; return; }
|
|
case N_ANY: emit(pr, OP_ANY); return;
|
|
case N_CLASS: {
|
|
int idx = (int)pr->classnodes.len;
|
|
Node **slot = da_push(&pr->classnodes); *slot = n;
|
|
int pc = emit(pr, OP_CLASS); pr->insts[pc].data = (uint32_t)idx;
|
|
return;
|
|
}
|
|
case N_CONCAT: compile_node(pr, n->a); compile_node(pr, n->b); return;
|
|
case N_ALT: {
|
|
int split_pc = emit(pr, OP_SPLIT);
|
|
int x0 = pr->n;
|
|
compile_node(pr, n->a);
|
|
int jmp_pc = emit(pr, OP_JMP);
|
|
int y0 = pr->n;
|
|
compile_node(pr, n->b);
|
|
int end = pr->n;
|
|
pr->insts[split_pc].x = x0; pr->insts[split_pc].y = y0;
|
|
pr->insts[jmp_pc].x = end;
|
|
return;
|
|
}
|
|
case N_REPEAT: compile_repeat(pr, n); return;
|
|
case N_GROUP: {
|
|
if (n->group_index >= 0) { int pc = emit(pr, OP_SAVE); pr->insts[pc].data = (uint32_t)(2 * n->group_index); }
|
|
compile_node(pr, n->a);
|
|
if (n->group_index >= 0) { int pc = emit(pr, OP_SAVE); pr->insts[pc].data = (uint32_t)(2 * n->group_index + 1); }
|
|
return;
|
|
}
|
|
case N_BACKREF: { int pc = emit(pr, OP_BACKREF); pr->insts[pc].data = (uint32_t)n->backref_index; return; }
|
|
case N_ANCHOR: { int pc = emit(pr, OP_ASSERT); pr->insts[pc].data = (uint32_t)n->anchor_kind; return; }
|
|
case N_LOOKAROUND: {
|
|
int op = (n->lookaround_kind == LK_AHEAD) ? OP_LOOKAHEAD : OP_LOOKBEHIND;
|
|
int pc = emit(pr, op);
|
|
pr->insts[pc].neg = n->lookaround_negate;
|
|
pr->insts[pc].width = n->width;
|
|
int sub_start = pr->n;
|
|
compile_node(pr, n->a);
|
|
emit(pr, OP_RETURN);
|
|
int after = pr->n;
|
|
pr->insts[pc].x = sub_start; pr->insts[pc].y = after;
|
|
return;
|
|
}
|
|
case N_ATOMIC: {
|
|
int pc = emit(pr, OP_ATOMIC);
|
|
int sub_start = pr->n;
|
|
compile_node(pr, n->a);
|
|
emit(pr, OP_RETURN);
|
|
int after = pr->n;
|
|
pr->insts[pc].x = sub_start; pr->insts[pc].y = after;
|
|
return;
|
|
}
|
|
}
|
|
}
|
|
|
|
/* ================================================================
|
|
* 6. Matcher (recursive backtracking VM, concept.md Section 7.3/7.1)
|
|
* ================================================================ */
|
|
|
|
#define MAX_DEPTH 60000
|
|
|
|
/* An earlier revision of this file capped compute_maxrun/
|
|
* compute_next_prevmatch's table size and fell back to the plain
|
|
* O(remaining length) scan above the cap, intending a bounded-memory
|
|
* degradation for huge inputs. Measured directly (a real 200MB search
|
|
* with no match) and reverted: the fallback does not degrade
|
|
* gracefully, it reintroduces the O(n^2) behavior these tables exist
|
|
* to fix, for exactly the pattern shape (a simple-atom quantifier
|
|
* followed by a required literal/class that never occurs) they matter
|
|
* most for, and O(n^2) at n in the hundreds of millions does not
|
|
* finish in any practical amount of time. A fast, diagnosable
|
|
* allocation failure is a better failure mode than a silent,
|
|
* effectively-unbounded hang, so these tables are allocated
|
|
* unconditionally again; README.md "Implementation status" states the
|
|
* resulting memory-per-input-byte multiplier plainly instead. There is
|
|
* no way to get both bounded memory and linear time out of this
|
|
* technique; only concept.md Section 7.2's actual streaming automaton
|
|
* (still unimplemented) gets both at once, by construction. */
|
|
|
|
typedef struct {
|
|
Prog *pr;
|
|
const uint32_t *text; /* UTF8 mode: real code points. NULL in BINARY/ASCII mode: use text8/text_at() instead. */
|
|
const uint8_t *text8; /* BINARY/ASCII mode: raw bytes, no widening copy, see MatBuf.text8 */
|
|
int64_t len;
|
|
int64_t *caps; /* 2*(ngroups+1) */
|
|
int64_t require_end; /* -1 unconstrained, -2 "sub-program, report on OP_RETURN", >=0 exact end required */
|
|
int64_t sub_end; /* scratch: end sp reported by a sub-program's OP_RETURN */
|
|
int depth_exceeded;
|
|
uint8_t *memo; /* NULL, or a (ninsts * (len+1))-bit "known to fail" cache, see run_memo */
|
|
int32_t **repeat_maxrun; /* NULL, or per-OP_REPEAT1-instruction run-length tables, see compute_maxrun */
|
|
int32_t **repeat_nextpm; /* NULL, or per-OP_REPEAT1-instruction "rightmost position <= X where
|
|
* the following atom matches" tables, see compute_next_prevmatch */
|
|
int forbid_empty; /* 1: OP_MATCH must not end at the attempt's own start, see
|
|
* Pattern_finditer/Pattern_split's same-position retry */
|
|
} MCtx;
|
|
|
|
/* Root-cause fix for the search-family quadratic behavior documented in
|
|
* README.md "Implementation status": a plain backtracking matcher has
|
|
* no memory that "starting at instruction pc, text position sp is
|
|
* hopeless", so re-trying nearby start positions (do_one's loop) or
|
|
* nearby alternative partitions of the same run of characters
|
|
* ((a+)+b-style catastrophic backtracking) re-derives the same failure
|
|
* over and over. For a pattern with no backreference, whether
|
|
* execution starting at (pc, sp) can ever reach OP_MATCH is a pure
|
|
* function of (pc, sp) alone whenever it is checked outside any active
|
|
* nullable-loop guard (guard_pc == -1) and outside any nested
|
|
* sub-program's constrained context (require_end == -1); this is
|
|
* exactly the "memoized backtracking" technique, provably the same
|
|
* complexity class as simulating the compiled program as an NFA
|
|
* (concept.md Section 7.2's Pike VM), just expressed recursively with
|
|
* a memo table instead of iteratively over an explicit thread list.
|
|
* Memoization never caches a *success*, only a proven failure, so it
|
|
* cannot change which match (or which captures) is found, only skip
|
|
* re-deriving failures already known. This is why it is always safe to
|
|
* enable, and it is enabled automatically whenever the compiled
|
|
* program has no OP_BACKREF (Prog.has_backref), regardless of which
|
|
* Pattern_ function is used. */
|
|
static int memo_get(MCtx *c, int pc, int64_t sp) {
|
|
if (!c->memo) return 0;
|
|
int64_t bit = (int64_t)pc * (c->len + 1) + sp;
|
|
return (c->memo[bit >> 3] >> (bit & 7)) & 1;
|
|
}
|
|
static void memo_set(MCtx *c, int pc, int64_t sp) {
|
|
if (!c->memo) return;
|
|
int64_t bit = (int64_t)pc * (c->len + 1) + sp;
|
|
c->memo[bit >> 3] |= (uint8_t)(1u << (bit & 7));
|
|
}
|
|
|
|
static int class_match(Node *cls, uint32_t cp, int mode, int flags) {
|
|
int hit = 0;
|
|
for (size_t k = 0; k < cls->items.len && !hit; k++) {
|
|
ClassItem *it = &((ClassItem *)cls->items.data)[k];
|
|
int m;
|
|
switch (it->kind) {
|
|
case CI_RANGE:
|
|
m = (cp >= it->lo && cp <= it->hi);
|
|
if (!m && (flags & IGNORECASE)) {
|
|
uint32_t alt = cls_swapcase(cp, mode, flags);
|
|
m = (alt >= it->lo && alt <= it->hi);
|
|
}
|
|
break;
|
|
case CI_D: m = cls_is_digit(cp, mode, flags); break;
|
|
case CI_W: m = cls_is_word(cp, mode, flags); break;
|
|
case CI_S: m = cls_is_space(cp, mode, flags); break;
|
|
default: m = 0;
|
|
}
|
|
if (it->kind != CI_RANGE && it->neg) m = !m;
|
|
if (m) hit = 1;
|
|
}
|
|
if (cls->class_negate) hit = !hit;
|
|
return hit;
|
|
}
|
|
static int char_eq(uint32_t a, uint32_t b, int mode, int flags) {
|
|
if (a == b) return 1;
|
|
if (flags & IGNORECASE) return cls_fold(a, mode, flags) == cls_fold(b, mode, flags);
|
|
return 0;
|
|
}
|
|
/* The one place BINARY/ASCII mode's "no widening copy" (MatBuf.text8,
|
|
* MCtx.text8) and UTF8 mode's real decoded code points (text) are
|
|
* reconciled into a single per-position value: c->text is NULL in
|
|
* BINARY/ASCII mode, so this reads c->text8 instead there. */
|
|
static inline uint32_t text_at(const MCtx *c, int64_t i) {
|
|
return c->text ? c->text[i] : c->text8[i];
|
|
}
|
|
static int is_word_at(MCtx *c, int64_t pos) {
|
|
if (pos < 0 || pos >= c->len) return 0;
|
|
return cls_is_word(text_at(c, pos), c->pr->mode, c->pr->flags);
|
|
}
|
|
|
|
static int run_memo(MCtx *c, int pc, int64_t sp, int64_t guard_sp, int32_t guard_pc, int depth);
|
|
|
|
static int run(MCtx *c, int pc, int64_t sp, int64_t guard_sp, int32_t guard_pc, int depth) {
|
|
if (depth > MAX_DEPTH) { c->depth_exceeded = 1; return 0; }
|
|
for (;;) {
|
|
Inst *in = &c->pr->insts[pc];
|
|
switch (in->op) {
|
|
case OP_CHAR:
|
|
if (sp >= c->len || !char_eq(text_at(c, sp), in->data, c->pr->mode, c->pr->flags)) return 0;
|
|
pc++; sp++; continue;
|
|
case OP_ANY:
|
|
if (sp >= c->len) return 0;
|
|
if (text_at(c, sp) == '\n' && !(c->pr->flags & DOTALL)) return 0;
|
|
pc++; sp++; continue;
|
|
case OP_CLASS: {
|
|
if (sp >= c->len) return 0;
|
|
Node *cls = ((Node **)c->pr->classnodes.data)[in->data];
|
|
if (!class_match(cls, text_at(c, sp), c->pr->mode, c->pr->flags)) return 0;
|
|
pc++; sp++; continue;
|
|
}
|
|
case OP_REPEAT1: {
|
|
int64_t maxc = (in->hi == -1) ? (c->len - sp) : in->hi;
|
|
if (maxc > c->len - sp) maxc = c->len - sp;
|
|
int64_t count;
|
|
if (c->repeat_maxrun && c->repeat_maxrun[pc]) {
|
|
/* O(1): see compute_maxrun. Without this, counting
|
|
* how many atoms match from sp is an O(remaining
|
|
* length) scan repeated at every start position
|
|
* do_one/finditer/split try, which is exactly the
|
|
* quadratic-time defect documented in README.md
|
|
* "Implementation status"; this is its fix. */
|
|
count = c->repeat_maxrun[pc][sp];
|
|
} else {
|
|
count = 0;
|
|
Node *cls = (in->atomkind == 1) ? ((Node **)c->pr->classnodes.data)[in->data] : NULL;
|
|
while (count < maxc) {
|
|
uint32_t ch = text_at(c, sp + count);
|
|
int m;
|
|
if (in->atomkind == 0) m = char_eq(ch, in->data, c->pr->mode, c->pr->flags);
|
|
else if (in->atomkind == 2) m = !(ch == '\n' && !(c->pr->flags & DOTALL));
|
|
else m = class_match(cls, ch, c->pr->mode, c->pr->flags);
|
|
if (!m) break;
|
|
count++;
|
|
}
|
|
}
|
|
if (count > maxc) count = maxc;
|
|
if (count < in->lo) return 0;
|
|
if (in->greedy && c->repeat_nextpm && c->repeat_nextpm[pc]) {
|
|
/* O(candidates actually worth trying), not O(count):
|
|
* see compute_next_prevmatch. Trying every k from
|
|
* count down to lo one at a time costs O(count)
|
|
* loop iterations even though each individual
|
|
* OP_CHAR/OP_CLASS/OP_ANY check at pc+1 is O(1),
|
|
* because the *number* of iterations, not the cost
|
|
* of any one of them, is what stayed quadratic
|
|
* across do_one/finditer/split's outer position
|
|
* loop; jumping straight to positions where pc+1
|
|
* can actually succeed is what fixes that. */
|
|
int32_t *pm = c->repeat_nextpm[pc];
|
|
int64_t hi = sp + count; if (hi > c->len - 1) hi = c->len - 1;
|
|
int64_t lo_bound = sp + in->lo;
|
|
int64_t p = (hi >= lo_bound && hi >= 0) ? pm[hi] : -1;
|
|
while (p >= lo_bound) {
|
|
if (run_memo(c, pc + 1, p, guard_sp, guard_pc, depth + 1)) return 1;
|
|
p = (p > 0) ? pm[p - 1] : -1;
|
|
}
|
|
return 0;
|
|
}
|
|
if (in->greedy) {
|
|
for (int64_t k = count; k >= in->lo; k--)
|
|
if (run_memo(c, pc + 1, sp + k, guard_sp, guard_pc, depth + 1)) return 1;
|
|
} else {
|
|
for (int64_t k = in->lo; k <= count; k++)
|
|
if (run_memo(c, pc + 1, sp + k, guard_sp, guard_pc, depth + 1)) return 1;
|
|
}
|
|
return 0;
|
|
}
|
|
case OP_ASSERT: {
|
|
int ok;
|
|
switch (in->data) {
|
|
case A_BOL: ok = (sp == 0) || ((c->pr->flags & MULTILINE) && sp > 0 && text_at(c, sp - 1) == '\n'); break;
|
|
case A_EOL: ok = (sp == c->len) || (text_at(c, sp) == '\n' && ((c->pr->flags & MULTILINE) || sp == c->len - 1)); break;
|
|
case A_BOS: ok = (sp == 0); break;
|
|
case A_EOS: ok = (sp == c->len); break;
|
|
case A_WB: ok = is_word_at(c, sp - 1) != is_word_at(c, sp); break;
|
|
case A_NWB: ok = is_word_at(c, sp - 1) == is_word_at(c, sp); break;
|
|
default: ok = 0;
|
|
}
|
|
if (!ok) return 0;
|
|
pc++; continue;
|
|
}
|
|
case OP_SAVE: {
|
|
int64_t old = c->caps[in->data];
|
|
c->caps[in->data] = sp;
|
|
if (run_memo(c, pc + 1, sp, guard_sp, guard_pc, depth + 1)) return 1;
|
|
c->caps[in->data] = old;
|
|
return 0;
|
|
}
|
|
case OP_SPLIT:
|
|
if (in->is_loop) {
|
|
if (run_memo(c, in->x, sp, sp, pc, depth + 1)) return 1;
|
|
pc = in->y; continue;
|
|
} else {
|
|
if (run_memo(c, in->x, sp, guard_sp, guard_pc, depth + 1)) return 1;
|
|
pc = in->y; continue;
|
|
}
|
|
case OP_JMP:
|
|
if (in->loop_close && in->x == guard_pc && sp == guard_sp) return 0;
|
|
pc = in->x; continue;
|
|
case OP_BACKREF: {
|
|
int gi = (int)in->data;
|
|
if (gi < 0 || gi > c->pr->ngroups) return 0;
|
|
int64_t s = c->caps[2 * gi], e = c->caps[2 * gi + 1];
|
|
if (s < 0 || e < 0) return 0;
|
|
int64_t rl = e - s;
|
|
if (sp + rl > c->len) return 0;
|
|
for (int64_t k = 0; k < rl; k++)
|
|
if (!char_eq(text_at(c, sp + k), text_at(c, s + k), c->pr->mode, c->pr->flags)) return 0;
|
|
pc++; sp += rl; continue;
|
|
}
|
|
case OP_LOOKAHEAD: {
|
|
size_t ncaps = 2 * (size_t)(c->pr->ngroups + 1);
|
|
int64_t *snap = malloc(ncaps * sizeof(int64_t));
|
|
memcpy(snap, c->caps, ncaps * sizeof(int64_t));
|
|
int64_t save_req = c->require_end; c->require_end = -2;
|
|
int ok = run_memo(c, in->x, sp, -1, -1, depth + 1);
|
|
c->require_end = save_req;
|
|
int accept = in->neg ? !ok : ok;
|
|
if (!accept) { memcpy(c->caps, snap, ncaps * sizeof(int64_t)); free(snap); return 0; }
|
|
free(snap);
|
|
pc = in->y; continue;
|
|
}
|
|
case OP_LOOKBEHIND: {
|
|
size_t ncaps = 2 * (size_t)(c->pr->ngroups + 1);
|
|
int64_t *snap = malloc(ncaps * sizeof(int64_t));
|
|
memcpy(snap, c->caps, ncaps * sizeof(int64_t));
|
|
int64_t start = sp - in->width;
|
|
int ok = 0;
|
|
if (start >= 0) {
|
|
int64_t save_req = c->require_end; c->require_end = sp;
|
|
ok = run_memo(c, in->x, start, -1, -1, depth + 1);
|
|
c->require_end = save_req;
|
|
}
|
|
int accept = in->neg ? !ok : ok;
|
|
if (!accept) { memcpy(c->caps, snap, ncaps * sizeof(int64_t)); free(snap); return 0; }
|
|
free(snap);
|
|
pc = in->y; continue;
|
|
}
|
|
case OP_ATOMIC: {
|
|
int64_t save_req = c->require_end; c->require_end = -2;
|
|
int ok = run_memo(c, in->x, sp, -1, -1, depth + 1);
|
|
c->require_end = save_req;
|
|
if (!ok) return 0;
|
|
sp = c->sub_end;
|
|
pc = in->y; continue;
|
|
}
|
|
case OP_RETURN:
|
|
if (c->require_end == -2) { c->sub_end = sp; return 1; }
|
|
if (c->require_end >= 0) return sp == c->require_end;
|
|
return 1;
|
|
case OP_MATCH:
|
|
if (c->require_end >= 0 && sp != c->require_end) return 0;
|
|
if (c->forbid_empty && sp == c->caps[0]) return 0;
|
|
c->caps[1] = sp;
|
|
return 1;
|
|
default:
|
|
return 0;
|
|
}
|
|
}
|
|
}
|
|
|
|
/* The only call site allowed to consult/populate the memo (see the
|
|
* comment on MCtx.memo above): every recursive call inside run(), and
|
|
* every external entry point, goes through here instead of run()
|
|
* directly. Caching is restricted to guard_pc == -1 (not inside an
|
|
* active nullable-loop guard) and c->require_end == -1 (not inside a
|
|
* fullmatch's exact-end constraint, nor inside a nested lookaround/
|
|
* atomic sub-program, which always sets require_end elsewhere first);
|
|
* outside that combination, whether (pc, sp) succeeds can depend on
|
|
* context beyond (pc, sp) itself, so it is never cached or consulted
|
|
* there, only computed directly by run(), exactly as before this
|
|
* optimization existed. */
|
|
static int run_memo(MCtx *c, int pc, int64_t sp, int64_t guard_sp, int32_t guard_pc, int depth) {
|
|
int cacheable = c->memo && guard_pc == -1 && c->require_end == -1 && !c->forbid_empty;
|
|
if (cacheable && memo_get(c, pc, sp)) return 0;
|
|
int result = run(c, pc, sp, guard_sp, guard_pc, depth);
|
|
if (cacheable && !result && !c->depth_exceeded) memo_set(c, pc, sp);
|
|
return result;
|
|
}
|
|
|
|
/* ================================================================
|
|
* 6b. Pike VM (regular engine over an NFA-only Prog, concept.md 7.7)
|
|
*
|
|
* Matches a Prog compiled with no_repeat1 set and known, by
|
|
* construction (the eligibility scan in re_compile), to contain none
|
|
* of OP_BACKREF/OP_LOOKAHEAD/OP_LOOKBEHIND/OP_ATOMIC/OP_REPEAT1: only
|
|
* OP_CHAR, OP_CLASS, OP_ANY, OP_SPLIT, OP_JMP, OP_SAVE, OP_MATCH,
|
|
* OP_ASSERT. Reuses MCtx purely as a bundle of the four fields that
|
|
* matter here (pr/text/text8/len); its other fields (memo, caps,
|
|
* repeat_maxrun/nextpm, depth_exceeded, sub_end) are backtracking
|
|
* engine state this VM never reads or writes.
|
|
* ================================================================ */
|
|
|
|
typedef struct {
|
|
int32_t pc;
|
|
int64_t *saved; /* owned by whichever thread list currently holds it;
|
|
* 2*(ngroups+1) entries */
|
|
} PikeThread;
|
|
|
|
typedef struct {
|
|
PikeThread *threads;
|
|
int count;
|
|
int cap; /* == nfa_prog->n; fixed for the whole search, see the
|
|
* "no growth logic, nothing can overflow it" argument
|
|
* in concept.md 7.7 (Cox's one-thread-per-PC bound) */
|
|
uint8_t *seen; /* size cap; which PCs already have a thread this step */
|
|
} PikeList;
|
|
|
|
static int pikelist_init(PikeList *l, int n) {
|
|
size_t cap = (size_t)(n > 0 ? n : 1);
|
|
l->threads = malloc(cap * sizeof(PikeThread));
|
|
l->seen = calloc(cap, 1);
|
|
l->count = 0;
|
|
l->cap = n;
|
|
return l->threads != NULL && l->seen != NULL;
|
|
}
|
|
/* Frees every currently-listed thread's saved array first: correct only
|
|
* when called at a point where l->count accurately reflects threads this
|
|
* list still owns (never right after a pike_step() call without first
|
|
* resetting count to 0, since pike_step() itself always resolves every
|
|
* entry it was given, by transferring, accepting, or freeing it). */
|
|
static void pikelist_free(PikeList *l) {
|
|
for (int i = 0; i < l->count; i++) free(l->threads[i].saved);
|
|
free(l->threads);
|
|
free(l->seen);
|
|
}
|
|
static void pikelist_clear(PikeList *l) {
|
|
l->count = 0;
|
|
memset(l->seen, 0, (size_t)(l->cap > 0 ? l->cap : 1));
|
|
}
|
|
|
|
typedef struct { int32_t pc; int64_t *saved; } PikeWork;
|
|
|
|
/* Epsilon closure from (pc0, saved0) at text position sp, added into
|
|
* list l (deduplicated by pc, first writer per step wins, exactly Cox's
|
|
* stated invariant). Specified recursively by the source material
|
|
* (concept.md 7.7) but implemented with an explicit heap stack instead
|
|
* of C call recursion, so a pattern with many alternations cannot
|
|
* recurse the C stack as deep as the instruction count.
|
|
*
|
|
* Contract: always resolves ownership of saved0 (frees it, or stores it
|
|
* into l, possibly after being duplicated further along the way) before
|
|
* returning, on every path, including failure. Returns 0 only on
|
|
* allocation failure, having freed everything it touched; l's existing
|
|
* entries (from before this call) are left valid and untouched. */
|
|
static int pike_addthread(PikeList *l, MCtx *sctx, int32_t pc0, int64_t *saved0, int64_t sp) {
|
|
int ncaps = 2 * (sctx->pr->ngroups + 1);
|
|
DArr stack; da_init(&stack, sizeof(PikeWork));
|
|
PikeWork *w0 = da_push(&stack);
|
|
if (!w0) { free(saved0); da_free(&stack); return 0; }
|
|
w0->pc = pc0; w0->saved = saved0;
|
|
|
|
while (stack.len > 0) {
|
|
PikeWork w = ((PikeWork *)stack.data)[stack.len - 1];
|
|
stack.len--;
|
|
if (l->seen[w.pc]) { free(w.saved); continue; }
|
|
l->seen[w.pc] = 1;
|
|
Inst *in = &sctx->pr->insts[w.pc];
|
|
switch (in->op) {
|
|
case OP_JMP: {
|
|
PikeWork *nw = da_push(&stack);
|
|
if (!nw) { free(w.saved); goto fail; }
|
|
nw->pc = in->x; nw->saved = w.saved;
|
|
break;
|
|
}
|
|
case OP_SPLIT: {
|
|
/* x is higher priority than y (compile_node already chose
|
|
* x/y per the quantifier's greedy/lazy sense, Section 6/
|
|
* 7.1); pushing y then x means x is popped next, giving
|
|
* the same DFS pre-order a recursive addthread(x) then
|
|
* addthread(y) would, so list order still encodes
|
|
* priority exactly as it does in the backtracking engine. */
|
|
int64_t *saved_y = malloc((size_t)ncaps * sizeof(int64_t));
|
|
if (!saved_y) { free(w.saved); goto fail; }
|
|
memcpy(saved_y, w.saved, (size_t)ncaps * sizeof(int64_t));
|
|
PikeWork *wy = da_push(&stack);
|
|
if (!wy) { free(saved_y); free(w.saved); goto fail; }
|
|
wy->pc = in->y; wy->saved = saved_y;
|
|
PikeWork *wx = da_push(&stack);
|
|
if (!wx) { free(w.saved); goto fail; } /* wy already on the
|
|
stack; the fail unwind below frees it along with
|
|
every other still-pending entry */
|
|
wx->pc = in->x; wx->saved = w.saved;
|
|
break;
|
|
}
|
|
case OP_SAVE: {
|
|
/* Safe to mutate w.saved in place, not duplicate: by
|
|
* construction the only place a saved array is ever
|
|
* shared between two in-flight branches is the instant
|
|
* OP_SPLIT above duplicates it, so past that point this
|
|
* copy is exclusively owned along this one path. */
|
|
w.saved[in->data] = sp;
|
|
PikeWork *nw = da_push(&stack);
|
|
if (!nw) { free(w.saved); goto fail; }
|
|
nw->pc = w.pc + 1; nw->saved = w.saved;
|
|
break;
|
|
}
|
|
case OP_ASSERT: {
|
|
int ok;
|
|
switch (in->data) {
|
|
case A_BOL: ok = (sp == 0) || ((sctx->pr->flags & MULTILINE) && sp > 0 && text_at(sctx, sp - 1) == '\n'); break;
|
|
case A_EOL: ok = (sp == sctx->len) || (text_at(sctx, sp) == '\n' && ((sctx->pr->flags & MULTILINE) || sp == sctx->len - 1)); break;
|
|
case A_BOS: ok = (sp == 0); break;
|
|
case A_EOS: ok = (sp == sctx->len); break;
|
|
case A_WB: ok = is_word_at(sctx, sp - 1) != is_word_at(sctx, sp); break;
|
|
case A_NWB: ok = is_word_at(sctx, sp - 1) == is_word_at(sctx, sp); break;
|
|
default: ok = 0;
|
|
}
|
|
if (!ok) { free(w.saved); break; }
|
|
PikeWork *nw = da_push(&stack);
|
|
if (!nw) { free(w.saved); goto fail; }
|
|
nw->pc = w.pc + 1; nw->saved = w.saved;
|
|
break;
|
|
}
|
|
case OP_CHAR: case OP_CLASS: case OP_ANY: case OP_MATCH:
|
|
if (l->count >= l->cap) { free(w.saved); goto fail; } /* unreachable
|
|
in practice, see the fixed-bound argument above; guarded
|
|
rather than assumed */
|
|
l->threads[l->count].pc = w.pc;
|
|
l->threads[l->count].saved = w.saved;
|
|
l->count++;
|
|
break;
|
|
default:
|
|
free(w.saved); /* unreachable: an eligible Prog contains no other op */
|
|
break;
|
|
}
|
|
}
|
|
da_free(&stack);
|
|
return 1;
|
|
fail:
|
|
for (size_t k = 0; k < stack.len; k++) free(((PikeWork *)stack.data)[k].saved);
|
|
da_free(&stack);
|
|
return 0;
|
|
}
|
|
|
|
/* Processes every thread in clist at position sp: a char/class/any
|
|
* thread that matches is carried into nlist at pc+1; a match thread is
|
|
* recorded as the best answer so far (subject to require_end and
|
|
* forbid_empty, reusing MCtx's exact same-named semantics) and every
|
|
* strictly lower priority thread remaining in clist this step is
|
|
* discarded without being run (the leftmost-first rule, concept.md
|
|
* 7.7, verified against rust-lang/regex's PikeVM). Always resolves
|
|
* ownership of every clist entry before returning, success or not.
|
|
* Returns 0 only on allocation failure. */
|
|
static int pike_step(PikeList *clist, PikeList *nlist, MCtx *sctx, int64_t sp,
|
|
int64_t require_end, int forbid_empty,
|
|
int64_t **best_saved, int *matched) {
|
|
int i;
|
|
for (i = 0; i < clist->count; i++) {
|
|
PikeThread th = clist->threads[i];
|
|
Inst *in = &sctx->pr->insts[th.pc];
|
|
int consume_ok;
|
|
switch (in->op) {
|
|
case OP_CHAR:
|
|
consume_ok = (sp < sctx->len) && char_eq(text_at(sctx, sp), in->data, sctx->pr->mode, sctx->pr->flags);
|
|
break;
|
|
case OP_ANY:
|
|
consume_ok = (sp < sctx->len) && !(text_at(sctx, sp) == '\n' && !(sctx->pr->flags & DOTALL));
|
|
break;
|
|
case OP_CLASS: {
|
|
Node *cls = ((Node **)sctx->pr->classnodes.data)[in->data];
|
|
consume_ok = (sp < sctx->len) && class_match(cls, text_at(sctx, sp), sctx->pr->mode, sctx->pr->flags);
|
|
break;
|
|
}
|
|
case OP_MATCH:
|
|
/* Group 0's end has no OP_SAVE (only explicit user groups
|
|
* do, Section 7.1's N_GROUP case); run()'s own OP_MATCH
|
|
* handler sets it directly the same way (Section 6), and
|
|
* this VM must too. */
|
|
if ((require_end < 0 || sp == require_end) && !(forbid_empty && sp == th.saved[0])) {
|
|
th.saved[1] = sp;
|
|
free(*best_saved);
|
|
*best_saved = th.saved;
|
|
*matched = 1;
|
|
i++;
|
|
goto discard_rest;
|
|
}
|
|
free(th.saved);
|
|
continue;
|
|
default:
|
|
free(th.saved); /* unreachable for an eligible Prog */
|
|
continue;
|
|
}
|
|
if (consume_ok) {
|
|
if (!pike_addthread(nlist, sctx, th.pc + 1, th.saved, sp + 1)) {
|
|
for (int j = i + 1; j < clist->count; j++) free(clist->threads[j].saved);
|
|
return 0;
|
|
}
|
|
} else {
|
|
free(th.saved);
|
|
}
|
|
}
|
|
return 1;
|
|
discard_rest:
|
|
for (; i < clist->count; i++) free(clist->threads[i].saved);
|
|
return 1;
|
|
}
|
|
|
|
/* Finds the single leftmost, highest priority match for an NFA-only
|
|
* Prog between pos and endpos (inclusive of trying a start at endpos
|
|
* itself, matching do_one's existing <= convention). anchored seeds
|
|
* exactly one start thread, at pos, and never injects another
|
|
* (Pattern_match/fullmatch); unanchored search (anchored == 0) injects
|
|
* a fresh start thread, at lowest priority, at every position up to
|
|
* endpos, until a match has been recorded (Pattern_search/finditer/
|
|
* split), giving one linear pass over [pos, endpos] rather than one
|
|
* attempt per candidate start position.
|
|
*
|
|
* require_end (-1 unconstrained, else the exact end position required)
|
|
* and forbid_empty mirror MCtx's identically named fields exactly
|
|
* (Section 6), so fullmatch and the empty-match retry rule (Pattern_
|
|
* finditer/Pattern_split) behave identically on both engines.
|
|
*
|
|
* Returns 1 (match found, caps_out filled with 2*(ngroups+1) entries),
|
|
* 0 (no match anywhere in range), or -1 (allocation failure partway
|
|
* through; mirrors run_memo's depth_exceeded signalling an
|
|
* unrecoverable condition through the same -1 every Pattern_/re_
|
|
* matching function already documents, docs/API.md Section 3). */
|
|
static int pike_find(Prog *nfa_pr, const uint32_t *text, const uint8_t *text8, int64_t pos, int64_t endpos,
|
|
int anchored, int64_t require_end, int forbid_empty, int64_t *caps_out) {
|
|
MCtx sctx; memset(&sctx, 0, sizeof sctx);
|
|
sctx.pr = nfa_pr; sctx.text = text; sctx.text8 = text8; sctx.len = endpos;
|
|
int ncaps = 2 * (nfa_pr->ngroups + 1);
|
|
|
|
PikeList clist, nlist;
|
|
if (!pikelist_init(&clist, nfa_pr->n)) return -1;
|
|
if (!pikelist_init(&nlist, nfa_pr->n)) { pikelist_free(&clist); return -1; }
|
|
|
|
int64_t *best_saved = NULL;
|
|
int matched = 0, oom = 0;
|
|
int64_t sp = pos;
|
|
|
|
/* Literal prefilter, added after v1 (concept.md 7.9): when the very
|
|
* first instruction is a mandatory OP_CHAR or OP_CLASS (a pattern
|
|
* that begins with a required literal or class, not a nullable loop
|
|
* or a leading assertion), a freshly injected start thread at a
|
|
* position that instruction does not accept is certain to die on
|
|
* the very next pike_step call, exactly the way an un-prefiltered
|
|
* one already does, just after paying for a malloc and a full
|
|
* pike_addthread call first. Checking the identical condition
|
|
* pike_step's own OP_CHAR/OP_CLASS case checks, before injecting
|
|
* rather than after, changes nothing about which threads ever exist
|
|
* (a position this skips would have produced a thread that dies
|
|
* unobserved on the next step regardless), only how much work is
|
|
* spent finding that out; this is the literal prefilter technique
|
|
* concept.md 7.5/README.md "Implementation status" already cite for
|
|
* the backtracking engine's own compute_next_prevmatch, applied
|
|
* here in its simplest form (per position, not precomputed). A
|
|
* pattern starting with `^`, a group, an alternation, or a nullable
|
|
* repeat gets no benefit and none of the risk: should_inject stays
|
|
* 1 and every position is injected exactly as in v1. */
|
|
int first_op = nfa_pr->n > 0 ? nfa_pr->insts[0].op : -1;
|
|
Node *first_cls = (first_op == OP_CLASS) ? ((Node **)nfa_pr->classnodes.data)[nfa_pr->insts[0].data] : NULL;
|
|
#define PIKE_SHOULD_INJECT(atpos) \
|
|
(first_op == OP_CHAR ? ((atpos) < endpos && char_eq(text_at(&sctx, (atpos)), nfa_pr->insts[0].data, nfa_pr->mode, nfa_pr->flags)) : \
|
|
first_op == OP_CLASS ? ((atpos) < endpos && class_match(first_cls, text_at(&sctx, (atpos)), nfa_pr->mode, nfa_pr->flags)) : \
|
|
1)
|
|
|
|
if (PIKE_SHOULD_INJECT(pos)) {
|
|
int64_t *seed = malloc((size_t)ncaps * sizeof(int64_t));
|
|
if (!seed) oom = 1;
|
|
else {
|
|
for (int k = 0; k < ncaps; k++) seed[k] = -1;
|
|
seed[0] = pos;
|
|
if (!pike_addthread(&clist, &sctx, 0, seed, sp)) oom = 1;
|
|
}
|
|
}
|
|
|
|
while (!oom) {
|
|
int step_ok = pike_step(&clist, &nlist, &sctx, sp, require_end, forbid_empty, &best_saved, &matched);
|
|
clist.count = 0; /* pike_step resolved every entry's ownership either way */
|
|
if (!step_ok) { oom = 1; break; }
|
|
if (sp >= endpos) break;
|
|
sp++;
|
|
if (!anchored && !matched && PIKE_SHOULD_INJECT(sp)) {
|
|
int64_t *s2 = malloc((size_t)ncaps * sizeof(int64_t));
|
|
if (!s2) { oom = 1; break; }
|
|
for (int k = 0; k < ncaps; k++) s2[k] = -1;
|
|
s2[0] = sp;
|
|
if (!pike_addthread(&nlist, &sctx, 0, s2, sp)) { oom = 1; break; }
|
|
}
|
|
{ PikeList tmp = clist; clist = nlist; nlist = tmp; }
|
|
pikelist_clear(&nlist);
|
|
/* An empty list only means "give up" when nothing will ever be
|
|
* injected into it again: once anchored (a single seed, never
|
|
* renewed) or matched (new starts stop being seeded, see above),
|
|
* an empty list can never become non-empty again. Otherwise, a
|
|
* momentarily empty list is not the same as "no more positions
|
|
* are worth trying": a fresh start seeded at some later position
|
|
* can die immediately during its own epsilon closure (its very
|
|
* first instruction, commonly \b, failing outright, exactly what
|
|
* happens repeatedly inside a longer word like "catalog" for the
|
|
* pattern \bcat\b) without that meaning every later position
|
|
* will too. Breaking here unconditionally was a real bug, found
|
|
* by the existing test suite: it stopped the whole search the
|
|
* first time this happened, before ever reaching a later
|
|
* position that would have matched. */
|
|
if (clist.count == 0 && (anchored || matched)) break;
|
|
}
|
|
#undef PIKE_SHOULD_INJECT
|
|
|
|
pikelist_free(&clist);
|
|
pikelist_free(&nlist);
|
|
if (oom) { free(best_saved); return -1; }
|
|
if (!matched) return 0;
|
|
memcpy(caps_out, best_saved, (size_t)ncaps * sizeof(int64_t));
|
|
free(best_saved);
|
|
return 1;
|
|
}
|
|
|
|
/* ================================================================
|
|
* 7. Pattern / Match / Input structures
|
|
* ================================================================ */
|
|
|
|
typedef struct { char **names; int *idx; int n; } GroupIndex;
|
|
|
|
typedef struct {
|
|
Prog prog;
|
|
GroupIndex gidx;
|
|
NodePool pool;
|
|
char *pattern_copy;
|
|
int has_nfa; /* 1 if nfa_prog below is valid and should be preferred, concept.md 7.7 */
|
|
Prog nfa_prog; /* compiled only when has_nfa; contains no OP_BACKREF/OP_LOOKAHEAD/
|
|
* OP_LOOKBEHIND/OP_ATOMIC/OP_REPEAT1, matched by pike_find() instead
|
|
* of run_memo(). Shares classnodes' underlying Node* with prog (both
|
|
* point into the one AST pool, freed once via PatternImpl.pool). */
|
|
} PatternImpl;
|
|
|
|
struct Input {
|
|
uint8_t *buf;
|
|
size_t len;
|
|
int owns_buf; /* free(buf) on Input_free */
|
|
int is_mmap; /* munmap(buf, len) on Input_free instead of free() */
|
|
};
|
|
|
|
typedef struct {
|
|
int64_t *pos; /* 2*(ngroups+1) */
|
|
int ngroups;
|
|
int64_t *cp_to_byte; /* NULL unless UTF8 mode; length ngroups-independent, sized to text length+1 */
|
|
} MatchSlots;
|
|
|
|
static int groupindex_lookup(GroupIndex *gi, const char *name) {
|
|
for (int k = 0; k < gi->n; k++) if (strcmp(gi->names[k], name) == 0) return gi->idx[k];
|
|
return -1;
|
|
}
|
|
static int resolve_mode(int flags) {
|
|
if (flags & BINARY) return MODE_BINARY;
|
|
if (flags & UTF8) return MODE_UTF8;
|
|
return MODE_ASCII;
|
|
}
|
|
static void pattern_error_fill(PatternError *err, const char *pattern, const char *msg, int64_t pos) {
|
|
if (!err) return;
|
|
err->msg = strdup(msg ? msg : "");
|
|
err->pattern = pattern ? strdup(pattern) : NULL;
|
|
err->pos = pos;
|
|
err->lineno = 1; err->colno = 1;
|
|
if (pattern) {
|
|
for (int64_t k = 0; k < pos && pattern[k]; k++) {
|
|
if (pattern[k] == '\n') { err->lineno++; err->colno = 1; } else err->colno++;
|
|
}
|
|
}
|
|
}
|
|
|
|
void PatternError_free(PatternError *err) {
|
|
if (!err) return;
|
|
free((void *)err->msg); free((void *)err->pattern);
|
|
err->msg = NULL; err->pattern = NULL;
|
|
}
|
|
|
|
/* ================================================================
|
|
* 8. Compile driver
|
|
* ================================================================ */
|
|
|
|
Pattern *re_compile(const char *pattern, size_t len, int flags, PatternError *err) {
|
|
/* Section 9.3: BINARY/UTF8 select their mode; otherwise ASCII. */
|
|
int mode = resolve_mode(flags);
|
|
|
|
/* Step 1: leading global inline flags (?aiLmsux), Section 2.5/2.6 */
|
|
int extra_flags = 0;
|
|
size_t body_off = 0;
|
|
if (len >= 3 && pattern[0] == '(' && pattern[1] == '?') {
|
|
size_t k = 2; int any = 0, ok = 1;
|
|
while (k < len && strchr("aiLmsux", (unsigned char)pattern[k])) {
|
|
switch (pattern[k]) {
|
|
case 'i': extra_flags |= IGNORECASE; break;
|
|
case 'm': extra_flags |= MULTILINE; break;
|
|
case 's': extra_flags |= DOTALL; break;
|
|
case 'x': extra_flags |= VERBOSE; break;
|
|
case 'a': extra_flags |= ASCII; break;
|
|
case 'L': extra_flags |= LOCALE; break;
|
|
default: break;
|
|
}
|
|
k++; any = 1;
|
|
}
|
|
if (any && k < len && pattern[k] == ')') { body_off = k + 1; }
|
|
else { extra_flags = 0; ok = 0; (void)ok; }
|
|
}
|
|
int all_flags = flags | extra_flags;
|
|
|
|
const char *body = pattern + body_off;
|
|
size_t bodylen = len - body_off;
|
|
char *stripped = NULL;
|
|
if (all_flags & VERBOSE) {
|
|
size_t nl;
|
|
stripped = strip_verbose(body, bodylen, &nl);
|
|
body = stripped; bodylen = nl;
|
|
}
|
|
|
|
Parser p; memset(&p, 0, sizeof p);
|
|
p.s = body; p.len = bodylen; p.i = 0; p.flags = all_flags; p.mode = mode;
|
|
da_init(&p.backrefs, sizeof(Node *));
|
|
da_init(&p.groupnames, sizeof(GroupName));
|
|
g_pool.data = NULL; g_pool.len = 0; g_pool.cap = 0;
|
|
|
|
Node *ast = parse_alt(&p);
|
|
if (!p.failed && !at_end(&p)) {
|
|
if (peek(&p) == ')') perr(&p, (int64_t)p.i, "unbalanced parenthesis");
|
|
else perr(&p, (int64_t)p.i, "unexpected character");
|
|
}
|
|
if (!p.failed) {
|
|
for (size_t k = 0; k < p.backrefs.len; k++) {
|
|
Node *b = ((Node **)p.backrefs.data)[k];
|
|
if (b->backref_name) {
|
|
int found = -1;
|
|
for (size_t j = 0; j < p.groupnames.len; j++) {
|
|
GroupName *gn = &((GroupName *)p.groupnames.data)[j];
|
|
if (strcmp(gn->name, b->backref_name) == 0) { found = gn->index; break; }
|
|
}
|
|
if (found < 0) { perr(&p, 0, "unknown group name '%s'", b->backref_name); break; }
|
|
b->backref_index = found;
|
|
} else if (b->backref_index < 1 || b->backref_index > p.ngroups) {
|
|
perr(&p, 0, "invalid group reference %d", b->backref_index);
|
|
break;
|
|
}
|
|
}
|
|
}
|
|
|
|
if (p.failed) {
|
|
pattern_error_fill(err, pattern, p.errmsg, p.errpos + (int64_t)body_off);
|
|
nodepool_free_all(&g_pool);
|
|
da_free(&p.backrefs);
|
|
for (size_t k = 0; k < p.groupnames.len; k++) free(((GroupName *)p.groupnames.data)[k].name);
|
|
da_free(&p.groupnames);
|
|
free(stripped);
|
|
return NULL;
|
|
}
|
|
|
|
Prog prog; memset(&prog, 0, sizeof prog);
|
|
prog.ngroups = p.ngroups; prog.mode = mode; prog.flags = all_flags;
|
|
da_init(&prog.classnodes, sizeof(Node *));
|
|
compile_node(&prog, ast);
|
|
emit(&prog, OP_MATCH);
|
|
/* Pike VM eligibility (concept.md 7.7): a pattern is ineligible iff
|
|
* its ordinary Prog contains a construct with no pure-NFA execution
|
|
* (a backreference, or a sub-program instruction: lookahead,
|
|
* lookbehind, atomic group; a possessive quantifier is already
|
|
* OP_ATOMIC by the time it reaches here, Section 4, so it needs no
|
|
* separate check). One scan serves both this and has_backref. */
|
|
int nfa_eligible = 1;
|
|
for (int k = 0; k < prog.n; k++) {
|
|
switch (prog.insts[k].op) {
|
|
case OP_BACKREF: prog.has_backref = 1; nfa_eligible = 0; break;
|
|
case OP_LOOKAHEAD: case OP_LOOKBEHIND: case OP_ATOMIC: nfa_eligible = 0; break;
|
|
default: break;
|
|
}
|
|
}
|
|
|
|
PatternImpl *impl = calloc(1, sizeof(PatternImpl));
|
|
impl->prog = prog;
|
|
impl->pool = g_pool;
|
|
if (nfa_eligible) {
|
|
Prog nprog; memset(&nprog, 0, sizeof nprog);
|
|
nprog.ngroups = p.ngroups; nprog.mode = mode; nprog.flags = all_flags;
|
|
nprog.no_repeat1 = 1;
|
|
da_init(&nprog.classnodes, sizeof(Node *));
|
|
compile_node(&nprog, ast);
|
|
emit(&nprog, OP_MATCH);
|
|
impl->nfa_prog = nprog;
|
|
impl->has_nfa = 1;
|
|
}
|
|
impl->pattern_copy = xstrndup(pattern, len);
|
|
impl->gidx.n = (int)p.groupnames.len;
|
|
impl->gidx.names = malloc(sizeof(char *) * (size_t)(impl->gidx.n ? impl->gidx.n : 1));
|
|
impl->gidx.idx = malloc(sizeof(int) * (size_t)(impl->gidx.n ? impl->gidx.n : 1));
|
|
for (int k = 0; k < impl->gidx.n; k++) {
|
|
GroupName *gn = &((GroupName *)p.groupnames.data)[k];
|
|
impl->gidx.names[k] = gn->name; /* transfer ownership */
|
|
impl->gidx.idx[k] = gn->index;
|
|
}
|
|
da_free(&p.groupnames);
|
|
da_free(&p.backrefs);
|
|
free(stripped);
|
|
|
|
Pattern *pat = calloc(1, sizeof(Pattern));
|
|
pat->pattern = impl->pattern_copy;
|
|
pat->flags = all_flags;
|
|
pat->groups = prog.ngroups;
|
|
pat->groupindex = &impl->gidx;
|
|
pat->program = impl;
|
|
return pat;
|
|
}
|
|
|
|
void Pattern_free(Pattern *self) {
|
|
if (!self) return;
|
|
PatternImpl *impl = (PatternImpl *)self->program;
|
|
if (impl) {
|
|
free(impl->prog.insts);
|
|
da_free(&impl->prog.classnodes);
|
|
if (impl->has_nfa) { free(impl->nfa_prog.insts); da_free(&impl->nfa_prog.classnodes); }
|
|
nodepool_free_all(&impl->pool);
|
|
for (int k = 0; k < impl->gidx.n; k++) free(impl->gidx.names[k]);
|
|
free(impl->gidx.names);
|
|
free(impl->gidx.idx);
|
|
free(impl->pattern_copy);
|
|
free(impl);
|
|
}
|
|
free(self);
|
|
}
|
|
|
|
/* ================================================================
|
|
* 9. Input
|
|
* ================================================================ */
|
|
|
|
Input *Input_from_buffer(const uint8_t *buf, size_t len) {
|
|
Input *in = calloc(1, sizeof(Input));
|
|
in->buf = (uint8_t *)buf; in->len = len; in->owns_buf = 0;
|
|
return in;
|
|
}
|
|
/* Reads the rest of an already-open fd (positioned at 0) into a fresh,
|
|
* owned, growable buffer, used both as Input_from_file's fallback when
|
|
* mmap is not applicable or fails, and for non-regular files (pipes,
|
|
* FIFOs, process substitution, stdin) whose size cannot be known
|
|
* upfront. Takes ownership of fd (always closes it). */
|
|
static Input *read_fd_incrementally(int fd, PatternError *err) {
|
|
size_t cap = 1 << 16, len = 0;
|
|
uint8_t *buf = malloc(cap);
|
|
if (!buf) { close(fd); pattern_error_fill(err, NULL, "out of memory reading file", 0); return NULL; }
|
|
ssize_t n;
|
|
while ((n = read(fd, buf + len, cap - len)) > 0) {
|
|
len += (size_t)n;
|
|
if (len == cap) {
|
|
size_t ncap = cap * 2;
|
|
uint8_t *nbuf = realloc(buf, ncap);
|
|
if (!nbuf) { free(buf); close(fd); pattern_error_fill(err, NULL, "out of memory reading file", (int64_t)len); return NULL; }
|
|
buf = nbuf; cap = ncap;
|
|
}
|
|
}
|
|
close(fd);
|
|
Input *in = calloc(1, sizeof(Input));
|
|
if (!in) { free(buf); pattern_error_fill(err, NULL, "out of memory reading file", 0); return NULL; }
|
|
in->buf = buf; in->len = len; in->owns_buf = 1;
|
|
return in;
|
|
}
|
|
|
|
/* Prefers mmap() over reading the file into a malloc'd copy: the
|
|
* mapped pages are backed directly by the file and stay clean (never
|
|
* written), so the kernel can reclaim them under memory pressure and
|
|
* page them back in from disk later, instead of them being pinned for
|
|
* the whole match attempt the way a malloc'd copy would be; it also
|
|
* skips one whole redundant copy of the file's bytes (README.md
|
|
* "Implementation status" records the memory-per-input-byte
|
|
* multiplier this and the other fixes around it were measured
|
|
* against). Only applies to regular, non-empty, seekable files;
|
|
* anything else (a pipe, a FIFO, an empty file) falls back to
|
|
* read_fd_incrementally, exactly as before mmap support existed. */
|
|
Input *Input_from_file(const char *path, PatternError *err) {
|
|
int fd = open(path, O_RDONLY);
|
|
if (fd < 0) { pattern_error_fill(err, NULL, strerror(errno), 0); return NULL; }
|
|
struct stat st;
|
|
if (fstat(fd, &st) == 0 && S_ISREG(st.st_mode) && st.st_size > 0) {
|
|
void *addr = mmap(NULL, (size_t)st.st_size, PROT_READ, MAP_PRIVATE, fd, 0);
|
|
if (addr != MAP_FAILED) {
|
|
close(fd);
|
|
Input *in = calloc(1, sizeof(Input));
|
|
if (!in) { munmap(addr, (size_t)st.st_size); pattern_error_fill(err, NULL, "out of memory reading file", 0); return NULL; }
|
|
in->buf = addr; in->len = (size_t)st.st_size; in->owns_buf = 0; in->is_mmap = 1;
|
|
return in;
|
|
}
|
|
/* mmap failed (unusual: an overcommit-restricted system, a
|
|
* filesystem that does not support it, and so on); the fd is
|
|
* still open and positioned at 0, so fall back to reading it
|
|
* the ordinary way instead of failing outright. */
|
|
return read_fd_incrementally(fd, err);
|
|
}
|
|
/* Empty regular file, or not a regular file at all: size is 0 or
|
|
* unknowable upfront, and mmap does not apply either way. */
|
|
return read_fd_incrementally(fd, err);
|
|
}
|
|
void Input_free(Input *in) {
|
|
if (!in) return;
|
|
if (in->is_mmap) munmap(in->buf, in->len);
|
|
else if (in->owns_buf) free(in->buf);
|
|
free(in);
|
|
}
|
|
|
|
/* ================================================================
|
|
* 10. Match buffer construction (UTF-8 pre-decode, Section 8.4/9.3)
|
|
* ================================================================ */
|
|
|
|
typedef struct {
|
|
uint32_t *text; /* UTF8 mode only: owned, decoded code points */
|
|
const uint8_t *text8; /* BINARY/ASCII mode only: borrowed, points directly
|
|
* into the Input's own buffer, no widening copy;
|
|
* see text_at() and README.md "Implementation
|
|
* status" for why this matters at real scale. */
|
|
int64_t len;
|
|
int64_t *cp_to_byte; /* len+1 entries, NULL for non-UTF8 */
|
|
} MatBuf;
|
|
|
|
static int build_matbuf(Pattern *pat, Input *in, MatBuf *mb, PatternError *err) {
|
|
int mode = resolve_mode(pat->flags);
|
|
if (mode == MODE_UTF8) {
|
|
DArr cps; da_init(&cps, sizeof(uint32_t));
|
|
DArr offs; da_init(&offs, sizeof(int64_t));
|
|
size_t i = 0;
|
|
while (i < in->len) {
|
|
uint32_t cp;
|
|
int n = utf8_decode(in->buf, in->len, i, &cp);
|
|
if (n == 0) {
|
|
pattern_error_fill(err, NULL, "invalid UTF-8 in subject", (int64_t)i);
|
|
da_free(&cps); da_free(&offs);
|
|
return 0;
|
|
}
|
|
uint32_t *cpp = da_push(&cps);
|
|
int64_t *op = da_push(&offs);
|
|
if (!cpp || !op) {
|
|
pattern_error_fill(err, NULL, "out of memory decoding UTF-8 subject", (int64_t)i);
|
|
da_free(&cps); da_free(&offs);
|
|
return 0;
|
|
}
|
|
*cpp = cp; *op = (int64_t)i;
|
|
i += (size_t)n;
|
|
}
|
|
int64_t *sentinel = da_push(&offs);
|
|
if (!sentinel) {
|
|
pattern_error_fill(err, NULL, "out of memory decoding UTF-8 subject", (int64_t)in->len);
|
|
da_free(&cps); da_free(&offs);
|
|
return 0;
|
|
}
|
|
*sentinel = (int64_t)in->len;
|
|
mb->text = (uint32_t *)cps.data;
|
|
mb->text8 = NULL;
|
|
mb->cp_to_byte = (int64_t *)offs.data;
|
|
mb->len = (int64_t)cps.len;
|
|
} else {
|
|
/* No widening copy: a byte never exceeds 255, so BINARY/ASCII
|
|
* mode reads the Input's own buffer directly. This is the
|
|
* single biggest lever found profiling a real search with
|
|
* Massif (README.md "Implementation status"): the widened
|
|
* uint32_t copy this replaces cost as much memory as the
|
|
* input itself, four times over, for every mode that never
|
|
* actually needed code points wider than a byte. mb->text8 is
|
|
* a borrowed pointer (into Input, which outlives the MatBuf
|
|
* that borrows it), never freed here. */
|
|
mb->text = NULL;
|
|
mb->text8 = in->buf;
|
|
mb->len = (int64_t)in->len;
|
|
mb->cp_to_byte = NULL;
|
|
}
|
|
return 1;
|
|
}
|
|
static void free_matbuf(MatBuf *mb) { free(mb->text); free(mb->cp_to_byte); }
|
|
|
|
/* ================================================================
|
|
* 11. Matching driver and Pattern_ methods
|
|
* ================================================================ */
|
|
|
|
static void alloc_caps(int64_t **caps, int ngroups) {
|
|
*caps = malloc(2 * (size_t)(ngroups + 1) * sizeof(int64_t));
|
|
}
|
|
static void reset_caps(int64_t *caps, int ngroups) {
|
|
for (int k = 0; k < 2 * (ngroups + 1); k++) caps[k] = -1;
|
|
}
|
|
|
|
static void fill_match_from_caps(Match *out, Pattern *self, Input *string, int64_t pos, int64_t endpos,
|
|
int64_t *caps, int ngroups, MatBuf *mb) {
|
|
MatchSlots *ms = calloc(1, sizeof(MatchSlots));
|
|
ms->pos = caps; ms->ngroups = ngroups;
|
|
if (mb->cp_to_byte) {
|
|
ms->cp_to_byte = malloc((size_t)(mb->len + 1) * sizeof(int64_t));
|
|
memcpy(ms->cp_to_byte, mb->cp_to_byte, (size_t)(mb->len + 1) * sizeof(int64_t));
|
|
}
|
|
out->re = self; out->string = string; out->pos = pos; out->endpos = endpos;
|
|
out->slots = ms; out->lastindex = -1; out->lastgroup = NULL;
|
|
PatternImpl *impl = (PatternImpl *)self->program;
|
|
for (int g = ngroups; g >= 1; g--) {
|
|
if (caps[2 * g] >= 0) {
|
|
out->lastindex = g;
|
|
for (int k = 0; k < impl->gidx.n; k++)
|
|
if (impl->gidx.idx[k] == g) { out->lastgroup = impl->gidx.names[k]; break; }
|
|
break;
|
|
}
|
|
}
|
|
}
|
|
|
|
static int64_t clamp(int64_t v, int64_t lo, int64_t hi) { return v < lo ? lo : (v > hi ? hi : v); }
|
|
|
|
/* Allocated once per top-level Pattern_/re_ call (not once per start
|
|
* position tried, and, for Pattern_finditer/Pattern_split, not once
|
|
* per match found either): see the comment on MCtx.memo. NULL, with no
|
|
* behavior change beyond the missing optimization, whenever the
|
|
* pattern contains a backreference anywhere. */
|
|
static uint8_t *alloc_memo(Prog *pr, int64_t textlen) {
|
|
if (pr->has_backref) return NULL;
|
|
int64_t bits = (int64_t)pr->n * (textlen + 1);
|
|
int64_t bytes = (bits + 7) / 8;
|
|
return calloc((size_t)(bytes > 0 ? bytes : 1), 1);
|
|
}
|
|
|
|
/* mr[i] = how many consecutive positions starting at text[i] satisfy
|
|
* in's atom predicate (0 if text[i] itself does not), computed in one
|
|
* backward O(textlen) pass instead of redone forward, from scratch, at
|
|
* every position a caller asks about it. This is what makes
|
|
* OP_REPEAT1 O(1) per position instead of O(remaining run length),
|
|
* which otherwise stays quadratic across do_one/finditer/split's outer
|
|
* position loop even with run_memo's (pc, sp) memoization, because the
|
|
* counting scan is an internal C loop, not expressed as (pc, sp)
|
|
* recursive calls at all. int32_t bounds a single repeat's run length
|
|
* to ~2 billion, far past any input this build's fully-materializing
|
|
* Input can hold in memory in the first place. */
|
|
/* Shared by compute_maxrun/compute_next_prevmatch, exactly like
|
|
* text_at() above them for run() itself: text32 is non-NULL only in
|
|
* UTF8 mode, in which case it holds real decoded code points; in
|
|
* BINARY/ASCII mode text32 is NULL and text8 (the Input's own buffer,
|
|
* no widening copy) is used instead. */
|
|
static inline uint32_t buf_at(const uint32_t *text32, const uint8_t *text8, int64_t i) {
|
|
return text32 ? text32[i] : text8[i];
|
|
}
|
|
|
|
static int32_t *compute_maxrun(Prog *pr, Inst *in, const uint32_t *text32, const uint8_t *text8, int64_t len) {
|
|
int32_t *mr = malloc((size_t)(len + 1) * sizeof(int32_t));
|
|
if (!mr) return NULL;
|
|
mr[len] = 0;
|
|
Node *cls = (in->atomkind == 1) ? ((Node **)pr->classnodes.data)[in->data] : NULL;
|
|
for (int64_t i = len - 1; i >= 0; i--) {
|
|
uint32_t ch = buf_at(text32, text8, i);
|
|
int m;
|
|
if (in->atomkind == 0) m = char_eq(ch, in->data, pr->mode, pr->flags);
|
|
else if (in->atomkind == 2) m = !(ch == '\n' && !(pr->flags & DOTALL));
|
|
else m = class_match(cls, ch, pr->mode, pr->flags);
|
|
int64_t next = mr[i + 1];
|
|
mr[i] = m ? (int32_t)(next < INT32_MAX ? next + 1 : INT32_MAX) : 0;
|
|
}
|
|
return mr;
|
|
}
|
|
|
|
/* One table per OP_REPEAT1 instruction actually present in the
|
|
* program, indexed by instruction number; only worth the O(text
|
|
* length) memory per instruction for the multi-position operations
|
|
* (search/finditer/split) that would otherwise redo the scan at every
|
|
* position, so match/fullmatch (a single attempt) do not allocate it. */
|
|
static int32_t **alloc_repeat_maxrun(Prog *pr, const uint32_t *text32, const uint8_t *text8, int64_t len) {
|
|
int32_t **arr = calloc((size_t)pr->n, sizeof(int32_t *));
|
|
if (!arr) return NULL;
|
|
for (int i = 0; i < pr->n; i++)
|
|
if (pr->insts[i].op == OP_REPEAT1)
|
|
arr[i] = compute_maxrun(pr, &pr->insts[i], text32, text8, len);
|
|
return arr;
|
|
}
|
|
static void free_repeat_maxrun(Prog *pr, int32_t **arr) {
|
|
if (!arr) return;
|
|
for (int i = 0; i < pr->n; i++) free(arr[i]);
|
|
free(arr);
|
|
}
|
|
|
|
static int inst_atom_match(Prog *pr, Inst *in, uint32_t ch) {
|
|
switch (in->op) {
|
|
case OP_CHAR: return char_eq(ch, in->data, pr->mode, pr->flags);
|
|
case OP_ANY: return !(ch == '\n' && !(pr->flags & DOTALL));
|
|
case OP_CLASS: return class_match(((Node **)pr->classnodes.data)[in->data], ch, pr->mode, pr->flags);
|
|
default: return 0;
|
|
}
|
|
}
|
|
|
|
/* pm[i] = the largest position <= i where the instruction right after
|
|
* an OP_REPEAT1 matches, or -1 if none exists in [0, i]. Only computed
|
|
* when that next instruction is itself a single simple atom (OP_CHAR/
|
|
* OP_CLASS/OP_ANY); this is what lets OP_REPEAT1's greedy backtrack
|
|
* jump straight from "the largest k worth trying" to "the next
|
|
* smaller one worth trying" instead of visiting every k in between
|
|
* (see the comment at its one call site). Built in one forward
|
|
* O(textlen) pass instead of walked freshly, backward, from every
|
|
* position a caller asks about it. */
|
|
static int32_t *compute_next_prevmatch(Prog *pr, Inst *next, const uint32_t *text32, const uint8_t *text8, int64_t len) {
|
|
if (len <= 0) return NULL;
|
|
int32_t *pm = malloc((size_t)len * sizeof(int32_t));
|
|
if (!pm) return NULL;
|
|
int32_t last = -1;
|
|
for (int64_t i = 0; i < len; i++) {
|
|
if (inst_atom_match(pr, next, buf_at(text32, text8, i))) last = (int32_t)i;
|
|
pm[i] = last;
|
|
}
|
|
return pm;
|
|
}
|
|
static int32_t **alloc_repeat_nextpm(Prog *pr, const uint32_t *text32, const uint8_t *text8, int64_t len) {
|
|
int32_t **arr = calloc((size_t)pr->n, sizeof(int32_t *));
|
|
if (!arr) return NULL;
|
|
for (int i = 0; i < pr->n; i++) {
|
|
if (pr->insts[i].op != OP_REPEAT1 || !pr->insts[i].greedy) continue;
|
|
if (i + 1 >= pr->n) continue;
|
|
int nextop = pr->insts[i + 1].op;
|
|
if (nextop == OP_CHAR || nextop == OP_ANY || nextop == OP_CLASS)
|
|
arr[i] = compute_next_prevmatch(pr, &pr->insts[i + 1], text32, text8, len);
|
|
}
|
|
return arr;
|
|
}
|
|
static void free_repeat_nextpm(Prog *pr, int32_t **arr) {
|
|
if (!arr) return;
|
|
for (int i = 0; i < pr->n; i++) free(arr[i]);
|
|
free(arr);
|
|
}
|
|
|
|
static int do_one(Pattern *self, Input *string, int64_t pos, int64_t endpos, int anchored, int fullmatch, Match *out) {
|
|
MatBuf mb;
|
|
if (!build_matbuf(self, string, &mb, NULL)) return -1;
|
|
int64_t ep = (endpos < 0) ? mb.len : clamp(endpos, 0, mb.len);
|
|
int64_t p0 = clamp(pos, 0, mb.len);
|
|
PatternImpl *impl = (PatternImpl *)self->program;
|
|
int ng = impl->prog.ngroups;
|
|
|
|
/* concept.md 7.7: an eligible pattern is matched by the Pike VM
|
|
* instead, in one call rather than the per-position loop below (that
|
|
* loop's entire job, finding the next match starting no earlier than
|
|
* a given position, is pike_find's job too, done in a single linear
|
|
* pass instead of one attempt per candidate start). Skips memo/
|
|
* maxrun/nextpm entirely: those exist only to make the loop below
|
|
* linear (Section 7.5), and this path was never quadratic in the
|
|
* first place. */
|
|
if (impl->has_nfa) {
|
|
int64_t *ncaps; alloc_caps(&ncaps, ng);
|
|
int r = pike_find(&impl->nfa_prog, mb.text, mb.text8, p0, ep, anchored, fullmatch ? ep : -1, 0, ncaps);
|
|
if (r < 0) { free(ncaps); free_matbuf(&mb); return -1; }
|
|
if (r == 0) { free(ncaps); free_matbuf(&mb); return 0; }
|
|
fill_match_from_caps(out, self, string, p0, ep, ncaps, ng, &mb);
|
|
free_matbuf(&mb);
|
|
return 1;
|
|
}
|
|
|
|
int64_t *caps; alloc_caps(&caps, ng);
|
|
uint8_t *memo = alloc_memo(&impl->prog, ep);
|
|
/* Only search (anchored == 0) tries more than one position, so
|
|
* only search pays for the run-length precompute (see
|
|
* compute_maxrun); match/fullmatch's single attempt gets no
|
|
* benefit from it and skips the O(text length) memory. */
|
|
int32_t **maxrun = anchored ? NULL : alloc_repeat_maxrun(&impl->prog, mb.text, mb.text8, ep);
|
|
int32_t **nextpm = anchored ? NULL : alloc_repeat_nextpm(&impl->prog, mb.text, mb.text8, ep);
|
|
int found = 0;
|
|
int64_t last_start = anchored ? p0 : ep;
|
|
for (int64_t start = p0; start <= last_start && !found; start++) {
|
|
reset_caps(caps, ng);
|
|
caps[0] = start;
|
|
MCtx c; c.pr = &impl->prog; c.text = mb.text; c.text8 = mb.text8; c.len = ep; c.caps = caps;
|
|
c.depth_exceeded = 0; c.require_end = fullmatch ? ep : -1; c.sub_end = -1; c.memo = memo;
|
|
c.repeat_maxrun = maxrun; c.repeat_nextpm = nextpm; c.forbid_empty = 0;
|
|
found = run_memo(&c, 0, start, -1, -1, 0);
|
|
if (!found && c.depth_exceeded) { free(caps); free(memo); free_repeat_maxrun(&impl->prog, maxrun); free_repeat_nextpm(&impl->prog, nextpm); free_matbuf(&mb); return -1; }
|
|
}
|
|
free(memo);
|
|
free_repeat_maxrun(&impl->prog, maxrun);
|
|
free_repeat_nextpm(&impl->prog, nextpm);
|
|
if (!found) { free(caps); free_matbuf(&mb); return 0; }
|
|
fill_match_from_caps(out, self, string, p0, ep, caps, ng, &mb);
|
|
free_matbuf(&mb);
|
|
return 1;
|
|
}
|
|
|
|
int Pattern_match(Pattern *self, Input *string, int64_t pos, int64_t endpos, Match *out) {
|
|
return do_one(self, string, pos, endpos, 1, 0, out);
|
|
}
|
|
int Pattern_fullmatch(Pattern *self, Input *string, int64_t pos, int64_t endpos, Match *out) {
|
|
return do_one(self, string, pos, endpos, 1, 1, out);
|
|
}
|
|
int Pattern_search(Pattern *self, Input *string, int64_t pos, int64_t endpos, Match *out) {
|
|
return do_one(self, string, pos, endpos, 0, 0, out);
|
|
}
|
|
|
|
/* Shared by Pattern_finditer/Pattern_split's two find points each (the
|
|
* primary "next match starting no earlier than `start`" search, and the
|
|
* "anchored, non-empty" empty-match retry at the exact same position,
|
|
* concept.md 7.7): dispatches to whichever engine this Pattern uses.
|
|
* anchored=0 is an unanchored scan of [start, ep]; anchored=1 is a
|
|
* single attempt at exactly `start` (the retry's actual shape: it is
|
|
* not a forward scan, matching the original backtracking call it
|
|
* replaces). memo/maxrun/nextpm are the backtracking engine's per-scan
|
|
* state and are ignored when impl->has_nfa (Section 7.7: the Pike VM
|
|
* needs none of them, having never been quadratic in the first place).
|
|
* Returns 1 (found, caps filled), 0 (no match), or -1 (unrecoverable;
|
|
* caller must clean up and return -1, exactly as a depth_exceeded
|
|
* backtracking failure already required before this engine existed). */
|
|
static int find_next(PatternImpl *impl, MatBuf *mb, int64_t ep, int64_t start, int anchored, int forbid_empty,
|
|
uint8_t *memo, int32_t **maxrun, int32_t **nextpm, int64_t *caps, int ng) {
|
|
if (impl->has_nfa)
|
|
return pike_find(&impl->nfa_prog, mb->text, mb->text8, start, ep, anchored, -1, forbid_empty, caps);
|
|
int found = 0;
|
|
int64_t last = anchored ? start : ep;
|
|
for (int64_t s = start; s <= last && !found; s++) {
|
|
reset_caps(caps, ng);
|
|
caps[0] = s;
|
|
MCtx c; c.pr = &impl->prog; c.text = mb->text; c.text8 = mb->text8; c.len = ep; c.caps = caps;
|
|
c.depth_exceeded = 0; c.require_end = -1; c.sub_end = -1; c.memo = memo;
|
|
c.repeat_maxrun = maxrun; c.repeat_nextpm = nextpm; c.forbid_empty = forbid_empty;
|
|
found = run_memo(&c, 0, s, -1, -1, 0);
|
|
if (!found && c.depth_exceeded) return -1;
|
|
}
|
|
return found ? 1 : 0;
|
|
}
|
|
|
|
int Pattern_finditer(Pattern *self, Input *string, int64_t pos, int64_t endpos, MatchIterCb cb, void *ctx) {
|
|
MatBuf mb;
|
|
if (!build_matbuf(self, string, &mb, NULL)) return -1;
|
|
int64_t ep = (endpos < 0) ? mb.len : clamp(endpos, 0, mb.len);
|
|
int64_t start = clamp(pos, 0, mb.len);
|
|
PatternImpl *impl = (PatternImpl *)self->program;
|
|
int ng = impl->prog.ngroups;
|
|
/* One memo table for the whole scan: every position tried, across
|
|
* every match found, shares it (see the comment on MCtx.memo). A
|
|
* fact it records ("(pc, sp) cannot reach OP_MATCH") never becomes
|
|
* false later in the same scan, so nothing here ever needs to
|
|
* invalidate or reset it between matches. */
|
|
uint8_t *memo = alloc_memo(&impl->prog, ep);
|
|
int32_t **maxrun = alloc_repeat_maxrun(&impl->prog, mb.text, mb.text8, ep); /* see compute_maxrun */
|
|
int32_t **nextpm = alloc_repeat_nextpm(&impl->prog, mb.text, mb.text8, ep); /* see compute_next_prevmatch */
|
|
int count = 0;
|
|
while (start <= ep) {
|
|
int64_t *caps; alloc_caps(&caps, ng);
|
|
int found = find_next(impl, &mb, ep, start, 0, 0, memo, maxrun, nextpm, caps, ng);
|
|
if (found < 0) { free(caps); free(memo); free_repeat_maxrun(&impl->prog, maxrun); free_repeat_nextpm(&impl->prog, nextpm); free_matbuf(&mb); return -1; }
|
|
if (!found) { free(caps); break; }
|
|
Match m; memset(&m, 0, sizeof m);
|
|
MatBuf shallow = mb; shallow.cp_to_byte = mb.cp_to_byte; /* share for lookup, copied inside fill */
|
|
fill_match_from_caps(&m, self, string, start, ep, caps, ng, &shallow);
|
|
cb(ctx, &m);
|
|
count++;
|
|
int64_t mstart = caps[0], mend = caps[1];
|
|
Match_free(&m);
|
|
/* CPython's finditer, reverse engineered against a real
|
|
* interpreter (not documented): if the match just reported was
|
|
* empty, it additionally looks for a second, non-empty match at
|
|
* that exact same start position before moving on, and reports
|
|
* that one too if it exists (README.md/docs/API.md carry the
|
|
* probe cases this was derived from). OP_MATCH's forbid_empty
|
|
* check forces exactly that search: the same attempt, with the
|
|
* empty solution excluded, so ordinary backtracking finds the
|
|
* next (longer) alternative on its own if one exists. */
|
|
if (mend == mstart) {
|
|
int64_t *caps2; alloc_caps(&caps2, ng);
|
|
int found2 = find_next(impl, &mb, ep, mstart, 1, 1, memo, maxrun, nextpm, caps2, ng);
|
|
if (found2 < 0) { free(caps2); free(memo); free_repeat_maxrun(&impl->prog, maxrun); free_repeat_nextpm(&impl->prog, nextpm); free_matbuf(&mb); return -1; }
|
|
if (found2) {
|
|
Match m2; memset(&m2, 0, sizeof m2);
|
|
fill_match_from_caps(&m2, self, string, mstart, ep, caps2, ng, &shallow);
|
|
cb(ctx, &m2);
|
|
count++;
|
|
mend = caps2[1];
|
|
Match_free(&m2);
|
|
} else {
|
|
free(caps2);
|
|
}
|
|
}
|
|
start = (mend > mstart) ? mend : mstart + 1;
|
|
}
|
|
free(memo);
|
|
free_repeat_maxrun(&impl->prog, maxrun);
|
|
free_repeat_nextpm(&impl->prog, nextpm);
|
|
free_matbuf(&mb);
|
|
return count;
|
|
}
|
|
int Pattern_findall(Pattern *self, Input *string, int64_t pos, int64_t endpos, MatchIterCb cb, void *ctx) {
|
|
return Pattern_finditer(self, string, pos, endpos, cb, ctx);
|
|
}
|
|
|
|
/* Pattern_split's contract: cb is invoked once per element of the list
|
|
* Python's re.split() would return, in order, each time as a Match
|
|
* whose group 0 (via Match_group(m, NULL, 0, ...)) is that element's
|
|
* text; a "None" element (an unparticipated capturing group between
|
|
* two matches) reports via Match_group returning 0, matching how an
|
|
* unparticipated group reports on any other Match. */
|
|
static void emit_split_item(MatchIterCb cb, void *ctx, Pattern *self, Input *string, MatBuf *mb, int64_t s, int64_t e) {
|
|
int64_t *caps = malloc(2 * sizeof(int64_t));
|
|
caps[0] = s; caps[1] = e;
|
|
Match m; memset(&m, 0, sizeof m);
|
|
fill_match_from_caps(&m, self, string, 0, mb->len, caps, 0, mb);
|
|
cb(ctx, &m);
|
|
Match_free(&m);
|
|
}
|
|
|
|
int Pattern_split(Pattern *self, Input *string, int maxsplit, MatchIterCb cb, void *ctx) {
|
|
MatBuf mb;
|
|
if (!build_matbuf(self, string, &mb, NULL)) return -1;
|
|
int64_t ep = mb.len;
|
|
int64_t start = 0, seg_start = 0;
|
|
PatternImpl *impl = (PatternImpl *)self->program;
|
|
int ng = impl->prog.ngroups;
|
|
uint8_t *memo = alloc_memo(&impl->prog, ep); /* one table for the whole scan, see MCtx.memo */
|
|
int32_t **maxrun = alloc_repeat_maxrun(&impl->prog, mb.text, mb.text8, ep); /* see compute_maxrun */
|
|
int32_t **nextpm = alloc_repeat_nextpm(&impl->prog, mb.text, mb.text8, ep); /* see compute_next_prevmatch */
|
|
int n = 0;
|
|
while (start <= ep) {
|
|
if (maxsplit > 0 && n >= maxsplit) break;
|
|
int64_t *caps; alloc_caps(&caps, ng);
|
|
int found = find_next(impl, &mb, ep, start, 0, 0, memo, maxrun, nextpm, caps, ng);
|
|
if (found < 0) { free(caps); free(memo); free_repeat_maxrun(&impl->prog, maxrun); free_repeat_nextpm(&impl->prog, nextpm); free_matbuf(&mb); return -1; }
|
|
if (!found) { free(caps); break; }
|
|
emit_split_item(cb, ctx, self, string, &mb, seg_start, caps[0]);
|
|
for (int g = 1; g <= ng; g++) emit_split_item(cb, ctx, self, string, &mb, caps[2 * g], caps[2 * g + 1]);
|
|
n++;
|
|
seg_start = caps[1];
|
|
int64_t mstart = caps[0], mend = caps[1];
|
|
free(caps);
|
|
/* Same same-position retry as Pattern_finditer; see its comment. */
|
|
if (mend == mstart) {
|
|
if (maxsplit <= 0 || n < maxsplit) {
|
|
int64_t *caps2; alloc_caps(&caps2, ng);
|
|
int found2 = find_next(impl, &mb, ep, mstart, 1, 1, memo, maxrun, nextpm, caps2, ng);
|
|
if (found2 < 0) { free(caps2); free(memo); free_repeat_maxrun(&impl->prog, maxrun); free_repeat_nextpm(&impl->prog, nextpm); free_matbuf(&mb); return -1; }
|
|
if (found2) {
|
|
emit_split_item(cb, ctx, self, string, &mb, mstart, caps2[0]);
|
|
for (int g = 1; g <= ng; g++) emit_split_item(cb, ctx, self, string, &mb, caps2[2 * g], caps2[2 * g + 1]);
|
|
n++;
|
|
seg_start = caps2[1];
|
|
mend = caps2[1];
|
|
}
|
|
free(caps2);
|
|
}
|
|
}
|
|
start = (mend > mstart) ? mend : mstart + 1;
|
|
}
|
|
free(memo);
|
|
free_repeat_maxrun(&impl->prog, maxrun);
|
|
free_repeat_nextpm(&impl->prog, nextpm);
|
|
emit_split_item(cb, ctx, self, string, &mb, seg_start, ep);
|
|
free_matbuf(&mb);
|
|
return n;
|
|
}
|
|
|
|
static char *expand_template(const char *repl, Match *m, size_t *outlen) {
|
|
size_t cap = 64, len = 0;
|
|
char *out = malloc(cap);
|
|
size_t rl = strlen(repl);
|
|
for (size_t i = 0; i < rl; i++) {
|
|
char c = repl[i];
|
|
const char *piece = NULL; size_t piecelen = 0;
|
|
int consumed_one_char = 1;
|
|
if (c == '\\' && i + 1 < rl) {
|
|
char nc = repl[i + 1];
|
|
if (nc == 'g' && i + 2 < rl && repl[i + 2] == '<') {
|
|
size_t j = i + 3, start = j;
|
|
while (j < rl && repl[j] != '>') j++;
|
|
char *name = xstrndup(repl + start, j - start);
|
|
int idx = -1;
|
|
if (strspn(name, "0123456789") == strlen(name) && name[0]) idx = atoi(name);
|
|
else idx = groupindex_lookup((GroupIndex *)m->re->groupindex, name);
|
|
free(name);
|
|
Match_group(m, NULL, idx, &piece, &piecelen);
|
|
i = j;
|
|
} else if (isdigit((unsigned char)nc)) {
|
|
size_t j = i + 1, start = j;
|
|
while (j < rl && isdigit((unsigned char)repl[j]) && j - start < 2) j++;
|
|
char *ns = xstrndup(repl + start, j - start);
|
|
int idx = atoi(ns); free(ns);
|
|
Match_group(m, NULL, idx, &piece, &piecelen);
|
|
i = j - 1;
|
|
} else if (nc == 'n') { char cc = '\n'; if (len+1>=cap){cap*=2;out=realloc(out,cap);} out[len++]=cc; i++; continue; }
|
|
else if (nc == 't') { char cc = '\t'; if (len+1>=cap){cap*=2;out=realloc(out,cap);} out[len++]=cc; i++; continue; }
|
|
else if (nc == '\\') { char cc = '\\'; if (len+1>=cap){cap*=2;out=realloc(out,cap);} out[len++]=cc; i++; continue; }
|
|
else { piece = repl + i + 1; piecelen = 1; i++; }
|
|
} else {
|
|
if (len + 1 >= cap) { cap *= 2; out = realloc(out, cap); }
|
|
out[len++] = c;
|
|
consumed_one_char = 1; (void)consumed_one_char;
|
|
continue;
|
|
}
|
|
if (piece && piecelen) {
|
|
while (len + piecelen >= cap) { cap *= 2; out = realloc(out, cap); }
|
|
memcpy(out + len, piece, piecelen);
|
|
len += piecelen;
|
|
}
|
|
}
|
|
out[len] = 0;
|
|
*outlen = len;
|
|
return out;
|
|
}
|
|
|
|
typedef struct { char *buf; size_t len, cap; int64_t last_end; Input *string; const char *repl; MatchSubCb cb; void *cbctx; int n; int count_limit; } SubCtx;
|
|
|
|
static void subctx_append(SubCtx *sc, const void *data, size_t len) {
|
|
while (sc->len + len + 1 > sc->cap) { sc->cap = sc->cap ? sc->cap * 2 : 256; sc->buf = realloc(sc->buf, sc->cap); }
|
|
memcpy(sc->buf + sc->len, data, len);
|
|
sc->len += len;
|
|
}
|
|
static void sub_cb(void *ctx, const Match *m_const) {
|
|
SubCtx *sc = ctx;
|
|
Match *m = (Match *)m_const;
|
|
if (sc->count_limit > 0 && sc->n >= sc->count_limit) return;
|
|
int64_t mstart_byte = Match_start_byte(m, 0);
|
|
subctx_append(sc, sc->string->buf + sc->last_end, (size_t)(mstart_byte - sc->last_end));
|
|
if (sc->cb) {
|
|
char *piece = NULL; size_t plen = 0;
|
|
sc->cb(sc->cbctx, m, &piece, &plen);
|
|
if (piece) { subctx_append(sc, piece, plen); free(piece); }
|
|
} else {
|
|
size_t plen; char *piece = expand_template(sc->repl, m, &plen);
|
|
subctx_append(sc, piece, plen);
|
|
free(piece);
|
|
}
|
|
sc->last_end = Match_end_byte(m, 0);
|
|
sc->n++;
|
|
}
|
|
|
|
int Pattern_subn(Pattern *self, Input *string, const char *repl, MatchSubCb cb, void *ctx, int count, char **out, size_t *outlen, int *n) {
|
|
SubCtx sc; memset(&sc, 0, sizeof sc);
|
|
sc.string = string; sc.repl = repl; sc.cb = cb; sc.cbctx = ctx; sc.count_limit = count; sc.last_end = 0;
|
|
int total = Pattern_finditer(self, string, 0, -1, sub_cb, &sc);
|
|
if (total < 0) { free(sc.buf); return -1; }
|
|
subctx_append(&sc, string->buf + sc.last_end, (size_t)((int64_t)string->len - sc.last_end));
|
|
if (!sc.buf) sc.buf = malloc(1);
|
|
sc.buf[sc.len] = 0;
|
|
*out = sc.buf; *outlen = sc.len;
|
|
if (n) *n = sc.n;
|
|
return 1;
|
|
}
|
|
|
|
int Pattern_sub(Pattern *self, Input *string, const char *repl, MatchSubCb cb, void *ctx, int count, char **out, size_t *outlen) {
|
|
int n;
|
|
return Pattern_subn(self, string, repl, cb, ctx, count, out, outlen, &n);
|
|
}
|
|
|
|
/* ================================================================
|
|
* 12. Match accessors
|
|
* ================================================================ */
|
|
|
|
int Pattern_groupindex_lookup(Pattern *self, const char *name) {
|
|
PatternImpl *impl = (PatternImpl *)self->program;
|
|
return groupindex_lookup(&impl->gidx, name);
|
|
}
|
|
|
|
int Match_group(Match *self, const char *name_or_null, int index, const char **out, size_t *outlen) {
|
|
MatchSlots *ms = self->slots;
|
|
int g = index;
|
|
if (name_or_null) {
|
|
PatternImpl *impl = (PatternImpl *)self->re->program;
|
|
g = groupindex_lookup(&impl->gidx, name_or_null);
|
|
if (g < 0) return -1;
|
|
}
|
|
if (g < 0 || g > ms->ngroups) return -1;
|
|
int64_t s = ms->pos[2 * g], e = ms->pos[2 * g + 1];
|
|
if (s < 0 || e < 0) { *out = NULL; *outlen = 0; return 0; }
|
|
int64_t bs = ms->cp_to_byte ? ms->cp_to_byte[s] : s;
|
|
int64_t be = ms->cp_to_byte ? ms->cp_to_byte[e] : e;
|
|
*out = (const char *)(self->string->buf + bs);
|
|
*outlen = (size_t)(be - bs);
|
|
return 1;
|
|
}
|
|
int64_t Match_start(Match *self, int group) {
|
|
MatchSlots *ms = self->slots;
|
|
if (group < 0 || group > ms->ngroups) return -1;
|
|
return ms->pos[2 * group];
|
|
}
|
|
int64_t Match_end(Match *self, int group) {
|
|
MatchSlots *ms = self->slots;
|
|
if (group < 0 || group > ms->ngroups) return -1;
|
|
return ms->pos[2 * group + 1];
|
|
}
|
|
void Match_span(Match *self, int group, int64_t *start, int64_t *end) {
|
|
*start = Match_start(self, group); *end = Match_end(self, group);
|
|
}
|
|
int64_t Match_start_byte(Match *self, int group) {
|
|
MatchSlots *ms = self->slots;
|
|
if (group < 0 || group > ms->ngroups) return -1;
|
|
int64_t v = ms->pos[2 * group];
|
|
if (v < 0) return -1;
|
|
return ms->cp_to_byte ? ms->cp_to_byte[v] : v;
|
|
}
|
|
int64_t Match_end_byte(Match *self, int group) {
|
|
MatchSlots *ms = self->slots;
|
|
if (group < 0 || group > ms->ngroups) return -1;
|
|
int64_t v = ms->pos[2 * group + 1];
|
|
if (v < 0) return -1;
|
|
return ms->cp_to_byte ? ms->cp_to_byte[v] : v;
|
|
}
|
|
void Match_span_byte(Match *self, int group, int64_t *start, int64_t *end) {
|
|
*start = Match_start_byte(self, group); *end = Match_end_byte(self, group);
|
|
}
|
|
void Match_free(Match *self) {
|
|
if (!self || !self->slots) return;
|
|
MatchSlots *ms = self->slots;
|
|
free(ms->pos); free(ms->cp_to_byte); free(ms);
|
|
self->slots = NULL;
|
|
}
|
|
|
|
/* ================================================================
|
|
* 13. Module level functions, cache, escape (Section 9.1)
|
|
* ================================================================ */
|
|
|
|
#define CACHE_CAP 512
|
|
typedef struct { char *key; int keylen; int flags; Pattern *pat; } CacheEnt;
|
|
static CacheEnt g_cache[CACHE_CAP];
|
|
static int g_cache_n = 0;
|
|
|
|
void re_purge(void) {
|
|
for (int i = 0; i < g_cache_n; i++) { free(g_cache[i].key); Pattern_free(g_cache[i].pat); }
|
|
g_cache_n = 0;
|
|
}
|
|
static Pattern *cache_compile(const char *pattern, size_t len, int flags, PatternError *err) {
|
|
for (int i = 0; i < g_cache_n; i++)
|
|
if (g_cache[i].flags == flags && (size_t)g_cache[i].keylen == len && memcmp(g_cache[i].key, pattern, len) == 0)
|
|
return g_cache[i].pat;
|
|
Pattern *pat = re_compile(pattern, len, flags, err);
|
|
if (!pat) return NULL;
|
|
if (g_cache_n >= CACHE_CAP) re_purge();
|
|
g_cache[g_cache_n].key = xstrndup(pattern, len);
|
|
g_cache[g_cache_n].keylen = (int)len;
|
|
g_cache[g_cache_n].flags = flags;
|
|
g_cache[g_cache_n].pat = pat;
|
|
g_cache_n++;
|
|
return pat;
|
|
}
|
|
|
|
int re_match(const char *pattern, size_t len, int flags, Input *string, Match *out) {
|
|
Pattern *pat = cache_compile(pattern, len, flags, NULL); if (!pat) return -1;
|
|
return Pattern_match(pat, string, 0, -1, out);
|
|
}
|
|
int re_fullmatch(const char *pattern, size_t len, int flags, Input *string, Match *out) {
|
|
Pattern *pat = cache_compile(pattern, len, flags, NULL); if (!pat) return -1;
|
|
return Pattern_fullmatch(pat, string, 0, -1, out);
|
|
}
|
|
int re_search(const char *pattern, size_t len, int flags, Input *string, Match *out) {
|
|
Pattern *pat = cache_compile(pattern, len, flags, NULL); if (!pat) return -1;
|
|
return Pattern_search(pat, string, 0, -1, out);
|
|
}
|
|
int re_finditer(const char *pattern, size_t len, int flags, Input *string, MatchIterCb cb, void *ctx) {
|
|
Pattern *pat = cache_compile(pattern, len, flags, NULL); if (!pat) return -1;
|
|
return Pattern_finditer(pat, string, 0, -1, cb, ctx);
|
|
}
|
|
int re_findall(const char *pattern, size_t len, int flags, Input *string, MatchIterCb cb, void *ctx) {
|
|
return re_finditer(pattern, len, flags, string, cb, ctx);
|
|
}
|
|
int re_split(const char *pattern, size_t len, int flags, Input *string, int maxsplit, MatchIterCb cb, void *ctx) {
|
|
Pattern *pat = cache_compile(pattern, len, flags, NULL); if (!pat) return -1;
|
|
return Pattern_split(pat, string, maxsplit, cb, ctx);
|
|
}
|
|
int re_sub(const char *pattern, size_t len, int flags, Input *string, const char *repl, MatchSubCb cb, void *ctx, int count, char **out, size_t *outlen) {
|
|
Pattern *pat = cache_compile(pattern, len, flags, NULL); if (!pat) return -1;
|
|
return Pattern_sub(pat, string, repl, cb, ctx, count, out, outlen);
|
|
}
|
|
int re_subn(const char *pattern, size_t len, int flags, Input *string, const char *repl, MatchSubCb cb, void *ctx, int count, char **out, size_t *outlen, int *n) {
|
|
Pattern *pat = cache_compile(pattern, len, flags, NULL); if (!pat) return -1;
|
|
return Pattern_subn(pat, string, repl, cb, ctx, count, out, outlen, n);
|
|
}
|
|
void re_escape(const char *in, size_t len, char **out, size_t *outlen) {
|
|
char *buf = malloc(len * 2 + 1);
|
|
size_t o = 0;
|
|
for (size_t i = 0; i < len; i++) {
|
|
unsigned char ch = (unsigned char)in[i];
|
|
int is_word = (ch >= 'a' && ch <= 'z') || (ch >= 'A' && ch <= 'Z') || (ch >= '0' && ch <= '9') || ch == '_';
|
|
if (!is_word && ch < 0x80) buf[o++] = '\\';
|
|
buf[o++] = (char)ch;
|
|
}
|
|
buf[o] = 0;
|
|
*out = buf; *outlen = o;
|
|
}
|