From 8f6afd6cd41f4c03562c714e004c9dcb3a16f7db Mon Sep 17 00:00:00 2001 From: retoor Date: Mon, 14 Sep 2026 05:56:16 +0000 Subject: [PATCH] Add regexx: a single-file C regex interpreter with Python re semantics concept.md is the full design document: the objective (Python re parity plus binary/ASCII/UTF-8 modes, gigabyte-scale input, single C file), the automata-theory argument for why unrestricted backreferences/lookaround are incompatible with strict single-pass constant memory, the resulting two-engine architecture, the exact Python-mirroring naming convention, and a full comparison against POSIX regex.h for C-background readers. regexx.c/regexx.h are the v1 implementation: parser, compiler to a Pike/backtracking-style bytecode, and a single recursive backtracking engine covering the pattern syntax and operations listed in README.md, validated against CPython's own re module output (tests/), clean under AddressSanitizer/UBSan, and stress-tested (50MB simple-quantifier match, graceful failure rather than a crash on complex repeats over large input, clean rejection of every intentionally unsupported construct). Also included: examples/rxgrep.c (a small grep-like program exercising all three data modes and the substitution API), the Makefile, the MIT LICENSE, and docs/API.md, an exhaustive reference for every type, flag, and function's exact return-value and memory-ownership convention, checked against the current source and against a real CPython interpreter rather than against memory. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01EjuMk8kY9SDus1wWe2K9xY --- .gitignore | 6 + CLAUDE.md | 3 + LICENSE | 21 + Makefile | 57 ++ README.md | 189 ++++++ concept.md | 345 ++++++++++ docs/API.md | 279 ++++++++ examples/rxgrep.c | 159 +++++ regexx.c | 1626 +++++++++++++++++++++++++++++++++++++++++++++ regexx.h | 132 ++++ tests/cases.py | 129 ++++ tests/gen.py | 181 +++++ tests/harness.c | 156 +++++ tests/harness.h | 36 + tests/main.c | 10 + 15 files changed, 3329 insertions(+) create mode 100644 .gitignore create mode 100644 CLAUDE.md create mode 100644 LICENSE create mode 100644 Makefile create mode 100644 README.md create mode 100644 concept.md create mode 100644 docs/API.md create mode 100644 examples/rxgrep.c create mode 100644 regexx.c create mode 100644 regexx.h create mode 100644 tests/cases.py create mode 100644 tests/gen.py create mode 100644 tests/harness.c create mode 100644 tests/harness.h create mode 100644 tests/main.c diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..b421d7e --- /dev/null +++ b/.gitignore @@ -0,0 +1,6 @@ +*.o +*.a +rxgrep +tests/generated_tests.c +/.claude/ +*.dSYM/ diff --git a/CLAUDE.md b/CLAUDE.md new file mode 100644 index 0000000..bcbc89a --- /dev/null +++ b/CLAUDE.md @@ -0,0 +1,3 @@ +Write always scientific when documenting. +No claude promotion. +No EM dashes. diff --git a/LICENSE b/LICENSE new file mode 100644 index 0000000..8e1e6a4 --- /dev/null +++ b/LICENSE @@ -0,0 +1,21 @@ +MIT License + +Copyright (c) 2026 retoor + +Permission is hereby granted, free of charge, to any person obtaining a copy +of this software and associated documentation files (the "Software"), to deal +in the Software without restriction, including without limitation the rights +to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +copies of the Software, and to permit persons to whom the Software is +furnished to do so, subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +SOFTWARE. diff --git a/Makefile b/Makefile new file mode 100644 index 0000000..84b13a1 --- /dev/null +++ b/Makefile @@ -0,0 +1,57 @@ +# Makefile for regexx: a single-file C regex interpreter with Python +# `re` semantics. See concept.md for the design and README.md for the +# implementation status. + +CC ?= cc +CSTD ?= -std=c11 +WARN ?= -Wall -Wextra +OPT ?= -O2 +CFLAGS ?= $(CSTD) $(WARN) $(OPT) -g +LDLIBS ?= + +PREFIX ?= /usr/local + +AR ?= ar + +.PHONY: all lib example test check clean install fuzz-smoke + +all: lib example + +# ---- library ----------------------------------------------------------- +libregexx.a: regexx.o + $(AR) rcs $@ $^ + +regexx.o: regexx.c regexx.h + $(CC) $(CFLAGS) -c regexx.c -o $@ + +lib: libregexx.a + +# ---- example ------------------------------------------------------------- +example: rxgrep + +rxgrep: examples/rxgrep.c libregexx.a regexx.h + $(CC) $(CFLAGS) -I. -o $@ examples/rxgrep.c libregexx.a $(LDLIBS) + +# ---- tests ----------------------------------------------------------------- +# tests/generated_tests.c is generated from tests/cases.py using +# CPython's own `re` module as ground truth (concept.md Section 11). +tests/generated_tests.c: tests/cases.py tests/gen.py + python3 tests/gen.py + +test: tests/generated_tests.c regexx.c regexx.h tests/harness.c tests/harness.h tests/main.c + $(CC) $(CFLAGS) -I. -o /tmp/regexx_test regexx.c tests/harness.c tests/generated_tests.c tests/main.c $(LDLIBS) + /tmp/regexx_test + +# check also runs the suite under AddressSanitizer + UBSan. +check: tests/generated_tests.c + $(CC) $(CSTD) $(WARN) -O0 -g -fsanitize=address,undefined -I. -o /tmp/regexx_test_san \ + regexx.c tests/harness.c tests/generated_tests.c tests/main.c $(LDLIBS) + /tmp/regexx_test_san + +install: libregexx.a regexx.h + install -d $(DESTDIR)$(PREFIX)/lib $(DESTDIR)$(PREFIX)/include + install -m644 libregexx.a $(DESTDIR)$(PREFIX)/lib/ + install -m644 regexx.h $(DESTDIR)$(PREFIX)/include/ + +clean: + rm -f regexx.o libregexx.a rxgrep tests/generated_tests.c diff --git a/README.md b/README.md new file mode 100644 index 0000000..c3b85bb --- /dev/null +++ b/README.md @@ -0,0 +1,189 @@ +# regexx + +A single-file C regular expression interpreter that reproduces the observable +behavior of Python's `re` module, including its exact identifier names +(`Pattern`, `Match`, `re_compile`, `re_sub`, `IGNORECASE`, and so on; see +Section 9 of `concept.md`), and that additionally supports binary data +(arbitrary byte streams, including embedded NUL) and UTF-8 text alongside +plain ASCII. + +The design rationale, the algorithmic trade-offs, and a full accounting of +what is and is not carried over from Python's `re` and from POSIX's native +`regex.h` are recorded in [`concept.md`](concept.md). This file documents the +implementation that exists today at a glance; [`docs/API.md`](docs/API.md) is +the exhaustive reference (every type, every flag, every function's exact +return-value and memory-ownership convention, checked against a real CPython +interpreter, not against memory). + +## Implementation status + +This is a v1 implementation. It is a complete, tested engine for the pattern +syntax and operations listed below, executed by a single recursive +backtracking engine (`concept.md` Section 7.3) over a fully materialized copy +of the input. + +**It does not yet implement the streaming, bounded-memory regular engine of +`concept.md` Section 7.2.** `Input` (the abstraction over "a source of +chunks", `concept.md` 9.2) is implemented, and `Input_from_file` reads a +whole file into memory before matching. Every public function signature is +already exactly what the streaming design in `concept.md` specifies, so the +non-streaming implementation underneath a given call can be replaced later +without changing any caller. Concretely, today: + +- A pattern with no backreference and no unbounded-width lookahead runs in + linear time and, for the common case of a single character, class, or `.` + repeated by a quantifier, in *O(1)* recursion depth regardless of input + size (the `OP_REPEAT1` fast path). A repeated *compound* sub-pattern (for + example `(ab)*`) still recurses once per repetition, bounded by a + configurable depth limit (`MAX_DEPTH` in `regexx.c`, currently 60000); past + that limit, matching fails with a reported error rather than a stack + overflow or a wrong answer. +- A pattern with a backreference or an unbounded-width lookahead can, like + CPython's own `_sre`, take worst-case exponential time on an adversarial + input (`concept.md` Section 5, 13.2); this is the same catastrophic + backtracking (ReDoS) behavior CPython itself exhibits on such patterns, not + a regression specific to this engine. + +### Pattern syntax supported + +Literals; `.` (with `DOTALL`); character classes with ranges, negation, and +`\d \D \w \W \s \S`; `\b \B`; anchors `^ $ \A \Z` (with `MULTILINE`); +quantifiers `* + ? {m,n} {m,} {,n} {m}`, greedy and lazy; possessive +quantifiers `*+ ++ ?+ {m,n}+`; groups `(...) (?:...) (?P...)`; +alternation `|`; backreferences `\1`-`\99`, `(?P=name)`, `\g`, +`\g`; lookahead `(?=...) (?!...)`; fixed-width lookbehind +`(?<=...) (?...)`; comments `(?#...)`; global +inline flags `(?aiLmsux)` at the start of a pattern; escapes +`\n \r \t \f \v \a`, octal `\0`-prefixed escapes, `\xhh`, `\uxxxx`, +`\Uxxxxxxxx`; flags `IGNORECASE`, `MULTILINE`, `DOTALL`, `VERBOSE`, `ASCII`, +`LOCALE` (all with observable effect; see `docs/API.md` Section 2 for exactly +what each one does), plus `UNICODE` and `DEBUG` (accepted for source +compatibility with Python, currently no-ops in this build). + +Rejected at compile time with a clear `PatternError`, rather than +mis-parsed: conditional groups `(?(id)yes|no)`, scoped inline flags +`(?flags:...)`, `\N{NAME}` named code points, and POSIX bracket classes +`[:alpha:]` (which are not part of Python `re` at all, `concept.md` 14.3). +Variable-width lookbehind is also rejected at compile time, matching +CPython. + +### Operations supported + +`Pattern_match/fullmatch/search/finditer/findall/split/sub/subn/free`, +`Pattern_groupindex_lookup`, +`Match_group/start/end/span/start_byte/end_byte/span_byte/free`, +`re_compile/match/fullmatch/search/finditer/findall/split/sub/subn/escape/purge`, +`PatternError_free`. See `regexx.h` for exact signatures, `docs/API.md` for +the full reference (return values, memory ownership, exact Python +correspondence for each one), and `concept.md` Section 9 for the naming +convention they follow. + +### Known deviations from `concept.md` and from CPython, beyond the items above + +- `\w`, `\s`, `IGNORECASE` case folding, and `\d` in `UTF8` mode are backed + by glibc's `wctype.h` functions under the `C.utf8` locale, not by a + hand-generated Unicode table (`concept.md` 13.3 anticipated a reduced + static table; using the C library's own tables turned out to be simpler + and more complete, at the cost of depending on the platform's Unicode + version rather than a pinned one). +- `lastindex`/`lastgroup` report the highest-numbered capturing group that + participated in the match, which coincides with CPython's "most recently + closed group" rule for straightforward patterns but can differ from it in + pathological cases (nested alternation re-executing a lower-numbered group + after a higher one). Not exercised by the test suite; documented here + rather than silently accepted. +- `Match_free`, `PatternError_free`, `Input_from_buffer`, `Input_from_file`, + and `Input_free` have no Python counterpart and are not mentioned in + `concept.md`'s API surface; they exist because C has no garbage collector. + `Pattern_sub`/`Pattern_subn` take the replacement template and the + callback as two separate parameters rather than one polymorphic argument, + for the same reason (`concept.md` 9.4 already anticipates and justifies + this one). +- Python's `Match.start(group)`/`.end(group)` raise `IndexError` for an + invalid group number and return `-1` only for a valid group that did not + participate; `Match_start`/`Match_end` return `-1` for both cases, since C + has no exception to raise. `Match_group` does distinguish them (`-1` for + no such group, `0` for an unparticipated one), see `docs/API.md` Section + 3.23. +- `Pattern.groupindex` has no enumeration function in this build, only + `Pattern_groupindex_lookup(pattern, name)`; there is no way to list every + name a compiled pattern defines without already knowing what to look for. + +## Building + +Requires a C11 compiler and, for the test suite, Python 3 (used only to +generate ground truth from CPython's own `re` module, `concept.md` Section +11; the library itself has no runtime dependency beyond the C standard +library and `libc`'s `wctype.h`/`locale.h`). + +```sh +make # builds libregexx.a and the rxgrep example +make test # regenerates tests/generated_tests.c from Python `re` + # ground truth and runs the full suite +make check # same, under AddressSanitizer + UndefinedBehaviorSanitizer +make clean +``` + +`make install` installs `libregexx.a` and `regexx.h` under `PREFIX` +(default `/usr/local`). + +## Using the library + +```c +#include "regexx.h" +#include + +const char *pattern = "(\\w+)@(\\w+)"; +Pattern *pat = re_compile(pattern, strlen(pattern), UTF8, NULL); +Input *in = Input_from_buffer((const uint8_t *)"user@host", strlen("user@host")); + +Match m; +if (Pattern_search(pat, in, 0, -1, &m) == 1) { + const char *g; size_t glen; + Match_group(&m, NULL, 1, &g, &glen); /* g/glen -> "user" */ + Match_free(&m); +} + +char *out; size_t outlen; +Pattern_sub(pat, in, "\\2@\\1", NULL, NULL, 0, &out, &outlen); /* "host@user" */ +free(out); + +Pattern_free(pat); +Input_free(in); +``` + +`flags` to `re_compile` combine a data mode, exactly one of `BINARY`, +`ASCII`, or `UTF8` (`concept.md` 9.3), with any of the Python-named flags +(`IGNORECASE`, `MULTILINE`, `DOTALL`, `VERBOSE`, `ASCII` as a flag also +forces ASCII-only `\w`/`\s`/`\d` inside `UTF8` mode, `LOCALE`, `DEBUG`). + +## Example: rxgrep + +`examples/rxgrep.c` is a small grep-like program built on the library, +demonstrating all three data modes and both the matching and substitution +API: + +```sh +./rxgrep -in 'hello' file.txt # case-insensitive, line numbers +./rxgrep -m utf8 -o '\w+' file.txt # print every UTF-8 word, one per line +./rxgrep -c 'error' log.txt # count matching lines +./rxgrep -m binary 'a.c' data.bin # match raw bytes, embedded NUL included +./rxgrep -m utf8 --sub 'REDACTED' '\d{3}-\d{4}' file.txt +``` + +Run `./rxgrep --help` for the full option list. + +## Testing + +`tests/cases.py` lists pattern/subject/operation triples. `tests/gen.py` +computes each one's expected result with CPython's own `re` module and +writes `tests/generated_tests.c`, which is then compiled against `regexx.c` +and checked. This is a direct implementation of the strategy `concept.md` +Section 11 describes: conformance is measured against what CPython actually +does, not against a re-derived reading of its documentation. `make check` +additionally runs the suite under AddressSanitizer and +UndefinedBehaviorSanitizer. + +## License + +MIT. See [`LICENSE`](LICENSE). diff --git a/concept.md b/concept.md new file mode 100644 index 0000000..77276b8 --- /dev/null +++ b/concept.md @@ -0,0 +1,345 @@ +# Concept: A Single File C Regex Interpreter with Python `re` Semantics + +## 1. Objective + +The objective is a regex interpreter, implemented as a single C source file, that reproduces the observable behavior of Python's `re` module (pattern syntax, flags, and the `match`, `search`, `fullmatch`, `findall`, `finditer`, `split`, `sub`, and `subn` operations) while being usable on inputs of arbitrary size, including multi-gigabyte files, without holding the entire input in memory. The implementation is required to operate in three data modes: binary (arbitrary byte streams), ASCII text, and UTF-8 text. Simplicity of the C code takes priority over raw execution speed. + +Sections 2 through 4 record what "full Python `re` support" concretely means. Section 5 records why an unrestricted single pass, constant memory implementation of that full feature set is not mathematically possible, and states the boundary precisely. Sections 6 through 12 describe the architecture chosen to get as close to the objective as that boundary allows, in the simplest C design found. Section 13 lists the residual gaps against CPython's `re`. Section 14 records, for a reader coming from C rather than Python, the respects in which the native C regular expression facility (POSIX `regex.h`) differs from Python `re`, and therefore from this design, none of which this design adopts beyond what Section 1 already asks for. + +## 2. Reference Semantics: Python `re` Pattern Syntax + +The engine parses the following syntax, matching CPython's documented and observed behavior. + +### 2.1 Atoms and literals +- Literal characters (bytes in binary/ASCII mode, decoded code points in UTF-8 mode). +- `.` matches any character except `\n`, or any character at all under `DOTALL`. +- `\` followed by a non-alphanumeric character is that literal character. +- Escapes: `\n \r \t \f \v \a \0`, octal `\ooo`, hexadecimal `\xhh`, `\uxxxx`, `\Uxxxxxxxx`, and named code points `\N{NAME}` (requires a Unicode name table; see 13.3). + +### 2.2 Character classes +- `[...]` with ranges (`a-z`), negation (`[^...]`), and literal `]`, `-`, `^` when escaped or positionally safe, matching CPython's class parser exactly (including that `]` as the first class member is literal). +- Shorthand classes `\d \D \w \W \s \S`, each with an ASCII definition and a Unicode definition, selected by mode and by the `ASCII`/`UNICODE` flags exactly as CPython selects them for `str` versus `bytes` patterns. +- `\b` and `\B` (word boundary and non boundary), defined with the same "word character" set as `\w` in the active mode. + +### 2.3 Anchors +- `^` and `$`: string boundaries by default; line boundaries under `MULTILINE`. +- `\A` and `\Z`: string boundaries, unaffected by `MULTILINE`. + +### 2.4 Quantifiers +- Greedy: `* + ? {m,n} {m,} {,n} {m}`. +- Lazy: `*? +? ?? {m,n}?`. +- Possessive (CPython 3.11 and later): `*+ ++ ?+ {m,n}+`. +- Atomic groups (CPython 3.11 and later): `(?>...)`. + +### 2.5 Groups and grouping constructs +- `(...)` capturing group, numbered left to right by opening parenthesis. +- `(?:...)` non capturing group. +- `(?P...)` named capturing group; `(?P=name)` named backreference; `\1`..`\99` numbered backreference; `\g` and `\g<1>` backreference forms usable inside a pattern as well as in a replacement string. +- `(?#...)` comment, discarded at parse time. +- `(?=...)`, `(?!...)`: lookahead, positive and negative, of unrestricted width. +- `(?<=...)`, `(?`, `\g<1>`, `\1`, and literal backslash escapes, or a callback invoked once per match with a match record and expected to return replacement bytes/text (the C equivalent of a Python callable, see 9.4). +- `escape(string)`: backslash escaping of all characters outside `[A-Za-z0-9_]` in the same way `re.escape` does since Python 3.7 (that version narrowed the escaped set relative to earlier Python releases; the engine follows the narrowed, current set). +- Match record fields: `group(n)`, `group(name)`, `groups()`, `groupdict()`, `start(n)`, `end(n)`, `span(n)`, `lastindex`, `lastgroup`. + +## 4. Feature Compatibility Table + +| Feature | Status | +|---|---| +| Literals, classes, anchors, quantifiers (greedy/lazy) | Full | +| Alternation, grouping, named groups | Full | +| Backreferences (pattern and replacement) | Full, bounded (Section 5) | +| Lookahead, fixed width | Full | +| Lookahead, unbounded width | Full, bounded (Section 5), same window as backreferences | +| Lookbehind, fixed width | Full | +| Lookbehind, variable width | Rejected at compile time, as in CPython | +| Conditional groups `(?(id)yes|no)` | Full | +| Possessive quantifiers, atomic groups | Full (`OP_ATOMIC`, Section 7.1) | +| Inline and scoped flags | Full | +| `\N{NAME}` named code points | Partial (Section 13.3) | +| Full Unicode `\w`/`\s`/`IGNORECASE` case folding | Partial (Section 13.3) | +| `LOCALE` flag beyond the "C" locale | Not supported (Section 13.4) | +| Streaming over unbounded input | Full for the regular subset, bounded window for backreferences/lookaround (Section 5) | + +## 5. The Single Pass, Bounded Memory Constraint + +This section states a limit that shapes the rest of the design, so it is recorded before the architecture. + +A pattern language restricted to literals, classes, anchors, quantifiers, grouping, and alternation is a regular language. Regular languages are recognized by a finite automaton, and a finite automaton processes an input stream in one pass, in time linear in the input length, using memory bounded by the automaton's state count, independent of input length. Thompson's construction (converting a pattern to a nondeterministic finite automaton) and its simulation without backtracking (the approach used by `grep -E`, `awk`, RE2, and Rob Pike's regular expression virtual machine) achieve exactly this, including for `findall` style capture extraction. + +Backreferences (`\1`, `(?P=name)`) break this property. No fixed size finite automaton recognizes a language defined with a backreference in general, because such an automaton would need to remember an arbitrarily long previously matched substring verbatim and compare it later, and a finite automaton has, by definition, only finitely many states with which to do so. The precise formal result here is a combined complexity result: deciding whether a string matches a pattern is NP-hard when both the pattern and the string are counted as part of the problem input. That result does not, by itself, say anything about a fixed pattern, compiled once, matched against a growing string, which is this design's actual situation; for a fixed pattern the practically relevant obstacle is different and better known by name, worst case exponential backtracking time in the length of the string, the mechanism behind catastrophic backtracking (commonly called ReDoS) in every production backtracking engine, including CPython's own `_sre`. Unbounded width lookahead and lookbehind create the same obstacle for a related reason: resolving them can require holding an unbounded span of the input, forward or backward, before the assertion's truth value is known. This is a property of the language class and of the evaluation strategy required to decide it, not of any particular implementation choice, and it means a literal reading of "full Python `re` support" and "does not remain in memory or hold the input" are mutually exclusive whenever a pattern actually uses a backreference or an unbounded width lookaround on unbounded input. + +The engine resolves this by splitting execution into two engines sharing one bytecode format: + +1. **Regular engine.** Any compiled pattern that contains no backreference and no lookaround whose body has unbounded width is executed by a Thompson/Pike style simulation: single pass, one buffered chunk of input at a time, memory bounded by the number of program instructions multiplied by the number of capture groups, independent of input length. This covers the large majority of patterns used in practice, including nested quantifiers, alternation, and fixed width lookaround. + +2. **Bounded backtracking engine.** Any compiled pattern that contains a backreference or an unbounded width lookahead (fixed width lookaround of either polarity always stays in the regular engine, per 7.1) is executed by a backtracking simulation over a sliding window of the input, of a fixed configurable size (default 1 MiB, see 7.3). This engine gives exact CPython semantics as long as the text a backreference or lookahead needs to inspect fits inside the window relative to the current match attempt. If it does not, the engine reports a recoverable error (`ERANGE`-style status) identifying the offending construct and offset, rather than silently returning a wrong answer or reading the whole file into memory. This is the same trade every production streaming text tool with backreference support makes; the engine documents the bound instead of hiding it. + +This split is decided once, at compile time, from the parsed pattern, before any input is read. A caller who needs a hard guarantee of bounded memory on arbitrary input can inspect the compiled pattern's engine selection before running it. + +## 6. Architecture Overview + +Single C file, four sections in this order, each independent of the ones after it: + +1. **Parser**: pattern text to abstract syntax tree (AST). Recursive descent, one function per grammar production (`parse_alt`, `parse_concat`, `parse_repeat`, `parse_atom`), matching the structure of the grammar in Section 2 directly, so the parser can be read as an executable grammar. +2. **Compiler**: AST to bytecode, by direct recursive translation (Thompson's construction), one code generation function per AST node kind. Same bytecode format is emitted regardless of which of the two engines (5.1/5.2) will run it; the engine choice is a separate flag computed from the AST (does it contain `OP_BACKREF`, or an `OP_LOOKAHEAD` whose body width is unbounded; CPython already forces `OP_LOOKBEHIND` to be fixed width, Section 2.5, so lookbehind never contributes to this flag). +3. **Engines**: the regular engine (Pike VM) and the bounded backtracking engine (recursive backtracking over the sliding window), described in Section 7. +4. **Public API**: the Python `re` equivalent entry points, described in Section 9. + +Keeping parser, compiler, and both engines as pure functions over explicit structs (no hidden global state except one user supplied allocator, see 8.1) is what keeps the single file simple to read despite covering the full grammar. + +## 7. Execution Engines + +### 7.1 Bytecode + +One flat instruction set, an array of tagged structs, used by both engines: + +``` +enum opcode { + OP_CHAR, /* match one literal byte/codepoint */ + OP_CLASS, /* match one byte/codepoint against a class */ + OP_ANY, /* match one byte/codepoint, DOTALL-sensitive */ + OP_SPLIT, /* two continuations (alternation, quantifiers) */ + OP_JMP, + OP_SAVE, /* record current offset into capture slot N */ + OP_MATCH, + OP_ASSERT, /* zero-width: ^ $ \b \B \A \Z */ + OP_BACKREF, /* forces bounded backtracking engine */ + OP_LOOKAHEAD, /* sub-program, zero-width, polarity flag */ + OP_LOOKBEHIND, /* sub-program, fixed width, zero-width, polarity*/ + OP_ATOMIC, /* sub-program, consuming, discards choice points */ + OP_COND, /* branch on whether group N has matched */ +}; + +struct inst { enum opcode op; int32_t x, y; uint32_t data; }; +``` + +This is the same instruction shape used by Pike's virtual machine and by RE2's bytecode; reusing it rather than inventing a new one is what keeps the compiler small (roughly one `case` per AST node). + +`OP_LOOKAHEAD` and `OP_LOOKBEHIND` each carry, alongside the sub-program pointer and polarity bit, a compile time computed width: a concrete integer for `OP_LOOKBEHIND` (CPython requires this to exist and be fixed, Section 2.5) and either a concrete integer or an explicit "unbounded" marker for `OP_LOOKAHEAD`. Only an unbounded width `OP_LOOKAHEAD`, together with `OP_BACKREF`, forces engine selection to the bounded backtracking engine (7.3); a fixed width `OP_LOOKAHEAD` or `OP_LOOKBEHIND`, of either polarity, is executed by the regular engine (7.2) using a peek buffer sized to exactly that width, never the full window `W`. + +Atomic groups (`(?>...)`) and possessive quantifiers (`*+ ++ ?+ {m,n}+`) both compile to `OP_ATOMIC`, which is the only new opcode either needs: a possessive quantifier is first desugared, at compile time, into the atomic group wrapping its ordinary greedy form (`X*+` becomes `(?>X*)`, `X{m,n}+` becomes `(?>X{m,n})`, and so on), so the compiler and both engines only ever have to implement `OP_ATOMIC` once. `OP_ATOMIC` runs its sub-program to its single highest priority success (one priority ordered thread simulation restricted to the sub-program in the regular engine, 7.2; one recursive match attempt in the bounded engine, 7.3), advances the current position past whatever it consumed, and then permanently discards every choice point created while matching the sub-program, so that if matching fails later in the overall pattern, the engine never backtracks into the atomic group looking for a different internal match, which is the defining behavior of both constructs in CPython. Unlike `OP_LOOKAHEAD`, `OP_ATOMIC` never forces the bounded backtracking engine, regardless of the width of its body: it advances the stream position as it matches, so, unlike a zero-width assertion, it never needs to hold matched text in memory for a decision made later. Only `OP_BACKREF` and an unbounded width `OP_LOOKAHEAD` trigger the bounded engine (Section 5, 6). + +### 7.2 Regular engine: Pike VM over a chunk stream + +Standard Thompson NFA simulation extended with capture slots, run breadth first ("all current threads advance over the same input character, in priority order, duplicate states are merged"). Per input character the engine holds at most `N` threads, `N` being the instruction count, each thread owning only its capture slot array (`2 * ngroups` offsets), so per character memory is `O(N * ngroups)`, not `O(input length)`. + +Streaming adaptation: input arrives as a sequence of chunks (see 7.3) rather than one buffer. Thread capture slots store absolute stream offsets (a 64 bit counter incremented across chunk boundaries), not pointers into the chunk buffer, so a thread survives a chunk boundary without copying. Once every live thread's earliest referenced offset has advanced past a chunk boundary, that chunk is released back to the caller supplied allocator. This is the entire mechanism that lets the regular engine run over an arbitrarily large file in bounded memory: it never needs to look backward, so it never needs to keep anything but the current chunk and the small thread list. + +Two small, constant size pieces of state cross a chunk boundary alongside the thread list. First, in `UTF8` mode, a partially read multi-byte sequence: at most 3 pending lead bytes (the longest UTF-8 sequence is 4 bytes), carried into the next chunk before character classification resumes; a chunk is only eligible for release once any sequence straddling its end has been completed by the following chunk. Second, under `MULTILINE`, one bit recording whether the byte immediately before the current chunk was `\n`, needed to classify `^` at the very first position of a new chunk without rereading the previous one; `\n` (`0x0A`) cannot appear as a non-initial byte of any valid multi-byte UTF-8 sequence, so this bit and the UTF-8 continuation state never interact with each other. Neither addition affects the O(instruction count x group count) memory bound of Section 10, since both are O(1) regardless of chunk size or input length. + +### 7.3 Bounded backtracking engine: sliding window + +Used only for the minority of patterns containing a backreference or an unbounded width lookahead (7.1). Maintains an explicit ring buffer window of `W` bytes (default `W = 1 MiB`, a run time parameter). The window always contains the current match attempt's start position and everything from there forward that has been read so far, up to `W` bytes. A recursive backtracking matcher, structurally the direct translation of the AST (one function per node kind, exactly as `_sre` and most textbook backtracking matchers are structured), walks the bytecode against the window. If a match attempt's required span would exceed `W`, the call returns the documented bounded-window error described in Section 5 instead of growing the window past its configured limit. + +Because this engine is only invoked for patterns that need it, ordinary patterns (the large majority) never pay for the ring buffer or for backtracking, and get the linear time guarantee of 7.2 instead. + +### 7.4 Anchoring across chunks + +Both engines expose the same chunk boundary contract: a match cannot be reported as final until either (a) `OP_MATCH` is reached, or (b) enough trailing context has been seen to prove that no continuation of the current input would change the answer for the leftmost still-open match attempt. For quantifiers this is decided directly by the bytecode's `OP_SPLIT`/`OP_JMP` shape; for anchors (`$`, `\Z`) the final chunk is distinguished by an explicit "end of stream" sentinel token, so `$` and `\Z` behave identically whether or not `MULTILINE` is set, matching CPython. + +## 8. Core Data Structures + +Kept intentionally minimal, all defined in the single file, no dependency beyond the C standard library (`stdint.h`, `stddef.h`, `string.h`): + +### 8.1 Allocator +One struct of three function pointers (`alloc`, `realloc`, `free`) passed once at engine creation, defaulting to the libc equivalents. This is the only piece of "infrastructure" abstraction in the file, and it exists so the sliding window (7.3) and chunk buffers (7.2) can be sized and released under caller control, which is a prerequisite for the gigabyte scale requirement. + +### 8.2 Dynamic array +One generic growable array (`{ void *data; size_t len, cap, elemsize; }`) with `push`/`get`, used for the instruction array, the capture slot array, and the AST node pool. Deliberately not a macro-heavy generic container; three functions (`da_init`, `da_push`, `da_free`) cover every use site in the file. + +### 8.3 Byte class table +A 256 bit set (`uint32_t bits[8]`) per compiled `[...]` class or shorthand class, precomputed at compile time. In UTF-8 mode, code points above 127 are matched against a small number of precompiled Unicode range tables (Section 13.3) instead of the 256 bit set. + +### 8.4 Capture slots +A flat array of `2 * groups` capture records per thread (regular engine) or per backtracking call frame (bounded engine), `-1` meaning unset. In `BINARY` and `ASCII` mode a record is a single `int64_t` byte offset. In `UTF8` mode a record is a pair, `{ int64_t byte_offset; int64_t codepoint_index; }`: the byte offset is what is needed to read the matched bytes back out of `Input`, and the code point index is what is needed to report `Match_start`/`Match_end`/`Match_span` in the same unit Python uses for `str` subjects (Section 9.1). The code point index is a running counter incremented once per decoded scalar value as the engine advances, so recording it at a `SAVE` costs one extra integer copy, not a second pass over the input; it changes the constant factor of the O(instruction count x group count) memory bound of Section 10 in `UTF8` mode, not its asymptotic class. This single representation backs `Match_group`, `Match_start`, `Match_end`, `Match_span`, `Match_start_byte`, `Match_end_byte`, `Match_span_byte`, and `Match_groupdict` in the public API. + +## 9. Public API (Python `re` equivalents) + +### 9.0 Naming convention + +Every public identifier that has a direct counterpart in Python's `re` module uses that counterpart's exact spelling, not a transliterated or prefixed variant. Concretely: + +- The two data types Python's `re` exposes, `Pattern` and `Match`, are C structs named `Pattern` and `Match`, not `regex_t` or `regexx_pattern`. +- A method Python calls as `pattern_obj.search(...)` is written `Pattern_search(Pattern *self, ...)`; a method called as `match_obj.group(...)` is written `Match_group(Match *self, ...)`. The `Type_method` shape is the direct C rendering of `type.method`, needed only because C has no bound methods; the two name fragments either side of the underscore are otherwise exactly the Python names. +- A module level function such as `re.compile(...)` or `re.sub(...)` is written `re_compile(...)`, `re_sub(...)`, and so on: the `re_` prefix stands for the module the function lives in in Python (`re.compile`), the same relationship `Pattern_` and `Match_` have to their types. +- Flag constants (`IGNORECASE`, `MULTILINE`, `DOTALL`, `VERBOSE`, `ASCII`, `UNICODE`, `LOCALE`, `DEBUG`) are `#define` or `enum` constants with exactly those names, no `RE_` or `REGEXX_` prefix, combined with bitwise `|` exactly as `re.IGNORECASE | re.MULTILINE` is combined with Python's `|`. +- `Pattern` fields are named `pattern`, `flags`, `groups`, `groupindex`, matching `re.Pattern.pattern`, `.flags`, `.groups`, `.groupindex` exactly. `Match` fields are named `string`, `pos`, `endpos`, `lastindex`, `lastgroup`, `re` (a pointer back to the owning `Pattern`, matching `re.Match.re`), matching `re.Match`'s attributes exactly. +- The exception is `Input` (9.1), which has no Python counterpart because Python's `re` never streams: it always operates on an in-memory `str` or `bytes` object. `Input` is the one C-only type the design adds, and it is named descriptively rather than after a nonexistent Python name, precisely so it stands out as the one addition a reader should not go looking for in the `re` documentation. +- The error type is named `PatternError`, matching the alias CPython itself introduced for `re.error` (`re.PatternError`), rather than a plain `error` (which would collide too easily with `errno.h`-style conventions) or an invented `regexx_error`. + +### 9.1 Declarations + +```c +typedef struct Pattern Pattern; /* re.Pattern */ +typedef struct Match Match; /* re.Match */ +typedef struct Input Input; /* no Python counterpart, see 9.0; left without a + * struct body here because its concrete layout is + * one of the two variants described in 9.2 and no + * code outside the Input implementation itself + * needs to see inside it, unlike Pattern and Match + * whose fields are part of the public, Python- + * mirroring surface. */ + +typedef void (*MatchIterCb)(void *ctx, const Match *m); +typedef void (*MatchSubCb)(void *ctx, const Match *m, char **out, size_t *outlen); + +typedef struct PatternError PatternError; /* re.error / re.PatternError */ +struct PatternError { + const char *msg; /* re.error.msg */ + const char *pattern; /* re.error.pattern */ + int64_t pos; /* re.error.pos */ + int64_t lineno; /* re.error.lineno */ + int64_t colno; /* re.error.colno */ +}; + +struct Pattern { + const char *pattern; /* re.Pattern.pattern */ + int flags; /* re.Pattern.flags */ + int groups; /* re.Pattern.groups */ + void *groupindex; /* re.Pattern.groupindex, name -> group number */ + void *program; /* compiled bytecode, private (7.1) */ +}; + +struct Match { + Pattern *re; /* re.Match.re */ + Input *string; /* re.Match.string */ + int64_t pos, endpos; /* re.Match.pos, re.Match.endpos; code point indices in UTF8 mode, byte offsets otherwise, see 8.4 */ + int lastindex; /* re.Match.lastindex */ + const char *lastgroup; /* re.Match.lastgroup */ + void *slots; /* private, see 8.4 */ +}; + +/* Pattern methods: the primitives, bound to an already compiled Pattern, + * mirroring re.Pattern.match / .search / .fullmatch / .finditer / .findall / + * .split / .sub / .subn exactly, including their pos/endpos parameters. */ +int Pattern_match(Pattern *self, Input *string, int64_t pos, int64_t endpos, Match *out); +int Pattern_fullmatch(Pattern *self, Input *string, int64_t pos, int64_t endpos, Match *out); +int Pattern_search(Pattern *self, Input *string, int64_t pos, int64_t endpos, Match *out); +int Pattern_finditer(Pattern *self, Input *string, int64_t pos, int64_t endpos, MatchIterCb cb, void *ctx); +int Pattern_findall(Pattern *self, Input *string, int64_t pos, int64_t endpos, MatchIterCb cb, void *ctx); +int Pattern_split(Pattern *self, Input *string, int maxsplit, MatchIterCb cb, void *ctx); +int Pattern_sub(Pattern *self, Input *string, const char *repl, MatchSubCb cb, void *ctx, int count, char **out, size_t *outlen); +int Pattern_subn(Pattern *self, Input *string, const char *repl, MatchSubCb cb, void *ctx, int count, char **out, size_t *outlen, int *n); +void Pattern_free(Pattern *self); + +/* Match accessors, matching re.Match's bound methods and attributes */ +int Match_group(Match *self, const char *name_or_null, int index, const char **out, size_t *outlen); +void Match_groups(Match *self, /* out array of (ptr,len) */ void *out); +void Match_groupdict(Match *self, /* out name -> (ptr,len) map */ void *out); +int64_t Match_start(Match *self, int group); /* code point index in UTF8 mode, byte offset otherwise */ +int64_t Match_end(Match *self, int group); +void Match_span(Match *self, int group, int64_t *start, int64_t *end); +int64_t Match_start_byte(Match *self, int group); /* always a byte offset into Input, see 8.4 */ +int64_t Match_end_byte(Match *self, int group); +void Match_span_byte(Match *self, int group, int64_t *start, int64_t *end); +void Match_expand(Match *self, const char *template, char **out, size_t *outlen); + +/* Module level functions, mirroring re.compile / re.match / re.search / ... + * exactly: each of the search-family functions compiles pattern through an + * internal bounded cache and then calls the matching Pattern_ function, + * exactly as CPython's re/__init__.py implements re.match as + * _compile(pattern, flags).match(string). */ +Pattern *re_compile(const char *pattern, size_t len, int flags, PatternError *err); +int re_match(const char *pattern, size_t len, int flags, Input *string, Match *out); +int re_fullmatch(const char *pattern, size_t len, int flags, Input *string, Match *out); +int re_search(const char *pattern, size_t len, int flags, Input *string, Match *out); +int re_finditer(const char *pattern, size_t len, int flags, Input *string, MatchIterCb cb, void *ctx); +int re_findall(const char *pattern, size_t len, int flags, Input *string, MatchIterCb cb, void *ctx); +int re_split(const char *pattern, size_t len, int flags, Input *string, int maxsplit, MatchIterCb cb, void *ctx); +int re_sub(const char *pattern, size_t len, int flags, Input *string, const char *repl, MatchSubCb cb, void *ctx, int count, char **out, size_t *outlen); +int re_subn(const char *pattern, size_t len, int flags, Input *string, const char *repl, MatchSubCb cb, void *ctx, int count, char **out, size_t *outlen, int *n); +void re_escape(const char *in, size_t len, char **out, size_t *outlen); +void re_purge(void); +``` + +The dependency runs from module level to `Pattern`, not the other way around, which is the same direction CPython itself uses: `re.py` defines `match`, `search`, and the rest as thin wrappers that call `_compile(pattern, flags)` and then the corresponding `Pattern` method. Each module level function above holds an internal cache keyed by `(pattern, len, flags)`, bounded to a fixed capacity and cleared entirely on overflow rather than evicting individual entries, again mirroring the strategy CPython's own `re` module cache uses. `re_purge()` clears that cache on demand, matching `re.purge()` exactly, including that it has no effect on any `Pattern *` a caller is still holding a direct reference to. `re_compile` bypasses the cache and always produces a fresh `Pattern`, matching `re.compile`. A caller working against a large or streaming `Input` and applying the same pattern repeatedly should call `re_compile` once and use the `Pattern_` functions directly, exactly as idiomatic Python precompiles a pattern that is reused in a loop rather than calling the module level function repeatedly; the module level functions exist for parity with `re.match`/`re.search`/etc., not as the recommended entry point for the gigabyte scale case this design targets. + +### 9.2 `Input` + +An abstraction over "a source of chunks": either a fixed buffer (small strings, a drop in replacement for the CPython `str`/`bytes` case) or a caller supplied `read(void *ctx, uint8_t *buf, size_t cap) -> size_t` callback (files, pipes, sockets), which is how gigabyte scale input is supplied without ever requiring the caller to load it fully into memory. See 9.0 for why this type does not carry a Python name. + +`pos` and `endpos` on the `Pattern_` functions are expressed in the same unit as `Match_start`/`Match_end` for the pattern's mode (code point index in `UTF8` mode, byte offset otherwise, Section 8.4). Honoring a nonzero `pos` against a streaming `Input` that only exposes sequential `read` requires decoding forward from the start of the stream until that position is reached, an O(pos) cost paid once per call, not a departure from the per-character bound of Section 10, which is stated per byte of input actually scanned. An `Input` that also exposes an optional `seek(void *ctx, int64_t byte_offset) -> int` callback lets the engine skip that decode pass in `ASCII`/`BINARY` mode, or in `UTF8` mode whenever the caller already knows the target byte offset, for example one returned earlier by `Match_start_byte` against the same `Input`. + +### 9.3 Encoding mode + +A field of the compile-time `flags` alongside `IGNORECASE`, `MULTILINE`, and the rest: `BINARY`, `ASCII`, `UTF8` (no `RE_` prefix, per 9.0; CPython has no equivalent constant because the choice between binary and text mode is implicit in whether a `bytes` or `str` pattern was compiled, so `Pattern_compile` here makes that same choice explicit through a flag instead). This flag selects, at compile time, which byte class tables (8.3), which `.`/`\w`/`\s` definitions (2.2), and which decoder (raw byte, or the UTF-8 decoder producing code points for classification while recording both a code point index and a byte offset per capture, Section 8.4) the compiled program uses. Binary mode never decodes: every byte value 0 to 255, including embedded NUL, is a valid atom, matching the behavior Python gets by compiling a `bytes` pattern against a `bytes` subject. + +A byte sequence in `UTF8` mode that is not valid UTF-8 at the point the decoder reaches it is handled the way `bytes.decode('utf-8', errors=...)` is in Python: the default policy, `strict`, surfaces a `PatternError` identifying the byte offset of the first invalid byte, matching the fact that CPython can never hand `re` a `str` that was not already validly decoded in the first place. An opt-in `replace` policy substitutes the Unicode replacement character `U+FFFD` for the offending bytes and continues, for callers that must process untrusted or partially corrupt streams without aborting. `BINARY` and `ASCII` mode have no decode step and so have no analogous failure mode; a byte outside `0`-`127` in `ASCII` mode is simply a byte no `ASCII`-mode class matches, not an error. + +### 9.4 Replacement callback + +`Pattern_sub`/`Pattern_subn` (and the module level `re_sub`/`re_subn` that wrap them, 9.1) take both a `repl` template string and a `cb` callback of type `MatchSubCb` as separate parameters, of which exactly one is non-`NULL` on any given call: a non-`NULL` `repl` is parsed once, at compile time, into a small list of literal/backreference segments, mirroring Section 3's replacement syntax; a non-`NULL` `cb` is invoked once per match with the match record and a caller supplied `ctx`, and is expected to write the replacement text through `out`/`outlen`, which is the C shape of Python's callable `repl` argument to `re.sub`. Two separate parameters, rather than one parameter serving both roles, is the direct C consequence of Python's single `repl` argument being polymorphic (string or callable) in a way C's static type system cannot express in one slot, the same kind of unavoidable, minimal departure from a one to one name and shape mapping that 9.0 already accepts for `Type_method`. + +## 10. Complexity Summary + +| Engine | Time | Memory | Applies to | +|---|---|---|---| +| Regular (Pike VM, 7.2) | O(input length x instruction count) | O(instruction count x group count), independent of input length | Patterns with no backreference and no unbounded lookaround | +| Bounded backtracking (7.3) | Worst case exponential in window size, as in CPython | O(window size `W`), independent of input length beyond `W` | Patterns with a backreference or an unbounded width lookahead | + +Both figures are stated relative to input length specifically because that is the axis the gigabyte scale requirement constrains; instruction count and group count are properties of the pattern, not the input, and are expected to stay small (tens to low hundreds) for realistically written patterns. + +## 11. Testing Strategy + +CPython ships its own `re` test suite (`Lib/test/test_re.py` / `re_tests.py`) as executable pattern, string, expected-result triples. The plan is to translate that suite mechanically into a table of C test cases run against `re_match`/`re_search`/`re_sub`, so the engine's conformance is measured against CPython's own stated behavior rather than against a re-derived interpretation of the documentation. Cases that exercise the explicitly out of scope items in Section 13 are recorded as known deviations rather than deleted, so the gap stays visible. + +## 12. Worked Example: Why This Is the Simplest Design That Reaches the Goal + +A single unified backtracking engine (matching CPython's own `_sre` design most closely) would be simpler to write than the two-engine split in Section 5 and Section 7, but it cannot satisfy the gigabyte scale, bounded memory requirement for the common case, because a naive backtracking matcher's stack depth and re-scan behavior scale with input length for ordinary patterns, not only for pattern using backreferences. Conversely, a single unified automaton engine (no backtracking at all) is simpler still, but cannot express backreferences, or the same unbounded width lookaround built from the widening this design already restricts to the bounded engine (Section 5, 7.1), which the objective in Section 1 requires. The two-engine split is the smallest design found that keeps the automaton engine's linear-time, bounded-memory property for the patterns that admit it, while still offering exact backreference and lookaround semantics for the patterns that need them, at an explicit and configurable memory cost. + +## 13. Known Gaps Against CPython `re` + +### 13.1 POSIX leftmost-longest matching +Not applicable; CPython's `re` itself uses ordered, first-alternative-wins backtracking semantics, and this engine matches that, not POSIX `grep -E` semantics. See Section 14.1 for the full comparison against native C `regex.h` behavior, which this section only touched on briefly before that comparison existed. + +### 13.2 Recursion limit parity +CPython raises `RecursionError` past a configurable backtracking depth. The bounded backtracking engine (7.3) instead bounds by window size and an explicit call depth counter with a similar default; exact error message parity is not a goal, only the presence of a safe failure mode. + +### 13.3 Full Unicode tables +`\N{NAME}` lookup, full `IGNORECASE` case folding (including special casing such as German `ß`), and complete `\w`/`\s` Unicode category coverage require the Unicode Character Database. The concept ships a reduced set of range tables covering the common categories (letters, digits, marks, common whitespace) rather than the full database, to keep the single file small; the compiled tables are generated from the Unicode Character Database offline and checked in as static arrays, with the generation script kept outside the single interpreter file. + +### 13.4 `LOCALE` flag +Only the "C" locale behavior is implemented (byte range `\x80`-`\xff` treated as word characters); full `locale.h` integration is out of scope because it reintroduces global, environment dependent state into an otherwise pure, single file design. + +### 13.5 `regex` third party module extensions +Constructs from the third party `regex` package (set operations inside classes such as `--`/`&&`, fuzzy matching, recursive patterns `(?R)`, variable width lookbehind) are not part of CPython's `re` and are out of scope by Section 1's own definition of "what Python supports." + +### 13.6 Concurrency of the module level pattern cache +The internal cache backing the module level functions of Section 9.1 (`re_match`, `re_search`, and the rest) is shared, mutable state. CPython's own equivalent cache is implicitly protected by the GIL; this design has no equivalent, so a caller invoking the module level functions from more than one thread concurrently must serialize access to the cache itself (a mutex around lookup and insertion, sized independently of the allocator in 8.1) or avoid the module level functions entirely and call `re_compile` once per pattern up front, sharing the resulting read only `Pattern *` across threads. The latter is already the recommended pattern for the gigabyte scale case (9.1's closing paragraph), so this limitation is expected to be inactive on the path the design is optimized for. + +## 14. POSIX / Native C Regex (`regex.h`) Capabilities Absent From Python `re` + +Because this project is delivered as a C module, a reader coming from C rather than from Python may reasonably expect it to behave like the regular expression facility native to the C standard library, POSIX `regex.h` (`regcomp`, `regexec`, `regfree`, `regerror`, specified by IEEE Std 1003.1). This section records, completely and for that reader specifically, the respects in which POSIX's native facility does something Python `re` does not do at all. Every item is either omitted by design, meaning it is incompatible with matching Python `re`'s own behavior and is therefore excluded by Section 1's own definition of the target, or out of scope, meaning it would not conflict with Python parity but was not requested and is not free to add under Section 1's simplicity priority. Section 14.6 closes with the one respect in which this design already exceeds POSIX `regex.h`, included so the comparison is not one sided. + +### 14.1 Leftmost-longest ("POSIX") matching +POSIX `regex.h` is specified to find the leftmost match and, among matches starting at that leftmost position, the longest one, applied recursively to subexpressions as well as to the overall match. Python `re`, like every Perl-derived engine, instead uses leftmost-first, ordered-alternation, backtracking semantics: the first alternative that leads to any overall match wins, even when a later alternative would consume more text, and quantifier greediness is resolved by backtracking order rather than by a global longest-match search. These are two different, mutually incompatible definitions of "the match" for the same pattern and string; `a|ab` against `"ab"` matches `"a"` under Python/Perl semantics and `"ab"` under POSIX semantics. This design follows Python's ordered semantics throughout (Section 2.5, Section 13.1), by the objective in Section 1. Omitted by design: Pike's priority ordered thread simulation (7.2), used here specifically to reproduce Perl style ordered semantics, is a different algorithm from the one POSIX-longest resolution requires, and running both simultaneously would cost the single, simple engine design Section 12 argues for, for a mode nothing in Section 1 asks for. + +### 14.2 POSIX bracket-expression collating symbols and equivalence classes +A POSIX bracket expression may contain `[.collating-symbol.]` (a named, possibly multi-character collating element treated as one unit, useful for ranges) and `[=equivalence-class=]` (every character the active locale's collation treats as primary equivalent to the given one). Python's `[...]` syntax has no counterpart to either: `[[.ch.]]` and `[[=e=]]` in a Python pattern parse as plain sets of the literal characters `[`, `.`, `c`, `h`, `]` and `[`, `=`, `e`, `]`, never as collating constructs, because Python bracket expressions are defined purely over literal characters and ranges, never over locale collation data. Out of scope: these constructs only have observable effect in locales with genuine multi-character collating elements, which is rare in practice, Python's `re` has never implemented them for `str` or `bytes` patterns, and adding them would require linking the compiled tables of Section 8.3 to the system's `LC_COLLATE` data, in direct tension with the dependency-free, pure-function design of Section 8. + +### 14.3 POSIX named character classes inside bracket expressions +POSIX bracket expressions accept the twelve standard named classes, `alpha`, `digit`, `alnum`, `upper`, `lower`, `space`, `blank`, `cntrl`, `graph`, `print`, `punct`, `xdigit`, written as `[:name:]` inside a bracket expression, for example `[[:alpha:][:digit:]]`. Python's `re` has no equivalent syntax; the same intent is expressed with ranges and the shorthand classes of Section 2.2 instead (`[a-zA-Z]`, `\d`, `\s`), and `[[:alpha:]]` in Python parses as a literal set containing `:`, `a`, `l`, `p`, `h`, `[`, `]`. Out of scope by direct consequence of Section 1: Python `re` genuinely has no such syntax, so parity with Python `re` already excludes it; it is recorded here only because a C-background reader is likely to look for it and be surprised to find it silently absent rather than documented. + +### 14.4 Locale collating sequence for bracket-expression ranges +POSIX specifies that a bracket-expression range such as `[a-z]` is resolved according to the current locale's collating sequence (`LC_COLLATE`), not according to raw code point or byte value order; in a locale whose collation is not a simple ascending code point order, `[a-z]` can therefore include, exclude, or reorder characters relative to what it means in the "C" locale, a well known source of surprising results in POSIX tools run under a non-"C" locale. Python's `re` never does this: a range in a Python pattern is always defined by code point value (`str` patterns) or byte value (`bytes` patterns), unconditionally, in every locale. This design follows Python exactly, ranges are always resolved by code point or byte value (Section 2.2), independent of the `LOCALE` flag (Section 13.4). Omitted by design, for the same reason `LOCALE` itself is restricted to the "C" locale in Section 13.4: honoring arbitrary system collation would reintroduce global, environment dependent state into a design that is otherwise a pure function of its inputs, and would make the meaning of a compiled `Pattern` depend on a process-wide setting instead of on the `flags` given to `re_compile`. + +### 14.5 `REG_NOSUB`: compiling a pattern that reports no subexpression positions +`regcomp(..., REG_NOSUB)` compiles a pattern that reports only whether it matched, not where its subexpressions matched, letting an implementation skip the capture bookkeeping of Section 8.4 entirely. Python's `re` has no equivalent compile-time flag; a compiled `Pattern` always reports full match and group data through `Match`. Out of scope: this is a pure performance optimization with no observable behavior difference, and Section 1 places convenience above performance throughout, so there is nothing here worth the added compile-time flag and the second code path it would require in both engines. + +### 14.6 Where this design already exceeds plain POSIX `regex.h` +`regexec` takes a nul-terminated C string, so plain POSIX `regex.h`, on most implementations, cannot search a subject containing an embedded NUL byte at all; the widely available but non-standard `REG_STARTEND` extension (present in glibc and the BSDs, not part of IEEE Std 1003.1 itself) works around this only on the platforms that provide it. Python's `re` has never had this limitation, since `str` and `bytes` are always explicit-length, never nul-terminated, and this design follows Python and inherits the same freedom from it directly: `Input` (Section 9.2) is always an explicit-length byte source, and `BINARY` mode (Section 9.3) explicitly allows an embedded NUL as an ordinary byte value, matching the gigabyte scale, arbitrary-binary-data objective of Section 1. This item is placed last, and out of sequence with the rest of Section 14's "absent from Python `re`" framing, specifically so the comparison in this section is accurate in both directions rather than reading as one sided. diff --git a/docs/API.md b/docs/API.md new file mode 100644 index 0000000..7fc554e --- /dev/null +++ b/docs/API.md @@ -0,0 +1,279 @@ +# regexx API Reference + +This document records, exhaustively, the complete public surface of +`regexx.h`/`regexx.c`: every type, every flag, every function, its exact +return-value convention, its memory ownership rule, and its relationship to +the corresponding Python `re` name. `concept.md` records the design +rationale; `README.md` is the short entry point (build, usage, implementation +status at a glance). This document is the complete reference the other two +point to when a precise answer is needed. + +Every fact below was checked against the current source (`regexx.c`, +`regexx.h`) and, where a Python behavior is cited, against a real CPython 3 +interpreter, not against memory or documentation alone. + +## 1. Types + +### 1.1 `Pattern` (`re.Pattern`) + +```c +struct Pattern { + const char *pattern; /* re.Pattern.pattern */ + int flags; /* re.Pattern.flags */ + int groups; /* re.Pattern.groups */ + void *groupindex; /* re.Pattern.groupindex, opaque, see 3.11 */ + void *program; /* private: compiled bytecode */ +}; +``` + +- `pattern`: the exact source text passed to `re_compile`, NUL-terminated, owned by the `Pattern` (freed by `Pattern_free`). Read-only for callers. +- `flags`: the exact `flags` value passed to `re_compile`, including the encoding-mode bits (`BINARY`/`ASCII`/`UTF8`) and any leading global inline flags folded in during parsing (Section 2.6). +- `groups`: the number of capturing groups, matching Python's `re.Pattern.groups` exactly (group 0, the whole match, is not counted). +- `groupindex`: opaque; see `Pattern_groupindex_lookup` (3.11) for the only supported access to it. +- `program`: private, never dereference directly. + +A `Pattern` is created only by `re_compile` (directly, or indirectly through the module level cache in `re_match`/`re_search`/etc.) and is freed only by `Pattern_free`. + +### 1.2 `Match` (`re.Match`) + +```c +struct Match { + Pattern *re; /* re.Match.re */ + Input *string; /* re.Match.string */ + int64_t pos, endpos; /* re.Match.pos, re.Match.endpos */ + int lastindex; /* re.Match.lastindex */ + const char *lastgroup; /* re.Match.lastgroup */ + void *slots; /* private */ +}; +``` + +- `re`: the `Pattern` that produced this match. Borrowed reference; do not free it through this pointer, and do not call `Pattern_free` on it while any `Match` from it is still in use. +- `string`: the `Input` the match was found in. Borrowed reference, same lifetime rule as `re`. +- `pos`, `endpos`: the effective search bounds used to produce this match (the `pos`/`endpos` arguments to whichever `Pattern_` function created it, clamped to `[0, length]`), in the same unit as `Match_start`/`Match_end` (code point index in `UTF8` mode, byte offset otherwise). Matches `re.Match.pos`/`.endpos` exactly. +- `lastindex`: the highest-numbered capturing group that participated in the match, or `-1` if none did. **Deviation from CPython:** Python's `lastindex` is "the index of the last group to match", which for a pattern with re-entrant alternation can differ from "the highest-numbered group that participated"; this build uses the latter, simpler rule. They coincide for every straightforward pattern (any pattern without a capturing group inside a repeated alternative that can also match via a different, lower-numbered branch later in the same attempt). +- `lastgroup`: the name of that group, or `NULL` if it is unnamed or `lastindex` is `-1`. Borrowed pointer into the `Pattern`'s group name table; valid as long as the `Pattern` is. +- `slots`: private. + +A `Match` is filled in by one of the `Pattern_`/`re_` matching functions (never allocated separately by the caller: pass the address of a stack or heap `Match` struct as `out`, zero-initialize it first). Its private `slots` are freed by `Match_free`, which does **not** free the `Match` struct itself (the caller owns that memory, stack or heap). + +### 1.3 `Input` (no Python counterpart) + +Opaque. Python's `re` never streams; it always operates on an in-memory `str`/`bytes` object already held by the caller. `Input` is the type this design adds so a C caller has an explicit thing to construct from a buffer or a file (`concept.md` 9.0/9.2). + +```c +Input *Input_from_buffer(const uint8_t *buf, size_t len); +Input *Input_from_file(const char *path, PatternError *err); +void Input_free(Input *in); +``` + +- `Input_from_buffer`: wraps an existing buffer. **Does not copy it and does not take ownership.** The buffer must outlive the `Input` and every `Match` produced from it (`Match_group` returns pointers directly into it, Section 3.9). Freeing an `Input_from_buffer` `Input` never frees the underlying buffer; the caller is responsible for that. +- `Input_from_file`: reads the whole file into a freshly allocated, owned buffer. Works on both seekable files and non-seekable sources (pipes, FIFOs, process substitution, `/dev/stdin`): a seekable source is read in one `fread` after `fseek`/`ftell` sizing it; a non-seekable source is read incrementally into a growable buffer. Either way, the entire input ends up in memory before any matching happens (`README.md` "Implementation status"). On failure (cannot open, cannot allocate) returns `NULL` and, if `err` is non-`NULL`, fills it with `strerror(errno)` as `msg`. +- `Input_free`: frees the `Input` and, only if it owns its buffer (true for `Input_from_file`, false for `Input_from_buffer`), the buffer too. + +### 1.4 `PatternError` (`re.error` / `re.PatternError`) + +```c +struct PatternError { + const char *msg; /* re.error.msg */ + const char *pattern; /* re.error.pattern */ + int64_t pos; /* re.error.pos */ + int64_t lineno; /* re.error.lineno */ + int64_t colno; /* re.error.colno */ +}; +void PatternError_free(PatternError *err); +``` + +Named after the alias CPython itself introduced for `re.error` (`re.PatternError`, `concept.md` 9.0). Fields match CPython's `re.error` attributes (added in CPython 3.5) exactly: `msg` is the human-readable message, `pattern` is the offending pattern text (or `NULL` when the error is not about pattern syntax, for example `Input_from_file` failing to open a file), `pos` is the byte offset into `pattern` the error was detected at, `lineno`/`colno` are computed from `pos` the same way CPython computes them (1-based, counting `\n` bytes in `pattern` up to `pos`). + +**Ownership:** `msg` and `pattern` are heap allocated (`strdup`) by whichever call filled the struct in. Zero-initialize a `PatternError` before passing its address in, and call `PatternError_free` on it once done reading it, whether or not the call that filled it in succeeded (a `NULL` `err` argument to any function is always safe to pass and simply skips error reporting). `PatternError_free` is safe to call on an all-zero or already-freed `PatternError`. + +Every function that can fail accepts an optional `PatternError *err` (pass `NULL` to ignore); `re_compile` and `Input_from_file` are the two that actually produce one today. `re_match`/`re_search`/etc. (the module level convenience functions, 3.12) do not expose a `PatternError` parameter at all, matching the fact that CPython's own `re.match`/`re.search`/etc. do not return one either (a syntax error there raises, which has no C equivalent; here it instead causes the call to return `-1` with no further diagnostic, which is why `Pattern`-based usage, not the module level convenience functions, is recommended whenever a compile error needs to be reported, `concept.md`/README "recommended entry point" note). + +### 1.5 `MatchIterCb` / `MatchSubCb` + +```c +typedef void (*MatchIterCb)(void *ctx, const Match *m); +typedef void (*MatchSubCb)(void *ctx, const Match *m, char **out, size_t *outlen); +``` + +`MatchIterCb` is invoked once per result by `Pattern_finditer`, `Pattern_findall` (an alias of `finditer` in this build, 3.7), and `Pattern_split` (3.8, with a different per-call meaning, documented there). The `Match` passed in is only valid for the duration of the call; it is freed immediately after the callback returns, so do not retain the pointer. + +`MatchSubCb` is the C shape of Python's callable `repl` argument to `re.sub`/`re.subn`: invoked once per match, expected to `malloc` a replacement buffer, write its address into `*out` and its length into `*outlen`. The callback's `*out` becomes owned by `Pattern_sub`/`Pattern_subn`, which frees it after copying its content into the final result buffer. + +## 2. Flags + +Passed as `flags` to `re_compile` or any `re_` module level function, combined with bitwise `|`, using the exact spelling CPython uses (`concept.md` 9.0). + +| Flag | Value | Effect | Status | +|---|---|---|---| +| `IGNORECASE` | `0x0001` | Case-insensitive literal, class, and backreference comparison. Class ranges are matched under both a character's original case and its swapped case (README "Known deviations": an approximation, not full Unicode case folding). | Implemented | +| `MULTILINE` | `0x0002` | `^`/`$` also match at the start/end of each line, not only the start/end of the string. | Implemented | +| `DOTALL` | `0x0004` | `.` matches `\n` too. | Implemented | +| `VERBOSE` | `0x0008` | Unescaped whitespace and `#`-to-end-of-line comments outside character classes are stripped from the pattern before parsing. | Implemented | +| `ASCII` | `0x0010` | Two roles: (a) with neither `BINARY` nor `UTF8` also given, selects the `ASCII` **encoding mode** (Section 2, below); (b) in `UTF8` mode, forces `\d`/`\w`/`\s` and `IGNORECASE` swapcase to their ASCII-only definitions instead of consulting `wctype.h`. | Implemented | +| `UNICODE` | `0x0020` | Accepted for source compatibility with Python. **No effect**: `UTF8` mode already behaves as CPython's default (Unicode) `str` matching, so there is no separate "unicode" flag needed the way `ASCII` needs one to opt out. | Accepted, no-op | +| `LOCALE` | `0x0040` | In `BINARY`/`ASCII` mode only, bytes `0x80`-`0xFF` are additionally treated as word characters for `\w`/`\b`/`\B` (`concept.md` 13.4's "C locale" approximation). No effect in `UTF8` mode. | Implemented (C-locale approximation only) | +| `DEBUG` | `0x0080` | Accepted for source compatibility with Python. **No effect** in this build: nothing is printed and matching is unaffected. | Accepted, no-op | +| `BINARY` | `0x0100` | Encoding mode: raw bytes, `0`-`255` all valid, no decoding, embedded `NUL` is an ordinary byte. | Implemented | +| `UTF8` | `0x0200` | Encoding mode: input is decoded as UTF-8 into code points before matching; `Match_start`/`Match_end` report code point indices (`Match_start_byte`/`Match_end_byte` report byte offsets, Section 3.9). | Implemented | + +Exactly one encoding mode is active for any compiled `Pattern`: `BINARY` and `UTF8` are checked first, in that order, and if neither is present the mode is `ASCII` regardless of whether the `ASCII` flag bit itself was set (so `re_compile(p, n, 0, &err)` and `re_compile(p, n, ASCII, &err)` compile to the same encoding mode; only combining `ASCII` with `UTF8` changes anything, per row (b) above). + +## 3. Functions + +Return value convention used throughout, unless noted otherwise for a specific function: `1` success/matched, `0` no match (not an error), `-1` an error occurred (invalid UTF-8 in the subject for a `UTF8`-mode `Pattern`, or the backtracking depth limit was reached, `README.md` "Implementation status"; no further detail is available through the return value itself in this build, only through the fact that it is negative rather than `0`). + +### 3.1 `re_compile` + +```c +Pattern *re_compile(const char *pattern, size_t len, int flags, PatternError *err); +``` + +Compiles `pattern` (`len` bytes, need not be `NUL`-terminated) under `flags` into a fresh, independent `Pattern`, matching `re.compile`. Returns `NULL` and fills `err` (if non-`NULL`) on a syntax error, an unsupported construct (Section 5 of this document lists all of them), or a lookbehind that is not fixed-width. Never consults or populates the module level cache (3.12); always allocates a new `Pattern`, freed only by `Pattern_free`. + +### 3.2-3.4 `Pattern_match` / `Pattern_fullmatch` / `Pattern_search` + +```c +int Pattern_match(Pattern *self, Input *string, int64_t pos, int64_t endpos, Match *out); +int Pattern_fullmatch(Pattern *self, Input *string, int64_t pos, int64_t endpos, Match *out); +int Pattern_search(Pattern *self, Input *string, int64_t pos, int64_t endpos, Match *out); +``` + +Mirror `re.Pattern.match`/`.fullmatch`/`.search` exactly, including the `pos`/`endpos` parameters (pass `0` and `-1` for CPython's own defaults, "search the whole string"). `pos`/`endpos` are in the pattern's native unit (code point index for `UTF8` mode, byte offset otherwise, Section 1.2); a negative `endpos` means "to the end". `out` must point to a zero-initialized `Match` (or one already released with `Match_free`); on a `0` or `-1` return it is left untouched. + +- `match`: anchored at `pos`, need not reach `endpos`. +- `fullmatch`: anchored at `pos`, must also reach exactly `endpos`. +- `search`: tries every start position from `pos` to `endpos` inclusive, left to right, and reports the first that admits any match (ordinary backtracking priority decides which match that is at that position, `concept.md` 2.5). + +### 3.5-3.6 `Pattern_finditer` / `Pattern_findall` + +```c +int Pattern_finditer(Pattern *self, Input *string, int64_t pos, int64_t endpos, MatchIterCb cb, void *ctx); +int Pattern_findall(Pattern *self, Input *string, int64_t pos, int64_t endpos, MatchIterCb cb, void *ctx); +``` + +`Pattern_findall` is defined as a call to `Pattern_finditer` with the same arguments; both invoke `cb(ctx, m)` once per non-overlapping match, left to right, applying CPython's own empty-match rule (an empty match advances one unit and is not merged with an adjacent non-empty match at the same start, `concept.md` 3). Returns the number of matches found, or `-1` on error. This build does not collapse a no-groups match down to "just the matched string" or a multi-group match to a Python tuple the way `re.findall` does at the Python level; the callback always receives a full `Match`, from which the caller reads whatever it needs via `Match_group`. This is a deliberate simplification: `findall`'s string/tuple collapsing is a Python-object-model convenience with no C equivalent to collapse into, so this build gives the caller the same, uniform `Match`-based access `finditer` does, and the two functions exist separately only for name-for-name parity with `re.findall`/`re.finditer`. + +### 3.7 `Pattern_split` + +```c +int Pattern_split(Pattern *self, Input *string, int maxsplit, MatchIterCb cb, void *ctx); +``` + +`cb` is invoked once per element of the list `re.split()` would return, **in order**: this is the entire contract, and it is unambiguous by construction (earlier drafts of this function called `cb` twice per match with the caller left to infer which call meant what; that design was replaced before release specifically because it was ambiguous). Each element's text is read via `Match_group(m, NULL, 0, &out, &outlen)`; a `0` return from that call means this element is Python's `None` (an unparticipated capturing group between two matches), matching how an unparticipated group reports on any other `Match`. `maxsplit` matches `re.split`'s parameter (`0` means unlimited). Returns the number of matches that were split on (not the number of list elements), or `-1` on error. + +### 3.8-3.9 `Pattern_sub` / `Pattern_subn` + +```c +int Pattern_sub(Pattern *self, Input *string, const char *repl, MatchSubCb cb, void *ctx, int count, char **out, size_t *outlen); +int Pattern_subn(Pattern *self, Input *string, const char *repl, MatchSubCb cb, void *ctx, int count, char **out, size_t *outlen, int *n); +``` + +Mirror `re.Pattern.sub`/`.subn`. Exactly one of `repl` (a template string) or `cb` (a callback) must be non-`NULL`; passing both or neither is a caller error with unspecified behavior. This two-parameter shape is the direct C consequence of Python's single `repl` argument being polymorphic (string or callable) in a way C's static type system cannot express in one slot (`concept.md` 9.4, 9.0). + +`repl` template syntax: `\g`, `\g`, `\N` (one or two digits), `\n`, `\t`, `\\`, and any other `\X` as the literal character `X`, matching `concept.md` Section 3. + +`count` matches `re.sub`'s `count` parameter (`0` means unlimited; a positive `count` stops substituting after that many matches, leaving the rest of the subject, including any further matches within it, untouched, exactly as CPython leaves it). + +`Pattern_sub` and `Pattern_subn` differ only in whether the number of substitutions actually made is reported back through `n` (mirroring `re.sub` returning just the string versus `re.subn` returning `(string, count)`); `Pattern_sub` is implemented as a call to `Pattern_subn` with a throwaway `n`. + +`*out` is a freshly `malloc`'d, `NUL`-terminated buffer of length `*outlen`; the caller must `free` it. On `-1` (error), `*out`/`*outlen` are left untouched. + +### 3.10 `Pattern_free` + +```c +void Pattern_free(Pattern *self); +``` + +Frees a `Pattern` and everything it owns (the compiled program, the retained parse tree, `groupindex`, the copy of the pattern text). Do not call this while any `Match` produced from this `Pattern` is still in use (`Match.re`/`Match.lastgroup` borrow from it, Section 1.2); free every such `Match` with `Match_free` first, or simply free them in the reverse order they were created, which is always safe. + +### 3.11 `Pattern_groupindex_lookup` + +```c +int Pattern_groupindex_lookup(Pattern *self, const char *name); +``` + +Looks up a named group in `re.Pattern.groupindex`; returns its 1-based group number, or `-1` if no group by that name exists in this pattern. This is the only supported access to `groupindex`'s content in this build: there is no enumeration function (no way to list every name a pattern defines without already knowing what to look for), unlike Python's `groupindex`, which is a full mapping object supporting iteration and `len()`. A caller that needs every name a pattern uses must track the names it compiled the pattern with itself. + +### 3.12-3.20 Module level functions + +```c +int re_match(const char *pattern, size_t len, int flags, Input *string, Match *out); +int re_fullmatch(const char *pattern, size_t len, int flags, Input *string, Match *out); +int re_search(const char *pattern, size_t len, int flags, Input *string, Match *out); +int re_finditer(const char *pattern, size_t len, int flags, Input *string, MatchIterCb cb, void *ctx); +int re_findall(const char *pattern, size_t len, int flags, Input *string, MatchIterCb cb, void *ctx); +int re_split(const char *pattern, size_t len, int flags, Input *string, int maxsplit, MatchIterCb cb, void *ctx); +int re_sub(const char *pattern, size_t len, int flags, Input *string, const char *repl, MatchSubCb cb, void *ctx, int count, char **out, size_t *outlen); +int re_subn(const char *pattern, size_t len, int flags, Input *string, const char *repl, MatchSubCb cb, void *ctx, int count, char **out, size_t *outlen, int *n); +``` + +Each compiles `pattern` through an internal cache and then calls the matching `Pattern_` function with `pos=0`, `endpos=-1` (module level `re.match`/`re.search`/etc. do not expose `pos`/`endpos` either, only the `Pattern` methods do, matching CPython exactly), exactly as CPython's own `re/__init__.py` implements `re.match` as `_compile(pattern, flags).match(string)`. The cache is keyed by `(pattern, len, flags)`, holds up to 512 entries, and is cleared entirely on overflow rather than evicting individual entries (mirroring the strategy CPython's own `re` module cache uses). A compile error inside these functions is silently reported as a `-1` return, with no `PatternError` available (Section 1.4); use `re_compile` plus a `Pattern_` function directly whenever a compile error needs to be diagnosed, or whenever the same pattern is applied more than once (idiomatic Python precompiles a pattern reused in a loop rather than calling the module level function repeatedly, and so should idiomatic use of this API, `concept.md` 9.1). + +**Concurrency:** the cache is shared, mutable, process-wide state with no internal locking (`concept.md` 13.6). Do not call any `re_`-prefixed module level function from more than one thread without external synchronization; `Pattern_`-prefixed functions on a `Pattern` no thread is concurrently modifying (which is all of them, since nothing here mutates a compiled `Pattern`) have no such restriction. + +### 3.21 `re_purge` + +```c +void re_purge(void); +``` + +Clears the module level cache, matching `re.purge()` exactly, including that it has no effect on any `Pattern *` a caller already holds a direct reference to (only the cache entry is dropped; already-returned pointers remain valid until `Pattern_free`d). + +### 3.22 `re_escape` + +```c +void re_escape(const char *in, size_t len, char **out, size_t *outlen); +``` + +Matches `re.escape` exactly, including the narrowed escaped-character set CPython adopted in 3.7 (only characters that are actually special in a regex, plus non-ASCII bytes are passed through unescaped rather than every non-alphanumeric character as in pre-3.7 Python). `*out` is a freshly `malloc`'d buffer the caller must `free`. + +### 3.23-3.29 `Match_` accessors + +```c +int Match_group(Match *self, const char *name_or_null, int index, const char **out, size_t *outlen); +int64_t Match_start(Match *self, int group); +int64_t Match_end(Match *self, int group); +void Match_span(Match *self, int group, int64_t *start, int64_t *end); +int64_t Match_start_byte(Match *self, int group); +int64_t Match_end_byte(Match *self, int group); +void Match_span_byte(Match *self, int group, int64_t *start, int64_t *end); +void Match_free(Match *self); +``` + +`Match_group`: if `name_or_null` is non-`NULL`, `index` is ignored and the group is looked up by name (via the same table `Pattern_groupindex_lookup` uses); otherwise `index` (`0` for the whole match) selects the group directly. Returns `1` and sets `*out`/`*outlen` to a borrowed pointer into the underlying `Input`'s buffer (valid as long as both the `Match` and the `Input` are) when the group matched; returns `0` and sets `*out = NULL, *outlen = 0` when the group exists but did not participate (Python's `None`); returns `-1`, leaving `*out`/`*outlen` untouched, when no such group exists at all (by index or by name). + +`Match_start`/`Match_end`/`Match_span`: report the group's span in the pattern's native unit (code point index in `UTF8` mode, byte offset otherwise). **Deviation from CPython:** Python's `Match.start(group)`/`.end(group)` raise `IndexError` for an out-of-range group number and return `-1` only for a valid, unparticipated group; this build returns `-1` for both cases uniformly, since C has no exception to raise. A caller that must tell "no such group" apart from "this group did not participate" should use `Match_group` instead, which does distinguish them (`-1` versus `0` above). + +`Match_start_byte`/`Match_end_byte`/`Match_span_byte`: always report a byte offset into the `Input`, regardless of encoding mode; identical to the non-`_byte` accessors in `BINARY`/`ASCII` mode, and the byte-offset translation of the same span in `UTF8` mode. Use these, not the code-point ones, whenever the result will be used to slice or seek into the raw `Input` buffer or file (`Match_group` already does this translation internally, so most callers only need these directly for cases `Match_group` does not cover, such as reporting a byte offset to an external tool). + +`Match_free`: frees the private per-match state (capture slots and, in `UTF8` mode, the code-point-to-byte-offset table). Does **not** free the `Match` struct itself, and does not affect `Match.re`/`Match.string` (borrowed, Section 1.2). Safe to call more than once on the same `Match` (the second call is a no-op, since the first sets `slots` to `NULL`). + +## 4. Memory ownership summary + +| Object | Created by | Freed by | Notes | +|---|---|---|---| +| `Pattern *` | `re_compile` | `Pattern_free` | Never returned by the `re_`-prefixed module level functions; those only take a pattern string, they do not hand back the `Pattern` they compiled internally. | +| `Input *` | `Input_from_buffer` / `Input_from_file` | `Input_free` | `Input_from_buffer` never owns the wrapped buffer; `Input_from_file` always owns the buffer it read. | +| `Match` (the struct) | The caller (stack or heap) | The caller | Never allocated by the library; only its private `slots` are, released by `Match_free`. | +| `PatternError` (the struct) | The caller (stack or heap) | The caller | Only `msg`/`pattern` are heap allocated; release them with `PatternError_free`. | +| `*out` from `Pattern_sub`/`subn`, `re_escape`, and a `MatchSubCb`'s own `*out` | The library (or, for the callback, the callback itself) | The caller (or, for the callback's `*out`, `Pattern_sub`/`subn`, immediately after copying it) | Plain `malloc`'d buffers; `free()` them normally. | +| Text returned by `Match_group` | Borrowed from the `Input`'s buffer | Nobody (not a separate allocation) | Valid exactly as long as the `Input` and the `Match` both are. | + +## 5. Rejected constructs + +These fail `re_compile` with a `PatternError` naming the construct, rather than being silently mis-parsed. See `concept.md` Section 4/13 for why each is out of scope for this build specifically (as opposed to out of scope for Python `re`, Section 6 below). + +- Conditional groups: `(?(id)yes|no)`, `(?(name)yes|no)`. +- Scoped inline flags: `(?i:...)`, `(?imsx-imsx:...)` (global inline flags at the very start of the pattern, `(?aiLmsux)`, are supported). +- Named code points: `\N{NAME}`. +- Variable-width lookbehind: `(?<=...)`/`(? +#include +#include + +static void usage(const char *prog) { + fprintf(stderr, + "usage: %s [-i] [-n] [-c] [-o] [-v] [-m MODE] PATTERN FILE\n" + " %s [-m MODE] --sub REPL PATTERN FILE\n" + "\n" + " -i ignore case (IGNORECASE)\n" + " -n prefix each match with its 1-based line number\n" + " -c print only a count of matching lines\n" + " -o print only the matched text, not the whole line\n" + " -v print lines that do NOT match\n" + " -m MODE ascii (default), utf8, or binary\n" + " --sub REPL replace every match with REPL and print the result\n", + prog, prog); +} + +static int mode_flag(const char *m) { + if (strcmp(m, "ascii") == 0) return ASCII; + if (strcmp(m, "utf8") == 0) return UTF8; + if (strcmp(m, "binary") == 0) return BINARY; + fprintf(stderr, "unknown mode '%s' (expected ascii, utf8, or binary)\n", m); + exit(2); +} + +typedef struct { long lineno; int with_lineno; } MatchLinePrint; +static void print_one_match(void *ctx, const Match *m_const) { + MatchLinePrint *p = ctx; + Match *m = (Match *)m_const; + const char *g; size_t glen; + Match_group(m, NULL, 0, &g, &glen); + if (p->with_lineno) printf("%ld:", p->lineno); + fwrite(g, 1, glen, stdout); + printf("\n"); +} + +int main(int argc, char **argv) { + int flags = 0; + int opt_n = 0, opt_c = 0, opt_o = 0, opt_v = 0; + const char *sub_repl = NULL; + int mode = ASCII; + int i = 1; + for (; i < argc; i++) { + const char *a = argv[i]; + if (strcmp(a, "-m") == 0 && i + 1 < argc) { mode = mode_flag(argv[++i]); continue; } + if (strcmp(a, "--sub") == 0 && i + 1 < argc) { sub_repl = argv[++i]; continue; } + if (strcmp(a, "-h") == 0 || strcmp(a, "--help") == 0) { usage(argv[0]); return 0; } + if (a[0] == '-' && a[1] != '-' && a[1] != '\0') { + int bad = 0; + for (const char *c = a + 1; *c; c++) { + switch (*c) { + case 'i': flags |= IGNORECASE; break; + case 'n': opt_n = 1; break; + case 'c': opt_c = 1; break; + case 'o': opt_o = 1; break; + case 'v': opt_v = 1; break; + default: bad = 1; break; + } + } + if (!bad) continue; + } + break; + } + if (argc - i != 2) { usage(argv[0]); return 2; } + const char *pattern = argv[i], *path = argv[i + 1]; + + PatternError err; memset(&err, 0, sizeof err); + Pattern *pat = re_compile(pattern, strlen(pattern), flags | mode, &err); + if (!pat) { + fprintf(stderr, "%s: pattern error at position %lld: %s\n", argv[0], (long long)err.pos, err.msg); + return 2; + } + + if (sub_repl) { + Input *file = Input_from_file(path, &err); + if (!file) { + fprintf(stderr, "%s: cannot read '%s': %s\n", argv[0], path, err.msg ? err.msg : "unknown error"); + Pattern_free(pat); + return 2; + } + char *out; size_t outlen; + int n; + if (Pattern_subn(pat, file, sub_repl, NULL, NULL, 0, &out, &outlen, &n) < 0) { + fprintf(stderr, "%s: matching failed (input too large/complex for this pattern, see README.md)\n", argv[0]); + Input_free(file); Pattern_free(pat); + return 1; + } + fwrite(out, 1, outlen, stdout); + fprintf(stderr, "%d replacement(s)\n", n); + free(out); + Input_free(file); Pattern_free(pat); + return n > 0 ? 0 : 1; + } + + /* Line oriented search: split on raw 0x0A, which is what every one + * of the three modes (BINARY, ASCII, UTF8) agrees is a line break + * (concept.md 7.2 notes 0x0A never appears inside a multi-byte + * UTF-8 sequence), then search within each line's [pos, endpos). + * Input's internals are intentionally opaque outside regexx.c + * (concept.md 9.0), so each line is handed to the library as its + * own small Input built with Input_from_buffer. */ + FILE *fp = fopen(path, "rb"); + if (!fp) { fprintf(stderr, "%s: cannot reopen '%s'\n", argv[0], path); return 2; } + fseek(fp, 0, SEEK_END); + long fsize = ftell(fp); + rewind(fp); + char *filebuf = malloc((size_t)fsize > 0 ? (size_t)fsize : 1); + size_t got = fread(filebuf, 1, (size_t)fsize, fp); + fclose(fp); + + long lineno = 0; + size_t line_start = 0; + int matched_lines = 0; + for (size_t pos = 0; pos <= got; pos++) { + if (pos == got || filebuf[pos] == '\n') { + lineno++; + size_t line_len = pos - line_start; + Input *lin = Input_from_buffer((const uint8_t *)(filebuf + line_start), line_len); + Match m; memset(&m, 0, sizeof m); + int r = Pattern_search(pat, lin, 0, -1, &m); + int is_match = (r == 1); + if (is_match) Match_free(&m); + if (is_match != opt_v) { + matched_lines++; + if (!opt_c) { + if (opt_o && is_match) { + /* -o: every match on the line gets its own + * output line, demonstrating Pattern_finditer. */ + MatchLinePrint ctx = { lineno, opt_n }; + Pattern_finditer(pat, lin, 0, -1, print_one_match, &ctx); + } else { + if (opt_n) printf("%ld:", lineno); + fwrite(filebuf + line_start, 1, line_len, stdout); + printf("\n"); + } + } + } + Input_free(lin); + line_start = pos + 1; + } + } + if (opt_c) printf("%d\n", matched_lines); + + free(filebuf); + Pattern_free(pat); + return matched_lines > 0 ? 0 : 1; +} diff --git a/regexx.c b/regexx.c new file mode 100644 index 0000000..6e199fb --- /dev/null +++ b/regexx.c @@ -0,0 +1,1626 @@ +/* regexx.c - a single-file C regex interpreter with Python `re` semantics. + * + * See concept.md for the full design rationale. This file is the v1 + * implementation: it covers the public API and pattern syntax listed in + * README.md "Implementation status" in full, executed by a single + * recursive backtracking engine (concept.md Section 7.3) over a fully + * materialized copy of the input. It does not yet implement the + * streaming, bounded-memory regular engine of concept.md Section 7.2; + * that remains future work, tracked in README.md. + * + * Naming convention (concept.md Section 9.0): every identifier with a + * direct Python `re` counterpart uses that counterpart's exact spelling. + * + * Not reentrant: re_compile() uses one process-wide parser scratch pool + * and the module level functions share one process-wide pattern cache + * (concept.md 13.6); do not call regexx functions from more than one + * thread without external synchronization. + */ +#define _DEFAULT_SOURCE +#include "regexx.h" +#include +#include +#include +#include +#include +#include +#include +#include +#include + +/* ================================================================ + * 0. Small utilities: dynamic array, allocation helpers + * ================================================================ */ + +typedef struct { void *data; size_t len, cap, elemsize; } DArr; + +static void da_init(DArr *a, size_t elemsize) { + a->data = NULL; a->len = 0; a->cap = 0; a->elemsize = elemsize; +} +static void *da_push(DArr *a) { + if (a->len == a->cap) { + size_t nc = a->cap ? a->cap * 2 : 8; + a->data = realloc(a->data, nc * a->elemsize); + a->cap = nc; + } + void *p = (char *)a->data + a->len * a->elemsize; + a->len++; + return p; +} +static void da_free(DArr *a) { free(a->data); a->data = NULL; a->len = a->cap = 0; } + +static char *xstrndup(const char *s, size_t n) { + char *r = malloc(n + 1); + memcpy(r, s, n); + r[n] = 0; + return r; +} + +static void ensure_unicode_locale(void) { + static int done = 0; + if (!done) { setlocale(LC_CTYPE, "C.utf8"); done = 1; } +} + +/* ================================================================ + * 1. UTF-8 decoding (concept.md Section 7.2's continuation handling is + * not needed here since v1 materializes the whole input up front, + * see README.md "Implementation status") + * ================================================================ */ + +static int utf8_decode(const uint8_t *buf, size_t len, size_t i, uint32_t *cp) { + uint8_t b0 = buf[i]; + if (b0 < 0x80) { *cp = b0; return 1; } + if ((b0 & 0xE0) == 0xC0) { + if (i + 1 >= len || (buf[i+1] & 0xC0) != 0x80) return 0; + uint32_t v = (uint32_t)((b0 & 0x1F) << 6) | (buf[i+1] & 0x3F); + if (v < 0x80) return 0; + *cp = v; return 2; + } + if ((b0 & 0xF0) == 0xE0) { + if (i + 2 >= len || (buf[i+1] & 0xC0) != 0x80 || (buf[i+2] & 0xC0) != 0x80) return 0; + uint32_t v = ((uint32_t)(b0 & 0x0F) << 12) | ((uint32_t)(buf[i+1] & 0x3F) << 6) | (buf[i+2] & 0x3F); + if (v < 0x800 || (v >= 0xD800 && v <= 0xDFFF)) return 0; + *cp = v; return 3; + } + if ((b0 & 0xF8) == 0xF0) { + if (i + 3 >= len || (buf[i+1] & 0xC0) != 0x80 || (buf[i+2] & 0xC0) != 0x80 || (buf[i+3] & 0xC0) != 0x80) return 0; + uint32_t v = ((uint32_t)(b0 & 0x07) << 18) | ((uint32_t)(buf[i+1] & 0x3F) << 12) | ((uint32_t)(buf[i+2] & 0x3F) << 6) | (buf[i+3] & 0x3F); + if (v < 0x10000 || v > 0x10FFFF) return 0; + *cp = v; return 4; + } + return 0; +} + +/* ================================================================ + * 2. Character classification predicates (mode aware, Section 2.2/14.4) + * ================================================================ */ + +#define MODE_BINARY 0 +#define MODE_ASCII 1 +#define MODE_UTF8 2 + +static int is_ascii_word(uint32_t cp) { + return (cp >= 'a' && cp <= 'z') || (cp >= 'A' && cp <= 'Z') || (cp >= '0' && cp <= '9') || cp == '_'; +} +static int is_ascii_space(uint32_t cp) { + return cp == ' ' || cp == '\t' || cp == '\n' || cp == '\r' || cp == '\f' || cp == '\v'; +} +static int is_ascii_digit(uint32_t cp) { return cp >= '0' && cp <= '9'; } + +static int cls_is_digit(uint32_t cp, int mode, int flags) { + if (mode == MODE_UTF8 && !(flags & ASCII)) { ensure_unicode_locale(); return iswdigit((wint_t)cp) ? 1 : 0; } + return is_ascii_digit(cp); +} +static int cls_is_space(uint32_t cp, int mode, int flags) { + if (mode == MODE_UTF8 && !(flags & ASCII)) { ensure_unicode_locale(); return iswspace((wint_t)cp) ? 1 : 0; } + return is_ascii_space(cp); +} +static int cls_is_word(uint32_t cp, int mode, int flags) { + if (mode == MODE_UTF8 && !(flags & ASCII)) { + ensure_unicode_locale(); + return (iswalnum((wint_t)cp) || cp == '_') ? 1 : 0; + } + if (is_ascii_word(cp)) return 1; + if ((flags & LOCALE) && mode != MODE_UTF8 && cp >= 0x80 && cp <= 0xFF) return 1; + return 0; +} +static uint32_t cls_swapcase(uint32_t cp, int mode, int flags) { + if (mode == MODE_UTF8 && !(flags & ASCII)) { + ensure_unicode_locale(); + wint_t lo = towlower((wint_t)cp), up = towupper((wint_t)cp); + if ((uint32_t)lo != cp) return (uint32_t)lo; + if ((uint32_t)up != cp) return (uint32_t)up; + return cp; + } + if (cp >= 'a' && cp <= 'z') return cp - 32; + if (cp >= 'A' && cp <= 'Z') return cp + 32; + return cp; +} +static uint32_t cls_fold(uint32_t cp, int mode, int flags) { + if (mode == MODE_UTF8 && !(flags & ASCII)) { ensure_unicode_locale(); return (uint32_t)towlower((wint_t)cp); } + if (cp >= 'A' && cp <= 'Z') return cp + 32; + return cp; +} + +/* ================================================================ + * 3. AST + * ================================================================ */ + +enum { + N_CHAR, N_ANY, N_CLASS, N_CONCAT, N_ALT, N_REPEAT, N_GROUP, + N_BACKREF, N_ANCHOR, N_LOOKAROUND, N_ATOMIC, N_EMPTY +}; +enum { A_BOL, A_EOL, A_BOS, A_EOS, A_WB, A_NWB }; +enum { LK_AHEAD, LK_BEHIND }; +enum { CI_RANGE, CI_D, CI_W, CI_S }; + +typedef struct { int kind; uint32_t lo, hi; int neg; } ClassItem; + +typedef struct Node { + int kind; + struct Node *a, *b; + int32_t lo, hi; + int greedy; + uint32_t ch; + DArr items; + int class_negate; + int group_index; + char *group_name; + int backref_index; + char *backref_name; + int anchor_kind; + int lookaround_kind; + int lookaround_negate; + int width; +} Node; + +typedef struct { Node **data; size_t len, cap; } NodePool; +static NodePool g_pool; + +static Node *node_new(int kind) { + if (g_pool.len == g_pool.cap) { + g_pool.cap = g_pool.cap ? g_pool.cap * 2 : 64; + g_pool.data = realloc(g_pool.data, g_pool.cap * sizeof(Node *)); + } + Node *n = calloc(1, sizeof(Node)); + n->kind = kind; n->group_index = -1; n->backref_index = -1; n->greedy = 1; n->width = -1; + da_init(&n->items, sizeof(ClassItem)); + g_pool.data[g_pool.len++] = n; + return n; +} +static void nodepool_free_all(NodePool *pool) { + for (size_t i = 0; i < pool->len; i++) { + Node *n = pool->data[i]; + da_free(&n->items); + free(n->group_name); + free(n->backref_name); + free(n); + } + free(pool->data); + pool->data = NULL; pool->len = pool->cap = 0; +} + +static int node_nullable(Node *n) { + if (!n) return 1; + switch (n->kind) { + case N_CHAR: case N_ANY: case N_CLASS: case N_BACKREF: return 0; + case N_CONCAT: return node_nullable(n->a) && node_nullable(n->b); + case N_ALT: return node_nullable(n->a) || node_nullable(n->b); + case N_REPEAT: return n->lo == 0 || node_nullable(n->a); + case N_GROUP: return node_nullable(n->a); + case N_ANCHOR: case N_LOOKAROUND: return 1; + case N_ATOMIC: return node_nullable(n->a); + default: return 1; + } +} +static int node_width(Node *n) { + if (!n) return 0; + switch (n->kind) { + case N_CHAR: case N_ANY: case N_CLASS: return 1; + case N_CONCAT: { int wa = node_width(n->a), wb = node_width(n->b); return (wa < 0 || wb < 0) ? -1 : wa + wb; } + case N_ALT: { int wa = node_width(n->a), wb = node_width(n->b); return (wa < 0 || wb < 0 || wa != wb) ? -1 : wa; } + case N_REPEAT: { if (n->lo != n->hi) return -1; int w = node_width(n->a); return w < 0 ? -1 : w * n->lo; } + case N_GROUP: return node_width(n->a); + case N_BACKREF: return -1; + case N_ANCHOR: case N_LOOKAROUND: return 0; + case N_ATOMIC: return node_width(n->a); + default: return 0; + } +} +static int is_simple_atom(Node *n) { return n->kind == N_CHAR || n->kind == N_ANY || n->kind == N_CLASS; } + +/* ================================================================ + * 4. Parser + * ================================================================ */ + +typedef struct { char *name; int index; } GroupName; + +typedef struct { + const char *s; + size_t len, i; + int flags; + int mode; + int ngroups; + DArr backrefs; /* Node* */ + DArr groupnames; /* GroupName */ + char errmsg[256]; + int64_t errpos; + int failed; +} Parser; + +static void perr(Parser *p, int64_t pos, const char *fmt, ...) { + if (p->failed) return; + p->failed = 1; p->errpos = pos; + va_list ap; va_start(ap, fmt); + vsnprintf(p->errmsg, sizeof p->errmsg, fmt, ap); + va_end(ap); +} +static int peek(Parser *p) { return p->i < p->len ? (unsigned char)p->s[p->i] : -1; } +static int peek2(Parser *p) { return p->i + 1 < p->len ? (unsigned char)p->s[p->i + 1] : -1; } +static int at_end(Parser *p) { return p->i >= p->len; } +static int hexval(int c) { + if (c >= '0' && c <= '9') return c - '0'; + if (c >= 'a' && c <= 'f') return c - 'a' + 10; + if (c >= 'A' && c <= 'F') return c - 'A' + 10; + return -1; +} + +static Node *parse_alt(Parser *p); + +static Node *parse_escape(Parser *p, int in_class, uint32_t *literal_cp, int *is_literal) { + *is_literal = 0; + if (at_end(p)) { perr(p, (int64_t)p->i, "bad escape (end of pattern)"); return NULL; } + int c = peek(p); p->i++; + switch (c) { + case 'n': *literal_cp = '\n'; *is_literal = 1; return NULL; + case 'r': *literal_cp = '\r'; *is_literal = 1; return NULL; + case 't': *literal_cp = '\t'; *is_literal = 1; return NULL; + case 'f': *literal_cp = '\f'; *is_literal = 1; return NULL; + case 'v': *literal_cp = '\v'; *is_literal = 1; return NULL; + case 'a': *literal_cp = '\a'; *is_literal = 1; return NULL; + case '0': { + uint32_t v = 0; int n = 0; + while (n < 2 && peek(p) >= '0' && peek(p) <= '7') { v = v * 8 + (uint32_t)(peek(p) - '0'); p->i++; n++; } + *literal_cp = v; *is_literal = 1; return NULL; + } + case 'x': { + int h1 = hexval(peek(p)); + if (h1 < 0) { perr(p, (int64_t)p->i, "incomplete escape \\x"); return NULL; } + p->i++; + int h2 = hexval(peek(p)); + if (h2 < 0) { perr(p, (int64_t)p->i, "incomplete escape \\x"); return NULL; } + p->i++; + *literal_cp = (uint32_t)(h1 * 16 + h2); *is_literal = 1; return NULL; + } + case 'u': case 'U': { + int ndig = (c == 'u') ? 4 : 8; + uint32_t v = 0; + for (int k = 0; k < ndig; k++) { + int h = hexval(peek(p)); + if (h < 0) { perr(p, (int64_t)p->i, "incomplete escape \\%c", c); return NULL; } + v = v * 16 + (uint32_t)h; p->i++; + } + *literal_cp = v; *is_literal = 1; return NULL; + } + case 'N': + perr(p, (int64_t)p->i, "\\N{...} named code points are not implemented in this build"); + return NULL; + case 'd': case 'D': case 'w': case 'W': case 's': case 'S': { + Node *n = node_new(N_CLASS); + ClassItem *it = da_push(&n->items); + it->kind = (c == 'd' || c == 'D') ? CI_D : (c == 'w' || c == 'W') ? CI_W : CI_S; + it->neg = (c == 'D' || c == 'W' || c == 'S') ? 1 : 0; + it->lo = it->hi = 0; + return n; + } + case 'b': + if (in_class) { *literal_cp = '\b'; *is_literal = 1; return NULL; } + { Node *n = node_new(N_ANCHOR); n->anchor_kind = A_WB; return n; } + case 'B': { Node *n = node_new(N_ANCHOR); n->anchor_kind = A_NWB; return n; } + case 'A': { Node *n = node_new(N_ANCHOR); n->anchor_kind = A_BOS; return n; } + case 'Z': { Node *n = node_new(N_ANCHOR); n->anchor_kind = A_EOS; return n; } + case 'g': { + if (in_class) { *literal_cp = 'g'; *is_literal = 1; return NULL; } + if (peek(p) != '<') { perr(p, (int64_t)p->i, "missing < in \\g<...>"); return NULL; } + p->i++; + size_t start = p->i; + while (!at_end(p) && peek(p) != '>') p->i++; + if (at_end(p)) { perr(p, (int64_t)p->i, "missing > in \\g<...>"); return NULL; } + char *ref = xstrndup(p->s + start, p->i - start); + p->i++; + Node *n = node_new(N_BACKREF); + size_t rl = strlen(ref); + int all_digit = rl > 0; + for (size_t k = 0; k < rl; k++) if (!isdigit((unsigned char)ref[k])) all_digit = 0; + if (all_digit) { n->backref_index = atoi(ref); free(ref); } + else n->backref_name = ref; + Node **slot = da_push(&p->backrefs); *slot = n; + return n; + } + case '1': case '2': case '3': case '4': case '5': case '6': case '7': case '8': case '9': { + if (in_class) { *literal_cp = (uint32_t)c; *is_literal = 1; return NULL; } + uint32_t v = (uint32_t)(c - '0'); + while (peek(p) >= '0' && peek(p) <= '9') { v = v * 10 + (uint32_t)(peek(p) - '0'); p->i++; } + Node *n = node_new(N_BACKREF); + n->backref_index = (int)v; + Node **slot = da_push(&p->backrefs); *slot = n; + return n; + } + default: + *literal_cp = (uint32_t)c; *is_literal = 1; return NULL; + } +} + +static uint32_t read_raw_char(Parser *p) { + if (p->mode == MODE_UTF8) { + uint32_t cp; + int n = utf8_decode((const uint8_t *)p->s, p->len, p->i, &cp); + if (n == 0) { perr(p, (int64_t)p->i, "invalid UTF-8 in pattern"); p->i++; return 0; } + p->i += (size_t)n; + return cp; + } + return (unsigned char)p->s[p->i++]; +} + +static Node *parse_class(Parser *p) { + Node *n = node_new(N_CLASS); + p->i++; + if (peek(p) == '^') { n->class_negate = 1; p->i++; } + int first = 1; + while (1) { + if (at_end(p)) { perr(p, (int64_t)p->i, "unterminated character set"); return n; } + if (peek(p) == ']' && !first) { p->i++; break; } + first = 0; + uint32_t lo; + if (peek(p) == '\\') { + p->i++; + uint32_t litcp; int is_lit; + Node *sub = parse_escape(p, 1, &litcp, &is_lit); + if (p->failed) return n; + if (!is_lit && sub) { + ClassItem *it = da_push(&n->items); + *it = ((ClassItem *)sub->items.data)[0]; + continue; + } + lo = litcp; + } else if (peek(p) == '[' && peek2(p) == ':') { + perr(p, (int64_t)p->i, "POSIX named classes ([:alpha:] etc) are not part of Python re, see concept.md 14.3"); + return n; + } else { + lo = read_raw_char(p); + if (p->failed) return n; + } + uint32_t hi = lo; + if (peek(p) == '-' && peek2(p) != ']' && peek2(p) != -1) { + p->i++; + if (peek(p) == '\\') { + p->i++; + uint32_t litcp; int is_lit; + parse_escape(p, 1, &litcp, &is_lit); + if (p->failed) return n; + if (!is_lit) { perr(p, (int64_t)p->i, "bad character range"); return n; } + hi = litcp; + } else { + hi = read_raw_char(p); + if (p->failed) return n; + } + if (hi < lo) { perr(p, (int64_t)p->i, "bad character range"); return n; } + } + ClassItem *it = da_push(&n->items); + it->kind = CI_RANGE; it->lo = lo; it->hi = hi; it->neg = 0; + } + return n; +} + +static Node *make_concat(Node *a, Node *b) { + if (!a) return b; + if (!b) return a; + Node *n = node_new(N_CONCAT); + n->a = a; n->b = b; + return n; +} + +static Node *parse_repeat_suffix(Parser *p, Node *atom) { + int lo = -1, hi = -1; + if (peek(p) == '*') { lo = 0; hi = -1; p->i++; } + else if (peek(p) == '+') { lo = 1; hi = -1; p->i++; } + else if (peek(p) == '?') { lo = 0; hi = 1; p->i++; } + else if (peek(p) == '{') { + size_t save = p->i; + p->i++; + int have_lo = 0, have_hi = 0; long vlo = 0, vhi = 0; + while (isdigit(peek(p))) { vlo = vlo * 10 + (peek(p) - '0'); p->i++; have_lo = 1; } + int has_comma = 0; + if (peek(p) == ',') { + has_comma = 1; p->i++; + while (isdigit(peek(p))) { vhi = vhi * 10 + (peek(p) - '0'); p->i++; have_hi = 1; } + } + if (peek(p) != '}' || (!have_lo && !has_comma)) { p->i = save; return atom; } + p->i++; + lo = have_lo ? (int)vlo : 0; + hi = has_comma ? (have_hi ? (int)vhi : -1) : lo; + if (hi != -1 && hi < lo) { perr(p, (int64_t)save, "min repeat greater than max repeat"); return atom; } + } else { + return atom; + } + Node *rep = node_new(N_REPEAT); + rep->a = atom; rep->lo = lo; rep->hi = hi; rep->greedy = 1; + if (peek(p) == '?') { rep->greedy = 0; p->i++; return rep; } + if (peek(p) == '+') { + p->i++; + Node *at = node_new(N_ATOMIC); + at->a = rep; + return at; + } + return rep; +} + +static Node *parse_group_or_special(Parser *p) { + size_t open_pos = p->i; + p->i++; + if (peek(p) == '?') { + p->i++; + int c = peek(p); + if (c == ':') { + p->i++; + Node *body = parse_alt(p); + if (peek(p) != ')') { perr(p, (int64_t)p->i, "missing ), unterminated subpattern"); return body; } + p->i++; + Node *g = node_new(N_GROUP); g->a = body; g->group_index = -1; + return g; + } + if (c == '#') { + p->i++; + while (!at_end(p) && peek(p) != ')') p->i++; + if (at_end(p)) { perr(p, (int64_t)p->i, "missing ), unterminated comment"); return node_new(N_EMPTY); } + p->i++; + return node_new(N_EMPTY); + } + if (c == '=' || c == '!') { + p->i++; + Node *body = parse_alt(p); + if (peek(p) != ')') { perr(p, (int64_t)p->i, "missing ), unterminated subpattern"); return body; } + p->i++; + Node *n = node_new(N_LOOKAROUND); + n->a = body; n->lookaround_kind = LK_AHEAD; n->lookaround_negate = (c == '!'); + return n; + } + if (c == '<' && (peek2(p) == '=' || peek2(p) == '!')) { + int neg = (peek2(p) == '!'); + p->i += 2; + Node *body = parse_alt(p); + if (peek(p) != ')') { perr(p, (int64_t)p->i, "missing ), unterminated subpattern"); return body; } + p->i++; + int w = node_width(body); + if (w < 0) { perr(p, (int64_t)open_pos, "look-behind requires fixed-width pattern"); return node_new(N_EMPTY); } + Node *n = node_new(N_LOOKAROUND); + n->a = body; n->lookaround_kind = LK_BEHIND; n->lookaround_negate = neg; n->width = w; + return n; + } + if (c == '>') { + p->i++; + Node *body = parse_alt(p); + if (peek(p) != ')') { perr(p, (int64_t)p->i, "missing ), unterminated subpattern"); return body; } + p->i++; + Node *n = node_new(N_ATOMIC); n->a = body; + return n; + } + if (c == 'P') { + p->i++; + if (peek(p) == '<') { + p->i++; + size_t start = p->i; + while (!at_end(p) && peek(p) != '>') p->i++; + if (at_end(p)) { perr(p, (int64_t)p->i, "missing >, unterminated name"); return node_new(N_EMPTY); } + char *name = xstrndup(p->s + start, p->i - start); + p->i++; + int idx = ++p->ngroups; + Node *body = parse_alt(p); + if (peek(p) != ')') { perr(p, (int64_t)p->i, "missing ), unterminated subpattern"); free(name); return body; } + p->i++; + Node *g = node_new(N_GROUP); g->a = body; g->group_index = idx; g->group_name = name; + GroupName *gn = da_push(&p->groupnames); + gn->name = xstrndup(name, strlen(name)); gn->index = idx; + return g; + } + if (peek(p) == '=') { + p->i++; + size_t start = p->i; + while (!at_end(p) && peek(p) != ')') p->i++; + if (at_end(p)) { perr(p, (int64_t)p->i, "missing ), unterminated name"); return node_new(N_EMPTY); } + char *name = xstrndup(p->s + start, p->i - start); + p->i++; + Node *n = node_new(N_BACKREF); n->backref_name = name; + Node **slot = da_push(&p->backrefs); *slot = n; + return n; + } + perr(p, (int64_t)p->i, "unknown extension ?P"); + return node_new(N_EMPTY); + } + if (c == '(') { + perr(p, (int64_t)open_pos, "conditional groups (?(id)yes|no) are not implemented in this build"); + return node_new(N_EMPTY); + } + { + size_t save = p->i; + int newflags = 0, any = 0; + while (!at_end(p) && strchr("aiLmsux", peek(p))) { + switch (peek(p)) { + case 'i': newflags |= IGNORECASE; break; + case 'm': newflags |= MULTILINE; break; + case 's': newflags |= DOTALL; break; + case 'x': newflags |= VERBOSE; break; + case 'a': newflags |= ASCII; break; + case 'L': newflags |= LOCALE; break; + default: break; + } + p->i++; any = 1; + } + if (any && peek(p) == ')') { + p->i++; + if (open_pos != 0) { perr(p, (int64_t)open_pos, "global flags not at the start of the expression"); return node_new(N_EMPTY); } + p->flags |= newflags; + return node_new(N_EMPTY); + } + p->i = save; + perr(p, (int64_t)open_pos, "scoped inline flags (?flags:...) are not implemented in this build"); + return node_new(N_EMPTY); + } + } + int idx = ++p->ngroups; + Node *body = parse_alt(p); + if (peek(p) != ')') { perr(p, (int64_t)p->i, "missing ), unterminated subpattern"); return body; } + p->i++; + Node *g = node_new(N_GROUP); g->a = body; g->group_index = idx; + return g; +} + +static Node *parse_atom(Parser *p) { + int c = peek(p); + if (c == '(') return parse_group_or_special(p); + if (c == '[') return parse_class(p); + if (c == '.') { p->i++; return node_new(N_ANY); } + if (c == '^') { p->i++; Node *n = node_new(N_ANCHOR); n->anchor_kind = A_BOL; return n; } + if (c == '$') { p->i++; Node *n = node_new(N_ANCHOR); n->anchor_kind = A_EOL; return n; } + if (c == '\\') { + p->i++; + uint32_t litcp; int is_lit; + Node *n = parse_escape(p, 0, &litcp, &is_lit); + if (p->failed) return node_new(N_EMPTY); + if (is_lit) { Node *cn = node_new(N_CHAR); cn->ch = litcp; return cn; } + return n; + } + if (c == '*' || c == '+' || c == '?') { + perr(p, (int64_t)p->i, "nothing to repeat"); + p->i++; + return node_new(N_EMPTY); + } + uint32_t cp = read_raw_char(p); + if (p->failed) return node_new(N_EMPTY); + Node *cn = node_new(N_CHAR); cn->ch = cp; + return cn; +} + +static Node *parse_concat(Parser *p) { + Node *result = NULL; + while (!at_end(p) && peek(p) != '|' && peek(p) != ')' && !p->failed) { + Node *atom = parse_atom(p); + if (p->failed) break; + atom = parse_repeat_suffix(p, atom); + result = make_concat(result, atom); + } + if (!result) result = node_new(N_EMPTY); + return result; +} + +static Node *parse_alt(Parser *p) { + Node *first = parse_concat(p); + if (peek(p) == '|') { + p->i++; + Node *rest = parse_alt(p); + Node *n = node_new(N_ALT); + n->a = first; n->b = rest; + return n; + } + return first; +} + +/* Strip VERBOSE-mode whitespace and #-comments, respecting [...] classes + * and backslash escapes. Only called once the VERBOSE flag is known. */ +static char *strip_verbose(const char *s, size_t len, size_t *outlen) { + char *out = malloc(len + 1); + size_t o = 0, i = 0; + int in_class = 0; + while (i < len) { + char c = s[i]; + if (c == '\\' && i + 1 < len) { out[o++] = s[i++]; out[o++] = s[i++]; continue; } + if (c == '[' && !in_class) { in_class = 1; out[o++] = s[i++]; continue; } + if (c == ']' && in_class) { in_class = 0; out[o++] = s[i++]; continue; } + if (!in_class && isspace((unsigned char)c)) { i++; continue; } + if (!in_class && c == '#') { while (i < len && s[i] != '\n') i++; continue; } + out[o++] = s[i++]; + } + out[o] = 0; + *outlen = o; + return out; +} + +/* ================================================================ + * 5. Bytecode + * ================================================================ */ + +enum { + OP_CHAR, OP_CLASS, OP_ANY, OP_SPLIT, OP_JMP, OP_SAVE, OP_MATCH, + OP_ASSERT, OP_BACKREF, OP_LOOKAHEAD, OP_LOOKBEHIND, OP_ATOMIC, + OP_RETURN, OP_REPEAT1 +}; + +typedef struct { + int op; + int32_t x, y; + uint32_t data; + int width; + int neg; + int is_loop; + int loop_close; + int32_t lo, hi; + int greedy; + int atomkind; /* OP_REPEAT1: 0=CHAR 1=CLASS 2=ANY */ +} Inst; + +typedef struct { + Inst *insts; + int n, cap; + DArr classnodes; /* Node* */ + int ngroups; + int mode; + int flags; +} Prog; + +static int emit(Prog *pr, int op) { + if (pr->n == pr->cap) { + pr->cap = pr->cap ? pr->cap * 2 : 64; + pr->insts = realloc(pr->insts, (size_t)pr->cap * sizeof(Inst)); + } + Inst *i = &pr->insts[pr->n]; + memset(i, 0, sizeof(*i)); + i->op = op; i->x = i->y = -1; + return pr->n++; +} + +static void compile_node(Prog *pr, Node *n); + +static void compile_repeat(Prog *pr, Node *n) { + if (is_simple_atom(n->a)) { + int pc = emit(pr, OP_REPEAT1); + Inst *in = &pr->insts[pc]; + in->lo = n->lo; in->hi = n->hi; in->greedy = n->greedy; + if (n->a->kind == N_CHAR) { in->atomkind = 0; in->data = n->a->ch; } + else if (n->a->kind == N_ANY) { in->atomkind = 2; } + else { + in->atomkind = 1; + int idx = (int)pr->classnodes.len; + Node **slot = da_push(&pr->classnodes); *slot = n->a; + in->data = (uint32_t)idx; + } + return; + } + int lo = n->lo, hi = n->hi; + for (int k = 0; k < lo; k++) compile_node(pr, n->a); + if (hi == -1) { + int nullable = node_nullable(n->a); + int split_pc = emit(pr, OP_SPLIT); + pr->insts[split_pc].is_loop = nullable; + int body_start = pr->n; + compile_node(pr, n->a); + int jmp_pc = emit(pr, OP_JMP); + pr->insts[jmp_pc].x = split_pc; + pr->insts[jmp_pc].loop_close = nullable; + int end = pr->n; + if (n->greedy) { pr->insts[split_pc].x = body_start; pr->insts[split_pc].y = end; } + else { pr->insts[split_pc].x = end; pr->insts[split_pc].y = body_start; } + } else if (hi > lo) { + DArr splits; da_init(&splits, sizeof(int)); + for (int k = 0; k < hi - lo; k++) { + int sp = emit(pr, OP_SPLIT); + int *slot = da_push(&splits); *slot = sp; + int body_start = pr->n; + compile_node(pr, n->a); + if (n->greedy) pr->insts[sp].x = body_start; else pr->insts[sp].y = body_start; + } + int end = pr->n; + for (size_t k = 0; k < splits.len; k++) { + int sp = ((int *)splits.data)[k]; + if (n->greedy) pr->insts[sp].y = end; else pr->insts[sp].x = end; + } + da_free(&splits); + } +} + +static void compile_node(Prog *pr, Node *n) { + switch (n->kind) { + case N_EMPTY: return; + case N_CHAR: { int pc = emit(pr, OP_CHAR); pr->insts[pc].data = n->ch; return; } + case N_ANY: emit(pr, OP_ANY); return; + case N_CLASS: { + int idx = (int)pr->classnodes.len; + Node **slot = da_push(&pr->classnodes); *slot = n; + int pc = emit(pr, OP_CLASS); pr->insts[pc].data = (uint32_t)idx; + return; + } + case N_CONCAT: compile_node(pr, n->a); compile_node(pr, n->b); return; + case N_ALT: { + int split_pc = emit(pr, OP_SPLIT); + int x0 = pr->n; + compile_node(pr, n->a); + int jmp_pc = emit(pr, OP_JMP); + int y0 = pr->n; + compile_node(pr, n->b); + int end = pr->n; + pr->insts[split_pc].x = x0; pr->insts[split_pc].y = y0; + pr->insts[jmp_pc].x = end; + return; + } + case N_REPEAT: compile_repeat(pr, n); return; + case N_GROUP: { + if (n->group_index >= 0) { int pc = emit(pr, OP_SAVE); pr->insts[pc].data = (uint32_t)(2 * n->group_index); } + compile_node(pr, n->a); + if (n->group_index >= 0) { int pc = emit(pr, OP_SAVE); pr->insts[pc].data = (uint32_t)(2 * n->group_index + 1); } + return; + } + case N_BACKREF: { int pc = emit(pr, OP_BACKREF); pr->insts[pc].data = (uint32_t)n->backref_index; return; } + case N_ANCHOR: { int pc = emit(pr, OP_ASSERT); pr->insts[pc].data = (uint32_t)n->anchor_kind; return; } + case N_LOOKAROUND: { + int op = (n->lookaround_kind == LK_AHEAD) ? OP_LOOKAHEAD : OP_LOOKBEHIND; + int pc = emit(pr, op); + pr->insts[pc].neg = n->lookaround_negate; + pr->insts[pc].width = n->width; + int sub_start = pr->n; + compile_node(pr, n->a); + emit(pr, OP_RETURN); + int after = pr->n; + pr->insts[pc].x = sub_start; pr->insts[pc].y = after; + return; + } + case N_ATOMIC: { + int pc = emit(pr, OP_ATOMIC); + int sub_start = pr->n; + compile_node(pr, n->a); + emit(pr, OP_RETURN); + int after = pr->n; + pr->insts[pc].x = sub_start; pr->insts[pc].y = after; + return; + } + } +} + +/* ================================================================ + * 6. Matcher (recursive backtracking VM, concept.md Section 7.3/7.1) + * ================================================================ */ + +#define MAX_DEPTH 60000 + +typedef struct { + Prog *pr; + const uint32_t *text; + int64_t len; + int64_t *caps; /* 2*(ngroups+1) */ + int64_t require_end; /* -1 unconstrained, -2 "sub-program, report on OP_RETURN", >=0 exact end required */ + int64_t sub_end; /* scratch: end sp reported by a sub-program's OP_RETURN */ + int depth_exceeded; +} MCtx; + +static int class_match(Node *cls, uint32_t cp, int mode, int flags) { + int hit = 0; + for (size_t k = 0; k < cls->items.len && !hit; k++) { + ClassItem *it = &((ClassItem *)cls->items.data)[k]; + int m; + switch (it->kind) { + case CI_RANGE: + m = (cp >= it->lo && cp <= it->hi); + if (!m && (flags & IGNORECASE)) { + uint32_t alt = cls_swapcase(cp, mode, flags); + m = (alt >= it->lo && alt <= it->hi); + } + break; + case CI_D: m = cls_is_digit(cp, mode, flags); break; + case CI_W: m = cls_is_word(cp, mode, flags); break; + case CI_S: m = cls_is_space(cp, mode, flags); break; + default: m = 0; + } + if (it->kind != CI_RANGE && it->neg) m = !m; + if (m) hit = 1; + } + if (cls->class_negate) hit = !hit; + return hit; +} +static int char_eq(uint32_t a, uint32_t b, int mode, int flags) { + if (a == b) return 1; + if (flags & IGNORECASE) return cls_fold(a, mode, flags) == cls_fold(b, mode, flags); + return 0; +} +static int is_word_at(MCtx *c, int64_t pos) { + if (pos < 0 || pos >= c->len) return 0; + return cls_is_word(c->text[pos], c->pr->mode, c->pr->flags); +} + +static int run(MCtx *c, int pc, int64_t sp, int64_t guard_sp, int32_t guard_pc, int depth) { + if (depth > MAX_DEPTH) { c->depth_exceeded = 1; return 0; } + for (;;) { + Inst *in = &c->pr->insts[pc]; + switch (in->op) { + case OP_CHAR: + if (sp >= c->len || !char_eq(c->text[sp], in->data, c->pr->mode, c->pr->flags)) return 0; + pc++; sp++; continue; + case OP_ANY: + if (sp >= c->len) return 0; + if (c->text[sp] == '\n' && !(c->pr->flags & DOTALL)) return 0; + pc++; sp++; continue; + case OP_CLASS: { + if (sp >= c->len) return 0; + Node *cls = ((Node **)c->pr->classnodes.data)[in->data]; + if (!class_match(cls, c->text[sp], c->pr->mode, c->pr->flags)) return 0; + pc++; sp++; continue; + } + case OP_REPEAT1: { + int64_t maxc = (in->hi == -1) ? (c->len - sp) : in->hi; + if (maxc > c->len - sp) maxc = c->len - sp; + int64_t count = 0; + Node *cls = (in->atomkind == 1) ? ((Node **)c->pr->classnodes.data)[in->data] : NULL; + while (count < maxc) { + uint32_t ch = c->text[sp + count]; + int m; + if (in->atomkind == 0) m = char_eq(ch, in->data, c->pr->mode, c->pr->flags); + else if (in->atomkind == 2) m = !(ch == '\n' && !(c->pr->flags & DOTALL)); + else m = class_match(cls, ch, c->pr->mode, c->pr->flags); + if (!m) break; + count++; + } + if (count < in->lo) return 0; + if (in->greedy) { + for (int64_t k = count; k >= in->lo; k--) + if (run(c, pc + 1, sp + k, guard_sp, guard_pc, depth + 1)) return 1; + } else { + for (int64_t k = in->lo; k <= count; k++) + if (run(c, pc + 1, sp + k, guard_sp, guard_pc, depth + 1)) return 1; + } + return 0; + } + case OP_ASSERT: { + int ok; + switch (in->data) { + case A_BOL: ok = (sp == 0) || ((c->pr->flags & MULTILINE) && sp > 0 && c->text[sp - 1] == '\n'); break; + case A_EOL: ok = (sp == c->len) || (c->text[sp] == '\n' && ((c->pr->flags & MULTILINE) || sp == c->len - 1)); break; + case A_BOS: ok = (sp == 0); break; + case A_EOS: ok = (sp == c->len); break; + case A_WB: ok = is_word_at(c, sp - 1) != is_word_at(c, sp); break; + case A_NWB: ok = is_word_at(c, sp - 1) == is_word_at(c, sp); break; + default: ok = 0; + } + if (!ok) return 0; + pc++; continue; + } + case OP_SAVE: { + int64_t old = c->caps[in->data]; + c->caps[in->data] = sp; + if (run(c, pc + 1, sp, guard_sp, guard_pc, depth + 1)) return 1; + c->caps[in->data] = old; + return 0; + } + case OP_SPLIT: + if (in->is_loop) { + if (run(c, in->x, sp, sp, pc, depth + 1)) return 1; + pc = in->y; continue; + } else { + if (run(c, in->x, sp, guard_sp, guard_pc, depth + 1)) return 1; + pc = in->y; continue; + } + case OP_JMP: + if (in->loop_close && in->x == guard_pc && sp == guard_sp) return 0; + pc = in->x; continue; + case OP_BACKREF: { + int gi = (int)in->data; + if (gi < 0 || gi > c->pr->ngroups) return 0; + int64_t s = c->caps[2 * gi], e = c->caps[2 * gi + 1]; + if (s < 0 || e < 0) return 0; + int64_t rl = e - s; + if (sp + rl > c->len) return 0; + for (int64_t k = 0; k < rl; k++) + if (!char_eq(c->text[sp + k], c->text[s + k], c->pr->mode, c->pr->flags)) return 0; + pc++; sp += rl; continue; + } + case OP_LOOKAHEAD: { + size_t ncaps = 2 * (size_t)(c->pr->ngroups + 1); + int64_t *snap = malloc(ncaps * sizeof(int64_t)); + memcpy(snap, c->caps, ncaps * sizeof(int64_t)); + int64_t save_req = c->require_end; c->require_end = -2; + int ok = run(c, in->x, sp, -1, -1, depth + 1); + c->require_end = save_req; + int accept = in->neg ? !ok : ok; + if (!accept) { memcpy(c->caps, snap, ncaps * sizeof(int64_t)); free(snap); return 0; } + free(snap); + pc = in->y; continue; + } + case OP_LOOKBEHIND: { + size_t ncaps = 2 * (size_t)(c->pr->ngroups + 1); + int64_t *snap = malloc(ncaps * sizeof(int64_t)); + memcpy(snap, c->caps, ncaps * sizeof(int64_t)); + int64_t start = sp - in->width; + int ok = 0; + if (start >= 0) { + int64_t save_req = c->require_end; c->require_end = sp; + ok = run(c, in->x, start, -1, -1, depth + 1); + c->require_end = save_req; + } + int accept = in->neg ? !ok : ok; + if (!accept) { memcpy(c->caps, snap, ncaps * sizeof(int64_t)); free(snap); return 0; } + free(snap); + pc = in->y; continue; + } + case OP_ATOMIC: { + int64_t save_req = c->require_end; c->require_end = -2; + int ok = run(c, in->x, sp, -1, -1, depth + 1); + c->require_end = save_req; + if (!ok) return 0; + sp = c->sub_end; + pc = in->y; continue; + } + case OP_RETURN: + if (c->require_end == -2) { c->sub_end = sp; return 1; } + if (c->require_end >= 0) return sp == c->require_end; + return 1; + case OP_MATCH: + if (c->require_end >= 0 && sp != c->require_end) return 0; + c->caps[1] = sp; + return 1; + default: + return 0; + } + } +} + +/* ================================================================ + * 7. Pattern / Match / Input structures + * ================================================================ */ + +typedef struct { char **names; int *idx; int n; } GroupIndex; + +typedef struct { + Prog prog; + GroupIndex gidx; + NodePool pool; + char *pattern_copy; +} PatternImpl; + +struct Input { + uint8_t *buf; + size_t len; + int owns_buf; +}; + +typedef struct { + int64_t *pos; /* 2*(ngroups+1) */ + int ngroups; + int64_t *cp_to_byte; /* NULL unless UTF8 mode; length ngroups-independent, sized to text length+1 */ +} MatchSlots; + +static int groupindex_lookup(GroupIndex *gi, const char *name) { + for (int k = 0; k < gi->n; k++) if (strcmp(gi->names[k], name) == 0) return gi->idx[k]; + return -1; +} +static int resolve_mode(int flags) { + if (flags & BINARY) return MODE_BINARY; + if (flags & UTF8) return MODE_UTF8; + return MODE_ASCII; +} +static void pattern_error_fill(PatternError *err, const char *pattern, const char *msg, int64_t pos) { + if (!err) return; + err->msg = strdup(msg ? msg : ""); + err->pattern = pattern ? strdup(pattern) : NULL; + err->pos = pos; + err->lineno = 1; err->colno = 1; + if (pattern) { + for (int64_t k = 0; k < pos && pattern[k]; k++) { + if (pattern[k] == '\n') { err->lineno++; err->colno = 1; } else err->colno++; + } + } +} + +void PatternError_free(PatternError *err) { + if (!err) return; + free((void *)err->msg); free((void *)err->pattern); + err->msg = NULL; err->pattern = NULL; +} + +/* ================================================================ + * 8. Compile driver + * ================================================================ */ + +Pattern *re_compile(const char *pattern, size_t len, int flags, PatternError *err) { + /* Section 9.3: BINARY/UTF8 select their mode; otherwise ASCII. */ + int mode = resolve_mode(flags); + + /* Step 1: leading global inline flags (?aiLmsux), Section 2.5/2.6 */ + int extra_flags = 0; + size_t body_off = 0; + if (len >= 3 && pattern[0] == '(' && pattern[1] == '?') { + size_t k = 2; int any = 0, ok = 1; + while (k < len && strchr("aiLmsux", (unsigned char)pattern[k])) { + switch (pattern[k]) { + case 'i': extra_flags |= IGNORECASE; break; + case 'm': extra_flags |= MULTILINE; break; + case 's': extra_flags |= DOTALL; break; + case 'x': extra_flags |= VERBOSE; break; + case 'a': extra_flags |= ASCII; break; + case 'L': extra_flags |= LOCALE; break; + default: break; + } + k++; any = 1; + } + if (any && k < len && pattern[k] == ')') { body_off = k + 1; } + else { extra_flags = 0; ok = 0; (void)ok; } + } + int all_flags = flags | extra_flags; + + const char *body = pattern + body_off; + size_t bodylen = len - body_off; + char *stripped = NULL; + if (all_flags & VERBOSE) { + size_t nl; + stripped = strip_verbose(body, bodylen, &nl); + body = stripped; bodylen = nl; + } + + Parser p; memset(&p, 0, sizeof p); + p.s = body; p.len = bodylen; p.i = 0; p.flags = all_flags; p.mode = mode; + da_init(&p.backrefs, sizeof(Node *)); + da_init(&p.groupnames, sizeof(GroupName)); + g_pool.data = NULL; g_pool.len = 0; g_pool.cap = 0; + + Node *ast = parse_alt(&p); + if (!p.failed && !at_end(&p)) { + if (peek(&p) == ')') perr(&p, (int64_t)p.i, "unbalanced parenthesis"); + else perr(&p, (int64_t)p.i, "unexpected character"); + } + if (!p.failed) { + for (size_t k = 0; k < p.backrefs.len; k++) { + Node *b = ((Node **)p.backrefs.data)[k]; + if (b->backref_name) { + int found = -1; + for (size_t j = 0; j < p.groupnames.len; j++) { + GroupName *gn = &((GroupName *)p.groupnames.data)[j]; + if (strcmp(gn->name, b->backref_name) == 0) { found = gn->index; break; } + } + if (found < 0) { perr(&p, 0, "unknown group name '%s'", b->backref_name); break; } + b->backref_index = found; + } else if (b->backref_index < 1 || b->backref_index > p.ngroups) { + perr(&p, 0, "invalid group reference %d", b->backref_index); + break; + } + } + } + + if (p.failed) { + pattern_error_fill(err, pattern, p.errmsg, p.errpos + (int64_t)body_off); + nodepool_free_all(&g_pool); + da_free(&p.backrefs); + for (size_t k = 0; k < p.groupnames.len; k++) free(((GroupName *)p.groupnames.data)[k].name); + da_free(&p.groupnames); + free(stripped); + return NULL; + } + + Prog prog; memset(&prog, 0, sizeof prog); + prog.ngroups = p.ngroups; prog.mode = mode; prog.flags = all_flags; + da_init(&prog.classnodes, sizeof(Node *)); + compile_node(&prog, ast); + emit(&prog, OP_MATCH); + + PatternImpl *impl = calloc(1, sizeof(PatternImpl)); + impl->prog = prog; + impl->pool = g_pool; + impl->pattern_copy = xstrndup(pattern, len); + impl->gidx.n = (int)p.groupnames.len; + impl->gidx.names = malloc(sizeof(char *) * (size_t)(impl->gidx.n ? impl->gidx.n : 1)); + impl->gidx.idx = malloc(sizeof(int) * (size_t)(impl->gidx.n ? impl->gidx.n : 1)); + for (int k = 0; k < impl->gidx.n; k++) { + GroupName *gn = &((GroupName *)p.groupnames.data)[k]; + impl->gidx.names[k] = gn->name; /* transfer ownership */ + impl->gidx.idx[k] = gn->index; + } + da_free(&p.groupnames); + da_free(&p.backrefs); + free(stripped); + + Pattern *pat = calloc(1, sizeof(Pattern)); + pat->pattern = impl->pattern_copy; + pat->flags = all_flags; + pat->groups = prog.ngroups; + pat->groupindex = &impl->gidx; + pat->program = impl; + return pat; +} + +void Pattern_free(Pattern *self) { + if (!self) return; + PatternImpl *impl = (PatternImpl *)self->program; + if (impl) { + free(impl->prog.insts); + da_free(&impl->prog.classnodes); + nodepool_free_all(&impl->pool); + for (int k = 0; k < impl->gidx.n; k++) free(impl->gidx.names[k]); + free(impl->gidx.names); + free(impl->gidx.idx); + free(impl->pattern_copy); + free(impl); + } + free(self); +} + +/* ================================================================ + * 9. Input + * ================================================================ */ + +Input *Input_from_buffer(const uint8_t *buf, size_t len) { + Input *in = calloc(1, sizeof(Input)); + in->buf = (uint8_t *)buf; in->len = len; in->owns_buf = 0; + return in; +} +Input *Input_from_file(const char *path, PatternError *err) { + FILE *f = fopen(path, "rb"); + if (!f) { pattern_error_fill(err, NULL, strerror(errno), 0); return NULL; } + if (fseek(f, 0, SEEK_END) == 0) { + long sz = ftell(f); + if (sz >= 0) { + rewind(f); + uint8_t *buf = malloc((size_t)sz > 0 ? (size_t)sz : 1); + size_t got = fread(buf, 1, (size_t)sz, f); + fclose(f); + Input *in = calloc(1, sizeof(Input)); + in->buf = buf; in->len = got; in->owns_buf = 1; + return in; + } + } + /* Not seekable (a pipe, a FIFO, process substitution, stdin): the + * size cannot be known upfront, so read incrementally into a + * growable buffer instead. Still materializes the whole input in + * memory, per this file's "Implementation status" note. */ + clearerr(f); + size_t cap = 1 << 16, len = 0; + uint8_t *buf = malloc(cap); + size_t n; + while ((n = fread(buf + len, 1, cap - len, f)) > 0) { + len += n; + if (len == cap) { cap *= 2; buf = realloc(buf, cap); } + } + fclose(f); + Input *in = calloc(1, sizeof(Input)); + in->buf = buf; in->len = len; in->owns_buf = 1; + return in; +} +void Input_free(Input *in) { + if (!in) return; + if (in->owns_buf) free(in->buf); + free(in); +} + +/* ================================================================ + * 10. Match buffer construction (UTF-8 pre-decode, Section 8.4/9.3) + * ================================================================ */ + +typedef struct { + uint32_t *text; + int64_t len; + int64_t *cp_to_byte; /* len+1 entries, NULL for non-UTF8 */ +} MatBuf; + +static int build_matbuf(Pattern *pat, Input *in, MatBuf *mb, PatternError *err) { + int mode = resolve_mode(pat->flags); + if (mode == MODE_UTF8) { + DArr cps; da_init(&cps, sizeof(uint32_t)); + DArr offs; da_init(&offs, sizeof(int64_t)); + size_t i = 0; + while (i < in->len) { + uint32_t cp; + int n = utf8_decode(in->buf, in->len, i, &cp); + if (n == 0) { + pattern_error_fill(err, NULL, "invalid UTF-8 in subject", (int64_t)i); + da_free(&cps); da_free(&offs); + return 0; + } + uint32_t *cpp = da_push(&cps); *cpp = cp; + int64_t *op = da_push(&offs); *op = (int64_t)i; + i += (size_t)n; + } + int64_t *sentinel = da_push(&offs); *sentinel = (int64_t)in->len; + mb->text = (uint32_t *)cps.data; + mb->cp_to_byte = (int64_t *)offs.data; + mb->len = (int64_t)cps.len; + } else { + uint32_t *text = malloc((in->len ? in->len : 1) * sizeof(uint32_t)); + for (size_t i = 0; i < in->len; i++) text[i] = in->buf[i]; + mb->text = text; mb->len = (int64_t)in->len; mb->cp_to_byte = NULL; + } + return 1; +} +static void free_matbuf(MatBuf *mb) { free(mb->text); free(mb->cp_to_byte); } + +/* ================================================================ + * 11. Matching driver and Pattern_ methods + * ================================================================ */ + +static void alloc_caps(int64_t **caps, int ngroups) { + *caps = malloc(2 * (size_t)(ngroups + 1) * sizeof(int64_t)); +} +static void reset_caps(int64_t *caps, int ngroups) { + for (int k = 0; k < 2 * (ngroups + 1); k++) caps[k] = -1; +} + +static void fill_match_from_caps(Match *out, Pattern *self, Input *string, int64_t pos, int64_t endpos, + int64_t *caps, int ngroups, MatBuf *mb) { + MatchSlots *ms = calloc(1, sizeof(MatchSlots)); + ms->pos = caps; ms->ngroups = ngroups; + if (mb->cp_to_byte) { + ms->cp_to_byte = malloc((size_t)(mb->len + 1) * sizeof(int64_t)); + memcpy(ms->cp_to_byte, mb->cp_to_byte, (size_t)(mb->len + 1) * sizeof(int64_t)); + } + out->re = self; out->string = string; out->pos = pos; out->endpos = endpos; + out->slots = ms; out->lastindex = -1; out->lastgroup = NULL; + PatternImpl *impl = (PatternImpl *)self->program; + for (int g = ngroups; g >= 1; g--) { + if (caps[2 * g] >= 0) { + out->lastindex = g; + for (int k = 0; k < impl->gidx.n; k++) + if (impl->gidx.idx[k] == g) { out->lastgroup = impl->gidx.names[k]; break; } + break; + } + } +} + +static int64_t clamp(int64_t v, int64_t lo, int64_t hi) { return v < lo ? lo : (v > hi ? hi : v); } + +static int do_one(Pattern *self, Input *string, int64_t pos, int64_t endpos, int anchored, int fullmatch, Match *out) { + MatBuf mb; + if (!build_matbuf(self, string, &mb, NULL)) return -1; + int64_t ep = (endpos < 0) ? mb.len : clamp(endpos, 0, mb.len); + int64_t p0 = clamp(pos, 0, mb.len); + PatternImpl *impl = (PatternImpl *)self->program; + int ng = impl->prog.ngroups; + int64_t *caps; alloc_caps(&caps, ng); + int found = 0; + int64_t last_start = anchored ? p0 : ep; + for (int64_t start = p0; start <= last_start && !found; start++) { + reset_caps(caps, ng); + caps[0] = start; + MCtx c; c.pr = &impl->prog; c.text = mb.text; c.len = ep; c.caps = caps; + c.depth_exceeded = 0; c.require_end = fullmatch ? ep : -1; c.sub_end = -1; + found = run(&c, 0, start, -1, -1, 0); + if (!found && c.depth_exceeded) { free(caps); free_matbuf(&mb); return -1; } + } + if (!found) { free(caps); free_matbuf(&mb); return 0; } + fill_match_from_caps(out, self, string, p0, ep, caps, ng, &mb); + free_matbuf(&mb); + return 1; +} + +int Pattern_match(Pattern *self, Input *string, int64_t pos, int64_t endpos, Match *out) { + return do_one(self, string, pos, endpos, 1, 0, out); +} +int Pattern_fullmatch(Pattern *self, Input *string, int64_t pos, int64_t endpos, Match *out) { + return do_one(self, string, pos, endpos, 1, 1, out); +} +int Pattern_search(Pattern *self, Input *string, int64_t pos, int64_t endpos, Match *out) { + return do_one(self, string, pos, endpos, 0, 0, out); +} + +int Pattern_finditer(Pattern *self, Input *string, int64_t pos, int64_t endpos, MatchIterCb cb, void *ctx) { + MatBuf mb; + if (!build_matbuf(self, string, &mb, NULL)) return -1; + int64_t ep = (endpos < 0) ? mb.len : clamp(endpos, 0, mb.len); + int64_t start = clamp(pos, 0, mb.len); + PatternImpl *impl = (PatternImpl *)self->program; + int ng = impl->prog.ngroups; + int count = 0; + while (start <= ep) { + int64_t *caps; alloc_caps(&caps, ng); + int found = 0; + int64_t s; + for (s = start; s <= ep && !found; s++) { + reset_caps(caps, ng); + caps[0] = s; + MCtx c; c.pr = &impl->prog; c.text = mb.text; c.len = ep; c.caps = caps; + c.depth_exceeded = 0; c.require_end = -1; c.sub_end = -1; + found = run(&c, 0, s, -1, -1, 0); + if (!found && c.depth_exceeded) { free(caps); free_matbuf(&mb); return -1; } + } + if (!found) { free(caps); break; } + Match m; memset(&m, 0, sizeof m); + MatBuf shallow = mb; shallow.cp_to_byte = mb.cp_to_byte; /* share for lookup, copied inside fill */ + fill_match_from_caps(&m, self, string, start, ep, caps, ng, &shallow); + cb(ctx, &m); + count++; + int64_t mend = caps[1]; + start = (mend > caps[0]) ? mend : caps[0] + 1; + Match_free(&m); + } + free_matbuf(&mb); + return count; +} +int Pattern_findall(Pattern *self, Input *string, int64_t pos, int64_t endpos, MatchIterCb cb, void *ctx) { + return Pattern_finditer(self, string, pos, endpos, cb, ctx); +} + +/* Pattern_split's contract: cb is invoked once per element of the list + * Python's re.split() would return, in order, each time as a Match + * whose group 0 (via Match_group(m, NULL, 0, ...)) is that element's + * text; a "None" element (an unparticipated capturing group between + * two matches) reports via Match_group returning 0, matching how an + * unparticipated group reports on any other Match. */ +static void emit_split_item(MatchIterCb cb, void *ctx, Pattern *self, Input *string, MatBuf *mb, int64_t s, int64_t e) { + int64_t *caps = malloc(2 * sizeof(int64_t)); + caps[0] = s; caps[1] = e; + Match m; memset(&m, 0, sizeof m); + fill_match_from_caps(&m, self, string, 0, mb->len, caps, 0, mb); + cb(ctx, &m); + Match_free(&m); +} + +int Pattern_split(Pattern *self, Input *string, int maxsplit, MatchIterCb cb, void *ctx) { + MatBuf mb; + if (!build_matbuf(self, string, &mb, NULL)) return -1; + int64_t ep = mb.len; + int64_t start = 0, seg_start = 0; + PatternImpl *impl = (PatternImpl *)self->program; + int ng = impl->prog.ngroups; + int n = 0; + while (start <= ep) { + if (maxsplit > 0 && n >= maxsplit) break; + int64_t *caps; alloc_caps(&caps, ng); + int found = 0; int64_t s; + for (s = start; s <= ep && !found; s++) { + reset_caps(caps, ng); + caps[0] = s; + MCtx c; c.pr = &impl->prog; c.text = mb.text; c.len = ep; c.caps = caps; + c.depth_exceeded = 0; c.require_end = -1; c.sub_end = -1; + found = run(&c, 0, s, -1, -1, 0); + if (!found && c.depth_exceeded) { free(caps); free_matbuf(&mb); return -1; } + } + if (!found) { free(caps); break; } + emit_split_item(cb, ctx, self, string, &mb, seg_start, caps[0]); + for (int g = 1; g <= ng; g++) emit_split_item(cb, ctx, self, string, &mb, caps[2 * g], caps[2 * g + 1]); + n++; + seg_start = caps[1]; + start = (caps[1] > caps[0]) ? caps[1] : caps[0] + 1; + free(caps); + } + emit_split_item(cb, ctx, self, string, &mb, seg_start, ep); + free_matbuf(&mb); + return n; +} + +static char *expand_template(const char *repl, Match *m, size_t *outlen) { + size_t cap = 64, len = 0; + char *out = malloc(cap); + size_t rl = strlen(repl); + for (size_t i = 0; i < rl; i++) { + char c = repl[i]; + const char *piece = NULL; size_t piecelen = 0; + int consumed_one_char = 1; + if (c == '\\' && i + 1 < rl) { + char nc = repl[i + 1]; + if (nc == 'g' && i + 2 < rl && repl[i + 2] == '<') { + size_t j = i + 3, start = j; + while (j < rl && repl[j] != '>') j++; + char *name = xstrndup(repl + start, j - start); + int idx = -1; + if (strspn(name, "0123456789") == strlen(name) && name[0]) idx = atoi(name); + else idx = groupindex_lookup((GroupIndex *)m->re->groupindex, name); + free(name); + Match_group(m, NULL, idx, &piece, &piecelen); + i = j; + } else if (isdigit((unsigned char)nc)) { + size_t j = i + 1, start = j; + while (j < rl && isdigit((unsigned char)repl[j]) && j - start < 2) j++; + char *ns = xstrndup(repl + start, j - start); + int idx = atoi(ns); free(ns); + Match_group(m, NULL, idx, &piece, &piecelen); + i = j - 1; + } else if (nc == 'n') { char cc = '\n'; if (len+1>=cap){cap*=2;out=realloc(out,cap);} out[len++]=cc; i++; continue; } + else if (nc == 't') { char cc = '\t'; if (len+1>=cap){cap*=2;out=realloc(out,cap);} out[len++]=cc; i++; continue; } + else if (nc == '\\') { char cc = '\\'; if (len+1>=cap){cap*=2;out=realloc(out,cap);} out[len++]=cc; i++; continue; } + else { piece = repl + i + 1; piecelen = 1; i++; } + } else { + if (len + 1 >= cap) { cap *= 2; out = realloc(out, cap); } + out[len++] = c; + consumed_one_char = 1; (void)consumed_one_char; + continue; + } + if (piece && piecelen) { + while (len + piecelen >= cap) { cap *= 2; out = realloc(out, cap); } + memcpy(out + len, piece, piecelen); + len += piecelen; + } + } + out[len] = 0; + *outlen = len; + return out; +} + +typedef struct { char *buf; size_t len, cap; int64_t last_end; Input *string; const char *repl; MatchSubCb cb; void *cbctx; int n; int count_limit; } SubCtx; + +static void subctx_append(SubCtx *sc, const void *data, size_t len) { + while (sc->len + len + 1 > sc->cap) { sc->cap = sc->cap ? sc->cap * 2 : 256; sc->buf = realloc(sc->buf, sc->cap); } + memcpy(sc->buf + sc->len, data, len); + sc->len += len; +} +static void sub_cb(void *ctx, const Match *m_const) { + SubCtx *sc = ctx; + Match *m = (Match *)m_const; + if (sc->count_limit > 0 && sc->n >= sc->count_limit) return; + int64_t mstart_byte = Match_start_byte(m, 0); + subctx_append(sc, sc->string->buf + sc->last_end, (size_t)(mstart_byte - sc->last_end)); + if (sc->cb) { + char *piece = NULL; size_t plen = 0; + sc->cb(sc->cbctx, m, &piece, &plen); + if (piece) { subctx_append(sc, piece, plen); free(piece); } + } else { + size_t plen; char *piece = expand_template(sc->repl, m, &plen); + subctx_append(sc, piece, plen); + free(piece); + } + sc->last_end = Match_end_byte(m, 0); + sc->n++; +} + +int Pattern_subn(Pattern *self, Input *string, const char *repl, MatchSubCb cb, void *ctx, int count, char **out, size_t *outlen, int *n) { + SubCtx sc; memset(&sc, 0, sizeof sc); + sc.string = string; sc.repl = repl; sc.cb = cb; sc.cbctx = ctx; sc.count_limit = count; sc.last_end = 0; + int total = Pattern_finditer(self, string, 0, -1, sub_cb, &sc); + if (total < 0) { free(sc.buf); return -1; } + subctx_append(&sc, string->buf + sc.last_end, (size_t)((int64_t)string->len - sc.last_end)); + if (!sc.buf) sc.buf = malloc(1); + sc.buf[sc.len] = 0; + *out = sc.buf; *outlen = sc.len; + if (n) *n = sc.n; + return 1; +} + +int Pattern_sub(Pattern *self, Input *string, const char *repl, MatchSubCb cb, void *ctx, int count, char **out, size_t *outlen) { + int n; + return Pattern_subn(self, string, repl, cb, ctx, count, out, outlen, &n); +} + +/* ================================================================ + * 12. Match accessors + * ================================================================ */ + +int Pattern_groupindex_lookup(Pattern *self, const char *name) { + PatternImpl *impl = (PatternImpl *)self->program; + return groupindex_lookup(&impl->gidx, name); +} + +int Match_group(Match *self, const char *name_or_null, int index, const char **out, size_t *outlen) { + MatchSlots *ms = self->slots; + int g = index; + if (name_or_null) { + PatternImpl *impl = (PatternImpl *)self->re->program; + g = groupindex_lookup(&impl->gidx, name_or_null); + if (g < 0) return -1; + } + if (g < 0 || g > ms->ngroups) return -1; + int64_t s = ms->pos[2 * g], e = ms->pos[2 * g + 1]; + if (s < 0 || e < 0) { *out = NULL; *outlen = 0; return 0; } + int64_t bs = ms->cp_to_byte ? ms->cp_to_byte[s] : s; + int64_t be = ms->cp_to_byte ? ms->cp_to_byte[e] : e; + *out = (const char *)(self->string->buf + bs); + *outlen = (size_t)(be - bs); + return 1; +} +int64_t Match_start(Match *self, int group) { + MatchSlots *ms = self->slots; + if (group < 0 || group > ms->ngroups) return -1; + return ms->pos[2 * group]; +} +int64_t Match_end(Match *self, int group) { + MatchSlots *ms = self->slots; + if (group < 0 || group > ms->ngroups) return -1; + return ms->pos[2 * group + 1]; +} +void Match_span(Match *self, int group, int64_t *start, int64_t *end) { + *start = Match_start(self, group); *end = Match_end(self, group); +} +int64_t Match_start_byte(Match *self, int group) { + MatchSlots *ms = self->slots; + if (group < 0 || group > ms->ngroups) return -1; + int64_t v = ms->pos[2 * group]; + if (v < 0) return -1; + return ms->cp_to_byte ? ms->cp_to_byte[v] : v; +} +int64_t Match_end_byte(Match *self, int group) { + MatchSlots *ms = self->slots; + if (group < 0 || group > ms->ngroups) return -1; + int64_t v = ms->pos[2 * group + 1]; + if (v < 0) return -1; + return ms->cp_to_byte ? ms->cp_to_byte[v] : v; +} +void Match_span_byte(Match *self, int group, int64_t *start, int64_t *end) { + *start = Match_start_byte(self, group); *end = Match_end_byte(self, group); +} +void Match_free(Match *self) { + if (!self || !self->slots) return; + MatchSlots *ms = self->slots; + free(ms->pos); free(ms->cp_to_byte); free(ms); + self->slots = NULL; +} + +/* ================================================================ + * 13. Module level functions, cache, escape (Section 9.1) + * ================================================================ */ + +#define CACHE_CAP 512 +typedef struct { char *key; int keylen; int flags; Pattern *pat; } CacheEnt; +static CacheEnt g_cache[CACHE_CAP]; +static int g_cache_n = 0; + +void re_purge(void) { + for (int i = 0; i < g_cache_n; i++) { free(g_cache[i].key); Pattern_free(g_cache[i].pat); } + g_cache_n = 0; +} +static Pattern *cache_compile(const char *pattern, size_t len, int flags, PatternError *err) { + for (int i = 0; i < g_cache_n; i++) + if (g_cache[i].flags == flags && (size_t)g_cache[i].keylen == len && memcmp(g_cache[i].key, pattern, len) == 0) + return g_cache[i].pat; + Pattern *pat = re_compile(pattern, len, flags, err); + if (!pat) return NULL; + if (g_cache_n >= CACHE_CAP) re_purge(); + g_cache[g_cache_n].key = xstrndup(pattern, len); + g_cache[g_cache_n].keylen = (int)len; + g_cache[g_cache_n].flags = flags; + g_cache[g_cache_n].pat = pat; + g_cache_n++; + return pat; +} + +int re_match(const char *pattern, size_t len, int flags, Input *string, Match *out) { + Pattern *pat = cache_compile(pattern, len, flags, NULL); if (!pat) return -1; + return Pattern_match(pat, string, 0, -1, out); +} +int re_fullmatch(const char *pattern, size_t len, int flags, Input *string, Match *out) { + Pattern *pat = cache_compile(pattern, len, flags, NULL); if (!pat) return -1; + return Pattern_fullmatch(pat, string, 0, -1, out); +} +int re_search(const char *pattern, size_t len, int flags, Input *string, Match *out) { + Pattern *pat = cache_compile(pattern, len, flags, NULL); if (!pat) return -1; + return Pattern_search(pat, string, 0, -1, out); +} +int re_finditer(const char *pattern, size_t len, int flags, Input *string, MatchIterCb cb, void *ctx) { + Pattern *pat = cache_compile(pattern, len, flags, NULL); if (!pat) return -1; + return Pattern_finditer(pat, string, 0, -1, cb, ctx); +} +int re_findall(const char *pattern, size_t len, int flags, Input *string, MatchIterCb cb, void *ctx) { + return re_finditer(pattern, len, flags, string, cb, ctx); +} +int re_split(const char *pattern, size_t len, int flags, Input *string, int maxsplit, MatchIterCb cb, void *ctx) { + Pattern *pat = cache_compile(pattern, len, flags, NULL); if (!pat) return -1; + return Pattern_split(pat, string, maxsplit, cb, ctx); +} +int re_sub(const char *pattern, size_t len, int flags, Input *string, const char *repl, MatchSubCb cb, void *ctx, int count, char **out, size_t *outlen) { + Pattern *pat = cache_compile(pattern, len, flags, NULL); if (!pat) return -1; + return Pattern_sub(pat, string, repl, cb, ctx, count, out, outlen); +} +int re_subn(const char *pattern, size_t len, int flags, Input *string, const char *repl, MatchSubCb cb, void *ctx, int count, char **out, size_t *outlen, int *n) { + Pattern *pat = cache_compile(pattern, len, flags, NULL); if (!pat) return -1; + return Pattern_subn(pat, string, repl, cb, ctx, count, out, outlen, n); +} +void re_escape(const char *in, size_t len, char **out, size_t *outlen) { + char *buf = malloc(len * 2 + 1); + size_t o = 0; + for (size_t i = 0; i < len; i++) { + unsigned char ch = (unsigned char)in[i]; + int is_word = (ch >= 'a' && ch <= 'z') || (ch >= 'A' && ch <= 'Z') || (ch >= '0' && ch <= '9') || ch == '_'; + if (!is_word && ch < 0x80) buf[o++] = '\\'; + buf[o++] = (char)ch; + } + buf[o] = 0; + *out = buf; *outlen = o; +} diff --git a/regexx.h b/regexx.h new file mode 100644 index 0000000..820f2e5 --- /dev/null +++ b/regexx.h @@ -0,0 +1,132 @@ +/* regexx.h - public API for regexx, a single-file C regex interpreter + * with Python `re` semantics. See concept.md for the design rationale + * and Section 9 in particular for the naming convention this header + * follows: every identifier with a direct Python `re` counterpart uses + * that counterpart's exact spelling. + * + * Implementation status relative to concept.md: this is the v1 + * implementation. It provides the full public API and pattern syntax + * described below, executed by a single recursive backtracking engine + * (concept.md Section 7.3) operating over a fully materialized copy of + * the input. The streaming, bounded-memory regular engine of Section + * 7.2 is not implemented yet; see README.md "Implementation status" + * for the complete, itemized list of what v1 does and does not cover. + */ +#ifndef REGEXX_H +#define REGEXX_H + +#include +#include + +#ifdef __cplusplus +extern "C" { +#endif + +/* ---- Flags (re.compile flags; bitwise OR, exact Python names) ---- */ +#define IGNORECASE 0x0001 +#define MULTILINE 0x0002 +#define DOTALL 0x0004 +#define VERBOSE 0x0008 +#define ASCII 0x0010 +#define UNICODE 0x0020 /* default for UTF8 mode; explicit for clarity */ +#define LOCALE 0x0040 +#define DEBUG 0x0080 + +/* Encoding mode, not a Python flag (Section 9.3): exactly one required. */ +#define BINARY 0x0100 +#define UTF8 0x0200 +/* ASCII (0x0010) doubles as the ASCII *encoding* mode when neither + * BINARY nor UTF8 is given; see regexx.c encoding-mode resolution. */ + +/* ---- Types ---- */ + +typedef struct Pattern Pattern; /* re.Pattern */ +typedef struct Match Match; /* re.Match */ +typedef struct Input Input; /* no Python counterpart, see concept.md 9.0/9.2 */ + +typedef void (*MatchIterCb)(void *ctx, const Match *m); +typedef void (*MatchSubCb)(void *ctx, const Match *m, char **out, size_t *outlen); + +typedef struct PatternError PatternError; /* re.error / re.PatternError */ +struct PatternError { + const char *msg; /* re.error.msg */ + const char *pattern; /* re.error.pattern */ + int64_t pos; /* re.error.pos */ + int64_t lineno; /* re.error.lineno */ + int64_t colno; /* re.error.colno */ +}; +/* msg and pattern are heap allocated by whichever call filled this + * struct in; call PatternError_free once you are done reading it + * (safe to call on a zero-initialized or already-freed PatternError). */ +void PatternError_free(PatternError *err); + +struct Pattern { + const char *pattern; /* re.Pattern.pattern */ + int flags; /* re.Pattern.flags */ + int groups; /* re.Pattern.groups */ + void *groupindex; /* re.Pattern.groupindex, name -> group number */ + void *program; /* compiled bytecode, private */ +}; + +struct Match { + Pattern *re; /* re.Match.re */ + Input *string; /* re.Match.string */ + int64_t pos, endpos; /* re.Match.pos, re.Match.endpos */ + int lastindex; /* re.Match.lastindex */ + const char *lastgroup; /* re.Match.lastgroup */ + void *slots; /* private */ +}; + +/* ---- Input construction (no Python counterpart) ---- */ +Input *Input_from_buffer(const uint8_t *buf, size_t len); /* does not copy or take ownership */ +Input *Input_from_file(const char *path, PatternError *err); /* seekable or not (pipes, "-" via /dev/stdin, etc. all work) */ +void Input_free(Input *in); + +/* ---- Pattern methods (bound to an already compiled Pattern) ---- */ +int Pattern_match(Pattern *self, Input *string, int64_t pos, int64_t endpos, Match *out); +int Pattern_fullmatch(Pattern *self, Input *string, int64_t pos, int64_t endpos, Match *out); +int Pattern_search(Pattern *self, Input *string, int64_t pos, int64_t endpos, Match *out); +int Pattern_finditer(Pattern *self, Input *string, int64_t pos, int64_t endpos, MatchIterCb cb, void *ctx); +int Pattern_findall(Pattern *self, Input *string, int64_t pos, int64_t endpos, MatchIterCb cb, void *ctx); +/* cb fires once per element of the list re.split() would return, in + * order; read each element's text via Match_group(m, NULL, 0, ...). */ +int Pattern_split(Pattern *self, Input *string, int maxsplit, MatchIterCb cb, void *ctx); +int Pattern_sub(Pattern *self, Input *string, const char *repl, MatchSubCb cb, void *ctx, int count, char **out, size_t *outlen); +int Pattern_subn(Pattern *self, Input *string, const char *repl, MatchSubCb cb, void *ctx, int count, char **out, size_t *outlen, int *n); +void Pattern_free(Pattern *self); + +/* Looks up a named group in re.Pattern.groupindex; returns its 1-based + * group number, or -1 if no group by that name exists. There is no + * enumeration function for groupindex as a whole in this build: a + * caller that needs every name must track the names it used to build + * the pattern itself, since Pattern->groupindex is otherwise opaque. */ +int Pattern_groupindex_lookup(Pattern *self, const char *name); + +/* ---- Match accessors ---- */ +int Match_group(Match *self, const char *name_or_null, int index, const char **out, size_t *outlen); +int64_t Match_start(Match *self, int group); +int64_t Match_end(Match *self, int group); +void Match_span(Match *self, int group, int64_t *start, int64_t *end); +int64_t Match_start_byte(Match *self, int group); +int64_t Match_end_byte(Match *self, int group); +void Match_span_byte(Match *self, int group, int64_t *start, int64_t *end); +void Match_free(Match *self); + +/* ---- Module level functions ---- */ +Pattern *re_compile(const char *pattern, size_t len, int flags, PatternError *err); +int re_match(const char *pattern, size_t len, int flags, Input *string, Match *out); +int re_fullmatch(const char *pattern, size_t len, int flags, Input *string, Match *out); +int re_search(const char *pattern, size_t len, int flags, Input *string, Match *out); +int re_finditer(const char *pattern, size_t len, int flags, Input *string, MatchIterCb cb, void *ctx); +int re_findall(const char *pattern, size_t len, int flags, Input *string, MatchIterCb cb, void *ctx); +int re_split(const char *pattern, size_t len, int flags, Input *string, int maxsplit, MatchIterCb cb, void *ctx); +int re_sub(const char *pattern, size_t len, int flags, Input *string, const char *repl, MatchSubCb cb, void *ctx, int count, char **out, size_t *outlen); +int re_subn(const char *pattern, size_t len, int flags, Input *string, const char *repl, MatchSubCb cb, void *ctx, int count, char **out, size_t *outlen, int *n); +void re_escape(const char *in, size_t len, char **out, size_t *outlen); +void re_purge(void); + +#ifdef __cplusplus +} +#endif + +#endif /* REGEXX_H */ diff --git a/tests/cases.py b/tests/cases.py new file mode 100644 index 0000000..a68d39e --- /dev/null +++ b/tests/cases.py @@ -0,0 +1,129 @@ +"""Test cases for regexx, checked against CPython's own `re` module +(concept.md Section 11). Each case records a pattern/subject pair and +an operation; gen.py computes the expected result with Python's `re` +and emits a C assertion that checks regexx against it. +""" + +CASES = [] + + +def add(op, pattern, subject, mode="utf8", flags=(), **kw): + CASES.append(dict(op=op, pattern=pattern, subject=subject, mode=mode, flags=tuple(flags), **kw)) + + +# ---- literals, concatenation, anchors -------------------------------- +add("search", r"abc", "xxabcxx") +add("fullmatch", r"abc", "abc") +add("fullmatch", r"abc", "xabc") +add("match", r"abc", "abcdef") +add("match", r"abc", "xabcdef") +add("search", r"^abc$", "abc") +add("search", r"^abc$", "xabc") +add("search", r"^abc", "abc\nabc", flags=["MULTILINE"]) +add("finditer", r"^abc", "abc\nabc", flags=["MULTILINE"]) +add("search", r"abc$", "xxabc\nabc", flags=["MULTILINE"]) +add("search", r"\Aabc", "abc") +add("search", r"abc\Z", "xxabc") +add("search", r"abc\Z", "xxabc\n") + +# ---- character classes ------------------------------------------------- +add("search", r"[a-z]+", "ABCdefGHI") +add("search", r"[^a-z]+", "abcDEFghi") +add("search", r"[a-z]+", "ABC", flags=["IGNORECASE"]) +add("finditer", r"[abc]", "xaybzc") +add("search", r"[]a]", "]a") # ']' literal as first class member +add("search", r"[a\-z]", "-") + +# ---- shorthand classes --------------------------------------------------- +add("finditer", r"\d+", "ab123cd45") +add("finditer", r"\D+", "ab123cd45") +add("search", r"\s+", "a b") +add("search", r"\S+", " ab ") +add("finditer", r"\w+", "hi there, friend!") +add("finditer", r"\W+", "hi there, friend!") +add("search", r"[\d\S]", " ") +add("search", r"[\d\S]", "5") + +# ---- word boundaries ------------------------------------------------------ +add("finditer", r"\bcat\b", "cat catalog cat") +add("finditer", r"\Bcat\B", "concatenate") + +# ---- quantifiers ------------------------------------------------------------ +add("search", r"a*", "aaab") +add("search", r"a+", "baaab") +add("search", r"a?b", "b") +add("search", r"a{2,4}", "aaaaa") +add("search", r"a{2,4}?", "aaaaa") +add("search", r"a{3}", "aa") +add("search", r"a.*b", "axxxbxxxb") +add("search", r"a.*?b", "axxxbxxxb") +add("finditer", r"a*", "baaab") # exercises empty-match advancement + +# ---- alternation and groups ------------------------------------------------- +add("search", r"(foo|bar)baz", "barbaz") +add("search", r"(?:foo|bar)baz", "foobaz") +add("finditer", r"(a)(b)?", "ab a") +add("search", r"(?P\d{4})-(?P\d{2})", "2024-09") + +# ---- backreferences --------------------------------------------------------- +add("search", r"(\w+) \1", "hello hello") +add("search", r"(\w+) \1", "hello world") +add("search", r"(?P\w+) (?P=w)", "abc abc") +add("search", r"(\w)\1", "aa") +add("search", r"(\w)\1", "ab") + +# ---- lookaround -------------------------------------------------------------- +add("search", r"foo(?=bar)", "foobar") +add("search", r"foo(?=bar)", "foobaz") +add("search", r"foo(?!bar)", "foobaz") +add("search", r"foo(?!bar)", "foobar") +add("search", r"(?<=foo)bar", "foobar") +add("search", r"(?<=foo)bar", "xxxbar") +add("search", r"(?bc|b)c", "abc") +add("search", r"a(bc|b)c", "abc") +add("search", r"a*+a", "aaaa") +add("search", r"a*a", "aaaa") + +# ---- DOTALL / VERBOSE ---------------------------------------------------------- +add("search", r"a.b", "a\nb") +add("search", r"a.b", "a\nb", flags=["DOTALL"]) +add("search", r""" + \d+ # the integer part + \. # the dot + \d+ # the fractional part +""", "pi is 3.14 roughly", flags=["VERBOSE"]) + +# ---- escapes --------------------------------------------------------------------- +add("search", r"a\tb", "a\tb") +add("search", r"\x41\x42", "AB") +add("search", r"é", "café") + +# ---- UTF-8 mode ------------------------------------------------------------------ +add("search", r"\w+", "café au lait") +add("finditer", r"\w+", "café 中文 word") +add("search", r"[à-ÿ]+", "éèê") + +# ---- sub / subn ------------------------------------------------------------------- +add("sub", r"a", "banana", repl="o", count=0) +add("sub", r"a", "banana", repl="o", count=2) +add("sub", r"(\w+)@(\w+)", "user@host", repl=r"\2@\1") +add("sub", r"(?P\w+)@(?P\w+)", "user@host", repl=r"\g@\g") +add("sub", r"\s+", "a b c", repl=" ") + +# ---- split ------------------------------------------------------------------------ +add("split", r"[,;]\s*", "a, b;c , d") +add("split", r"(,)", "a,b,c") +add("split", r"\s*", "abc") +add("split", r",", "a,b,c", maxsplit=1) + +# ---- ASCII vs UTF8 mode differences for \w ---------------------------------------- +add("search", r"\w+", "café", mode="ascii") + +# ---- BINARY mode -------------------------------------------------------------------- +add("search", r"a.c", "a\x00c", mode="binary") +add("finditer", r"\d+", "12ab34", mode="binary") diff --git a/tests/gen.py b/tests/gen.py new file mode 100644 index 0000000..09ccab0 --- /dev/null +++ b/tests/gen.py @@ -0,0 +1,181 @@ +#!/usr/bin/env python3 +"""Generate tests/generated_tests.c from tests/cases.py, using CPython's +own `re` module as ground truth (concept.md Section 11).""" +import re +import sys +import os + +sys.path.insert(0, os.path.dirname(__file__)) +from cases import CASES # noqa: E402 + +FLAGMAP = { + "IGNORECASE": re.IGNORECASE, + "MULTILINE": re.MULTILINE, + "DOTALL": re.DOTALL, + "VERBOSE": re.VERBOSE, + "LOCALE": re.LOCALE, + "ASCII": re.ASCII, +} + + +def c_str(b): + if isinstance(b, str): + b = b.encode("utf-8") + out = ['"'] + for byte in b: + c = chr(byte) + if c == '"': + out.append('\\"') + elif c == "\\": + out.append("\\\\") + elif 32 <= byte < 127: + out.append(c) + else: + out.append("\\%03o" % byte) + out.append('"') + return "".join(out) + + +def py_flags(case): + f = 0 + if case["mode"] == "ascii": + f |= re.ASCII + for name in case["flags"]: + f |= FLAGMAP[name] + return f + + +def c_flags(case): + names = [] + if case["mode"] == "ascii": + names.append("ASCII") + elif case["mode"] == "utf8": + names.append("UTF8") + elif case["mode"] == "binary": + names.append("BINARY") + for name in case["flags"]: + names.append(name) + return "|".join(names) if names else "0" + + +def encode(case, s): + if case["mode"] == "binary": + return s.encode("utf-8") + return s + + +def blen(case, s): + """Byte length of the encoded subject, as the C API always wants a + byte count (UTF8 mode still stores a byte buffer; only Match_start + reports code points), never a Python character count.""" + e = encode(case, s) + return len(e) if isinstance(e, bytes) else len(e.encode("utf-8")) + + +def desc(case, i): + return "#%d %s /%s/ on %r" % (i, case["op"], case["pattern"], case["subject"]) + + +def gen_search_family(case, i, out): + kind = {"search": 0, "match": 1, "fullmatch": 2}[case["op"]] + flags = py_flags(case) + subj = encode(case, case["subject"]) + pat = encode(case, case["pattern"]) + compiled = re.compile(pat, flags) + fn = {"search": compiled.search, "match": compiled.match, "fullmatch": compiled.fullmatch}[case["op"]] + m = fn(subj) + d = desc(case, i) + if m is None: + out.append(' tc_search_family(%s, %s, %s, %s, %d, %d, 0, 0, NULL);' % ( + c_str(d), c_str(case["pattern"]), c_flags(case), c_str(case["subject"]), blen(case, case["subject"]), kind)) + return + spans = [m.start(0), m.end(0)] + for g in range(1, compiled.groups + 1): + try: + spans += [m.start(g), m.end(g)] + except IndexError: + spans += [-1, -1] + n_spans = compiled.groups + 1 + arr = ",".join(str(v) for v in spans) + out.append(' { static const long long sp[] = {%s}; tc_search_family(%s, %s, %s, %s, %d, %d, 1, %d, sp); }' % ( + arr, c_str(d), c_str(case["pattern"]), c_flags(case), c_str(case["subject"]), blen(case, case["subject"]), kind, n_spans)) + + +def gen_finditer(case, i, out): + flags = py_flags(case) + subj = encode(case, case["subject"]) + pat = encode(case, case["pattern"]) + ms = list(re.finditer(pat, subj, flags)) + spans = [] + for m in ms: + spans += [m.start(0), m.end(0)] + d = desc(case, i) + if not spans: + out.append(' tc_finditer(%s, %s, %s, %s, %d, 0, NULL);' % ( + c_str(d), c_str(case["pattern"]), c_flags(case), c_str(case["subject"]), blen(case, case["subject"]))) + return + arr = ",".join(str(v) for v in spans) + out.append(' { static const long long sp[] = {%s}; tc_finditer(%s, %s, %s, %s, %d, %d, sp); }' % ( + arr, c_str(d), c_str(case["pattern"]), c_flags(case), c_str(case["subject"]), blen(case, case["subject"]), len(ms))) + + +def gen_sub(case, i, out): + flags = py_flags(case) + subj = encode(case, case["subject"]) + pat = encode(case, case["pattern"]) + repl = encode(case, case["repl"]) + count = case.get("count", 0) + result, n = re.subn(pat, repl, subj, count=count, flags=flags) + d = desc(case, i) + result_str = result.decode("utf-8") if isinstance(result, bytes) else result + out.append(' tc_sub(%s, %s, %s, %s, %d, %s, %d, %s, %d);' % ( + c_str(d), c_str(case["pattern"]), c_flags(case), c_str(case["subject"]), blen(case, case["subject"]), + c_str(case["repl"]), count, c_str(result_str), n)) + + +def gen_split(case, i, out): + flags = py_flags(case) + subj = encode(case, case["subject"]) + pat = encode(case, case["pattern"]) + maxsplit = case.get("maxsplit", 0) + items = re.split(pat, subj, maxsplit=maxsplit, flags=flags) + d = desc(case, i) + parts = [] + for it in items: + if it is None: + parts.append("NULL") + else: + s = it.decode("utf-8") if isinstance(it, bytes) else it + parts.append(c_str(s)) + arr = ",".join(parts) if parts else "0" + out.append(' { static const char *const items[] = {%s}; tc_split(%s, %s, %s, %s, %d, %d, %d, items); }' % ( + arr, c_str(d), c_str(case["pattern"]), c_flags(case), c_str(case["subject"]), blen(case, case["subject"]), + maxsplit, len(items))) + + +def main(): + out = [] + out.append('/* GENERATED by tests/gen.py from tests/cases.py. Do not edit by hand. */') + out.append('#include "harness.h"') + out.append('void run_generated_tests(void) {') + for i, case in enumerate(CASES): + op = case["op"] + if op in ("search", "match", "fullmatch"): + gen_search_family(case, i, out) + elif op == "finditer": + gen_finditer(case, i, out) + elif op == "sub": + gen_sub(case, i, out) + elif op == "split": + gen_split(case, i, out) + else: + raise ValueError("unknown op %r" % op) + out.append('}') + dest = os.path.join(os.path.dirname(__file__), "generated_tests.c") + with open(dest, "w") as f: + f.write("\n".join(out) + "\n") + print("wrote %s (%d cases)" % (dest, len(CASES))) + + +if __name__ == "__main__": + main() diff --git a/tests/harness.c b/tests/harness.c new file mode 100644 index 0000000..0237587 --- /dev/null +++ b/tests/harness.c @@ -0,0 +1,156 @@ +#include "harness.h" +#include +#include +#include + +int tc_pass = 0, tc_fail = 0; + +void tc_search_family(const char *desc, const char *pat, int flags, + const char *subj, size_t subjlen, int kind, + int exp_matched, int n_spans, const long long *exp_spans) { + PatternError err; memset(&err, 0, sizeof err); + Pattern *p = re_compile(pat, strlen(pat), flags, &err); + if (!p) { + printf("FAIL [%s]: compile error: %s\n", desc, err.msg ? err.msg : "?"); + tc_fail++; + return; + } + Input *in = Input_from_buffer((const uint8_t *)subj, subjlen); + Match m; memset(&m, 0, sizeof m); + int r; + if (kind == 0) r = Pattern_search(p, in, 0, -1, &m); + else if (kind == 1) r = Pattern_match(p, in, 0, -1, &m); + else r = Pattern_fullmatch(p, in, 0, -1, &m); + + if (r < 0) { + printf("FAIL [%s]: internal error (depth exceeded or bad UTF-8)\n", desc); + tc_fail++; + } else if ((r == 1) != (exp_matched != 0)) { + printf("FAIL [%s]: matched=%d expected=%d\n", desc, r, exp_matched); + tc_fail++; + } else if (r == 1) { + int ok = 1; + for (int g = 0; g < n_spans; g++) { + long long es = exp_spans[2 * g], ee = exp_spans[2 * g + 1]; + long long as = Match_start(&m, g), ae = Match_end(&m, g); + if (as != es || ae != ee) { + printf("FAIL [%s]: group %d span (%lld,%lld) expected (%lld,%lld)\n", desc, g, as, ae, es, ee); + ok = 0; + } + } + if (ok) tc_pass++; else tc_fail++; + Match_free(&m); + } else { + tc_pass++; + } + Input_free(in); + Pattern_free(p); +} + +typedef struct { long long *spans; int n, cap; } SpanList; +static void span_cb(void *ctx, const Match *m_const) { + SpanList *sl = ctx; + Match *m = (Match *)m_const; + if (sl->n == sl->cap) { sl->cap = sl->cap ? sl->cap * 2 : 8; sl->spans = realloc(sl->spans, (size_t)sl->cap * 2 * sizeof(long long)); } + sl->spans[2 * sl->n] = Match_start(m, 0); + sl->spans[2 * sl->n + 1] = Match_end(m, 0); + sl->n++; +} + +void tc_finditer(const char *desc, const char *pat, int flags, + const char *subj, size_t subjlen, + int exp_n_matches, const long long *exp_spans) { + PatternError err; memset(&err, 0, sizeof err); + Pattern *p = re_compile(pat, strlen(pat), flags, &err); + if (!p) { printf("FAIL [%s]: compile error: %s\n", desc, err.msg ? err.msg : "?"); tc_fail++; return; } + Input *in = Input_from_buffer((const uint8_t *)subj, subjlen); + SpanList sl; memset(&sl, 0, sizeof sl); + int total = Pattern_finditer(p, in, 0, -1, span_cb, &sl); + if (total < 0) { printf("FAIL [%s]: internal error\n", desc); tc_fail++; } + else if (sl.n != exp_n_matches) { printf("FAIL [%s]: %d matches, expected %d\n", desc, sl.n, exp_n_matches); tc_fail++; } + else { + int ok = 1; + for (int i = 0; i < sl.n; i++) + if (sl.spans[2 * i] != exp_spans[2 * i] || sl.spans[2 * i + 1] != exp_spans[2 * i + 1]) { + printf("FAIL [%s]: match %d span (%lld,%lld) expected (%lld,%lld)\n", desc, i, + sl.spans[2 * i], sl.spans[2 * i + 1], exp_spans[2 * i], exp_spans[2 * i + 1]); + ok = 0; + } + if (ok) tc_pass++; else tc_fail++; + } + free(sl.spans); + Input_free(in); + Pattern_free(p); +} + +void tc_sub(const char *desc, const char *pat, int flags, + const char *subj, size_t subjlen, const char *repl, int count, + const char *exp_result, int exp_n) { + PatternError err; memset(&err, 0, sizeof err); + Pattern *p = re_compile(pat, strlen(pat), flags, &err); + if (!p) { printf("FAIL [%s]: compile error: %s\n", desc, err.msg ? err.msg : "?"); tc_fail++; return; } + Input *in = Input_from_buffer((const uint8_t *)subj, subjlen); + char *out = NULL; size_t outlen = 0; int n = 0; + int r = Pattern_subn(p, in, repl, NULL, NULL, count, &out, &outlen, &n); + if (r < 0) { printf("FAIL [%s]: internal error\n", desc); tc_fail++; } + else { + size_t explen = strlen(exp_result); + if (outlen != explen || memcmp(out, exp_result, explen) != 0 || n != exp_n) { + printf("FAIL [%s]: sub result '%.*s' (n=%d) expected '%s' (n=%d)\n", desc, (int)outlen, out, n, exp_result, exp_n); + tc_fail++; + } else tc_pass++; + } + free(out); + Input_free(in); + Pattern_free(p); +} + +typedef struct { int present; const char *ptr; size_t len; } SplitItem; +typedef struct { SplitItem *items; int n, cap; } SplitList; +static void split_cb(void *ctx, const Match *m_const) { + SplitList *sl = ctx; + Match *m = (Match *)m_const; + if (sl->n == sl->cap) { sl->cap = sl->cap ? sl->cap * 2 : 8; sl->items = realloc(sl->items, (size_t)sl->cap * sizeof(SplitItem)); } + const char *out; size_t outlen; + int r = Match_group(m, NULL, 0, &out, &outlen); + sl->items[sl->n].present = (r == 1); + sl->items[sl->n].ptr = out; sl->items[sl->n].len = outlen; + sl->n++; +} + +void tc_split(const char *desc, const char *pat, int flags, + const char *subj, size_t subjlen, int maxsplit, + int n_items, const char *const *texts) { + PatternError err; memset(&err, 0, sizeof err); + Pattern *p = re_compile(pat, strlen(pat), flags, &err); + if (!p) { printf("FAIL [%s]: compile error: %s\n", desc, err.msg ? err.msg : "?"); tc_fail++; return; } + Input *in = Input_from_buffer((const uint8_t *)subj, subjlen); + SplitList sl; memset(&sl, 0, sizeof sl); + int r = Pattern_split(p, in, maxsplit, split_cb, &sl); + if (r < 0) { printf("FAIL [%s]: internal error\n", desc); tc_fail++; } + else if (sl.n != n_items) { printf("FAIL [%s]: %d items, expected %d\n", desc, sl.n, n_items); tc_fail++; } + else { + int ok = 1; + for (int i = 0; i < sl.n; i++) { + int exp_none = (texts[i] == NULL); + if (exp_none && sl.items[i].present) { printf("FAIL [%s]: item %d expected None\n", desc, i); ok = 0; } + else if (!exp_none) { + size_t explen = strlen(texts[i]); + if (!sl.items[i].present || sl.items[i].len != explen || memcmp(sl.items[i].ptr, texts[i], explen) != 0) { + printf("FAIL [%s]: item %d = '%.*s' expected '%s'\n", desc, i, + sl.items[i].present ? (int)sl.items[i].len : 4, + sl.items[i].present ? sl.items[i].ptr : "NONE", texts[i]); + ok = 0; + } + } + } + if (ok) tc_pass++; else tc_fail++; + } + free(sl.items); + Input_free(in); + Pattern_free(p); +} + +void harness_report(void) { + printf("\n%d passed, %d failed\n", tc_pass, tc_fail); +} diff --git a/tests/harness.h b/tests/harness.h new file mode 100644 index 0000000..709e40e --- /dev/null +++ b/tests/harness.h @@ -0,0 +1,36 @@ +/* tests/harness.h - shared by hand-written and generated test cases. + * Ground truth for generated cases comes from CPython's own `re` + * module (concept.md Section 11); see gen.py. */ +#ifndef REGEXX_TEST_HARNESS_H +#define REGEXX_TEST_HARNESS_H +#include "../regexx.h" +#include + +extern int tc_pass, tc_fail; + +/* search/match/fullmatch: exp_spans holds 2*n_spans int64_t values, + * (start,end) for group 0, then group 1, 2, ... in order; -1,-1 means + * the group did not participate (or, for group 0 alone, that + * exp_matched is 0 and no spans are checked). */ +void tc_search_family(const char *desc, const char *pat, int flags, + const char *subj, size_t subjlen, int kind, + int exp_matched, int n_spans, const long long *exp_spans); + +/* finditer: exp_spans holds 2*n_matches int64_t values, (start,end) of + * the whole match for each match found, in order. */ +void tc_finditer(const char *desc, const char *pat, int flags, + const char *subj, size_t subjlen, + int exp_n_matches, const long long *exp_spans); + +void tc_sub(const char *desc, const char *pat, int flags, + const char *subj, size_t subjlen, const char *repl, int count, + const char *exp_result, int exp_n); + +/* split: texts[i] == NULL means that list element is None in Python. */ +void tc_split(const char *desc, const char *pat, int flags, + const char *subj, size_t subjlen, int maxsplit, + int n_items, const char *const *texts); + +void harness_report(void); + +#endif diff --git a/tests/main.c b/tests/main.c new file mode 100644 index 0000000..09baba6 --- /dev/null +++ b/tests/main.c @@ -0,0 +1,10 @@ +#include "harness.h" +#include + +extern void run_generated_tests(void); + +int main(void) { + run_generated_tests(); + harness_report(); + return tc_fail ? 1 : 0; +}