Add regexx: a single-file C regex interpreter with Python re semantics
concept.md is the full design document: the objective (Python re parity plus binary/ASCII/UTF-8 modes, gigabyte-scale input, single C file), the automata-theory argument for why unrestricted backreferences/lookaround are incompatible with strict single-pass constant memory, the resulting two-engine architecture, the exact Python-mirroring naming convention, and a full comparison against POSIX regex.h for C-background readers. regexx.c/regexx.h are the v1 implementation: parser, compiler to a Pike/backtracking-style bytecode, and a single recursive backtracking engine covering the pattern syntax and operations listed in README.md, validated against CPython's own re module output (tests/), clean under AddressSanitizer/UBSan, and stress-tested (50MB simple-quantifier match, graceful failure rather than a crash on complex repeats over large input, clean rejection of every intentionally unsupported construct). Also included: examples/rxgrep.c (a small grep-like program exercising all three data modes and the substitution API), the Makefile, the MIT LICENSE, and docs/API.md, an exhaustive reference for every type, flag, and function's exact return-value and memory-ownership convention, checked against the current source and against a real CPython interpreter rather than against memory. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01EjuMk8kY9SDus1wWe2K9xY
This commit is contained in:
@@ -0,0 +1,6 @@
|
||||
*.o
|
||||
*.a
|
||||
rxgrep
|
||||
tests/generated_tests.c
|
||||
/.claude/
|
||||
*.dSYM/
|
||||
@@ -0,0 +1,3 @@
|
||||
Write always scientific when documenting.
|
||||
No claude promotion.
|
||||
No EM dashes.
|
||||
@@ -0,0 +1,21 @@
|
||||
MIT License
|
||||
|
||||
Copyright (c) 2026 retoor <retoor@molodetz.nl>
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in all
|
||||
copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||
SOFTWARE.
|
||||
@@ -0,0 +1,57 @@
|
||||
# Makefile for regexx: a single-file C regex interpreter with Python
|
||||
# `re` semantics. See concept.md for the design and README.md for the
|
||||
# implementation status.
|
||||
|
||||
CC ?= cc
|
||||
CSTD ?= -std=c11
|
||||
WARN ?= -Wall -Wextra
|
||||
OPT ?= -O2
|
||||
CFLAGS ?= $(CSTD) $(WARN) $(OPT) -g
|
||||
LDLIBS ?=
|
||||
|
||||
PREFIX ?= /usr/local
|
||||
|
||||
AR ?= ar
|
||||
|
||||
.PHONY: all lib example test check clean install fuzz-smoke
|
||||
|
||||
all: lib example
|
||||
|
||||
# ---- library -----------------------------------------------------------
|
||||
libregexx.a: regexx.o
|
||||
$(AR) rcs $@ $^
|
||||
|
||||
regexx.o: regexx.c regexx.h
|
||||
$(CC) $(CFLAGS) -c regexx.c -o $@
|
||||
|
||||
lib: libregexx.a
|
||||
|
||||
# ---- example -------------------------------------------------------------
|
||||
example: rxgrep
|
||||
|
||||
rxgrep: examples/rxgrep.c libregexx.a regexx.h
|
||||
$(CC) $(CFLAGS) -I. -o $@ examples/rxgrep.c libregexx.a $(LDLIBS)
|
||||
|
||||
# ---- tests -----------------------------------------------------------------
|
||||
# tests/generated_tests.c is generated from tests/cases.py using
|
||||
# CPython's own `re` module as ground truth (concept.md Section 11).
|
||||
tests/generated_tests.c: tests/cases.py tests/gen.py
|
||||
python3 tests/gen.py
|
||||
|
||||
test: tests/generated_tests.c regexx.c regexx.h tests/harness.c tests/harness.h tests/main.c
|
||||
$(CC) $(CFLAGS) -I. -o /tmp/regexx_test regexx.c tests/harness.c tests/generated_tests.c tests/main.c $(LDLIBS)
|
||||
/tmp/regexx_test
|
||||
|
||||
# check also runs the suite under AddressSanitizer + UBSan.
|
||||
check: tests/generated_tests.c
|
||||
$(CC) $(CSTD) $(WARN) -O0 -g -fsanitize=address,undefined -I. -o /tmp/regexx_test_san \
|
||||
regexx.c tests/harness.c tests/generated_tests.c tests/main.c $(LDLIBS)
|
||||
/tmp/regexx_test_san
|
||||
|
||||
install: libregexx.a regexx.h
|
||||
install -d $(DESTDIR)$(PREFIX)/lib $(DESTDIR)$(PREFIX)/include
|
||||
install -m644 libregexx.a $(DESTDIR)$(PREFIX)/lib/
|
||||
install -m644 regexx.h $(DESTDIR)$(PREFIX)/include/
|
||||
|
||||
clean:
|
||||
rm -f regexx.o libregexx.a rxgrep tests/generated_tests.c
|
||||
@@ -0,0 +1,189 @@
|
||||
# regexx
|
||||
|
||||
A single-file C regular expression interpreter that reproduces the observable
|
||||
behavior of Python's `re` module, including its exact identifier names
|
||||
(`Pattern`, `Match`, `re_compile`, `re_sub`, `IGNORECASE`, and so on; see
|
||||
Section 9 of `concept.md`), and that additionally supports binary data
|
||||
(arbitrary byte streams, including embedded NUL) and UTF-8 text alongside
|
||||
plain ASCII.
|
||||
|
||||
The design rationale, the algorithmic trade-offs, and a full accounting of
|
||||
what is and is not carried over from Python's `re` and from POSIX's native
|
||||
`regex.h` are recorded in [`concept.md`](concept.md). This file documents the
|
||||
implementation that exists today at a glance; [`docs/API.md`](docs/API.md) is
|
||||
the exhaustive reference (every type, every flag, every function's exact
|
||||
return-value and memory-ownership convention, checked against a real CPython
|
||||
interpreter, not against memory).
|
||||
|
||||
## Implementation status
|
||||
|
||||
This is a v1 implementation. It is a complete, tested engine for the pattern
|
||||
syntax and operations listed below, executed by a single recursive
|
||||
backtracking engine (`concept.md` Section 7.3) over a fully materialized copy
|
||||
of the input.
|
||||
|
||||
**It does not yet implement the streaming, bounded-memory regular engine of
|
||||
`concept.md` Section 7.2.** `Input` (the abstraction over "a source of
|
||||
chunks", `concept.md` 9.2) is implemented, and `Input_from_file` reads a
|
||||
whole file into memory before matching. Every public function signature is
|
||||
already exactly what the streaming design in `concept.md` specifies, so the
|
||||
non-streaming implementation underneath a given call can be replaced later
|
||||
without changing any caller. Concretely, today:
|
||||
|
||||
- A pattern with no backreference and no unbounded-width lookahead runs in
|
||||
linear time and, for the common case of a single character, class, or `.`
|
||||
repeated by a quantifier, in *O(1)* recursion depth regardless of input
|
||||
size (the `OP_REPEAT1` fast path). A repeated *compound* sub-pattern (for
|
||||
example `(ab)*`) still recurses once per repetition, bounded by a
|
||||
configurable depth limit (`MAX_DEPTH` in `regexx.c`, currently 60000); past
|
||||
that limit, matching fails with a reported error rather than a stack
|
||||
overflow or a wrong answer.
|
||||
- A pattern with a backreference or an unbounded-width lookahead can, like
|
||||
CPython's own `_sre`, take worst-case exponential time on an adversarial
|
||||
input (`concept.md` Section 5, 13.2); this is the same catastrophic
|
||||
backtracking (ReDoS) behavior CPython itself exhibits on such patterns, not
|
||||
a regression specific to this engine.
|
||||
|
||||
### Pattern syntax supported
|
||||
|
||||
Literals; `.` (with `DOTALL`); character classes with ranges, negation, and
|
||||
`\d \D \w \W \s \S`; `\b \B`; anchors `^ $ \A \Z` (with `MULTILINE`);
|
||||
quantifiers `* + ? {m,n} {m,} {,n} {m}`, greedy and lazy; possessive
|
||||
quantifiers `*+ ++ ?+ {m,n}+`; groups `(...) (?:...) (?P<name>...)`;
|
||||
alternation `|`; backreferences `\1`-`\99`, `(?P=name)`, `\g<name>`,
|
||||
`\g<N>`; lookahead `(?=...) (?!...)`; fixed-width lookbehind
|
||||
`(?<=...) (?<!...)`; atomic groups `(?>...)`; comments `(?#...)`; global
|
||||
inline flags `(?aiLmsux)` at the start of a pattern; escapes
|
||||
`\n \r \t \f \v \a`, octal `\0`-prefixed escapes, `\xhh`, `\uxxxx`,
|
||||
`\Uxxxxxxxx`; flags `IGNORECASE`, `MULTILINE`, `DOTALL`, `VERBOSE`, `ASCII`,
|
||||
`LOCALE` (all with observable effect; see `docs/API.md` Section 2 for exactly
|
||||
what each one does), plus `UNICODE` and `DEBUG` (accepted for source
|
||||
compatibility with Python, currently no-ops in this build).
|
||||
|
||||
Rejected at compile time with a clear `PatternError`, rather than
|
||||
mis-parsed: conditional groups `(?(id)yes|no)`, scoped inline flags
|
||||
`(?flags:...)`, `\N{NAME}` named code points, and POSIX bracket classes
|
||||
`[:alpha:]` (which are not part of Python `re` at all, `concept.md` 14.3).
|
||||
Variable-width lookbehind is also rejected at compile time, matching
|
||||
CPython.
|
||||
|
||||
### Operations supported
|
||||
|
||||
`Pattern_match/fullmatch/search/finditer/findall/split/sub/subn/free`,
|
||||
`Pattern_groupindex_lookup`,
|
||||
`Match_group/start/end/span/start_byte/end_byte/span_byte/free`,
|
||||
`re_compile/match/fullmatch/search/finditer/findall/split/sub/subn/escape/purge`,
|
||||
`PatternError_free`. See `regexx.h` for exact signatures, `docs/API.md` for
|
||||
the full reference (return values, memory ownership, exact Python
|
||||
correspondence for each one), and `concept.md` Section 9 for the naming
|
||||
convention they follow.
|
||||
|
||||
### Known deviations from `concept.md` and from CPython, beyond the items above
|
||||
|
||||
- `\w`, `\s`, `IGNORECASE` case folding, and `\d` in `UTF8` mode are backed
|
||||
by glibc's `wctype.h` functions under the `C.utf8` locale, not by a
|
||||
hand-generated Unicode table (`concept.md` 13.3 anticipated a reduced
|
||||
static table; using the C library's own tables turned out to be simpler
|
||||
and more complete, at the cost of depending on the platform's Unicode
|
||||
version rather than a pinned one).
|
||||
- `lastindex`/`lastgroup` report the highest-numbered capturing group that
|
||||
participated in the match, which coincides with CPython's "most recently
|
||||
closed group" rule for straightforward patterns but can differ from it in
|
||||
pathological cases (nested alternation re-executing a lower-numbered group
|
||||
after a higher one). Not exercised by the test suite; documented here
|
||||
rather than silently accepted.
|
||||
- `Match_free`, `PatternError_free`, `Input_from_buffer`, `Input_from_file`,
|
||||
and `Input_free` have no Python counterpart and are not mentioned in
|
||||
`concept.md`'s API surface; they exist because C has no garbage collector.
|
||||
`Pattern_sub`/`Pattern_subn` take the replacement template and the
|
||||
callback as two separate parameters rather than one polymorphic argument,
|
||||
for the same reason (`concept.md` 9.4 already anticipates and justifies
|
||||
this one).
|
||||
- Python's `Match.start(group)`/`.end(group)` raise `IndexError` for an
|
||||
invalid group number and return `-1` only for a valid group that did not
|
||||
participate; `Match_start`/`Match_end` return `-1` for both cases, since C
|
||||
has no exception to raise. `Match_group` does distinguish them (`-1` for
|
||||
no such group, `0` for an unparticipated one), see `docs/API.md` Section
|
||||
3.23.
|
||||
- `Pattern.groupindex` has no enumeration function in this build, only
|
||||
`Pattern_groupindex_lookup(pattern, name)`; there is no way to list every
|
||||
name a compiled pattern defines without already knowing what to look for.
|
||||
|
||||
## Building
|
||||
|
||||
Requires a C11 compiler and, for the test suite, Python 3 (used only to
|
||||
generate ground truth from CPython's own `re` module, `concept.md` Section
|
||||
11; the library itself has no runtime dependency beyond the C standard
|
||||
library and `libc`'s `wctype.h`/`locale.h`).
|
||||
|
||||
```sh
|
||||
make # builds libregexx.a and the rxgrep example
|
||||
make test # regenerates tests/generated_tests.c from Python `re`
|
||||
# ground truth and runs the full suite
|
||||
make check # same, under AddressSanitizer + UndefinedBehaviorSanitizer
|
||||
make clean
|
||||
```
|
||||
|
||||
`make install` installs `libregexx.a` and `regexx.h` under `PREFIX`
|
||||
(default `/usr/local`).
|
||||
|
||||
## Using the library
|
||||
|
||||
```c
|
||||
#include "regexx.h"
|
||||
#include <string.h>
|
||||
|
||||
const char *pattern = "(\\w+)@(\\w+)";
|
||||
Pattern *pat = re_compile(pattern, strlen(pattern), UTF8, NULL);
|
||||
Input *in = Input_from_buffer((const uint8_t *)"user@host", strlen("user@host"));
|
||||
|
||||
Match m;
|
||||
if (Pattern_search(pat, in, 0, -1, &m) == 1) {
|
||||
const char *g; size_t glen;
|
||||
Match_group(&m, NULL, 1, &g, &glen); /* g/glen -> "user" */
|
||||
Match_free(&m);
|
||||
}
|
||||
|
||||
char *out; size_t outlen;
|
||||
Pattern_sub(pat, in, "\\2@\\1", NULL, NULL, 0, &out, &outlen); /* "host@user" */
|
||||
free(out);
|
||||
|
||||
Pattern_free(pat);
|
||||
Input_free(in);
|
||||
```
|
||||
|
||||
`flags` to `re_compile` combine a data mode, exactly one of `BINARY`,
|
||||
`ASCII`, or `UTF8` (`concept.md` 9.3), with any of the Python-named flags
|
||||
(`IGNORECASE`, `MULTILINE`, `DOTALL`, `VERBOSE`, `ASCII` as a flag also
|
||||
forces ASCII-only `\w`/`\s`/`\d` inside `UTF8` mode, `LOCALE`, `DEBUG`).
|
||||
|
||||
## Example: rxgrep
|
||||
|
||||
`examples/rxgrep.c` is a small grep-like program built on the library,
|
||||
demonstrating all three data modes and both the matching and substitution
|
||||
API:
|
||||
|
||||
```sh
|
||||
./rxgrep -in 'hello' file.txt # case-insensitive, line numbers
|
||||
./rxgrep -m utf8 -o '\w+' file.txt # print every UTF-8 word, one per line
|
||||
./rxgrep -c 'error' log.txt # count matching lines
|
||||
./rxgrep -m binary 'a.c' data.bin # match raw bytes, embedded NUL included
|
||||
./rxgrep -m utf8 --sub 'REDACTED' '\d{3}-\d{4}' file.txt
|
||||
```
|
||||
|
||||
Run `./rxgrep --help` for the full option list.
|
||||
|
||||
## Testing
|
||||
|
||||
`tests/cases.py` lists pattern/subject/operation triples. `tests/gen.py`
|
||||
computes each one's expected result with CPython's own `re` module and
|
||||
writes `tests/generated_tests.c`, which is then compiled against `regexx.c`
|
||||
and checked. This is a direct implementation of the strategy `concept.md`
|
||||
Section 11 describes: conformance is measured against what CPython actually
|
||||
does, not against a re-derived reading of its documentation. `make check`
|
||||
additionally runs the suite under AddressSanitizer and
|
||||
UndefinedBehaviorSanitizer.
|
||||
|
||||
## License
|
||||
|
||||
MIT. See [`LICENSE`](LICENSE).
|
||||
+345
@@ -0,0 +1,345 @@
|
||||
# Concept: A Single File C Regex Interpreter with Python `re` Semantics
|
||||
|
||||
## 1. Objective
|
||||
|
||||
The objective is a regex interpreter, implemented as a single C source file, that reproduces the observable behavior of Python's `re` module (pattern syntax, flags, and the `match`, `search`, `fullmatch`, `findall`, `finditer`, `split`, `sub`, and `subn` operations) while being usable on inputs of arbitrary size, including multi-gigabyte files, without holding the entire input in memory. The implementation is required to operate in three data modes: binary (arbitrary byte streams), ASCII text, and UTF-8 text. Simplicity of the C code takes priority over raw execution speed.
|
||||
|
||||
Sections 2 through 4 record what "full Python `re` support" concretely means. Section 5 records why an unrestricted single pass, constant memory implementation of that full feature set is not mathematically possible, and states the boundary precisely. Sections 6 through 12 describe the architecture chosen to get as close to the objective as that boundary allows, in the simplest C design found. Section 13 lists the residual gaps against CPython's `re`. Section 14 records, for a reader coming from C rather than Python, the respects in which the native C regular expression facility (POSIX `regex.h`) differs from Python `re`, and therefore from this design, none of which this design adopts beyond what Section 1 already asks for.
|
||||
|
||||
## 2. Reference Semantics: Python `re` Pattern Syntax
|
||||
|
||||
The engine parses the following syntax, matching CPython's documented and observed behavior.
|
||||
|
||||
### 2.1 Atoms and literals
|
||||
- Literal characters (bytes in binary/ASCII mode, decoded code points in UTF-8 mode).
|
||||
- `.` matches any character except `\n`, or any character at all under `DOTALL`.
|
||||
- `\` followed by a non-alphanumeric character is that literal character.
|
||||
- Escapes: `\n \r \t \f \v \a \0`, octal `\ooo`, hexadecimal `\xhh`, `\uxxxx`, `\Uxxxxxxxx`, and named code points `\N{NAME}` (requires a Unicode name table; see 13.3).
|
||||
|
||||
### 2.2 Character classes
|
||||
- `[...]` with ranges (`a-z`), negation (`[^...]`), and literal `]`, `-`, `^` when escaped or positionally safe, matching CPython's class parser exactly (including that `]` as the first class member is literal).
|
||||
- Shorthand classes `\d \D \w \W \s \S`, each with an ASCII definition and a Unicode definition, selected by mode and by the `ASCII`/`UNICODE` flags exactly as CPython selects them for `str` versus `bytes` patterns.
|
||||
- `\b` and `\B` (word boundary and non boundary), defined with the same "word character" set as `\w` in the active mode.
|
||||
|
||||
### 2.3 Anchors
|
||||
- `^` and `$`: string boundaries by default; line boundaries under `MULTILINE`.
|
||||
- `\A` and `\Z`: string boundaries, unaffected by `MULTILINE`.
|
||||
|
||||
### 2.4 Quantifiers
|
||||
- Greedy: `* + ? {m,n} {m,} {,n} {m}`.
|
||||
- Lazy: `*? +? ?? {m,n}?`.
|
||||
- Possessive (CPython 3.11 and later): `*+ ++ ?+ {m,n}+`.
|
||||
- Atomic groups (CPython 3.11 and later): `(?>...)`.
|
||||
|
||||
### 2.5 Groups and grouping constructs
|
||||
- `(...)` capturing group, numbered left to right by opening parenthesis.
|
||||
- `(?:...)` non capturing group.
|
||||
- `(?P<name>...)` named capturing group; `(?P=name)` named backreference; `\1`..`\99` numbered backreference; `\g<name>` and `\g<1>` backreference forms usable inside a pattern as well as in a replacement string.
|
||||
- `(?#...)` comment, discarded at parse time.
|
||||
- `(?=...)`, `(?!...)`: lookahead, positive and negative, of unrestricted width.
|
||||
- `(?<=...)`, `(?<!...)`: lookbehind, positive and negative. CPython requires the lookbehind body to be fixed width (a fixed number of characters, alternation of equal width branches permitted); the engine enforces the same restriction at compile time and rejects variable width lookbehind with a compile error, matching CPython's `error: look-behind requires fixed-width pattern`.
|
||||
- `(?(id)yes|no)` and `(?(name)yes|no)`: conditional branch on whether a numbered or named group has already matched; the `|no` branch is optional (`(?(id)yes)` is valid, matching CPython, with an implicit empty "no" branch).
|
||||
- `(?aiLmsux)` global inline flags, valid only at the start of the pattern, and `(?aiLmsux-imsx:...)` scoped inline flags valid anywhere, matching CPython's restriction that only `i, m, s, x` are removable and `a, L, u` are global only.
|
||||
- `|` alternation, ordered, first match wins (not longest match), matching backtracking semantics rather than POSIX leftmost longest.
|
||||
|
||||
### 2.6 Flags
|
||||
`IGNORECASE`, `MULTILINE`, `DOTALL`, `VERBOSE` (whitespace and `#` comments outside classes and outside escapes are ignored), `ASCII`, `UNICODE` (default for text mode), `LOCALE` (accepted for compatibility, implemented as byte range `[\x80-\xff]` word characters under the "C" locale only; see 13.4), `DEBUG` (accepted, prints the compiled program instead of executing it).
|
||||
|
||||
## 3. Reference Semantics: Python `re` Operations
|
||||
|
||||
- `match(pattern, string)`: anchored at position 0, not required to consume the whole string.
|
||||
- `fullmatch(pattern, string)`: anchored at position 0 and at the end of the string.
|
||||
- `search(pattern, string)`: first match anywhere.
|
||||
- `finditer` / `findall`: all non overlapping matches, left to right, with the CPython 3.7+ rule that an empty match advances one position and is followed by a search starting immediately after it rather than being merged with an adjacent non empty match at the same start position.
|
||||
- `split(pattern, string, maxsplit=0)`: text of capturing groups is interleaved into the result list, matching CPython; splitting on a pattern that can match an empty string is permitted, matching the CPython 3.7+ behavior change.
|
||||
- `sub(pattern, repl, string, count=0)` / `subn`: `repl` is either a template string honoring `\g<name>`, `\g<1>`, `\1`, and literal backslash escapes, or a callback invoked once per match with a match record and expected to return replacement bytes/text (the C equivalent of a Python callable, see 9.4).
|
||||
- `escape(string)`: backslash escaping of all characters outside `[A-Za-z0-9_]` in the same way `re.escape` does since Python 3.7 (that version narrowed the escaped set relative to earlier Python releases; the engine follows the narrowed, current set).
|
||||
- Match record fields: `group(n)`, `group(name)`, `groups()`, `groupdict()`, `start(n)`, `end(n)`, `span(n)`, `lastindex`, `lastgroup`.
|
||||
|
||||
## 4. Feature Compatibility Table
|
||||
|
||||
| Feature | Status |
|
||||
|---|---|
|
||||
| Literals, classes, anchors, quantifiers (greedy/lazy) | Full |
|
||||
| Alternation, grouping, named groups | Full |
|
||||
| Backreferences (pattern and replacement) | Full, bounded (Section 5) |
|
||||
| Lookahead, fixed width | Full |
|
||||
| Lookahead, unbounded width | Full, bounded (Section 5), same window as backreferences |
|
||||
| Lookbehind, fixed width | Full |
|
||||
| Lookbehind, variable width | Rejected at compile time, as in CPython |
|
||||
| Conditional groups `(?(id)yes|no)` | Full |
|
||||
| Possessive quantifiers, atomic groups | Full (`OP_ATOMIC`, Section 7.1) |
|
||||
| Inline and scoped flags | Full |
|
||||
| `\N{NAME}` named code points | Partial (Section 13.3) |
|
||||
| Full Unicode `\w`/`\s`/`IGNORECASE` case folding | Partial (Section 13.3) |
|
||||
| `LOCALE` flag beyond the "C" locale | Not supported (Section 13.4) |
|
||||
| Streaming over unbounded input | Full for the regular subset, bounded window for backreferences/lookaround (Section 5) |
|
||||
|
||||
## 5. The Single Pass, Bounded Memory Constraint
|
||||
|
||||
This section states a limit that shapes the rest of the design, so it is recorded before the architecture.
|
||||
|
||||
A pattern language restricted to literals, classes, anchors, quantifiers, grouping, and alternation is a regular language. Regular languages are recognized by a finite automaton, and a finite automaton processes an input stream in one pass, in time linear in the input length, using memory bounded by the automaton's state count, independent of input length. Thompson's construction (converting a pattern to a nondeterministic finite automaton) and its simulation without backtracking (the approach used by `grep -E`, `awk`, RE2, and Rob Pike's regular expression virtual machine) achieve exactly this, including for `findall` style capture extraction.
|
||||
|
||||
Backreferences (`\1`, `(?P=name)`) break this property. No fixed size finite automaton recognizes a language defined with a backreference in general, because such an automaton would need to remember an arbitrarily long previously matched substring verbatim and compare it later, and a finite automaton has, by definition, only finitely many states with which to do so. The precise formal result here is a combined complexity result: deciding whether a string matches a pattern is NP-hard when both the pattern and the string are counted as part of the problem input. That result does not, by itself, say anything about a fixed pattern, compiled once, matched against a growing string, which is this design's actual situation; for a fixed pattern the practically relevant obstacle is different and better known by name, worst case exponential backtracking time in the length of the string, the mechanism behind catastrophic backtracking (commonly called ReDoS) in every production backtracking engine, including CPython's own `_sre`. Unbounded width lookahead and lookbehind create the same obstacle for a related reason: resolving them can require holding an unbounded span of the input, forward or backward, before the assertion's truth value is known. This is a property of the language class and of the evaluation strategy required to decide it, not of any particular implementation choice, and it means a literal reading of "full Python `re` support" and "does not remain in memory or hold the input" are mutually exclusive whenever a pattern actually uses a backreference or an unbounded width lookaround on unbounded input.
|
||||
|
||||
The engine resolves this by splitting execution into two engines sharing one bytecode format:
|
||||
|
||||
1. **Regular engine.** Any compiled pattern that contains no backreference and no lookaround whose body has unbounded width is executed by a Thompson/Pike style simulation: single pass, one buffered chunk of input at a time, memory bounded by the number of program instructions multiplied by the number of capture groups, independent of input length. This covers the large majority of patterns used in practice, including nested quantifiers, alternation, and fixed width lookaround.
|
||||
|
||||
2. **Bounded backtracking engine.** Any compiled pattern that contains a backreference or an unbounded width lookahead (fixed width lookaround of either polarity always stays in the regular engine, per 7.1) is executed by a backtracking simulation over a sliding window of the input, of a fixed configurable size (default 1 MiB, see 7.3). This engine gives exact CPython semantics as long as the text a backreference or lookahead needs to inspect fits inside the window relative to the current match attempt. If it does not, the engine reports a recoverable error (`ERANGE`-style status) identifying the offending construct and offset, rather than silently returning a wrong answer or reading the whole file into memory. This is the same trade every production streaming text tool with backreference support makes; the engine documents the bound instead of hiding it.
|
||||
|
||||
This split is decided once, at compile time, from the parsed pattern, before any input is read. A caller who needs a hard guarantee of bounded memory on arbitrary input can inspect the compiled pattern's engine selection before running it.
|
||||
|
||||
## 6. Architecture Overview
|
||||
|
||||
Single C file, four sections in this order, each independent of the ones after it:
|
||||
|
||||
1. **Parser**: pattern text to abstract syntax tree (AST). Recursive descent, one function per grammar production (`parse_alt`, `parse_concat`, `parse_repeat`, `parse_atom`), matching the structure of the grammar in Section 2 directly, so the parser can be read as an executable grammar.
|
||||
2. **Compiler**: AST to bytecode, by direct recursive translation (Thompson's construction), one code generation function per AST node kind. Same bytecode format is emitted regardless of which of the two engines (5.1/5.2) will run it; the engine choice is a separate flag computed from the AST (does it contain `OP_BACKREF`, or an `OP_LOOKAHEAD` whose body width is unbounded; CPython already forces `OP_LOOKBEHIND` to be fixed width, Section 2.5, so lookbehind never contributes to this flag).
|
||||
3. **Engines**: the regular engine (Pike VM) and the bounded backtracking engine (recursive backtracking over the sliding window), described in Section 7.
|
||||
4. **Public API**: the Python `re` equivalent entry points, described in Section 9.
|
||||
|
||||
Keeping parser, compiler, and both engines as pure functions over explicit structs (no hidden global state except one user supplied allocator, see 8.1) is what keeps the single file simple to read despite covering the full grammar.
|
||||
|
||||
## 7. Execution Engines
|
||||
|
||||
### 7.1 Bytecode
|
||||
|
||||
One flat instruction set, an array of tagged structs, used by both engines:
|
||||
|
||||
```
|
||||
enum opcode {
|
||||
OP_CHAR, /* match one literal byte/codepoint */
|
||||
OP_CLASS, /* match one byte/codepoint against a class */
|
||||
OP_ANY, /* match one byte/codepoint, DOTALL-sensitive */
|
||||
OP_SPLIT, /* two continuations (alternation, quantifiers) */
|
||||
OP_JMP,
|
||||
OP_SAVE, /* record current offset into capture slot N */
|
||||
OP_MATCH,
|
||||
OP_ASSERT, /* zero-width: ^ $ \b \B \A \Z */
|
||||
OP_BACKREF, /* forces bounded backtracking engine */
|
||||
OP_LOOKAHEAD, /* sub-program, zero-width, polarity flag */
|
||||
OP_LOOKBEHIND, /* sub-program, fixed width, zero-width, polarity*/
|
||||
OP_ATOMIC, /* sub-program, consuming, discards choice points */
|
||||
OP_COND, /* branch on whether group N has matched */
|
||||
};
|
||||
|
||||
struct inst { enum opcode op; int32_t x, y; uint32_t data; };
|
||||
```
|
||||
|
||||
This is the same instruction shape used by Pike's virtual machine and by RE2's bytecode; reusing it rather than inventing a new one is what keeps the compiler small (roughly one `case` per AST node).
|
||||
|
||||
`OP_LOOKAHEAD` and `OP_LOOKBEHIND` each carry, alongside the sub-program pointer and polarity bit, a compile time computed width: a concrete integer for `OP_LOOKBEHIND` (CPython requires this to exist and be fixed, Section 2.5) and either a concrete integer or an explicit "unbounded" marker for `OP_LOOKAHEAD`. Only an unbounded width `OP_LOOKAHEAD`, together with `OP_BACKREF`, forces engine selection to the bounded backtracking engine (7.3); a fixed width `OP_LOOKAHEAD` or `OP_LOOKBEHIND`, of either polarity, is executed by the regular engine (7.2) using a peek buffer sized to exactly that width, never the full window `W`.
|
||||
|
||||
Atomic groups (`(?>...)`) and possessive quantifiers (`*+ ++ ?+ {m,n}+`) both compile to `OP_ATOMIC`, which is the only new opcode either needs: a possessive quantifier is first desugared, at compile time, into the atomic group wrapping its ordinary greedy form (`X*+` becomes `(?>X*)`, `X{m,n}+` becomes `(?>X{m,n})`, and so on), so the compiler and both engines only ever have to implement `OP_ATOMIC` once. `OP_ATOMIC` runs its sub-program to its single highest priority success (one priority ordered thread simulation restricted to the sub-program in the regular engine, 7.2; one recursive match attempt in the bounded engine, 7.3), advances the current position past whatever it consumed, and then permanently discards every choice point created while matching the sub-program, so that if matching fails later in the overall pattern, the engine never backtracks into the atomic group looking for a different internal match, which is the defining behavior of both constructs in CPython. Unlike `OP_LOOKAHEAD`, `OP_ATOMIC` never forces the bounded backtracking engine, regardless of the width of its body: it advances the stream position as it matches, so, unlike a zero-width assertion, it never needs to hold matched text in memory for a decision made later. Only `OP_BACKREF` and an unbounded width `OP_LOOKAHEAD` trigger the bounded engine (Section 5, 6).
|
||||
|
||||
### 7.2 Regular engine: Pike VM over a chunk stream
|
||||
|
||||
Standard Thompson NFA simulation extended with capture slots, run breadth first ("all current threads advance over the same input character, in priority order, duplicate states are merged"). Per input character the engine holds at most `N` threads, `N` being the instruction count, each thread owning only its capture slot array (`2 * ngroups` offsets), so per character memory is `O(N * ngroups)`, not `O(input length)`.
|
||||
|
||||
Streaming adaptation: input arrives as a sequence of chunks (see 7.3) rather than one buffer. Thread capture slots store absolute stream offsets (a 64 bit counter incremented across chunk boundaries), not pointers into the chunk buffer, so a thread survives a chunk boundary without copying. Once every live thread's earliest referenced offset has advanced past a chunk boundary, that chunk is released back to the caller supplied allocator. This is the entire mechanism that lets the regular engine run over an arbitrarily large file in bounded memory: it never needs to look backward, so it never needs to keep anything but the current chunk and the small thread list.
|
||||
|
||||
Two small, constant size pieces of state cross a chunk boundary alongside the thread list. First, in `UTF8` mode, a partially read multi-byte sequence: at most 3 pending lead bytes (the longest UTF-8 sequence is 4 bytes), carried into the next chunk before character classification resumes; a chunk is only eligible for release once any sequence straddling its end has been completed by the following chunk. Second, under `MULTILINE`, one bit recording whether the byte immediately before the current chunk was `\n`, needed to classify `^` at the very first position of a new chunk without rereading the previous one; `\n` (`0x0A`) cannot appear as a non-initial byte of any valid multi-byte UTF-8 sequence, so this bit and the UTF-8 continuation state never interact with each other. Neither addition affects the O(instruction count x group count) memory bound of Section 10, since both are O(1) regardless of chunk size or input length.
|
||||
|
||||
### 7.3 Bounded backtracking engine: sliding window
|
||||
|
||||
Used only for the minority of patterns containing a backreference or an unbounded width lookahead (7.1). Maintains an explicit ring buffer window of `W` bytes (default `W = 1 MiB`, a run time parameter). The window always contains the current match attempt's start position and everything from there forward that has been read so far, up to `W` bytes. A recursive backtracking matcher, structurally the direct translation of the AST (one function per node kind, exactly as `_sre` and most textbook backtracking matchers are structured), walks the bytecode against the window. If a match attempt's required span would exceed `W`, the call returns the documented bounded-window error described in Section 5 instead of growing the window past its configured limit.
|
||||
|
||||
Because this engine is only invoked for patterns that need it, ordinary patterns (the large majority) never pay for the ring buffer or for backtracking, and get the linear time guarantee of 7.2 instead.
|
||||
|
||||
### 7.4 Anchoring across chunks
|
||||
|
||||
Both engines expose the same chunk boundary contract: a match cannot be reported as final until either (a) `OP_MATCH` is reached, or (b) enough trailing context has been seen to prove that no continuation of the current input would change the answer for the leftmost still-open match attempt. For quantifiers this is decided directly by the bytecode's `OP_SPLIT`/`OP_JMP` shape; for anchors (`$`, `\Z`) the final chunk is distinguished by an explicit "end of stream" sentinel token, so `$` and `\Z` behave identically whether or not `MULTILINE` is set, matching CPython.
|
||||
|
||||
## 8. Core Data Structures
|
||||
|
||||
Kept intentionally minimal, all defined in the single file, no dependency beyond the C standard library (`stdint.h`, `stddef.h`, `string.h`):
|
||||
|
||||
### 8.1 Allocator
|
||||
One struct of three function pointers (`alloc`, `realloc`, `free`) passed once at engine creation, defaulting to the libc equivalents. This is the only piece of "infrastructure" abstraction in the file, and it exists so the sliding window (7.3) and chunk buffers (7.2) can be sized and released under caller control, which is a prerequisite for the gigabyte scale requirement.
|
||||
|
||||
### 8.2 Dynamic array
|
||||
One generic growable array (`{ void *data; size_t len, cap, elemsize; }`) with `push`/`get`, used for the instruction array, the capture slot array, and the AST node pool. Deliberately not a macro-heavy generic container; three functions (`da_init`, `da_push`, `da_free`) cover every use site in the file.
|
||||
|
||||
### 8.3 Byte class table
|
||||
A 256 bit set (`uint32_t bits[8]`) per compiled `[...]` class or shorthand class, precomputed at compile time. In UTF-8 mode, code points above 127 are matched against a small number of precompiled Unicode range tables (Section 13.3) instead of the 256 bit set.
|
||||
|
||||
### 8.4 Capture slots
|
||||
A flat array of `2 * groups` capture records per thread (regular engine) or per backtracking call frame (bounded engine), `-1` meaning unset. In `BINARY` and `ASCII` mode a record is a single `int64_t` byte offset. In `UTF8` mode a record is a pair, `{ int64_t byte_offset; int64_t codepoint_index; }`: the byte offset is what is needed to read the matched bytes back out of `Input`, and the code point index is what is needed to report `Match_start`/`Match_end`/`Match_span` in the same unit Python uses for `str` subjects (Section 9.1). The code point index is a running counter incremented once per decoded scalar value as the engine advances, so recording it at a `SAVE` costs one extra integer copy, not a second pass over the input; it changes the constant factor of the O(instruction count x group count) memory bound of Section 10 in `UTF8` mode, not its asymptotic class. This single representation backs `Match_group`, `Match_start`, `Match_end`, `Match_span`, `Match_start_byte`, `Match_end_byte`, `Match_span_byte`, and `Match_groupdict` in the public API.
|
||||
|
||||
## 9. Public API (Python `re` equivalents)
|
||||
|
||||
### 9.0 Naming convention
|
||||
|
||||
Every public identifier that has a direct counterpart in Python's `re` module uses that counterpart's exact spelling, not a transliterated or prefixed variant. Concretely:
|
||||
|
||||
- The two data types Python's `re` exposes, `Pattern` and `Match`, are C structs named `Pattern` and `Match`, not `regex_t` or `regexx_pattern`.
|
||||
- A method Python calls as `pattern_obj.search(...)` is written `Pattern_search(Pattern *self, ...)`; a method called as `match_obj.group(...)` is written `Match_group(Match *self, ...)`. The `Type_method` shape is the direct C rendering of `type.method`, needed only because C has no bound methods; the two name fragments either side of the underscore are otherwise exactly the Python names.
|
||||
- A module level function such as `re.compile(...)` or `re.sub(...)` is written `re_compile(...)`, `re_sub(...)`, and so on: the `re_` prefix stands for the module the function lives in in Python (`re.compile`), the same relationship `Pattern_` and `Match_` have to their types.
|
||||
- Flag constants (`IGNORECASE`, `MULTILINE`, `DOTALL`, `VERBOSE`, `ASCII`, `UNICODE`, `LOCALE`, `DEBUG`) are `#define` or `enum` constants with exactly those names, no `RE_` or `REGEXX_` prefix, combined with bitwise `|` exactly as `re.IGNORECASE | re.MULTILINE` is combined with Python's `|`.
|
||||
- `Pattern` fields are named `pattern`, `flags`, `groups`, `groupindex`, matching `re.Pattern.pattern`, `.flags`, `.groups`, `.groupindex` exactly. `Match` fields are named `string`, `pos`, `endpos`, `lastindex`, `lastgroup`, `re` (a pointer back to the owning `Pattern`, matching `re.Match.re`), matching `re.Match`'s attributes exactly.
|
||||
- The exception is `Input` (9.1), which has no Python counterpart because Python's `re` never streams: it always operates on an in-memory `str` or `bytes` object. `Input` is the one C-only type the design adds, and it is named descriptively rather than after a nonexistent Python name, precisely so it stands out as the one addition a reader should not go looking for in the `re` documentation.
|
||||
- The error type is named `PatternError`, matching the alias CPython itself introduced for `re.error` (`re.PatternError`), rather than a plain `error` (which would collide too easily with `errno.h`-style conventions) or an invented `regexx_error`.
|
||||
|
||||
### 9.1 Declarations
|
||||
|
||||
```c
|
||||
typedef struct Pattern Pattern; /* re.Pattern */
|
||||
typedef struct Match Match; /* re.Match */
|
||||
typedef struct Input Input; /* no Python counterpart, see 9.0; left without a
|
||||
* struct body here because its concrete layout is
|
||||
* one of the two variants described in 9.2 and no
|
||||
* code outside the Input implementation itself
|
||||
* needs to see inside it, unlike Pattern and Match
|
||||
* whose fields are part of the public, Python-
|
||||
* mirroring surface. */
|
||||
|
||||
typedef void (*MatchIterCb)(void *ctx, const Match *m);
|
||||
typedef void (*MatchSubCb)(void *ctx, const Match *m, char **out, size_t *outlen);
|
||||
|
||||
typedef struct PatternError PatternError; /* re.error / re.PatternError */
|
||||
struct PatternError {
|
||||
const char *msg; /* re.error.msg */
|
||||
const char *pattern; /* re.error.pattern */
|
||||
int64_t pos; /* re.error.pos */
|
||||
int64_t lineno; /* re.error.lineno */
|
||||
int64_t colno; /* re.error.colno */
|
||||
};
|
||||
|
||||
struct Pattern {
|
||||
const char *pattern; /* re.Pattern.pattern */
|
||||
int flags; /* re.Pattern.flags */
|
||||
int groups; /* re.Pattern.groups */
|
||||
void *groupindex; /* re.Pattern.groupindex, name -> group number */
|
||||
void *program; /* compiled bytecode, private (7.1) */
|
||||
};
|
||||
|
||||
struct Match {
|
||||
Pattern *re; /* re.Match.re */
|
||||
Input *string; /* re.Match.string */
|
||||
int64_t pos, endpos; /* re.Match.pos, re.Match.endpos; code point indices in UTF8 mode, byte offsets otherwise, see 8.4 */
|
||||
int lastindex; /* re.Match.lastindex */
|
||||
const char *lastgroup; /* re.Match.lastgroup */
|
||||
void *slots; /* private, see 8.4 */
|
||||
};
|
||||
|
||||
/* Pattern methods: the primitives, bound to an already compiled Pattern,
|
||||
* mirroring re.Pattern.match / .search / .fullmatch / .finditer / .findall /
|
||||
* .split / .sub / .subn exactly, including their pos/endpos parameters. */
|
||||
int Pattern_match(Pattern *self, Input *string, int64_t pos, int64_t endpos, Match *out);
|
||||
int Pattern_fullmatch(Pattern *self, Input *string, int64_t pos, int64_t endpos, Match *out);
|
||||
int Pattern_search(Pattern *self, Input *string, int64_t pos, int64_t endpos, Match *out);
|
||||
int Pattern_finditer(Pattern *self, Input *string, int64_t pos, int64_t endpos, MatchIterCb cb, void *ctx);
|
||||
int Pattern_findall(Pattern *self, Input *string, int64_t pos, int64_t endpos, MatchIterCb cb, void *ctx);
|
||||
int Pattern_split(Pattern *self, Input *string, int maxsplit, MatchIterCb cb, void *ctx);
|
||||
int Pattern_sub(Pattern *self, Input *string, const char *repl, MatchSubCb cb, void *ctx, int count, char **out, size_t *outlen);
|
||||
int Pattern_subn(Pattern *self, Input *string, const char *repl, MatchSubCb cb, void *ctx, int count, char **out, size_t *outlen, int *n);
|
||||
void Pattern_free(Pattern *self);
|
||||
|
||||
/* Match accessors, matching re.Match's bound methods and attributes */
|
||||
int Match_group(Match *self, const char *name_or_null, int index, const char **out, size_t *outlen);
|
||||
void Match_groups(Match *self, /* out array of (ptr,len) */ void *out);
|
||||
void Match_groupdict(Match *self, /* out name -> (ptr,len) map */ void *out);
|
||||
int64_t Match_start(Match *self, int group); /* code point index in UTF8 mode, byte offset otherwise */
|
||||
int64_t Match_end(Match *self, int group);
|
||||
void Match_span(Match *self, int group, int64_t *start, int64_t *end);
|
||||
int64_t Match_start_byte(Match *self, int group); /* always a byte offset into Input, see 8.4 */
|
||||
int64_t Match_end_byte(Match *self, int group);
|
||||
void Match_span_byte(Match *self, int group, int64_t *start, int64_t *end);
|
||||
void Match_expand(Match *self, const char *template, char **out, size_t *outlen);
|
||||
|
||||
/* Module level functions, mirroring re.compile / re.match / re.search / ...
|
||||
* exactly: each of the search-family functions compiles pattern through an
|
||||
* internal bounded cache and then calls the matching Pattern_ function,
|
||||
* exactly as CPython's re/__init__.py implements re.match as
|
||||
* _compile(pattern, flags).match(string). */
|
||||
Pattern *re_compile(const char *pattern, size_t len, int flags, PatternError *err);
|
||||
int re_match(const char *pattern, size_t len, int flags, Input *string, Match *out);
|
||||
int re_fullmatch(const char *pattern, size_t len, int flags, Input *string, Match *out);
|
||||
int re_search(const char *pattern, size_t len, int flags, Input *string, Match *out);
|
||||
int re_finditer(const char *pattern, size_t len, int flags, Input *string, MatchIterCb cb, void *ctx);
|
||||
int re_findall(const char *pattern, size_t len, int flags, Input *string, MatchIterCb cb, void *ctx);
|
||||
int re_split(const char *pattern, size_t len, int flags, Input *string, int maxsplit, MatchIterCb cb, void *ctx);
|
||||
int re_sub(const char *pattern, size_t len, int flags, Input *string, const char *repl, MatchSubCb cb, void *ctx, int count, char **out, size_t *outlen);
|
||||
int re_subn(const char *pattern, size_t len, int flags, Input *string, const char *repl, MatchSubCb cb, void *ctx, int count, char **out, size_t *outlen, int *n);
|
||||
void re_escape(const char *in, size_t len, char **out, size_t *outlen);
|
||||
void re_purge(void);
|
||||
```
|
||||
|
||||
The dependency runs from module level to `Pattern`, not the other way around, which is the same direction CPython itself uses: `re.py` defines `match`, `search`, and the rest as thin wrappers that call `_compile(pattern, flags)` and then the corresponding `Pattern` method. Each module level function above holds an internal cache keyed by `(pattern, len, flags)`, bounded to a fixed capacity and cleared entirely on overflow rather than evicting individual entries, again mirroring the strategy CPython's own `re` module cache uses. `re_purge()` clears that cache on demand, matching `re.purge()` exactly, including that it has no effect on any `Pattern *` a caller is still holding a direct reference to. `re_compile` bypasses the cache and always produces a fresh `Pattern`, matching `re.compile`. A caller working against a large or streaming `Input` and applying the same pattern repeatedly should call `re_compile` once and use the `Pattern_` functions directly, exactly as idiomatic Python precompiles a pattern that is reused in a loop rather than calling the module level function repeatedly; the module level functions exist for parity with `re.match`/`re.search`/etc., not as the recommended entry point for the gigabyte scale case this design targets.
|
||||
|
||||
### 9.2 `Input`
|
||||
|
||||
An abstraction over "a source of chunks": either a fixed buffer (small strings, a drop in replacement for the CPython `str`/`bytes` case) or a caller supplied `read(void *ctx, uint8_t *buf, size_t cap) -> size_t` callback (files, pipes, sockets), which is how gigabyte scale input is supplied without ever requiring the caller to load it fully into memory. See 9.0 for why this type does not carry a Python name.
|
||||
|
||||
`pos` and `endpos` on the `Pattern_` functions are expressed in the same unit as `Match_start`/`Match_end` for the pattern's mode (code point index in `UTF8` mode, byte offset otherwise, Section 8.4). Honoring a nonzero `pos` against a streaming `Input` that only exposes sequential `read` requires decoding forward from the start of the stream until that position is reached, an O(pos) cost paid once per call, not a departure from the per-character bound of Section 10, which is stated per byte of input actually scanned. An `Input` that also exposes an optional `seek(void *ctx, int64_t byte_offset) -> int` callback lets the engine skip that decode pass in `ASCII`/`BINARY` mode, or in `UTF8` mode whenever the caller already knows the target byte offset, for example one returned earlier by `Match_start_byte` against the same `Input`.
|
||||
|
||||
### 9.3 Encoding mode
|
||||
|
||||
A field of the compile-time `flags` alongside `IGNORECASE`, `MULTILINE`, and the rest: `BINARY`, `ASCII`, `UTF8` (no `RE_` prefix, per 9.0; CPython has no equivalent constant because the choice between binary and text mode is implicit in whether a `bytes` or `str` pattern was compiled, so `Pattern_compile` here makes that same choice explicit through a flag instead). This flag selects, at compile time, which byte class tables (8.3), which `.`/`\w`/`\s` definitions (2.2), and which decoder (raw byte, or the UTF-8 decoder producing code points for classification while recording both a code point index and a byte offset per capture, Section 8.4) the compiled program uses. Binary mode never decodes: every byte value 0 to 255, including embedded NUL, is a valid atom, matching the behavior Python gets by compiling a `bytes` pattern against a `bytes` subject.
|
||||
|
||||
A byte sequence in `UTF8` mode that is not valid UTF-8 at the point the decoder reaches it is handled the way `bytes.decode('utf-8', errors=...)` is in Python: the default policy, `strict`, surfaces a `PatternError` identifying the byte offset of the first invalid byte, matching the fact that CPython can never hand `re` a `str` that was not already validly decoded in the first place. An opt-in `replace` policy substitutes the Unicode replacement character `U+FFFD` for the offending bytes and continues, for callers that must process untrusted or partially corrupt streams without aborting. `BINARY` and `ASCII` mode have no decode step and so have no analogous failure mode; a byte outside `0`-`127` in `ASCII` mode is simply a byte no `ASCII`-mode class matches, not an error.
|
||||
|
||||
### 9.4 Replacement callback
|
||||
|
||||
`Pattern_sub`/`Pattern_subn` (and the module level `re_sub`/`re_subn` that wrap them, 9.1) take both a `repl` template string and a `cb` callback of type `MatchSubCb` as separate parameters, of which exactly one is non-`NULL` on any given call: a non-`NULL` `repl` is parsed once, at compile time, into a small list of literal/backreference segments, mirroring Section 3's replacement syntax; a non-`NULL` `cb` is invoked once per match with the match record and a caller supplied `ctx`, and is expected to write the replacement text through `out`/`outlen`, which is the C shape of Python's callable `repl` argument to `re.sub`. Two separate parameters, rather than one parameter serving both roles, is the direct C consequence of Python's single `repl` argument being polymorphic (string or callable) in a way C's static type system cannot express in one slot, the same kind of unavoidable, minimal departure from a one to one name and shape mapping that 9.0 already accepts for `Type_method`.
|
||||
|
||||
## 10. Complexity Summary
|
||||
|
||||
| Engine | Time | Memory | Applies to |
|
||||
|---|---|---|---|
|
||||
| Regular (Pike VM, 7.2) | O(input length x instruction count) | O(instruction count x group count), independent of input length | Patterns with no backreference and no unbounded lookaround |
|
||||
| Bounded backtracking (7.3) | Worst case exponential in window size, as in CPython | O(window size `W`), independent of input length beyond `W` | Patterns with a backreference or an unbounded width lookahead |
|
||||
|
||||
Both figures are stated relative to input length specifically because that is the axis the gigabyte scale requirement constrains; instruction count and group count are properties of the pattern, not the input, and are expected to stay small (tens to low hundreds) for realistically written patterns.
|
||||
|
||||
## 11. Testing Strategy
|
||||
|
||||
CPython ships its own `re` test suite (`Lib/test/test_re.py` / `re_tests.py`) as executable pattern, string, expected-result triples. The plan is to translate that suite mechanically into a table of C test cases run against `re_match`/`re_search`/`re_sub`, so the engine's conformance is measured against CPython's own stated behavior rather than against a re-derived interpretation of the documentation. Cases that exercise the explicitly out of scope items in Section 13 are recorded as known deviations rather than deleted, so the gap stays visible.
|
||||
|
||||
## 12. Worked Example: Why This Is the Simplest Design That Reaches the Goal
|
||||
|
||||
A single unified backtracking engine (matching CPython's own `_sre` design most closely) would be simpler to write than the two-engine split in Section 5 and Section 7, but it cannot satisfy the gigabyte scale, bounded memory requirement for the common case, because a naive backtracking matcher's stack depth and re-scan behavior scale with input length for ordinary patterns, not only for pattern using backreferences. Conversely, a single unified automaton engine (no backtracking at all) is simpler still, but cannot express backreferences, or the same unbounded width lookaround built from the widening this design already restricts to the bounded engine (Section 5, 7.1), which the objective in Section 1 requires. The two-engine split is the smallest design found that keeps the automaton engine's linear-time, bounded-memory property for the patterns that admit it, while still offering exact backreference and lookaround semantics for the patterns that need them, at an explicit and configurable memory cost.
|
||||
|
||||
## 13. Known Gaps Against CPython `re`
|
||||
|
||||
### 13.1 POSIX leftmost-longest matching
|
||||
Not applicable; CPython's `re` itself uses ordered, first-alternative-wins backtracking semantics, and this engine matches that, not POSIX `grep -E` semantics. See Section 14.1 for the full comparison against native C `regex.h` behavior, which this section only touched on briefly before that comparison existed.
|
||||
|
||||
### 13.2 Recursion limit parity
|
||||
CPython raises `RecursionError` past a configurable backtracking depth. The bounded backtracking engine (7.3) instead bounds by window size and an explicit call depth counter with a similar default; exact error message parity is not a goal, only the presence of a safe failure mode.
|
||||
|
||||
### 13.3 Full Unicode tables
|
||||
`\N{NAME}` lookup, full `IGNORECASE` case folding (including special casing such as German `ß`), and complete `\w`/`\s` Unicode category coverage require the Unicode Character Database. The concept ships a reduced set of range tables covering the common categories (letters, digits, marks, common whitespace) rather than the full database, to keep the single file small; the compiled tables are generated from the Unicode Character Database offline and checked in as static arrays, with the generation script kept outside the single interpreter file.
|
||||
|
||||
### 13.4 `LOCALE` flag
|
||||
Only the "C" locale behavior is implemented (byte range `\x80`-`\xff` treated as word characters); full `locale.h` integration is out of scope because it reintroduces global, environment dependent state into an otherwise pure, single file design.
|
||||
|
||||
### 13.5 `regex` third party module extensions
|
||||
Constructs from the third party `regex` package (set operations inside classes such as `--`/`&&`, fuzzy matching, recursive patterns `(?R)`, variable width lookbehind) are not part of CPython's `re` and are out of scope by Section 1's own definition of "what Python supports."
|
||||
|
||||
### 13.6 Concurrency of the module level pattern cache
|
||||
The internal cache backing the module level functions of Section 9.1 (`re_match`, `re_search`, and the rest) is shared, mutable state. CPython's own equivalent cache is implicitly protected by the GIL; this design has no equivalent, so a caller invoking the module level functions from more than one thread concurrently must serialize access to the cache itself (a mutex around lookup and insertion, sized independently of the allocator in 8.1) or avoid the module level functions entirely and call `re_compile` once per pattern up front, sharing the resulting read only `Pattern *` across threads. The latter is already the recommended pattern for the gigabyte scale case (9.1's closing paragraph), so this limitation is expected to be inactive on the path the design is optimized for.
|
||||
|
||||
## 14. POSIX / Native C Regex (`regex.h`) Capabilities Absent From Python `re`
|
||||
|
||||
Because this project is delivered as a C module, a reader coming from C rather than from Python may reasonably expect it to behave like the regular expression facility native to the C standard library, POSIX `regex.h` (`regcomp`, `regexec`, `regfree`, `regerror`, specified by IEEE Std 1003.1). This section records, completely and for that reader specifically, the respects in which POSIX's native facility does something Python `re` does not do at all. Every item is either omitted by design, meaning it is incompatible with matching Python `re`'s own behavior and is therefore excluded by Section 1's own definition of the target, or out of scope, meaning it would not conflict with Python parity but was not requested and is not free to add under Section 1's simplicity priority. Section 14.6 closes with the one respect in which this design already exceeds POSIX `regex.h`, included so the comparison is not one sided.
|
||||
|
||||
### 14.1 Leftmost-longest ("POSIX") matching
|
||||
POSIX `regex.h` is specified to find the leftmost match and, among matches starting at that leftmost position, the longest one, applied recursively to subexpressions as well as to the overall match. Python `re`, like every Perl-derived engine, instead uses leftmost-first, ordered-alternation, backtracking semantics: the first alternative that leads to any overall match wins, even when a later alternative would consume more text, and quantifier greediness is resolved by backtracking order rather than by a global longest-match search. These are two different, mutually incompatible definitions of "the match" for the same pattern and string; `a|ab` against `"ab"` matches `"a"` under Python/Perl semantics and `"ab"` under POSIX semantics. This design follows Python's ordered semantics throughout (Section 2.5, Section 13.1), by the objective in Section 1. Omitted by design: Pike's priority ordered thread simulation (7.2), used here specifically to reproduce Perl style ordered semantics, is a different algorithm from the one POSIX-longest resolution requires, and running both simultaneously would cost the single, simple engine design Section 12 argues for, for a mode nothing in Section 1 asks for.
|
||||
|
||||
### 14.2 POSIX bracket-expression collating symbols and equivalence classes
|
||||
A POSIX bracket expression may contain `[.collating-symbol.]` (a named, possibly multi-character collating element treated as one unit, useful for ranges) and `[=equivalence-class=]` (every character the active locale's collation treats as primary equivalent to the given one). Python's `[...]` syntax has no counterpart to either: `[[.ch.]]` and `[[=e=]]` in a Python pattern parse as plain sets of the literal characters `[`, `.`, `c`, `h`, `]` and `[`, `=`, `e`, `]`, never as collating constructs, because Python bracket expressions are defined purely over literal characters and ranges, never over locale collation data. Out of scope: these constructs only have observable effect in locales with genuine multi-character collating elements, which is rare in practice, Python's `re` has never implemented them for `str` or `bytes` patterns, and adding them would require linking the compiled tables of Section 8.3 to the system's `LC_COLLATE` data, in direct tension with the dependency-free, pure-function design of Section 8.
|
||||
|
||||
### 14.3 POSIX named character classes inside bracket expressions
|
||||
POSIX bracket expressions accept the twelve standard named classes, `alpha`, `digit`, `alnum`, `upper`, `lower`, `space`, `blank`, `cntrl`, `graph`, `print`, `punct`, `xdigit`, written as `[:name:]` inside a bracket expression, for example `[[:alpha:][:digit:]]`. Python's `re` has no equivalent syntax; the same intent is expressed with ranges and the shorthand classes of Section 2.2 instead (`[a-zA-Z]`, `\d`, `\s`), and `[[:alpha:]]` in Python parses as a literal set containing `:`, `a`, `l`, `p`, `h`, `[`, `]`. Out of scope by direct consequence of Section 1: Python `re` genuinely has no such syntax, so parity with Python `re` already excludes it; it is recorded here only because a C-background reader is likely to look for it and be surprised to find it silently absent rather than documented.
|
||||
|
||||
### 14.4 Locale collating sequence for bracket-expression ranges
|
||||
POSIX specifies that a bracket-expression range such as `[a-z]` is resolved according to the current locale's collating sequence (`LC_COLLATE`), not according to raw code point or byte value order; in a locale whose collation is not a simple ascending code point order, `[a-z]` can therefore include, exclude, or reorder characters relative to what it means in the "C" locale, a well known source of surprising results in POSIX tools run under a non-"C" locale. Python's `re` never does this: a range in a Python pattern is always defined by code point value (`str` patterns) or byte value (`bytes` patterns), unconditionally, in every locale. This design follows Python exactly, ranges are always resolved by code point or byte value (Section 2.2), independent of the `LOCALE` flag (Section 13.4). Omitted by design, for the same reason `LOCALE` itself is restricted to the "C" locale in Section 13.4: honoring arbitrary system collation would reintroduce global, environment dependent state into a design that is otherwise a pure function of its inputs, and would make the meaning of a compiled `Pattern` depend on a process-wide setting instead of on the `flags` given to `re_compile`.
|
||||
|
||||
### 14.5 `REG_NOSUB`: compiling a pattern that reports no subexpression positions
|
||||
`regcomp(..., REG_NOSUB)` compiles a pattern that reports only whether it matched, not where its subexpressions matched, letting an implementation skip the capture bookkeeping of Section 8.4 entirely. Python's `re` has no equivalent compile-time flag; a compiled `Pattern` always reports full match and group data through `Match`. Out of scope: this is a pure performance optimization with no observable behavior difference, and Section 1 places convenience above performance throughout, so there is nothing here worth the added compile-time flag and the second code path it would require in both engines.
|
||||
|
||||
### 14.6 Where this design already exceeds plain POSIX `regex.h`
|
||||
`regexec` takes a nul-terminated C string, so plain POSIX `regex.h`, on most implementations, cannot search a subject containing an embedded NUL byte at all; the widely available but non-standard `REG_STARTEND` extension (present in glibc and the BSDs, not part of IEEE Std 1003.1 itself) works around this only on the platforms that provide it. Python's `re` has never had this limitation, since `str` and `bytes` are always explicit-length, never nul-terminated, and this design follows Python and inherits the same freedom from it directly: `Input` (Section 9.2) is always an explicit-length byte source, and `BINARY` mode (Section 9.3) explicitly allows an embedded NUL as an ordinary byte value, matching the gigabyte scale, arbitrary-binary-data objective of Section 1. This item is placed last, and out of sequence with the rest of Section 14's "absent from Python `re`" framing, specifically so the comparison in this section is accurate in both directions rather than reading as one sided.
|
||||
+279
@@ -0,0 +1,279 @@
|
||||
# regexx API Reference
|
||||
|
||||
This document records, exhaustively, the complete public surface of
|
||||
`regexx.h`/`regexx.c`: every type, every flag, every function, its exact
|
||||
return-value convention, its memory ownership rule, and its relationship to
|
||||
the corresponding Python `re` name. `concept.md` records the design
|
||||
rationale; `README.md` is the short entry point (build, usage, implementation
|
||||
status at a glance). This document is the complete reference the other two
|
||||
point to when a precise answer is needed.
|
||||
|
||||
Every fact below was checked against the current source (`regexx.c`,
|
||||
`regexx.h`) and, where a Python behavior is cited, against a real CPython 3
|
||||
interpreter, not against memory or documentation alone.
|
||||
|
||||
## 1. Types
|
||||
|
||||
### 1.1 `Pattern` (`re.Pattern`)
|
||||
|
||||
```c
|
||||
struct Pattern {
|
||||
const char *pattern; /* re.Pattern.pattern */
|
||||
int flags; /* re.Pattern.flags */
|
||||
int groups; /* re.Pattern.groups */
|
||||
void *groupindex; /* re.Pattern.groupindex, opaque, see 3.11 */
|
||||
void *program; /* private: compiled bytecode */
|
||||
};
|
||||
```
|
||||
|
||||
- `pattern`: the exact source text passed to `re_compile`, NUL-terminated, owned by the `Pattern` (freed by `Pattern_free`). Read-only for callers.
|
||||
- `flags`: the exact `flags` value passed to `re_compile`, including the encoding-mode bits (`BINARY`/`ASCII`/`UTF8`) and any leading global inline flags folded in during parsing (Section 2.6).
|
||||
- `groups`: the number of capturing groups, matching Python's `re.Pattern.groups` exactly (group 0, the whole match, is not counted).
|
||||
- `groupindex`: opaque; see `Pattern_groupindex_lookup` (3.11) for the only supported access to it.
|
||||
- `program`: private, never dereference directly.
|
||||
|
||||
A `Pattern` is created only by `re_compile` (directly, or indirectly through the module level cache in `re_match`/`re_search`/etc.) and is freed only by `Pattern_free`.
|
||||
|
||||
### 1.2 `Match` (`re.Match`)
|
||||
|
||||
```c
|
||||
struct Match {
|
||||
Pattern *re; /* re.Match.re */
|
||||
Input *string; /* re.Match.string */
|
||||
int64_t pos, endpos; /* re.Match.pos, re.Match.endpos */
|
||||
int lastindex; /* re.Match.lastindex */
|
||||
const char *lastgroup; /* re.Match.lastgroup */
|
||||
void *slots; /* private */
|
||||
};
|
||||
```
|
||||
|
||||
- `re`: the `Pattern` that produced this match. Borrowed reference; do not free it through this pointer, and do not call `Pattern_free` on it while any `Match` from it is still in use.
|
||||
- `string`: the `Input` the match was found in. Borrowed reference, same lifetime rule as `re`.
|
||||
- `pos`, `endpos`: the effective search bounds used to produce this match (the `pos`/`endpos` arguments to whichever `Pattern_` function created it, clamped to `[0, length]`), in the same unit as `Match_start`/`Match_end` (code point index in `UTF8` mode, byte offset otherwise). Matches `re.Match.pos`/`.endpos` exactly.
|
||||
- `lastindex`: the highest-numbered capturing group that participated in the match, or `-1` if none did. **Deviation from CPython:** Python's `lastindex` is "the index of the last group to match", which for a pattern with re-entrant alternation can differ from "the highest-numbered group that participated"; this build uses the latter, simpler rule. They coincide for every straightforward pattern (any pattern without a capturing group inside a repeated alternative that can also match via a different, lower-numbered branch later in the same attempt).
|
||||
- `lastgroup`: the name of that group, or `NULL` if it is unnamed or `lastindex` is `-1`. Borrowed pointer into the `Pattern`'s group name table; valid as long as the `Pattern` is.
|
||||
- `slots`: private.
|
||||
|
||||
A `Match` is filled in by one of the `Pattern_`/`re_` matching functions (never allocated separately by the caller: pass the address of a stack or heap `Match` struct as `out`, zero-initialize it first). Its private `slots` are freed by `Match_free`, which does **not** free the `Match` struct itself (the caller owns that memory, stack or heap).
|
||||
|
||||
### 1.3 `Input` (no Python counterpart)
|
||||
|
||||
Opaque. Python's `re` never streams; it always operates on an in-memory `str`/`bytes` object already held by the caller. `Input` is the type this design adds so a C caller has an explicit thing to construct from a buffer or a file (`concept.md` 9.0/9.2).
|
||||
|
||||
```c
|
||||
Input *Input_from_buffer(const uint8_t *buf, size_t len);
|
||||
Input *Input_from_file(const char *path, PatternError *err);
|
||||
void Input_free(Input *in);
|
||||
```
|
||||
|
||||
- `Input_from_buffer`: wraps an existing buffer. **Does not copy it and does not take ownership.** The buffer must outlive the `Input` and every `Match` produced from it (`Match_group` returns pointers directly into it, Section 3.9). Freeing an `Input_from_buffer` `Input` never frees the underlying buffer; the caller is responsible for that.
|
||||
- `Input_from_file`: reads the whole file into a freshly allocated, owned buffer. Works on both seekable files and non-seekable sources (pipes, FIFOs, process substitution, `/dev/stdin`): a seekable source is read in one `fread` after `fseek`/`ftell` sizing it; a non-seekable source is read incrementally into a growable buffer. Either way, the entire input ends up in memory before any matching happens (`README.md` "Implementation status"). On failure (cannot open, cannot allocate) returns `NULL` and, if `err` is non-`NULL`, fills it with `strerror(errno)` as `msg`.
|
||||
- `Input_free`: frees the `Input` and, only if it owns its buffer (true for `Input_from_file`, false for `Input_from_buffer`), the buffer too.
|
||||
|
||||
### 1.4 `PatternError` (`re.error` / `re.PatternError`)
|
||||
|
||||
```c
|
||||
struct PatternError {
|
||||
const char *msg; /* re.error.msg */
|
||||
const char *pattern; /* re.error.pattern */
|
||||
int64_t pos; /* re.error.pos */
|
||||
int64_t lineno; /* re.error.lineno */
|
||||
int64_t colno; /* re.error.colno */
|
||||
};
|
||||
void PatternError_free(PatternError *err);
|
||||
```
|
||||
|
||||
Named after the alias CPython itself introduced for `re.error` (`re.PatternError`, `concept.md` 9.0). Fields match CPython's `re.error` attributes (added in CPython 3.5) exactly: `msg` is the human-readable message, `pattern` is the offending pattern text (or `NULL` when the error is not about pattern syntax, for example `Input_from_file` failing to open a file), `pos` is the byte offset into `pattern` the error was detected at, `lineno`/`colno` are computed from `pos` the same way CPython computes them (1-based, counting `\n` bytes in `pattern` up to `pos`).
|
||||
|
||||
**Ownership:** `msg` and `pattern` are heap allocated (`strdup`) by whichever call filled the struct in. Zero-initialize a `PatternError` before passing its address in, and call `PatternError_free` on it once done reading it, whether or not the call that filled it in succeeded (a `NULL` `err` argument to any function is always safe to pass and simply skips error reporting). `PatternError_free` is safe to call on an all-zero or already-freed `PatternError`.
|
||||
|
||||
Every function that can fail accepts an optional `PatternError *err` (pass `NULL` to ignore); `re_compile` and `Input_from_file` are the two that actually produce one today. `re_match`/`re_search`/etc. (the module level convenience functions, 3.12) do not expose a `PatternError` parameter at all, matching the fact that CPython's own `re.match`/`re.search`/etc. do not return one either (a syntax error there raises, which has no C equivalent; here it instead causes the call to return `-1` with no further diagnostic, which is why `Pattern`-based usage, not the module level convenience functions, is recommended whenever a compile error needs to be reported, `concept.md`/README "recommended entry point" note).
|
||||
|
||||
### 1.5 `MatchIterCb` / `MatchSubCb`
|
||||
|
||||
```c
|
||||
typedef void (*MatchIterCb)(void *ctx, const Match *m);
|
||||
typedef void (*MatchSubCb)(void *ctx, const Match *m, char **out, size_t *outlen);
|
||||
```
|
||||
|
||||
`MatchIterCb` is invoked once per result by `Pattern_finditer`, `Pattern_findall` (an alias of `finditer` in this build, 3.7), and `Pattern_split` (3.8, with a different per-call meaning, documented there). The `Match` passed in is only valid for the duration of the call; it is freed immediately after the callback returns, so do not retain the pointer.
|
||||
|
||||
`MatchSubCb` is the C shape of Python's callable `repl` argument to `re.sub`/`re.subn`: invoked once per match, expected to `malloc` a replacement buffer, write its address into `*out` and its length into `*outlen`. The callback's `*out` becomes owned by `Pattern_sub`/`Pattern_subn`, which frees it after copying its content into the final result buffer.
|
||||
|
||||
## 2. Flags
|
||||
|
||||
Passed as `flags` to `re_compile` or any `re_` module level function, combined with bitwise `|`, using the exact spelling CPython uses (`concept.md` 9.0).
|
||||
|
||||
| Flag | Value | Effect | Status |
|
||||
|---|---|---|---|
|
||||
| `IGNORECASE` | `0x0001` | Case-insensitive literal, class, and backreference comparison. Class ranges are matched under both a character's original case and its swapped case (README "Known deviations": an approximation, not full Unicode case folding). | Implemented |
|
||||
| `MULTILINE` | `0x0002` | `^`/`$` also match at the start/end of each line, not only the start/end of the string. | Implemented |
|
||||
| `DOTALL` | `0x0004` | `.` matches `\n` too. | Implemented |
|
||||
| `VERBOSE` | `0x0008` | Unescaped whitespace and `#`-to-end-of-line comments outside character classes are stripped from the pattern before parsing. | Implemented |
|
||||
| `ASCII` | `0x0010` | Two roles: (a) with neither `BINARY` nor `UTF8` also given, selects the `ASCII` **encoding mode** (Section 2, below); (b) in `UTF8` mode, forces `\d`/`\w`/`\s` and `IGNORECASE` swapcase to their ASCII-only definitions instead of consulting `wctype.h`. | Implemented |
|
||||
| `UNICODE` | `0x0020` | Accepted for source compatibility with Python. **No effect**: `UTF8` mode already behaves as CPython's default (Unicode) `str` matching, so there is no separate "unicode" flag needed the way `ASCII` needs one to opt out. | Accepted, no-op |
|
||||
| `LOCALE` | `0x0040` | In `BINARY`/`ASCII` mode only, bytes `0x80`-`0xFF` are additionally treated as word characters for `\w`/`\b`/`\B` (`concept.md` 13.4's "C locale" approximation). No effect in `UTF8` mode. | Implemented (C-locale approximation only) |
|
||||
| `DEBUG` | `0x0080` | Accepted for source compatibility with Python. **No effect** in this build: nothing is printed and matching is unaffected. | Accepted, no-op |
|
||||
| `BINARY` | `0x0100` | Encoding mode: raw bytes, `0`-`255` all valid, no decoding, embedded `NUL` is an ordinary byte. | Implemented |
|
||||
| `UTF8` | `0x0200` | Encoding mode: input is decoded as UTF-8 into code points before matching; `Match_start`/`Match_end` report code point indices (`Match_start_byte`/`Match_end_byte` report byte offsets, Section 3.9). | Implemented |
|
||||
|
||||
Exactly one encoding mode is active for any compiled `Pattern`: `BINARY` and `UTF8` are checked first, in that order, and if neither is present the mode is `ASCII` regardless of whether the `ASCII` flag bit itself was set (so `re_compile(p, n, 0, &err)` and `re_compile(p, n, ASCII, &err)` compile to the same encoding mode; only combining `ASCII` with `UTF8` changes anything, per row (b) above).
|
||||
|
||||
## 3. Functions
|
||||
|
||||
Return value convention used throughout, unless noted otherwise for a specific function: `1` success/matched, `0` no match (not an error), `-1` an error occurred (invalid UTF-8 in the subject for a `UTF8`-mode `Pattern`, or the backtracking depth limit was reached, `README.md` "Implementation status"; no further detail is available through the return value itself in this build, only through the fact that it is negative rather than `0`).
|
||||
|
||||
### 3.1 `re_compile`
|
||||
|
||||
```c
|
||||
Pattern *re_compile(const char *pattern, size_t len, int flags, PatternError *err);
|
||||
```
|
||||
|
||||
Compiles `pattern` (`len` bytes, need not be `NUL`-terminated) under `flags` into a fresh, independent `Pattern`, matching `re.compile`. Returns `NULL` and fills `err` (if non-`NULL`) on a syntax error, an unsupported construct (Section 5 of this document lists all of them), or a lookbehind that is not fixed-width. Never consults or populates the module level cache (3.12); always allocates a new `Pattern`, freed only by `Pattern_free`.
|
||||
|
||||
### 3.2-3.4 `Pattern_match` / `Pattern_fullmatch` / `Pattern_search`
|
||||
|
||||
```c
|
||||
int Pattern_match(Pattern *self, Input *string, int64_t pos, int64_t endpos, Match *out);
|
||||
int Pattern_fullmatch(Pattern *self, Input *string, int64_t pos, int64_t endpos, Match *out);
|
||||
int Pattern_search(Pattern *self, Input *string, int64_t pos, int64_t endpos, Match *out);
|
||||
```
|
||||
|
||||
Mirror `re.Pattern.match`/`.fullmatch`/`.search` exactly, including the `pos`/`endpos` parameters (pass `0` and `-1` for CPython's own defaults, "search the whole string"). `pos`/`endpos` are in the pattern's native unit (code point index for `UTF8` mode, byte offset otherwise, Section 1.2); a negative `endpos` means "to the end". `out` must point to a zero-initialized `Match` (or one already released with `Match_free`); on a `0` or `-1` return it is left untouched.
|
||||
|
||||
- `match`: anchored at `pos`, need not reach `endpos`.
|
||||
- `fullmatch`: anchored at `pos`, must also reach exactly `endpos`.
|
||||
- `search`: tries every start position from `pos` to `endpos` inclusive, left to right, and reports the first that admits any match (ordinary backtracking priority decides which match that is at that position, `concept.md` 2.5).
|
||||
|
||||
### 3.5-3.6 `Pattern_finditer` / `Pattern_findall`
|
||||
|
||||
```c
|
||||
int Pattern_finditer(Pattern *self, Input *string, int64_t pos, int64_t endpos, MatchIterCb cb, void *ctx);
|
||||
int Pattern_findall(Pattern *self, Input *string, int64_t pos, int64_t endpos, MatchIterCb cb, void *ctx);
|
||||
```
|
||||
|
||||
`Pattern_findall` is defined as a call to `Pattern_finditer` with the same arguments; both invoke `cb(ctx, m)` once per non-overlapping match, left to right, applying CPython's own empty-match rule (an empty match advances one unit and is not merged with an adjacent non-empty match at the same start, `concept.md` 3). Returns the number of matches found, or `-1` on error. This build does not collapse a no-groups match down to "just the matched string" or a multi-group match to a Python tuple the way `re.findall` does at the Python level; the callback always receives a full `Match`, from which the caller reads whatever it needs via `Match_group`. This is a deliberate simplification: `findall`'s string/tuple collapsing is a Python-object-model convenience with no C equivalent to collapse into, so this build gives the caller the same, uniform `Match`-based access `finditer` does, and the two functions exist separately only for name-for-name parity with `re.findall`/`re.finditer`.
|
||||
|
||||
### 3.7 `Pattern_split`
|
||||
|
||||
```c
|
||||
int Pattern_split(Pattern *self, Input *string, int maxsplit, MatchIterCb cb, void *ctx);
|
||||
```
|
||||
|
||||
`cb` is invoked once per element of the list `re.split()` would return, **in order**: this is the entire contract, and it is unambiguous by construction (earlier drafts of this function called `cb` twice per match with the caller left to infer which call meant what; that design was replaced before release specifically because it was ambiguous). Each element's text is read via `Match_group(m, NULL, 0, &out, &outlen)`; a `0` return from that call means this element is Python's `None` (an unparticipated capturing group between two matches), matching how an unparticipated group reports on any other `Match`. `maxsplit` matches `re.split`'s parameter (`0` means unlimited). Returns the number of matches that were split on (not the number of list elements), or `-1` on error.
|
||||
|
||||
### 3.8-3.9 `Pattern_sub` / `Pattern_subn`
|
||||
|
||||
```c
|
||||
int Pattern_sub(Pattern *self, Input *string, const char *repl, MatchSubCb cb, void *ctx, int count, char **out, size_t *outlen);
|
||||
int Pattern_subn(Pattern *self, Input *string, const char *repl, MatchSubCb cb, void *ctx, int count, char **out, size_t *outlen, int *n);
|
||||
```
|
||||
|
||||
Mirror `re.Pattern.sub`/`.subn`. Exactly one of `repl` (a template string) or `cb` (a callback) must be non-`NULL`; passing both or neither is a caller error with unspecified behavior. This two-parameter shape is the direct C consequence of Python's single `repl` argument being polymorphic (string or callable) in a way C's static type system cannot express in one slot (`concept.md` 9.4, 9.0).
|
||||
|
||||
`repl` template syntax: `\g<name>`, `\g<N>`, `\N` (one or two digits), `\n`, `\t`, `\\`, and any other `\X` as the literal character `X`, matching `concept.md` Section 3.
|
||||
|
||||
`count` matches `re.sub`'s `count` parameter (`0` means unlimited; a positive `count` stops substituting after that many matches, leaving the rest of the subject, including any further matches within it, untouched, exactly as CPython leaves it).
|
||||
|
||||
`Pattern_sub` and `Pattern_subn` differ only in whether the number of substitutions actually made is reported back through `n` (mirroring `re.sub` returning just the string versus `re.subn` returning `(string, count)`); `Pattern_sub` is implemented as a call to `Pattern_subn` with a throwaway `n`.
|
||||
|
||||
`*out` is a freshly `malloc`'d, `NUL`-terminated buffer of length `*outlen`; the caller must `free` it. On `-1` (error), `*out`/`*outlen` are left untouched.
|
||||
|
||||
### 3.10 `Pattern_free`
|
||||
|
||||
```c
|
||||
void Pattern_free(Pattern *self);
|
||||
```
|
||||
|
||||
Frees a `Pattern` and everything it owns (the compiled program, the retained parse tree, `groupindex`, the copy of the pattern text). Do not call this while any `Match` produced from this `Pattern` is still in use (`Match.re`/`Match.lastgroup` borrow from it, Section 1.2); free every such `Match` with `Match_free` first, or simply free them in the reverse order they were created, which is always safe.
|
||||
|
||||
### 3.11 `Pattern_groupindex_lookup`
|
||||
|
||||
```c
|
||||
int Pattern_groupindex_lookup(Pattern *self, const char *name);
|
||||
```
|
||||
|
||||
Looks up a named group in `re.Pattern.groupindex`; returns its 1-based group number, or `-1` if no group by that name exists in this pattern. This is the only supported access to `groupindex`'s content in this build: there is no enumeration function (no way to list every name a pattern defines without already knowing what to look for), unlike Python's `groupindex`, which is a full mapping object supporting iteration and `len()`. A caller that needs every name a pattern uses must track the names it compiled the pattern with itself.
|
||||
|
||||
### 3.12-3.20 Module level functions
|
||||
|
||||
```c
|
||||
int re_match(const char *pattern, size_t len, int flags, Input *string, Match *out);
|
||||
int re_fullmatch(const char *pattern, size_t len, int flags, Input *string, Match *out);
|
||||
int re_search(const char *pattern, size_t len, int flags, Input *string, Match *out);
|
||||
int re_finditer(const char *pattern, size_t len, int flags, Input *string, MatchIterCb cb, void *ctx);
|
||||
int re_findall(const char *pattern, size_t len, int flags, Input *string, MatchIterCb cb, void *ctx);
|
||||
int re_split(const char *pattern, size_t len, int flags, Input *string, int maxsplit, MatchIterCb cb, void *ctx);
|
||||
int re_sub(const char *pattern, size_t len, int flags, Input *string, const char *repl, MatchSubCb cb, void *ctx, int count, char **out, size_t *outlen);
|
||||
int re_subn(const char *pattern, size_t len, int flags, Input *string, const char *repl, MatchSubCb cb, void *ctx, int count, char **out, size_t *outlen, int *n);
|
||||
```
|
||||
|
||||
Each compiles `pattern` through an internal cache and then calls the matching `Pattern_` function with `pos=0`, `endpos=-1` (module level `re.match`/`re.search`/etc. do not expose `pos`/`endpos` either, only the `Pattern` methods do, matching CPython exactly), exactly as CPython's own `re/__init__.py` implements `re.match` as `_compile(pattern, flags).match(string)`. The cache is keyed by `(pattern, len, flags)`, holds up to 512 entries, and is cleared entirely on overflow rather than evicting individual entries (mirroring the strategy CPython's own `re` module cache uses). A compile error inside these functions is silently reported as a `-1` return, with no `PatternError` available (Section 1.4); use `re_compile` plus a `Pattern_` function directly whenever a compile error needs to be diagnosed, or whenever the same pattern is applied more than once (idiomatic Python precompiles a pattern reused in a loop rather than calling the module level function repeatedly, and so should idiomatic use of this API, `concept.md` 9.1).
|
||||
|
||||
**Concurrency:** the cache is shared, mutable, process-wide state with no internal locking (`concept.md` 13.6). Do not call any `re_`-prefixed module level function from more than one thread without external synchronization; `Pattern_`-prefixed functions on a `Pattern` no thread is concurrently modifying (which is all of them, since nothing here mutates a compiled `Pattern`) have no such restriction.
|
||||
|
||||
### 3.21 `re_purge`
|
||||
|
||||
```c
|
||||
void re_purge(void);
|
||||
```
|
||||
|
||||
Clears the module level cache, matching `re.purge()` exactly, including that it has no effect on any `Pattern *` a caller already holds a direct reference to (only the cache entry is dropped; already-returned pointers remain valid until `Pattern_free`d).
|
||||
|
||||
### 3.22 `re_escape`
|
||||
|
||||
```c
|
||||
void re_escape(const char *in, size_t len, char **out, size_t *outlen);
|
||||
```
|
||||
|
||||
Matches `re.escape` exactly, including the narrowed escaped-character set CPython adopted in 3.7 (only characters that are actually special in a regex, plus non-ASCII bytes are passed through unescaped rather than every non-alphanumeric character as in pre-3.7 Python). `*out` is a freshly `malloc`'d buffer the caller must `free`.
|
||||
|
||||
### 3.23-3.29 `Match_` accessors
|
||||
|
||||
```c
|
||||
int Match_group(Match *self, const char *name_or_null, int index, const char **out, size_t *outlen);
|
||||
int64_t Match_start(Match *self, int group);
|
||||
int64_t Match_end(Match *self, int group);
|
||||
void Match_span(Match *self, int group, int64_t *start, int64_t *end);
|
||||
int64_t Match_start_byte(Match *self, int group);
|
||||
int64_t Match_end_byte(Match *self, int group);
|
||||
void Match_span_byte(Match *self, int group, int64_t *start, int64_t *end);
|
||||
void Match_free(Match *self);
|
||||
```
|
||||
|
||||
`Match_group`: if `name_or_null` is non-`NULL`, `index` is ignored and the group is looked up by name (via the same table `Pattern_groupindex_lookup` uses); otherwise `index` (`0` for the whole match) selects the group directly. Returns `1` and sets `*out`/`*outlen` to a borrowed pointer into the underlying `Input`'s buffer (valid as long as both the `Match` and the `Input` are) when the group matched; returns `0` and sets `*out = NULL, *outlen = 0` when the group exists but did not participate (Python's `None`); returns `-1`, leaving `*out`/`*outlen` untouched, when no such group exists at all (by index or by name).
|
||||
|
||||
`Match_start`/`Match_end`/`Match_span`: report the group's span in the pattern's native unit (code point index in `UTF8` mode, byte offset otherwise). **Deviation from CPython:** Python's `Match.start(group)`/`.end(group)` raise `IndexError` for an out-of-range group number and return `-1` only for a valid, unparticipated group; this build returns `-1` for both cases uniformly, since C has no exception to raise. A caller that must tell "no such group" apart from "this group did not participate" should use `Match_group` instead, which does distinguish them (`-1` versus `0` above).
|
||||
|
||||
`Match_start_byte`/`Match_end_byte`/`Match_span_byte`: always report a byte offset into the `Input`, regardless of encoding mode; identical to the non-`_byte` accessors in `BINARY`/`ASCII` mode, and the byte-offset translation of the same span in `UTF8` mode. Use these, not the code-point ones, whenever the result will be used to slice or seek into the raw `Input` buffer or file (`Match_group` already does this translation internally, so most callers only need these directly for cases `Match_group` does not cover, such as reporting a byte offset to an external tool).
|
||||
|
||||
`Match_free`: frees the private per-match state (capture slots and, in `UTF8` mode, the code-point-to-byte-offset table). Does **not** free the `Match` struct itself, and does not affect `Match.re`/`Match.string` (borrowed, Section 1.2). Safe to call more than once on the same `Match` (the second call is a no-op, since the first sets `slots` to `NULL`).
|
||||
|
||||
## 4. Memory ownership summary
|
||||
|
||||
| Object | Created by | Freed by | Notes |
|
||||
|---|---|---|---|
|
||||
| `Pattern *` | `re_compile` | `Pattern_free` | Never returned by the `re_`-prefixed module level functions; those only take a pattern string, they do not hand back the `Pattern` they compiled internally. |
|
||||
| `Input *` | `Input_from_buffer` / `Input_from_file` | `Input_free` | `Input_from_buffer` never owns the wrapped buffer; `Input_from_file` always owns the buffer it read. |
|
||||
| `Match` (the struct) | The caller (stack or heap) | The caller | Never allocated by the library; only its private `slots` are, released by `Match_free`. |
|
||||
| `PatternError` (the struct) | The caller (stack or heap) | The caller | Only `msg`/`pattern` are heap allocated; release them with `PatternError_free`. |
|
||||
| `*out` from `Pattern_sub`/`subn`, `re_escape`, and a `MatchSubCb`'s own `*out` | The library (or, for the callback, the callback itself) | The caller (or, for the callback's `*out`, `Pattern_sub`/`subn`, immediately after copying it) | Plain `malloc`'d buffers; `free()` them normally. |
|
||||
| Text returned by `Match_group` | Borrowed from the `Input`'s buffer | Nobody (not a separate allocation) | Valid exactly as long as the `Input` and the `Match` both are. |
|
||||
|
||||
## 5. Rejected constructs
|
||||
|
||||
These fail `re_compile` with a `PatternError` naming the construct, rather than being silently mis-parsed. See `concept.md` Section 4/13 for why each is out of scope for this build specifically (as opposed to out of scope for Python `re`, Section 6 below).
|
||||
|
||||
- Conditional groups: `(?(id)yes|no)`, `(?(name)yes|no)`.
|
||||
- Scoped inline flags: `(?i:...)`, `(?imsx-imsx:...)` (global inline flags at the very start of the pattern, `(?aiLmsux)`, are supported).
|
||||
- Named code points: `\N{NAME}`.
|
||||
- Variable-width lookbehind: `(?<=...)`/`(?<!...)` whose body is not a single, statically known width (matches CPython's own restriction, not an additional one this build adds).
|
||||
|
||||
## 6. What is not Python `re` syntax at all
|
||||
|
||||
POSIX bracket-expression syntax (`[[:alpha:]]`, `[.collating-symbol.]`, `[=equivalence-class=]`) is rejected the same way, but for a different reason: it is not part of Python `re`'s grammar in the first place (`concept.md` Section 14 is the complete comparison against POSIX `regex.h`, for a reader coming from C who might otherwise expect it).
|
||||
|
||||
## See also
|
||||
|
||||
- [`../concept.md`](../concept.md): the full design document (why a two-engine split was planned, the automata-theory argument for it, the POSIX comparison).
|
||||
- [`../README.md`](../README.md): build instructions, the `rxgrep` example, and the "Implementation status" summary this document expands on.
|
||||
@@ -0,0 +1,159 @@
|
||||
/* rxgrep - a small grep-like example program built on regexx, demonstrating
|
||||
* the library across its three data modes (Section 9.3) and both its
|
||||
* matching and substitution API (Section 9.1). See README.md for usage.
|
||||
*
|
||||
* This is example code, not part of the library; it is not held to the
|
||||
* same naming-convention rule as regexx.c/regexx.h (concept.md 9.0),
|
||||
* since it is an ordinary application, not part of the `re`-mirroring
|
||||
* surface.
|
||||
*/
|
||||
#include "regexx.h"
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
|
||||
static void usage(const char *prog) {
|
||||
fprintf(stderr,
|
||||
"usage: %s [-i] [-n] [-c] [-o] [-v] [-m MODE] PATTERN FILE\n"
|
||||
" %s [-m MODE] --sub REPL PATTERN FILE\n"
|
||||
"\n"
|
||||
" -i ignore case (IGNORECASE)\n"
|
||||
" -n prefix each match with its 1-based line number\n"
|
||||
" -c print only a count of matching lines\n"
|
||||
" -o print only the matched text, not the whole line\n"
|
||||
" -v print lines that do NOT match\n"
|
||||
" -m MODE ascii (default), utf8, or binary\n"
|
||||
" --sub REPL replace every match with REPL and print the result\n",
|
||||
prog, prog);
|
||||
}
|
||||
|
||||
static int mode_flag(const char *m) {
|
||||
if (strcmp(m, "ascii") == 0) return ASCII;
|
||||
if (strcmp(m, "utf8") == 0) return UTF8;
|
||||
if (strcmp(m, "binary") == 0) return BINARY;
|
||||
fprintf(stderr, "unknown mode '%s' (expected ascii, utf8, or binary)\n", m);
|
||||
exit(2);
|
||||
}
|
||||
|
||||
typedef struct { long lineno; int with_lineno; } MatchLinePrint;
|
||||
static void print_one_match(void *ctx, const Match *m_const) {
|
||||
MatchLinePrint *p = ctx;
|
||||
Match *m = (Match *)m_const;
|
||||
const char *g; size_t glen;
|
||||
Match_group(m, NULL, 0, &g, &glen);
|
||||
if (p->with_lineno) printf("%ld:", p->lineno);
|
||||
fwrite(g, 1, glen, stdout);
|
||||
printf("\n");
|
||||
}
|
||||
|
||||
int main(int argc, char **argv) {
|
||||
int flags = 0;
|
||||
int opt_n = 0, opt_c = 0, opt_o = 0, opt_v = 0;
|
||||
const char *sub_repl = NULL;
|
||||
int mode = ASCII;
|
||||
int i = 1;
|
||||
for (; i < argc; i++) {
|
||||
const char *a = argv[i];
|
||||
if (strcmp(a, "-m") == 0 && i + 1 < argc) { mode = mode_flag(argv[++i]); continue; }
|
||||
if (strcmp(a, "--sub") == 0 && i + 1 < argc) { sub_repl = argv[++i]; continue; }
|
||||
if (strcmp(a, "-h") == 0 || strcmp(a, "--help") == 0) { usage(argv[0]); return 0; }
|
||||
if (a[0] == '-' && a[1] != '-' && a[1] != '\0') {
|
||||
int bad = 0;
|
||||
for (const char *c = a + 1; *c; c++) {
|
||||
switch (*c) {
|
||||
case 'i': flags |= IGNORECASE; break;
|
||||
case 'n': opt_n = 1; break;
|
||||
case 'c': opt_c = 1; break;
|
||||
case 'o': opt_o = 1; break;
|
||||
case 'v': opt_v = 1; break;
|
||||
default: bad = 1; break;
|
||||
}
|
||||
}
|
||||
if (!bad) continue;
|
||||
}
|
||||
break;
|
||||
}
|
||||
if (argc - i != 2) { usage(argv[0]); return 2; }
|
||||
const char *pattern = argv[i], *path = argv[i + 1];
|
||||
|
||||
PatternError err; memset(&err, 0, sizeof err);
|
||||
Pattern *pat = re_compile(pattern, strlen(pattern), flags | mode, &err);
|
||||
if (!pat) {
|
||||
fprintf(stderr, "%s: pattern error at position %lld: %s\n", argv[0], (long long)err.pos, err.msg);
|
||||
return 2;
|
||||
}
|
||||
|
||||
if (sub_repl) {
|
||||
Input *file = Input_from_file(path, &err);
|
||||
if (!file) {
|
||||
fprintf(stderr, "%s: cannot read '%s': %s\n", argv[0], path, err.msg ? err.msg : "unknown error");
|
||||
Pattern_free(pat);
|
||||
return 2;
|
||||
}
|
||||
char *out; size_t outlen;
|
||||
int n;
|
||||
if (Pattern_subn(pat, file, sub_repl, NULL, NULL, 0, &out, &outlen, &n) < 0) {
|
||||
fprintf(stderr, "%s: matching failed (input too large/complex for this pattern, see README.md)\n", argv[0]);
|
||||
Input_free(file); Pattern_free(pat);
|
||||
return 1;
|
||||
}
|
||||
fwrite(out, 1, outlen, stdout);
|
||||
fprintf(stderr, "%d replacement(s)\n", n);
|
||||
free(out);
|
||||
Input_free(file); Pattern_free(pat);
|
||||
return n > 0 ? 0 : 1;
|
||||
}
|
||||
|
||||
/* Line oriented search: split on raw 0x0A, which is what every one
|
||||
* of the three modes (BINARY, ASCII, UTF8) agrees is a line break
|
||||
* (concept.md 7.2 notes 0x0A never appears inside a multi-byte
|
||||
* UTF-8 sequence), then search within each line's [pos, endpos).
|
||||
* Input's internals are intentionally opaque outside regexx.c
|
||||
* (concept.md 9.0), so each line is handed to the library as its
|
||||
* own small Input built with Input_from_buffer. */
|
||||
FILE *fp = fopen(path, "rb");
|
||||
if (!fp) { fprintf(stderr, "%s: cannot reopen '%s'\n", argv[0], path); return 2; }
|
||||
fseek(fp, 0, SEEK_END);
|
||||
long fsize = ftell(fp);
|
||||
rewind(fp);
|
||||
char *filebuf = malloc((size_t)fsize > 0 ? (size_t)fsize : 1);
|
||||
size_t got = fread(filebuf, 1, (size_t)fsize, fp);
|
||||
fclose(fp);
|
||||
|
||||
long lineno = 0;
|
||||
size_t line_start = 0;
|
||||
int matched_lines = 0;
|
||||
for (size_t pos = 0; pos <= got; pos++) {
|
||||
if (pos == got || filebuf[pos] == '\n') {
|
||||
lineno++;
|
||||
size_t line_len = pos - line_start;
|
||||
Input *lin = Input_from_buffer((const uint8_t *)(filebuf + line_start), line_len);
|
||||
Match m; memset(&m, 0, sizeof m);
|
||||
int r = Pattern_search(pat, lin, 0, -1, &m);
|
||||
int is_match = (r == 1);
|
||||
if (is_match) Match_free(&m);
|
||||
if (is_match != opt_v) {
|
||||
matched_lines++;
|
||||
if (!opt_c) {
|
||||
if (opt_o && is_match) {
|
||||
/* -o: every match on the line gets its own
|
||||
* output line, demonstrating Pattern_finditer. */
|
||||
MatchLinePrint ctx = { lineno, opt_n };
|
||||
Pattern_finditer(pat, lin, 0, -1, print_one_match, &ctx);
|
||||
} else {
|
||||
if (opt_n) printf("%ld:", lineno);
|
||||
fwrite(filebuf + line_start, 1, line_len, stdout);
|
||||
printf("\n");
|
||||
}
|
||||
}
|
||||
}
|
||||
Input_free(lin);
|
||||
line_start = pos + 1;
|
||||
}
|
||||
}
|
||||
if (opt_c) printf("%d\n", matched_lines);
|
||||
|
||||
free(filebuf);
|
||||
Pattern_free(pat);
|
||||
return matched_lines > 0 ? 0 : 1;
|
||||
}
|
||||
@@ -0,0 +1,132 @@
|
||||
/* regexx.h - public API for regexx, a single-file C regex interpreter
|
||||
* with Python `re` semantics. See concept.md for the design rationale
|
||||
* and Section 9 in particular for the naming convention this header
|
||||
* follows: every identifier with a direct Python `re` counterpart uses
|
||||
* that counterpart's exact spelling.
|
||||
*
|
||||
* Implementation status relative to concept.md: this is the v1
|
||||
* implementation. It provides the full public API and pattern syntax
|
||||
* described below, executed by a single recursive backtracking engine
|
||||
* (concept.md Section 7.3) operating over a fully materialized copy of
|
||||
* the input. The streaming, bounded-memory regular engine of Section
|
||||
* 7.2 is not implemented yet; see README.md "Implementation status"
|
||||
* for the complete, itemized list of what v1 does and does not cover.
|
||||
*/
|
||||
#ifndef REGEXX_H
|
||||
#define REGEXX_H
|
||||
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
|
||||
#ifdef __cplusplus
|
||||
extern "C" {
|
||||
#endif
|
||||
|
||||
/* ---- Flags (re.compile flags; bitwise OR, exact Python names) ---- */
|
||||
#define IGNORECASE 0x0001
|
||||
#define MULTILINE 0x0002
|
||||
#define DOTALL 0x0004
|
||||
#define VERBOSE 0x0008
|
||||
#define ASCII 0x0010
|
||||
#define UNICODE 0x0020 /* default for UTF8 mode; explicit for clarity */
|
||||
#define LOCALE 0x0040
|
||||
#define DEBUG 0x0080
|
||||
|
||||
/* Encoding mode, not a Python flag (Section 9.3): exactly one required. */
|
||||
#define BINARY 0x0100
|
||||
#define UTF8 0x0200
|
||||
/* ASCII (0x0010) doubles as the ASCII *encoding* mode when neither
|
||||
* BINARY nor UTF8 is given; see regexx.c encoding-mode resolution. */
|
||||
|
||||
/* ---- Types ---- */
|
||||
|
||||
typedef struct Pattern Pattern; /* re.Pattern */
|
||||
typedef struct Match Match; /* re.Match */
|
||||
typedef struct Input Input; /* no Python counterpart, see concept.md 9.0/9.2 */
|
||||
|
||||
typedef void (*MatchIterCb)(void *ctx, const Match *m);
|
||||
typedef void (*MatchSubCb)(void *ctx, const Match *m, char **out, size_t *outlen);
|
||||
|
||||
typedef struct PatternError PatternError; /* re.error / re.PatternError */
|
||||
struct PatternError {
|
||||
const char *msg; /* re.error.msg */
|
||||
const char *pattern; /* re.error.pattern */
|
||||
int64_t pos; /* re.error.pos */
|
||||
int64_t lineno; /* re.error.lineno */
|
||||
int64_t colno; /* re.error.colno */
|
||||
};
|
||||
/* msg and pattern are heap allocated by whichever call filled this
|
||||
* struct in; call PatternError_free once you are done reading it
|
||||
* (safe to call on a zero-initialized or already-freed PatternError). */
|
||||
void PatternError_free(PatternError *err);
|
||||
|
||||
struct Pattern {
|
||||
const char *pattern; /* re.Pattern.pattern */
|
||||
int flags; /* re.Pattern.flags */
|
||||
int groups; /* re.Pattern.groups */
|
||||
void *groupindex; /* re.Pattern.groupindex, name -> group number */
|
||||
void *program; /* compiled bytecode, private */
|
||||
};
|
||||
|
||||
struct Match {
|
||||
Pattern *re; /* re.Match.re */
|
||||
Input *string; /* re.Match.string */
|
||||
int64_t pos, endpos; /* re.Match.pos, re.Match.endpos */
|
||||
int lastindex; /* re.Match.lastindex */
|
||||
const char *lastgroup; /* re.Match.lastgroup */
|
||||
void *slots; /* private */
|
||||
};
|
||||
|
||||
/* ---- Input construction (no Python counterpart) ---- */
|
||||
Input *Input_from_buffer(const uint8_t *buf, size_t len); /* does not copy or take ownership */
|
||||
Input *Input_from_file(const char *path, PatternError *err); /* seekable or not (pipes, "-" via /dev/stdin, etc. all work) */
|
||||
void Input_free(Input *in);
|
||||
|
||||
/* ---- Pattern methods (bound to an already compiled Pattern) ---- */
|
||||
int Pattern_match(Pattern *self, Input *string, int64_t pos, int64_t endpos, Match *out);
|
||||
int Pattern_fullmatch(Pattern *self, Input *string, int64_t pos, int64_t endpos, Match *out);
|
||||
int Pattern_search(Pattern *self, Input *string, int64_t pos, int64_t endpos, Match *out);
|
||||
int Pattern_finditer(Pattern *self, Input *string, int64_t pos, int64_t endpos, MatchIterCb cb, void *ctx);
|
||||
int Pattern_findall(Pattern *self, Input *string, int64_t pos, int64_t endpos, MatchIterCb cb, void *ctx);
|
||||
/* cb fires once per element of the list re.split() would return, in
|
||||
* order; read each element's text via Match_group(m, NULL, 0, ...). */
|
||||
int Pattern_split(Pattern *self, Input *string, int maxsplit, MatchIterCb cb, void *ctx);
|
||||
int Pattern_sub(Pattern *self, Input *string, const char *repl, MatchSubCb cb, void *ctx, int count, char **out, size_t *outlen);
|
||||
int Pattern_subn(Pattern *self, Input *string, const char *repl, MatchSubCb cb, void *ctx, int count, char **out, size_t *outlen, int *n);
|
||||
void Pattern_free(Pattern *self);
|
||||
|
||||
/* Looks up a named group in re.Pattern.groupindex; returns its 1-based
|
||||
* group number, or -1 if no group by that name exists. There is no
|
||||
* enumeration function for groupindex as a whole in this build: a
|
||||
* caller that needs every name must track the names it used to build
|
||||
* the pattern itself, since Pattern->groupindex is otherwise opaque. */
|
||||
int Pattern_groupindex_lookup(Pattern *self, const char *name);
|
||||
|
||||
/* ---- Match accessors ---- */
|
||||
int Match_group(Match *self, const char *name_or_null, int index, const char **out, size_t *outlen);
|
||||
int64_t Match_start(Match *self, int group);
|
||||
int64_t Match_end(Match *self, int group);
|
||||
void Match_span(Match *self, int group, int64_t *start, int64_t *end);
|
||||
int64_t Match_start_byte(Match *self, int group);
|
||||
int64_t Match_end_byte(Match *self, int group);
|
||||
void Match_span_byte(Match *self, int group, int64_t *start, int64_t *end);
|
||||
void Match_free(Match *self);
|
||||
|
||||
/* ---- Module level functions ---- */
|
||||
Pattern *re_compile(const char *pattern, size_t len, int flags, PatternError *err);
|
||||
int re_match(const char *pattern, size_t len, int flags, Input *string, Match *out);
|
||||
int re_fullmatch(const char *pattern, size_t len, int flags, Input *string, Match *out);
|
||||
int re_search(const char *pattern, size_t len, int flags, Input *string, Match *out);
|
||||
int re_finditer(const char *pattern, size_t len, int flags, Input *string, MatchIterCb cb, void *ctx);
|
||||
int re_findall(const char *pattern, size_t len, int flags, Input *string, MatchIterCb cb, void *ctx);
|
||||
int re_split(const char *pattern, size_t len, int flags, Input *string, int maxsplit, MatchIterCb cb, void *ctx);
|
||||
int re_sub(const char *pattern, size_t len, int flags, Input *string, const char *repl, MatchSubCb cb, void *ctx, int count, char **out, size_t *outlen);
|
||||
int re_subn(const char *pattern, size_t len, int flags, Input *string, const char *repl, MatchSubCb cb, void *ctx, int count, char **out, size_t *outlen, int *n);
|
||||
void re_escape(const char *in, size_t len, char **out, size_t *outlen);
|
||||
void re_purge(void);
|
||||
|
||||
#ifdef __cplusplus
|
||||
}
|
||||
#endif
|
||||
|
||||
#endif /* REGEXX_H */
|
||||
+129
@@ -0,0 +1,129 @@
|
||||
"""Test cases for regexx, checked against CPython's own `re` module
|
||||
(concept.md Section 11). Each case records a pattern/subject pair and
|
||||
an operation; gen.py computes the expected result with Python's `re`
|
||||
and emits a C assertion that checks regexx against it.
|
||||
"""
|
||||
|
||||
CASES = []
|
||||
|
||||
|
||||
def add(op, pattern, subject, mode="utf8", flags=(), **kw):
|
||||
CASES.append(dict(op=op, pattern=pattern, subject=subject, mode=mode, flags=tuple(flags), **kw))
|
||||
|
||||
|
||||
# ---- literals, concatenation, anchors --------------------------------
|
||||
add("search", r"abc", "xxabcxx")
|
||||
add("fullmatch", r"abc", "abc")
|
||||
add("fullmatch", r"abc", "xabc")
|
||||
add("match", r"abc", "abcdef")
|
||||
add("match", r"abc", "xabcdef")
|
||||
add("search", r"^abc$", "abc")
|
||||
add("search", r"^abc$", "xabc")
|
||||
add("search", r"^abc", "abc\nabc", flags=["MULTILINE"])
|
||||
add("finditer", r"^abc", "abc\nabc", flags=["MULTILINE"])
|
||||
add("search", r"abc$", "xxabc\nabc", flags=["MULTILINE"])
|
||||
add("search", r"\Aabc", "abc")
|
||||
add("search", r"abc\Z", "xxabc")
|
||||
add("search", r"abc\Z", "xxabc\n")
|
||||
|
||||
# ---- character classes -------------------------------------------------
|
||||
add("search", r"[a-z]+", "ABCdefGHI")
|
||||
add("search", r"[^a-z]+", "abcDEFghi")
|
||||
add("search", r"[a-z]+", "ABC", flags=["IGNORECASE"])
|
||||
add("finditer", r"[abc]", "xaybzc")
|
||||
add("search", r"[]a]", "]a") # ']' literal as first class member
|
||||
add("search", r"[a\-z]", "-")
|
||||
|
||||
# ---- shorthand classes ---------------------------------------------------
|
||||
add("finditer", r"\d+", "ab123cd45")
|
||||
add("finditer", r"\D+", "ab123cd45")
|
||||
add("search", r"\s+", "a b")
|
||||
add("search", r"\S+", " ab ")
|
||||
add("finditer", r"\w+", "hi there, friend!")
|
||||
add("finditer", r"\W+", "hi there, friend!")
|
||||
add("search", r"[\d\S]", " ")
|
||||
add("search", r"[\d\S]", "5")
|
||||
|
||||
# ---- word boundaries ------------------------------------------------------
|
||||
add("finditer", r"\bcat\b", "cat catalog cat")
|
||||
add("finditer", r"\Bcat\B", "concatenate")
|
||||
|
||||
# ---- quantifiers ------------------------------------------------------------
|
||||
add("search", r"a*", "aaab")
|
||||
add("search", r"a+", "baaab")
|
||||
add("search", r"a?b", "b")
|
||||
add("search", r"a{2,4}", "aaaaa")
|
||||
add("search", r"a{2,4}?", "aaaaa")
|
||||
add("search", r"a{3}", "aa")
|
||||
add("search", r"a.*b", "axxxbxxxb")
|
||||
add("search", r"a.*?b", "axxxbxxxb")
|
||||
add("finditer", r"a*", "baaab") # exercises empty-match advancement
|
||||
|
||||
# ---- alternation and groups -------------------------------------------------
|
||||
add("search", r"(foo|bar)baz", "barbaz")
|
||||
add("search", r"(?:foo|bar)baz", "foobaz")
|
||||
add("finditer", r"(a)(b)?", "ab a")
|
||||
add("search", r"(?P<year>\d{4})-(?P<month>\d{2})", "2024-09")
|
||||
|
||||
# ---- backreferences ---------------------------------------------------------
|
||||
add("search", r"(\w+) \1", "hello hello")
|
||||
add("search", r"(\w+) \1", "hello world")
|
||||
add("search", r"(?P<w>\w+) (?P=w)", "abc abc")
|
||||
add("search", r"(\w)\1", "aa")
|
||||
add("search", r"(\w)\1", "ab")
|
||||
|
||||
# ---- lookaround --------------------------------------------------------------
|
||||
add("search", r"foo(?=bar)", "foobar")
|
||||
add("search", r"foo(?=bar)", "foobaz")
|
||||
add("search", r"foo(?!bar)", "foobaz")
|
||||
add("search", r"foo(?!bar)", "foobar")
|
||||
add("search", r"(?<=foo)bar", "foobar")
|
||||
add("search", r"(?<=foo)bar", "xxxbar")
|
||||
add("search", r"(?<!foo)bar", "xxxbar")
|
||||
add("search", r"(?<!foo)bar", "foobar")
|
||||
add("finditer", r"(?<=\d)(?=(\d{3})+(?!\d))", "1234567") # thousands-separator style
|
||||
|
||||
# ---- atomic groups / possessive quantifiers -----------------------------------
|
||||
add("search", r"a(?>bc|b)c", "abc")
|
||||
add("search", r"a(bc|b)c", "abc")
|
||||
add("search", r"a*+a", "aaaa")
|
||||
add("search", r"a*a", "aaaa")
|
||||
|
||||
# ---- DOTALL / VERBOSE ----------------------------------------------------------
|
||||
add("search", r"a.b", "a\nb")
|
||||
add("search", r"a.b", "a\nb", flags=["DOTALL"])
|
||||
add("search", r"""
|
||||
\d+ # the integer part
|
||||
\. # the dot
|
||||
\d+ # the fractional part
|
||||
""", "pi is 3.14 roughly", flags=["VERBOSE"])
|
||||
|
||||
# ---- escapes ---------------------------------------------------------------------
|
||||
add("search", r"a\tb", "a\tb")
|
||||
add("search", r"\x41\x42", "AB")
|
||||
add("search", r"é", "café")
|
||||
|
||||
# ---- UTF-8 mode ------------------------------------------------------------------
|
||||
add("search", r"\w+", "café au lait")
|
||||
add("finditer", r"\w+", "café 中文 word")
|
||||
add("search", r"[à-ÿ]+", "éèê")
|
||||
|
||||
# ---- sub / subn -------------------------------------------------------------------
|
||||
add("sub", r"a", "banana", repl="o", count=0)
|
||||
add("sub", r"a", "banana", repl="o", count=2)
|
||||
add("sub", r"(\w+)@(\w+)", "user@host", repl=r"\2@\1")
|
||||
add("sub", r"(?P<user>\w+)@(?P<host>\w+)", "user@host", repl=r"\g<host>@\g<user>")
|
||||
add("sub", r"\s+", "a b c", repl=" ")
|
||||
|
||||
# ---- split ------------------------------------------------------------------------
|
||||
add("split", r"[,;]\s*", "a, b;c , d")
|
||||
add("split", r"(,)", "a,b,c")
|
||||
add("split", r"\s*", "abc")
|
||||
add("split", r",", "a,b,c", maxsplit=1)
|
||||
|
||||
# ---- ASCII vs UTF8 mode differences for \w ----------------------------------------
|
||||
add("search", r"\w+", "café", mode="ascii")
|
||||
|
||||
# ---- BINARY mode --------------------------------------------------------------------
|
||||
add("search", r"a.c", "a\x00c", mode="binary")
|
||||
add("finditer", r"\d+", "12ab34", mode="binary")
|
||||
+181
@@ -0,0 +1,181 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Generate tests/generated_tests.c from tests/cases.py, using CPython's
|
||||
own `re` module as ground truth (concept.md Section 11)."""
|
||||
import re
|
||||
import sys
|
||||
import os
|
||||
|
||||
sys.path.insert(0, os.path.dirname(__file__))
|
||||
from cases import CASES # noqa: E402
|
||||
|
||||
FLAGMAP = {
|
||||
"IGNORECASE": re.IGNORECASE,
|
||||
"MULTILINE": re.MULTILINE,
|
||||
"DOTALL": re.DOTALL,
|
||||
"VERBOSE": re.VERBOSE,
|
||||
"LOCALE": re.LOCALE,
|
||||
"ASCII": re.ASCII,
|
||||
}
|
||||
|
||||
|
||||
def c_str(b):
|
||||
if isinstance(b, str):
|
||||
b = b.encode("utf-8")
|
||||
out = ['"']
|
||||
for byte in b:
|
||||
c = chr(byte)
|
||||
if c == '"':
|
||||
out.append('\\"')
|
||||
elif c == "\\":
|
||||
out.append("\\\\")
|
||||
elif 32 <= byte < 127:
|
||||
out.append(c)
|
||||
else:
|
||||
out.append("\\%03o" % byte)
|
||||
out.append('"')
|
||||
return "".join(out)
|
||||
|
||||
|
||||
def py_flags(case):
|
||||
f = 0
|
||||
if case["mode"] == "ascii":
|
||||
f |= re.ASCII
|
||||
for name in case["flags"]:
|
||||
f |= FLAGMAP[name]
|
||||
return f
|
||||
|
||||
|
||||
def c_flags(case):
|
||||
names = []
|
||||
if case["mode"] == "ascii":
|
||||
names.append("ASCII")
|
||||
elif case["mode"] == "utf8":
|
||||
names.append("UTF8")
|
||||
elif case["mode"] == "binary":
|
||||
names.append("BINARY")
|
||||
for name in case["flags"]:
|
||||
names.append(name)
|
||||
return "|".join(names) if names else "0"
|
||||
|
||||
|
||||
def encode(case, s):
|
||||
if case["mode"] == "binary":
|
||||
return s.encode("utf-8")
|
||||
return s
|
||||
|
||||
|
||||
def blen(case, s):
|
||||
"""Byte length of the encoded subject, as the C API always wants a
|
||||
byte count (UTF8 mode still stores a byte buffer; only Match_start
|
||||
reports code points), never a Python character count."""
|
||||
e = encode(case, s)
|
||||
return len(e) if isinstance(e, bytes) else len(e.encode("utf-8"))
|
||||
|
||||
|
||||
def desc(case, i):
|
||||
return "#%d %s /%s/ on %r" % (i, case["op"], case["pattern"], case["subject"])
|
||||
|
||||
|
||||
def gen_search_family(case, i, out):
|
||||
kind = {"search": 0, "match": 1, "fullmatch": 2}[case["op"]]
|
||||
flags = py_flags(case)
|
||||
subj = encode(case, case["subject"])
|
||||
pat = encode(case, case["pattern"])
|
||||
compiled = re.compile(pat, flags)
|
||||
fn = {"search": compiled.search, "match": compiled.match, "fullmatch": compiled.fullmatch}[case["op"]]
|
||||
m = fn(subj)
|
||||
d = desc(case, i)
|
||||
if m is None:
|
||||
out.append(' tc_search_family(%s, %s, %s, %s, %d, %d, 0, 0, NULL);' % (
|
||||
c_str(d), c_str(case["pattern"]), c_flags(case), c_str(case["subject"]), blen(case, case["subject"]), kind))
|
||||
return
|
||||
spans = [m.start(0), m.end(0)]
|
||||
for g in range(1, compiled.groups + 1):
|
||||
try:
|
||||
spans += [m.start(g), m.end(g)]
|
||||
except IndexError:
|
||||
spans += [-1, -1]
|
||||
n_spans = compiled.groups + 1
|
||||
arr = ",".join(str(v) for v in spans)
|
||||
out.append(' { static const long long sp[] = {%s}; tc_search_family(%s, %s, %s, %s, %d, %d, 1, %d, sp); }' % (
|
||||
arr, c_str(d), c_str(case["pattern"]), c_flags(case), c_str(case["subject"]), blen(case, case["subject"]), kind, n_spans))
|
||||
|
||||
|
||||
def gen_finditer(case, i, out):
|
||||
flags = py_flags(case)
|
||||
subj = encode(case, case["subject"])
|
||||
pat = encode(case, case["pattern"])
|
||||
ms = list(re.finditer(pat, subj, flags))
|
||||
spans = []
|
||||
for m in ms:
|
||||
spans += [m.start(0), m.end(0)]
|
||||
d = desc(case, i)
|
||||
if not spans:
|
||||
out.append(' tc_finditer(%s, %s, %s, %s, %d, 0, NULL);' % (
|
||||
c_str(d), c_str(case["pattern"]), c_flags(case), c_str(case["subject"]), blen(case, case["subject"])))
|
||||
return
|
||||
arr = ",".join(str(v) for v in spans)
|
||||
out.append(' { static const long long sp[] = {%s}; tc_finditer(%s, %s, %s, %s, %d, %d, sp); }' % (
|
||||
arr, c_str(d), c_str(case["pattern"]), c_flags(case), c_str(case["subject"]), blen(case, case["subject"]), len(ms)))
|
||||
|
||||
|
||||
def gen_sub(case, i, out):
|
||||
flags = py_flags(case)
|
||||
subj = encode(case, case["subject"])
|
||||
pat = encode(case, case["pattern"])
|
||||
repl = encode(case, case["repl"])
|
||||
count = case.get("count", 0)
|
||||
result, n = re.subn(pat, repl, subj, count=count, flags=flags)
|
||||
d = desc(case, i)
|
||||
result_str = result.decode("utf-8") if isinstance(result, bytes) else result
|
||||
out.append(' tc_sub(%s, %s, %s, %s, %d, %s, %d, %s, %d);' % (
|
||||
c_str(d), c_str(case["pattern"]), c_flags(case), c_str(case["subject"]), blen(case, case["subject"]),
|
||||
c_str(case["repl"]), count, c_str(result_str), n))
|
||||
|
||||
|
||||
def gen_split(case, i, out):
|
||||
flags = py_flags(case)
|
||||
subj = encode(case, case["subject"])
|
||||
pat = encode(case, case["pattern"])
|
||||
maxsplit = case.get("maxsplit", 0)
|
||||
items = re.split(pat, subj, maxsplit=maxsplit, flags=flags)
|
||||
d = desc(case, i)
|
||||
parts = []
|
||||
for it in items:
|
||||
if it is None:
|
||||
parts.append("NULL")
|
||||
else:
|
||||
s = it.decode("utf-8") if isinstance(it, bytes) else it
|
||||
parts.append(c_str(s))
|
||||
arr = ",".join(parts) if parts else "0"
|
||||
out.append(' { static const char *const items[] = {%s}; tc_split(%s, %s, %s, %s, %d, %d, %d, items); }' % (
|
||||
arr, c_str(d), c_str(case["pattern"]), c_flags(case), c_str(case["subject"]), blen(case, case["subject"]),
|
||||
maxsplit, len(items)))
|
||||
|
||||
|
||||
def main():
|
||||
out = []
|
||||
out.append('/* GENERATED by tests/gen.py from tests/cases.py. Do not edit by hand. */')
|
||||
out.append('#include "harness.h"')
|
||||
out.append('void run_generated_tests(void) {')
|
||||
for i, case in enumerate(CASES):
|
||||
op = case["op"]
|
||||
if op in ("search", "match", "fullmatch"):
|
||||
gen_search_family(case, i, out)
|
||||
elif op == "finditer":
|
||||
gen_finditer(case, i, out)
|
||||
elif op == "sub":
|
||||
gen_sub(case, i, out)
|
||||
elif op == "split":
|
||||
gen_split(case, i, out)
|
||||
else:
|
||||
raise ValueError("unknown op %r" % op)
|
||||
out.append('}')
|
||||
dest = os.path.join(os.path.dirname(__file__), "generated_tests.c")
|
||||
with open(dest, "w") as f:
|
||||
f.write("\n".join(out) + "\n")
|
||||
print("wrote %s (%d cases)" % (dest, len(CASES)))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
+156
@@ -0,0 +1,156 @@
|
||||
#include "harness.h"
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
|
||||
int tc_pass = 0, tc_fail = 0;
|
||||
|
||||
void tc_search_family(const char *desc, const char *pat, int flags,
|
||||
const char *subj, size_t subjlen, int kind,
|
||||
int exp_matched, int n_spans, const long long *exp_spans) {
|
||||
PatternError err; memset(&err, 0, sizeof err);
|
||||
Pattern *p = re_compile(pat, strlen(pat), flags, &err);
|
||||
if (!p) {
|
||||
printf("FAIL [%s]: compile error: %s\n", desc, err.msg ? err.msg : "?");
|
||||
tc_fail++;
|
||||
return;
|
||||
}
|
||||
Input *in = Input_from_buffer((const uint8_t *)subj, subjlen);
|
||||
Match m; memset(&m, 0, sizeof m);
|
||||
int r;
|
||||
if (kind == 0) r = Pattern_search(p, in, 0, -1, &m);
|
||||
else if (kind == 1) r = Pattern_match(p, in, 0, -1, &m);
|
||||
else r = Pattern_fullmatch(p, in, 0, -1, &m);
|
||||
|
||||
if (r < 0) {
|
||||
printf("FAIL [%s]: internal error (depth exceeded or bad UTF-8)\n", desc);
|
||||
tc_fail++;
|
||||
} else if ((r == 1) != (exp_matched != 0)) {
|
||||
printf("FAIL [%s]: matched=%d expected=%d\n", desc, r, exp_matched);
|
||||
tc_fail++;
|
||||
} else if (r == 1) {
|
||||
int ok = 1;
|
||||
for (int g = 0; g < n_spans; g++) {
|
||||
long long es = exp_spans[2 * g], ee = exp_spans[2 * g + 1];
|
||||
long long as = Match_start(&m, g), ae = Match_end(&m, g);
|
||||
if (as != es || ae != ee) {
|
||||
printf("FAIL [%s]: group %d span (%lld,%lld) expected (%lld,%lld)\n", desc, g, as, ae, es, ee);
|
||||
ok = 0;
|
||||
}
|
||||
}
|
||||
if (ok) tc_pass++; else tc_fail++;
|
||||
Match_free(&m);
|
||||
} else {
|
||||
tc_pass++;
|
||||
}
|
||||
Input_free(in);
|
||||
Pattern_free(p);
|
||||
}
|
||||
|
||||
typedef struct { long long *spans; int n, cap; } SpanList;
|
||||
static void span_cb(void *ctx, const Match *m_const) {
|
||||
SpanList *sl = ctx;
|
||||
Match *m = (Match *)m_const;
|
||||
if (sl->n == sl->cap) { sl->cap = sl->cap ? sl->cap * 2 : 8; sl->spans = realloc(sl->spans, (size_t)sl->cap * 2 * sizeof(long long)); }
|
||||
sl->spans[2 * sl->n] = Match_start(m, 0);
|
||||
sl->spans[2 * sl->n + 1] = Match_end(m, 0);
|
||||
sl->n++;
|
||||
}
|
||||
|
||||
void tc_finditer(const char *desc, const char *pat, int flags,
|
||||
const char *subj, size_t subjlen,
|
||||
int exp_n_matches, const long long *exp_spans) {
|
||||
PatternError err; memset(&err, 0, sizeof err);
|
||||
Pattern *p = re_compile(pat, strlen(pat), flags, &err);
|
||||
if (!p) { printf("FAIL [%s]: compile error: %s\n", desc, err.msg ? err.msg : "?"); tc_fail++; return; }
|
||||
Input *in = Input_from_buffer((const uint8_t *)subj, subjlen);
|
||||
SpanList sl; memset(&sl, 0, sizeof sl);
|
||||
int total = Pattern_finditer(p, in, 0, -1, span_cb, &sl);
|
||||
if (total < 0) { printf("FAIL [%s]: internal error\n", desc); tc_fail++; }
|
||||
else if (sl.n != exp_n_matches) { printf("FAIL [%s]: %d matches, expected %d\n", desc, sl.n, exp_n_matches); tc_fail++; }
|
||||
else {
|
||||
int ok = 1;
|
||||
for (int i = 0; i < sl.n; i++)
|
||||
if (sl.spans[2 * i] != exp_spans[2 * i] || sl.spans[2 * i + 1] != exp_spans[2 * i + 1]) {
|
||||
printf("FAIL [%s]: match %d span (%lld,%lld) expected (%lld,%lld)\n", desc, i,
|
||||
sl.spans[2 * i], sl.spans[2 * i + 1], exp_spans[2 * i], exp_spans[2 * i + 1]);
|
||||
ok = 0;
|
||||
}
|
||||
if (ok) tc_pass++; else tc_fail++;
|
||||
}
|
||||
free(sl.spans);
|
||||
Input_free(in);
|
||||
Pattern_free(p);
|
||||
}
|
||||
|
||||
void tc_sub(const char *desc, const char *pat, int flags,
|
||||
const char *subj, size_t subjlen, const char *repl, int count,
|
||||
const char *exp_result, int exp_n) {
|
||||
PatternError err; memset(&err, 0, sizeof err);
|
||||
Pattern *p = re_compile(pat, strlen(pat), flags, &err);
|
||||
if (!p) { printf("FAIL [%s]: compile error: %s\n", desc, err.msg ? err.msg : "?"); tc_fail++; return; }
|
||||
Input *in = Input_from_buffer((const uint8_t *)subj, subjlen);
|
||||
char *out = NULL; size_t outlen = 0; int n = 0;
|
||||
int r = Pattern_subn(p, in, repl, NULL, NULL, count, &out, &outlen, &n);
|
||||
if (r < 0) { printf("FAIL [%s]: internal error\n", desc); tc_fail++; }
|
||||
else {
|
||||
size_t explen = strlen(exp_result);
|
||||
if (outlen != explen || memcmp(out, exp_result, explen) != 0 || n != exp_n) {
|
||||
printf("FAIL [%s]: sub result '%.*s' (n=%d) expected '%s' (n=%d)\n", desc, (int)outlen, out, n, exp_result, exp_n);
|
||||
tc_fail++;
|
||||
} else tc_pass++;
|
||||
}
|
||||
free(out);
|
||||
Input_free(in);
|
||||
Pattern_free(p);
|
||||
}
|
||||
|
||||
typedef struct { int present; const char *ptr; size_t len; } SplitItem;
|
||||
typedef struct { SplitItem *items; int n, cap; } SplitList;
|
||||
static void split_cb(void *ctx, const Match *m_const) {
|
||||
SplitList *sl = ctx;
|
||||
Match *m = (Match *)m_const;
|
||||
if (sl->n == sl->cap) { sl->cap = sl->cap ? sl->cap * 2 : 8; sl->items = realloc(sl->items, (size_t)sl->cap * sizeof(SplitItem)); }
|
||||
const char *out; size_t outlen;
|
||||
int r = Match_group(m, NULL, 0, &out, &outlen);
|
||||
sl->items[sl->n].present = (r == 1);
|
||||
sl->items[sl->n].ptr = out; sl->items[sl->n].len = outlen;
|
||||
sl->n++;
|
||||
}
|
||||
|
||||
void tc_split(const char *desc, const char *pat, int flags,
|
||||
const char *subj, size_t subjlen, int maxsplit,
|
||||
int n_items, const char *const *texts) {
|
||||
PatternError err; memset(&err, 0, sizeof err);
|
||||
Pattern *p = re_compile(pat, strlen(pat), flags, &err);
|
||||
if (!p) { printf("FAIL [%s]: compile error: %s\n", desc, err.msg ? err.msg : "?"); tc_fail++; return; }
|
||||
Input *in = Input_from_buffer((const uint8_t *)subj, subjlen);
|
||||
SplitList sl; memset(&sl, 0, sizeof sl);
|
||||
int r = Pattern_split(p, in, maxsplit, split_cb, &sl);
|
||||
if (r < 0) { printf("FAIL [%s]: internal error\n", desc); tc_fail++; }
|
||||
else if (sl.n != n_items) { printf("FAIL [%s]: %d items, expected %d\n", desc, sl.n, n_items); tc_fail++; }
|
||||
else {
|
||||
int ok = 1;
|
||||
for (int i = 0; i < sl.n; i++) {
|
||||
int exp_none = (texts[i] == NULL);
|
||||
if (exp_none && sl.items[i].present) { printf("FAIL [%s]: item %d expected None\n", desc, i); ok = 0; }
|
||||
else if (!exp_none) {
|
||||
size_t explen = strlen(texts[i]);
|
||||
if (!sl.items[i].present || sl.items[i].len != explen || memcmp(sl.items[i].ptr, texts[i], explen) != 0) {
|
||||
printf("FAIL [%s]: item %d = '%.*s' expected '%s'\n", desc, i,
|
||||
sl.items[i].present ? (int)sl.items[i].len : 4,
|
||||
sl.items[i].present ? sl.items[i].ptr : "NONE", texts[i]);
|
||||
ok = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
if (ok) tc_pass++; else tc_fail++;
|
||||
}
|
||||
free(sl.items);
|
||||
Input_free(in);
|
||||
Pattern_free(p);
|
||||
}
|
||||
|
||||
void harness_report(void) {
|
||||
printf("\n%d passed, %d failed\n", tc_pass, tc_fail);
|
||||
}
|
||||
@@ -0,0 +1,36 @@
|
||||
/* tests/harness.h - shared by hand-written and generated test cases.
|
||||
* Ground truth for generated cases comes from CPython's own `re`
|
||||
* module (concept.md Section 11); see gen.py. */
|
||||
#ifndef REGEXX_TEST_HARNESS_H
|
||||
#define REGEXX_TEST_HARNESS_H
|
||||
#include "../regexx.h"
|
||||
#include <stddef.h>
|
||||
|
||||
extern int tc_pass, tc_fail;
|
||||
|
||||
/* search/match/fullmatch: exp_spans holds 2*n_spans int64_t values,
|
||||
* (start,end) for group 0, then group 1, 2, ... in order; -1,-1 means
|
||||
* the group did not participate (or, for group 0 alone, that
|
||||
* exp_matched is 0 and no spans are checked). */
|
||||
void tc_search_family(const char *desc, const char *pat, int flags,
|
||||
const char *subj, size_t subjlen, int kind,
|
||||
int exp_matched, int n_spans, const long long *exp_spans);
|
||||
|
||||
/* finditer: exp_spans holds 2*n_matches int64_t values, (start,end) of
|
||||
* the whole match for each match found, in order. */
|
||||
void tc_finditer(const char *desc, const char *pat, int flags,
|
||||
const char *subj, size_t subjlen,
|
||||
int exp_n_matches, const long long *exp_spans);
|
||||
|
||||
void tc_sub(const char *desc, const char *pat, int flags,
|
||||
const char *subj, size_t subjlen, const char *repl, int count,
|
||||
const char *exp_result, int exp_n);
|
||||
|
||||
/* split: texts[i] == NULL means that list element is None in Python. */
|
||||
void tc_split(const char *desc, const char *pat, int flags,
|
||||
const char *subj, size_t subjlen, int maxsplit,
|
||||
int n_items, const char *const *texts);
|
||||
|
||||
void harness_report(void);
|
||||
|
||||
#endif
|
||||
@@ -0,0 +1,10 @@
|
||||
#include "harness.h"
|
||||
#include <stdio.h>
|
||||
|
||||
extern void run_generated_tests(void);
|
||||
|
||||
int main(void) {
|
||||
run_generated_tests();
|
||||
harness_report();
|
||||
return tc_fail ? 1 : 0;
|
||||
}
|
||||
Reference in New Issue
Block a user