diff --git a/.gitignore b/.gitignore index b421d7e..a5fbcc8 100644 --- a/.gitignore +++ b/.gitignore @@ -1,6 +1,13 @@ *.o *.a rxgrep +binary_scan +utf8_scripts +ascii_logparse +redos_atomic +empty_match_rule +large_file_search +bench_vs_posix tests/generated_tests.c /.claude/ *.dSYM/ diff --git a/Makefile b/Makefile index 84b13a1..676cfe1 100644 --- a/Makefile +++ b/Makefile @@ -13,7 +13,7 @@ PREFIX ?= /usr/local AR ?= ar -.PHONY: all lib example test check clean install fuzz-smoke +.PHONY: all lib example examples test check clean install fuzz-smoke all: lib example @@ -32,6 +32,36 @@ example: rxgrep rxgrep: examples/rxgrep.c libregexx.a regexx.h $(CC) $(CFLAGS) -I. -o $@ examples/rxgrep.c libregexx.a $(LDLIBS) +# ---- feature examples ------------------------------------------------------ +# Each program under examples/ (besides rxgrep) isolates one distinct, +# less-obvious feature (a data mode, an empty-match rule, a ReDoS +# mitigation, mmap-backed large file input) rather than being a general +# purpose tool; see examples/README.md for what each one demonstrates. +EXAMPLE_BINS = binary_scan utf8_scripts ascii_logparse redos_atomic empty_match_rule large_file_search bench_vs_posix + +examples: rxgrep $(EXAMPLE_BINS) + +binary_scan: examples/binary_scan.c libregexx.a regexx.h + $(CC) $(CFLAGS) -I. -o $@ examples/binary_scan.c libregexx.a $(LDLIBS) + +utf8_scripts: examples/utf8_scripts.c libregexx.a regexx.h + $(CC) $(CFLAGS) -I. -o $@ examples/utf8_scripts.c libregexx.a $(LDLIBS) + +ascii_logparse: examples/ascii_logparse.c libregexx.a regexx.h + $(CC) $(CFLAGS) -I. -o $@ examples/ascii_logparse.c libregexx.a $(LDLIBS) + +redos_atomic: examples/redos_atomic.c libregexx.a regexx.h + $(CC) $(CFLAGS) -I. -o $@ examples/redos_atomic.c libregexx.a $(LDLIBS) + +empty_match_rule: examples/empty_match_rule.c libregexx.a regexx.h + $(CC) $(CFLAGS) -I. -o $@ examples/empty_match_rule.c libregexx.a $(LDLIBS) + +large_file_search: examples/large_file_search.c libregexx.a regexx.h + $(CC) $(CFLAGS) -I. -o $@ examples/large_file_search.c libregexx.a $(LDLIBS) + +bench_vs_posix: examples/bench_vs_posix.c libregexx.a regexx.h + $(CC) $(CFLAGS) -I. -o $@ examples/bench_vs_posix.c libregexx.a $(LDLIBS) + # ---- tests ----------------------------------------------------------------- # tests/generated_tests.c is generated from tests/cases.py using # CPython's own `re` module as ground truth (concept.md Section 11). @@ -54,4 +84,4 @@ install: libregexx.a regexx.h install -m644 regexx.h $(DESTDIR)$(PREFIX)/include/ clean: - rm -f regexx.o libregexx.a rxgrep tests/generated_tests.c + rm -f regexx.o libregexx.a rxgrep tests/generated_tests.c $(EXAMPLE_BINS) diff --git a/README.md b/README.md index 1aec56d..0b6d37e 100644 --- a/README.md +++ b/README.md @@ -367,6 +367,36 @@ API: Run `./rxgrep --help` for the full option list. +`examples/` has six further programs, each isolating one distinct feature +(a data mode, the undocumented CPython empty-match rule, atomic-group ReDoS +mitigation, `mmap`-backed large file input) rather than being a general +purpose tool; `make examples` builds all of them, and +[`examples/README.md`](examples/README.md) lists what each one demonstrates. + +## Benchmarks + +`examples/bench_vs_posix.c` measures this library directly against the C +standard library's own `` (POSIX `regcomp`/`regexec`, glibc's +DFA-backed implementation), on six scenarios at multi-megabyte or +multi-hundred-thousand-line scale, using only ERE pattern syntax that +regexx also accepts (no `\d`/`\w`/`\s`, no POSIX bracket classes), so both +engines run the identical pattern text against the identical subject. Reported without adjustment in either direction: on this +machine, glibc's engine is roughly 2x to 35x faster across the five +scenarios both engines can run, which is the expected outcome of a +roughly 2000-line backtracking interpreter built for Python `re` +compatibility competing against a DFA-backed engine with decades of +production optimization behind it and a much smaller feature set (no named +groups, no lazy quantifiers, no lookaround, no atomic groups, and no way to +search past an embedded `NUL` byte at all). The sharpest case is the +textbook ReDoS shape `(a+)+b`: measured at roughly 3x slower than glibc in +its plain form (a real cost, not a hang, since memoization keeps it +quadratic rather than exponential, "Implementation status" above), and, in +the same measurement, effectively free once rewritten with an atomic group, +a fix POSIX ERE cannot even express. Every scenario's match count is +cross-checked between the two engines and reported as agreeing or +differing, an independent correctness check beyond the CPython-derived test +suite below. + ## Testing `tests/cases.py` lists pattern/subject/operation triples, both hand-written diff --git a/USAGE.md b/USAGE.md index ada1b3e..901f312 100644 --- a/USAGE.md +++ b/USAGE.md @@ -719,3 +719,20 @@ Reading its source alongside this document is a reasonable next step once the examples above are familiar: it shows every function here used together in one program, including the parts this document simplified away for clarity (option parsing, line splitting, output formatting). + +`examples/` also has six smaller, single-purpose programs, each isolating +one distinct feature this section-by-section walkthrough only touches +briefly: `binary_scan.c` (raw byte-range classes including embedded `NUL`), +`utf8_scripts.c` (`\w` across non-Latin scripts, code point versus byte +offsets), `ascii_logparse.c` (named groups against structured log text), +`redos_atomic.c` (atomic groups/possessive quantifiers timed directly +against the unprotected form of a catastrophic-backtracking pattern), +`empty_match_rule.c` (the undocumented CPython empty-match retry rule, +verified against a real interpreter), `large_file_search.c` (`mmap`-backed +file input at 100MB, with elapsed time and peak memory printed), and +`bench_vs_posix.c` (a direct, honestly-reported timing comparison against +the C standard library's own `` on six scenarios, including the +one adversarial pattern shape where the gap is largest and how an atomic +group closes it). See [`examples/README.md`](examples/README.md) for the +complete list with what each one demonstrates; `make examples` builds all +of them. diff --git a/examples/README.md b/examples/README.md new file mode 100644 index 0000000..1bd6962 --- /dev/null +++ b/examples/README.md @@ -0,0 +1,32 @@ +# Examples + +Each program here is self-contained (`main`, no shared helper code) and +isolates one distinct feature rather than being a general purpose tool; the +one exception is `rxgrep.c`, a small grep-like program that ties several of +the same features together into something actually usable from a shell. +[`../USAGE.md`](../USAGE.md) walks through the same API surface function by +function, with smaller inline snippets; these programs are complete, +runnable, and closer to what real code doing this specific thing looks +like. + +Build all of them with `make examples` from the repository root (or +`make all` / `make example` for just `rxgrep`, the default). Each is also a +single `cc -I. file.c libregexx.a -o file` invocation away from being built +directly, no build system required. + +| Program | Demonstrates | +|---|---| +| [`rxgrep.c`](rxgrep.c) | A complete grep-like CLI: all three data modes via `-m`, `Pattern_search`/`finditer` for line and per-match output, `Pattern_subn` for `--sub`, reading from a real file or a pipe. Run `./rxgrep --help`. | +| [`binary_scan.c`](binary_scan.c) | `BINARY` mode: a byte-range character class (`[\x00-\xff]`) matching raw bytes 0-255, including an embedded `NUL` and an embedded `0x0A`, neither of which is a terminator or a line break in this mode the way it would be to a C string function or a line-oriented reader. This is the one thing `ASCII`/`UTF8` mode cannot do, since both introduce either a restricted classification or a decoding step. | +| [`utf8_scripts.c`](utf8_scripts.c) | `UTF8` mode: `\w` recognizing letters across Latin, Greek, Cyrillic, and CJK text (not only ASCII), and the resulting difference between `Match_span` (code point units) and `Match_span_byte` (byte units) once a match spans characters that are more than one byte wide. | +| [`ascii_logparse.c`](ascii_logparse.c) | `ASCII` mode doing what it is ordinarily used for: parsing structured, line-oriented text with named groups (`(?P...)`) and reading fields back out by name via `Match_group`/`Pattern_groupindex_lookup`, not by a numeric position the caller has to remember. | +| [`redos_atomic.c`](redos_atomic.c) | Atomic groups (`(?>...)`) and possessive quantifiers (`a++`) as a real, pattern-author-controlled defense against catastrophic backtracking, timed directly against the plain, unprotected form of the textbook `(a+)+b` shape on a non-matching adversarial input. | +| [`empty_match_rule.c`](empty_match_rule.c) | `Pattern_finditer`/`Pattern_split` reproducing a real CPython interpreter's undocumented empty-match retry rule exactly (`\d*?` against `"123abc456"` yields 16 matches, not 9), found by probing a real interpreter directly rather than by reading its documentation. | +| [`large_file_search.c`](large_file_search.c) | `Input_from_file`'s `mmap`-backed reading on a generated 100MB file, with the elapsed time and peak resident memory printed directly so they can be checked against the measured figures in `../README.md` "Memory footprint" rather than taken on faith. | +| [`bench_vs_posix.c`](bench_vs_posix.c) | A direct, timed comparison against the C standard library's own `` (POSIX `regcomp`/`regexec`) on six scenarios at multi-megabyte/multi-hundred-thousand-line scale, using only pattern syntax valid in both engines. Reported honestly: glibc's DFA-backed engine wins most scenarios by a wide margin, and the one adversarial pattern shape (`(a+)+b`) where this library's backtracking cost is most visible is shown both in its plain form and fixed with an atomic group, the latter having no POSIX ERE equivalent to compare against at all. | + +Every program below was compiled and actually run while writing it; the +claims in its top-of-file comment (what a real CPython interpreter does, +what timing difference an atomic group makes, and so on) are checked +against that program's own printed output, not written by hand and left +unverified. diff --git a/examples/ascii_logparse.c b/examples/ascii_logparse.c new file mode 100644 index 0000000..50300c5 --- /dev/null +++ b/examples/ascii_logparse.c @@ -0,0 +1,71 @@ +/* ascii_logparse - demonstrates ASCII mode doing what it is actually for: + * parsing structured, line-oriented text with named groups, then reading + * fields back out by name rather than by remembering a numeric position. + * This is the ordinary, common case the other examples in this directory + * deliberately are not (they each isolate one less-common feature + * instead); it is included so the directory has one example that looks + * like typical real-world usage: log filtering and field extraction. + */ +#include "regexx.h" +#include +#include + +static const char *LOG = + "2024-01-15 10:23:45 INFO [auth] user 'alice' logged in from 10.0.0.5\n" + "2024-01-15 10:24:02 ERROR [db] connection refused to 10.0.0.9:5432\n" + "2024-01-15 10:24:03 WARN [auth] repeated login failure for user 'bob'\n" + "2024-01-15 10:25:11 ERROR [auth] user 'bob' locked out after 5 failures\n" + "2024-01-15 10:26:00 INFO [db] connection pool resized to 32\n"; + +int main(void) { + const char *pattern = + "(?P\\d{4}-\\d{2}-\\d{2}) (?P