concept.md is the full design document: the objective (Python re parity plus binary/ASCII/UTF-8 modes, gigabyte-scale input, single C file), the automata-theory argument for why unrestricted backreferences/lookaround are incompatible with strict single-pass constant memory, the resulting two-engine architecture, the exact Python-mirroring naming convention, and a full comparison against POSIX regex.h for C-background readers. regexx.c/regexx.h are the v1 implementation: parser, compiler to a Pike/backtracking-style bytecode, and a single recursive backtracking engine covering the pattern syntax and operations listed in README.md, validated against CPython's own re module output (tests/), clean under AddressSanitizer/UBSan, and stress-tested (50MB simple-quantifier match, graceful failure rather than a crash on complex repeats over large input, clean rejection of every intentionally unsupported construct). Also included: examples/rxgrep.c (a small grep-like program exercising all three data modes and the substitution API), the Makefile, the MIT LICENSE, and docs/API.md, an exhaustive reference for every type, flag, and function's exact return-value and memory-ownership convention, checked against the current source and against a real CPython interpreter rather than against memory. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01EjuMk8kY9SDus1wWe2K9xY
130 lines
5.0 KiB
Python
130 lines
5.0 KiB
Python
"""Test cases for regexx, checked against CPython's own `re` module
|
|
(concept.md Section 11). Each case records a pattern/subject pair and
|
|
an operation; gen.py computes the expected result with Python's `re`
|
|
and emits a C assertion that checks regexx against it.
|
|
"""
|
|
|
|
CASES = []
|
|
|
|
|
|
def add(op, pattern, subject, mode="utf8", flags=(), **kw):
|
|
CASES.append(dict(op=op, pattern=pattern, subject=subject, mode=mode, flags=tuple(flags), **kw))
|
|
|
|
|
|
# ---- literals, concatenation, anchors --------------------------------
|
|
add("search", r"abc", "xxabcxx")
|
|
add("fullmatch", r"abc", "abc")
|
|
add("fullmatch", r"abc", "xabc")
|
|
add("match", r"abc", "abcdef")
|
|
add("match", r"abc", "xabcdef")
|
|
add("search", r"^abc$", "abc")
|
|
add("search", r"^abc$", "xabc")
|
|
add("search", r"^abc", "abc\nabc", flags=["MULTILINE"])
|
|
add("finditer", r"^abc", "abc\nabc", flags=["MULTILINE"])
|
|
add("search", r"abc$", "xxabc\nabc", flags=["MULTILINE"])
|
|
add("search", r"\Aabc", "abc")
|
|
add("search", r"abc\Z", "xxabc")
|
|
add("search", r"abc\Z", "xxabc\n")
|
|
|
|
# ---- character classes -------------------------------------------------
|
|
add("search", r"[a-z]+", "ABCdefGHI")
|
|
add("search", r"[^a-z]+", "abcDEFghi")
|
|
add("search", r"[a-z]+", "ABC", flags=["IGNORECASE"])
|
|
add("finditer", r"[abc]", "xaybzc")
|
|
add("search", r"[]a]", "]a") # ']' literal as first class member
|
|
add("search", r"[a\-z]", "-")
|
|
|
|
# ---- shorthand classes ---------------------------------------------------
|
|
add("finditer", r"\d+", "ab123cd45")
|
|
add("finditer", r"\D+", "ab123cd45")
|
|
add("search", r"\s+", "a b")
|
|
add("search", r"\S+", " ab ")
|
|
add("finditer", r"\w+", "hi there, friend!")
|
|
add("finditer", r"\W+", "hi there, friend!")
|
|
add("search", r"[\d\S]", " ")
|
|
add("search", r"[\d\S]", "5")
|
|
|
|
# ---- word boundaries ------------------------------------------------------
|
|
add("finditer", r"\bcat\b", "cat catalog cat")
|
|
add("finditer", r"\Bcat\B", "concatenate")
|
|
|
|
# ---- quantifiers ------------------------------------------------------------
|
|
add("search", r"a*", "aaab")
|
|
add("search", r"a+", "baaab")
|
|
add("search", r"a?b", "b")
|
|
add("search", r"a{2,4}", "aaaaa")
|
|
add("search", r"a{2,4}?", "aaaaa")
|
|
add("search", r"a{3}", "aa")
|
|
add("search", r"a.*b", "axxxbxxxb")
|
|
add("search", r"a.*?b", "axxxbxxxb")
|
|
add("finditer", r"a*", "baaab") # exercises empty-match advancement
|
|
|
|
# ---- alternation and groups -------------------------------------------------
|
|
add("search", r"(foo|bar)baz", "barbaz")
|
|
add("search", r"(?:foo|bar)baz", "foobaz")
|
|
add("finditer", r"(a)(b)?", "ab a")
|
|
add("search", r"(?P<year>\d{4})-(?P<month>\d{2})", "2024-09")
|
|
|
|
# ---- backreferences ---------------------------------------------------------
|
|
add("search", r"(\w+) \1", "hello hello")
|
|
add("search", r"(\w+) \1", "hello world")
|
|
add("search", r"(?P<w>\w+) (?P=w)", "abc abc")
|
|
add("search", r"(\w)\1", "aa")
|
|
add("search", r"(\w)\1", "ab")
|
|
|
|
# ---- lookaround --------------------------------------------------------------
|
|
add("search", r"foo(?=bar)", "foobar")
|
|
add("search", r"foo(?=bar)", "foobaz")
|
|
add("search", r"foo(?!bar)", "foobaz")
|
|
add("search", r"foo(?!bar)", "foobar")
|
|
add("search", r"(?<=foo)bar", "foobar")
|
|
add("search", r"(?<=foo)bar", "xxxbar")
|
|
add("search", r"(?<!foo)bar", "xxxbar")
|
|
add("search", r"(?<!foo)bar", "foobar")
|
|
add("finditer", r"(?<=\d)(?=(\d{3})+(?!\d))", "1234567") # thousands-separator style
|
|
|
|
# ---- atomic groups / possessive quantifiers -----------------------------------
|
|
add("search", r"a(?>bc|b)c", "abc")
|
|
add("search", r"a(bc|b)c", "abc")
|
|
add("search", r"a*+a", "aaaa")
|
|
add("search", r"a*a", "aaaa")
|
|
|
|
# ---- DOTALL / VERBOSE ----------------------------------------------------------
|
|
add("search", r"a.b", "a\nb")
|
|
add("search", r"a.b", "a\nb", flags=["DOTALL"])
|
|
add("search", r"""
|
|
\d+ # the integer part
|
|
\. # the dot
|
|
\d+ # the fractional part
|
|
""", "pi is 3.14 roughly", flags=["VERBOSE"])
|
|
|
|
# ---- escapes ---------------------------------------------------------------------
|
|
add("search", r"a\tb", "a\tb")
|
|
add("search", r"\x41\x42", "AB")
|
|
add("search", r"é", "café")
|
|
|
|
# ---- UTF-8 mode ------------------------------------------------------------------
|
|
add("search", r"\w+", "café au lait")
|
|
add("finditer", r"\w+", "café 中文 word")
|
|
add("search", r"[à-ÿ]+", "éèê")
|
|
|
|
# ---- sub / subn -------------------------------------------------------------------
|
|
add("sub", r"a", "banana", repl="o", count=0)
|
|
add("sub", r"a", "banana", repl="o", count=2)
|
|
add("sub", r"(\w+)@(\w+)", "user@host", repl=r"\2@\1")
|
|
add("sub", r"(?P<user>\w+)@(?P<host>\w+)", "user@host", repl=r"\g<host>@\g<user>")
|
|
add("sub", r"\s+", "a b c", repl=" ")
|
|
|
|
# ---- split ------------------------------------------------------------------------
|
|
add("split", r"[,;]\s*", "a, b;c , d")
|
|
add("split", r"(,)", "a,b,c")
|
|
add("split", r"\s*", "abc")
|
|
add("split", r",", "a,b,c", maxsplit=1)
|
|
|
|
# ---- ASCII vs UTF8 mode differences for \w ----------------------------------------
|
|
add("search", r"\w+", "café", mode="ascii")
|
|
|
|
# ---- BINARY mode --------------------------------------------------------------------
|
|
add("search", r"a.c", "a\x00c", mode="binary")
|
|
add("finditer", r"\d+", "12ab34", mode="binary")
|