Files
regexx/tests/cases.py
T
retoorandClaude Sonnet 5 8f6afd6cd4 Add regexx: a single-file C regex interpreter with Python re semantics
concept.md is the full design document: the objective (Python re parity
plus binary/ASCII/UTF-8 modes, gigabyte-scale input, single C file), the
automata-theory argument for why unrestricted backreferences/lookaround
are incompatible with strict single-pass constant memory, the resulting
two-engine architecture, the exact Python-mirroring naming convention,
and a full comparison against POSIX regex.h for C-background readers.

regexx.c/regexx.h are the v1 implementation: parser, compiler to a
Pike/backtracking-style bytecode, and a single recursive backtracking
engine covering the pattern syntax and operations listed in README.md,
validated against CPython's own re module output (tests/), clean under
AddressSanitizer/UBSan, and stress-tested (50MB simple-quantifier match,
graceful failure rather than a crash on complex repeats over large
input, clean rejection of every intentionally unsupported construct).

Also included: examples/rxgrep.c (a small grep-like program exercising
all three data modes and the substitution API), the Makefile, the MIT
LICENSE, and docs/API.md, an exhaustive reference for every type, flag,
and function's exact return-value and memory-ownership convention,
checked against the current source and against a real CPython
interpreter rather than against memory.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01EjuMk8kY9SDus1wWe2K9xY
2026-09-14 05:56:16 +00:00

130 lines
5.0 KiB
Python

"""Test cases for regexx, checked against CPython's own `re` module
(concept.md Section 11). Each case records a pattern/subject pair and
an operation; gen.py computes the expected result with Python's `re`
and emits a C assertion that checks regexx against it.
"""
CASES = []
def add(op, pattern, subject, mode="utf8", flags=(), **kw):
CASES.append(dict(op=op, pattern=pattern, subject=subject, mode=mode, flags=tuple(flags), **kw))
# ---- literals, concatenation, anchors --------------------------------
add("search", r"abc", "xxabcxx")
add("fullmatch", r"abc", "abc")
add("fullmatch", r"abc", "xabc")
add("match", r"abc", "abcdef")
add("match", r"abc", "xabcdef")
add("search", r"^abc$", "abc")
add("search", r"^abc$", "xabc")
add("search", r"^abc", "abc\nabc", flags=["MULTILINE"])
add("finditer", r"^abc", "abc\nabc", flags=["MULTILINE"])
add("search", r"abc$", "xxabc\nabc", flags=["MULTILINE"])
add("search", r"\Aabc", "abc")
add("search", r"abc\Z", "xxabc")
add("search", r"abc\Z", "xxabc\n")
# ---- character classes -------------------------------------------------
add("search", r"[a-z]+", "ABCdefGHI")
add("search", r"[^a-z]+", "abcDEFghi")
add("search", r"[a-z]+", "ABC", flags=["IGNORECASE"])
add("finditer", r"[abc]", "xaybzc")
add("search", r"[]a]", "]a") # ']' literal as first class member
add("search", r"[a\-z]", "-")
# ---- shorthand classes ---------------------------------------------------
add("finditer", r"\d+", "ab123cd45")
add("finditer", r"\D+", "ab123cd45")
add("search", r"\s+", "a b")
add("search", r"\S+", " ab ")
add("finditer", r"\w+", "hi there, friend!")
add("finditer", r"\W+", "hi there, friend!")
add("search", r"[\d\S]", " ")
add("search", r"[\d\S]", "5")
# ---- word boundaries ------------------------------------------------------
add("finditer", r"\bcat\b", "cat catalog cat")
add("finditer", r"\Bcat\B", "concatenate")
# ---- quantifiers ------------------------------------------------------------
add("search", r"a*", "aaab")
add("search", r"a+", "baaab")
add("search", r"a?b", "b")
add("search", r"a{2,4}", "aaaaa")
add("search", r"a{2,4}?", "aaaaa")
add("search", r"a{3}", "aa")
add("search", r"a.*b", "axxxbxxxb")
add("search", r"a.*?b", "axxxbxxxb")
add("finditer", r"a*", "baaab") # exercises empty-match advancement
# ---- alternation and groups -------------------------------------------------
add("search", r"(foo|bar)baz", "barbaz")
add("search", r"(?:foo|bar)baz", "foobaz")
add("finditer", r"(a)(b)?", "ab a")
add("search", r"(?P<year>\d{4})-(?P<month>\d{2})", "2024-09")
# ---- backreferences ---------------------------------------------------------
add("search", r"(\w+) \1", "hello hello")
add("search", r"(\w+) \1", "hello world")
add("search", r"(?P<w>\w+) (?P=w)", "abc abc")
add("search", r"(\w)\1", "aa")
add("search", r"(\w)\1", "ab")
# ---- lookaround --------------------------------------------------------------
add("search", r"foo(?=bar)", "foobar")
add("search", r"foo(?=bar)", "foobaz")
add("search", r"foo(?!bar)", "foobaz")
add("search", r"foo(?!bar)", "foobar")
add("search", r"(?<=foo)bar", "foobar")
add("search", r"(?<=foo)bar", "xxxbar")
add("search", r"(?<!foo)bar", "xxxbar")
add("search", r"(?<!foo)bar", "foobar")
add("finditer", r"(?<=\d)(?=(\d{3})+(?!\d))", "1234567") # thousands-separator style
# ---- atomic groups / possessive quantifiers -----------------------------------
add("search", r"a(?>bc|b)c", "abc")
add("search", r"a(bc|b)c", "abc")
add("search", r"a*+a", "aaaa")
add("search", r"a*a", "aaaa")
# ---- DOTALL / VERBOSE ----------------------------------------------------------
add("search", r"a.b", "a\nb")
add("search", r"a.b", "a\nb", flags=["DOTALL"])
add("search", r"""
\d+ # the integer part
\. # the dot
\d+ # the fractional part
""", "pi is 3.14 roughly", flags=["VERBOSE"])
# ---- escapes ---------------------------------------------------------------------
add("search", r"a\tb", "a\tb")
add("search", r"\x41\x42", "AB")
add("search", r"é", "café")
# ---- UTF-8 mode ------------------------------------------------------------------
add("search", r"\w+", "café au lait")
add("finditer", r"\w+", "café 中文 word")
add("search", r"[à-ÿ]+", "éèê")
# ---- sub / subn -------------------------------------------------------------------
add("sub", r"a", "banana", repl="o", count=0)
add("sub", r"a", "banana", repl="o", count=2)
add("sub", r"(\w+)@(\w+)", "user@host", repl=r"\2@\1")
add("sub", r"(?P<user>\w+)@(?P<host>\w+)", "user@host", repl=r"\g<host>@\g<user>")
add("sub", r"\s+", "a b c", repl=" ")
# ---- split ------------------------------------------------------------------------
add("split", r"[,;]\s*", "a, b;c , d")
add("split", r"(,)", "a,b,c")
add("split", r"\s*", "abc")
add("split", r",", "a,b,c", maxsplit=1)
# ---- ASCII vs UTF8 mode differences for \w ----------------------------------------
add("search", r"\w+", "café", mode="ascii")
# ---- BINARY mode --------------------------------------------------------------------
add("search", r"a.c", "a\x00c", mode="binary")
add("finditer", r"\d+", "12ab34", mode="binary")