"""Test cases for regexx, checked against CPython's own `re` module (concept.md Section 11). Each case records a pattern/subject pair and an operation; gen.py computes the expected result with Python's `re` and emits a C assertion that checks regexx against it. """ import re CASES = [] def add(op, pattern, subject, mode="utf8", flags=(), **kw): CASES.append(dict(op=op, pattern=pattern, subject=subject, mode=mode, flags=tuple(flags), **kw)) # ---- literals, concatenation, anchors -------------------------------- add("search", r"abc", "xxabcxx") add("fullmatch", r"abc", "abc") add("fullmatch", r"abc", "xabc") add("match", r"abc", "abcdef") add("match", r"abc", "xabcdef") add("search", r"^abc$", "abc") add("search", r"^abc$", "xabc") add("search", r"^abc", "abc\nabc", flags=["MULTILINE"]) add("finditer", r"^abc", "abc\nabc", flags=["MULTILINE"]) add("search", r"abc$", "xxabc\nabc", flags=["MULTILINE"]) add("search", r"\Aabc", "abc") add("search", r"abc\Z", "xxabc") add("search", r"abc\Z", "xxabc\n") # ---- character classes ------------------------------------------------- add("search", r"[a-z]+", "ABCdefGHI") add("search", r"[^a-z]+", "abcDEFghi") add("search", r"[a-z]+", "ABC", flags=["IGNORECASE"]) add("finditer", r"[abc]", "xaybzc") add("search", r"[]a]", "]a") # ']' literal as first class member add("search", r"[a\-z]", "-") # ---- shorthand classes --------------------------------------------------- add("finditer", r"\d+", "ab123cd45") add("finditer", r"\D+", "ab123cd45") add("search", r"\s+", "a b") add("search", r"\S+", " ab ") add("finditer", r"\w+", "hi there, friend!") add("finditer", r"\W+", "hi there, friend!") add("search", r"[\d\S]", " ") add("search", r"[\d\S]", "5") # ---- word boundaries ------------------------------------------------------ add("finditer", r"\bcat\b", "cat catalog cat") add("finditer", r"\Bcat\B", "concatenate") # ---- quantifiers ------------------------------------------------------------ add("search", r"a*", "aaab") add("search", r"a+", "baaab") add("search", r"a?b", "b") add("search", r"a{2,4}", "aaaaa") add("search", r"a{2,4}?", "aaaaa") add("search", r"a{3}", "aa") add("search", r"a.*b", "axxxbxxxb") add("search", r"a.*?b", "axxxbxxxb") add("finditer", r"a*", "baaab") # exercises empty-match advancement # ---- alternation and groups ------------------------------------------------- add("search", r"(foo|bar)baz", "barbaz") add("search", r"(?:foo|bar)baz", "foobaz") add("finditer", r"(a)(b)?", "ab a") add("search", r"(?P\d{4})-(?P\d{2})", "2024-09") # ---- backreferences --------------------------------------------------------- add("search", r"(\w+) \1", "hello hello") add("search", r"(\w+) \1", "hello world") add("search", r"(?P\w+) (?P=w)", "abc abc") add("search", r"(\w)\1", "aa") add("search", r"(\w)\1", "ab") # ---- lookaround -------------------------------------------------------------- add("search", r"foo(?=bar)", "foobar") add("search", r"foo(?=bar)", "foobaz") add("search", r"foo(?!bar)", "foobaz") add("search", r"foo(?!bar)", "foobar") add("search", r"(?<=foo)bar", "foobar") add("search", r"(?<=foo)bar", "xxxbar") add("search", r"(?bc|b)c", "abc") add("search", r"a(bc|b)c", "abc") add("search", r"a*+a", "aaaa") add("search", r"a*a", "aaaa") # ---- DOTALL / VERBOSE ---------------------------------------------------------- add("search", r"a.b", "a\nb") add("search", r"a.b", "a\nb", flags=["DOTALL"]) add("search", r""" \d+ # the integer part \. # the dot \d+ # the fractional part """, "pi is 3.14 roughly", flags=["VERBOSE"]) # ---- escapes --------------------------------------------------------------------- add("search", r"a\tb", "a\tb") add("search", r"\x41\x42", "AB") add("search", r"é", "café") # ---- UTF-8 mode ------------------------------------------------------------------ add("search", r"\w+", "café au lait") add("finditer", r"\w+", "café 中文 word") add("search", r"[à-ÿ]+", "éèê") # ---- sub / subn ------------------------------------------------------------------- add("sub", r"a", "banana", repl="o", count=0) add("sub", r"a", "banana", repl="o", count=2) add("sub", r"(\w+)@(\w+)", "user@host", repl=r"\2@\1") add("sub", r"(?P\w+)@(?P\w+)", "user@host", repl=r"\g@\g") add("sub", r"\s+", "a b c", repl=" ") # ---- split ------------------------------------------------------------------------ add("split", r"[,;]\s*", "a, b;c , d") add("split", r"(,)", "a,b,c") add("split", r"\s*", "abc") add("split", r",", "a,b,c", maxsplit=1) # ---- ASCII vs UTF8 mode differences for \w ---------------------------------------- add("search", r"\w+", "café", mode="ascii") # ---- BINARY mode -------------------------------------------------------------------- add("search", r"a.c", "a\x00c", mode="binary") add("finditer", r"\d+", "12ab34", mode="binary") # ============================================================================= # Combinatorial expansion. See tests/TEST_PLAN.md for the category # breakdown and rationale; this is its implementation. Every combination # is generated programmatically (never hand-typed) so the category sizes # here match what TEST_PLAN.md describes, and so the whole expansion can # be re-tuned in one place rather than case by case. # ============================================================================= # ---- Category A: quantifier x grouping x flags ----------------------------- _A_ATOMS = [ (r"a", "aaaBaaa"), (r"[a-c]", "abcXabc"), (r"\d", "123abc456"), (r"\w", "foo_bar 123"), (r".", "ab\ncd"), ] _A_QUANTS = ["*", "+", "?", "{2,3}", "{1,}", "*?", "+?", "??"] _A_GROUPS = ["{0}", "({0})", "(?:{0})", "(?P{0})"] _A_FLAGS = [[], ["IGNORECASE"], ["MULTILINE", "DOTALL"]] for _atom, _subj in _A_ATOMS: for _q in _A_QUANTS: for _g in _A_GROUPS: _pat = _g.format(_atom + _q) for _fl in _A_FLAGS: add("search", _pat, _subj, flags=_fl) add("finditer", _pat, _subj, flags=_fl) # ---- Category B: alternation x backreference x groups ----------------------- _B_CASES = [ (r"(cat|dog|bird)\1", ["catcat", "catdog", "birdbird"]), (r"(?Pfoo|bar)-(?P=x)", ["foo-foo", "bar-baz", "foo-bar"]), (r"(a|b){2,3}\1?", ["ababab", "bbb", "aab"]), (r"(red|green|blue) \1", ["red red", "green blue", "blue blue"]), (r"(?P\w+)\s+(?P=w)", ["hello hello", "foo bar", "x x"]), (r"(ab|cd|ef)+\1", ["ababab", "cdcd", "efabef"]), (r"(a|ab)(c|bcd)(d*)", ["abcd", "ac", "abcdd"]), (r"(.)(.)\2\1", ["abba", "xyzy", "aa"]), (r"(\d{2})-(\d{2})-\1", ["12-34-12", "12-34-56", "99-01-99"]), (r"((a)|(b))\2?\3?", ["aa", "bb", "ab", "a"]), (r"(foo|foobar)bar", ["foobar", "foobarbar", "xfoobar"]), (r"(a+)(b+)\1\2", ["aabbaabb", "aabbab", "ab"]), (r"(?:(a)|(b))\1?\2?", ["aa", "bb", "a", "b"]), (r"(x|y|z)\1{1,2}", ["xxx", "yy", "z", "xyz"]), (r"(?P\w+),\s*(?P\w+)", ["Doe, John", "Smith,Jane", "noComma"]), (r"(the|a|an) (\w+)", ["the cat", "a dog", "an apple", "xyz abc"]), (r"(\w)(\w)(\w)\3\2\1", ["abccba", "xyzzyx", "abcabc"]), (r"(ab)+(\1)?", ["ababab", "abab", "ab"]), (r"(a{2}|b{3})\1", ["aaaa", "bbbbbb", "aabb"]), (r"(?P\d+)\.(?P\d+)", ["3.14", "0.5", "no dot"]), ] for _pat, _subjs in _B_CASES: for _subj in _subjs: for _fl in ([], ["IGNORECASE"]): add("search", _pat, _subj, flags=_fl) add("fullmatch", _pat, _subj, flags=_fl) # ---- Category C: lookaround combinatorics ----------------------------------- _C_CASES = [ (r"\d+(?=px)", ["100px", "100em", "px100"]), (r"\d+(?!px)", ["100px", "100em", "100"]), (r"(?<=\$)\d+", ["$100", "100", "€100"]), (r"(?\w+)@(?P\w+)", "user@host", r"\g@\g"), (r"\s+", "a b\tc\nd", " "), (r"[aeiou]", "hello world", "*"), (r"[aeiou]", "hello world", "*", 3), (r"(a)(b)?", "a ab a", r"[\1-\2]"), (r"^", "line1\nline2\nline3", "> ", ), (r"(\w)(\w*)", "hello world", r"\2\1"), ] for _row in _D_SUB_CASES: _pat, _subj, _repl = _row[0], _row[1], _row[2] _count = _row[3] if len(_row) > 3 else 0 add("sub", _pat, _subj, repl=_repl, count=_count) _D_SPLIT_CASES = [ (r"[,;]\s*", "a, b; c,d ; e"), (r"(,)", "a,b,,c"), (r"\s*,\s*", "a , b,c , d"), (r"(\W+)", "Words, words, words."), (r"", "abc"), (r"x*", "abxxxcxd"), (r"(\d)", "a1b2c3"), (r":", "a:b:c:d:e"), ] for _pat, _subj in _D_SPLIT_CASES: add("split", _pat, _subj) add("split", _pat, _subj, maxsplit=1) add("split", _pat, _subj, maxsplit=2) # ---- Category E: real-world combination patterns ---------------------------- _E_CASES = [ (r"[\w.+-]+@[\w-]+\.[\w.-]+", ["contact us at jane.doe+test@example.co.uk please", "no email here"], ("search", "finditer")), (r"https?://[\w.-]+(?:/[\w./?%&=-]*)?", ["visit https://example.com/path?x=1&y=2 now", "no url"], ("search", "finditer")), (r"\b(?:\d{1,3}\.){3}\d{1,3}\b", ["server at 192.168.1.1 responded", "not an ip 999.999.999.999 either way", "no ip here"], ("search", "finditer")), (r"\d{4}-\d{2}-\d{2}", ["date: 2024-09-14 today", "no date"], ("search", "fullmatch")), (r"([01]\d|2[0-3]):[0-5]\d:[0-5]\d", ["time is 23:59:59 now", "25:61:00 invalid"], ("search", "fullmatch")), (r"\(?\d{3}\)?[-.\s]?\d{3}[-.\s]?\d{4}", ["call (555) 123-4567 today", "555.123.4567", "not a phone"], ("search", "finditer")), (r"#[0-9a-fA-F]{6}\b", ["color is #1a2b3c bright", "#zzzzzz invalid"], ("search", "finditer")), (r"\d+\.\d+\.\d+(?:-\w+)?", ["version 1.2.3-beta released", "version 1.2.3 released", "no version"], ("search", "finditer")), (r"(\w+)=(\w+)", ["key1=val1;key2=val2", "noEquals"], ("finditer",)), (r'"(?:[^"\\]|\\.)*"', ['say "hello \\"world\\"" now', 'no quotes'], ("search", "finditer")), (r"(?P[\w.]+) (?PGET|POST) (?P/\S*) (?P\d{3})", ["10.0.0.1 GET /index.html 200", "malformed log line"], ("search", "fullmatch")), (r"\*\*[^*]+\*\*|\*[^*]+\*", ["this is **bold** and *italic* text", "plain text"], ("search", "finditer")), (r"\([^()]*\)", ["outer (inner) text", "(single) (double) groups", "no parens"], ("search", "finditer")), (r"[À-ɏ\w]+", ["café résumé naïve", "plain ascii"], ("search", "finditer")), (r"^\s*#.*$", [" # a comment\ncode here", "no comment"], ("search",)), ] for _pat, _subjs, _ops in _E_CASES: for _subj in _subjs: for _op in _ops: if _op == "search": add("search", _pat, _subj) elif _op == "finditer": add("finditer", _pat, _subj) elif _op == "fullmatch": add("fullmatch", _pat, _subj) # ---- Category F: mode cross-checks (ascii/utf8/binary) ---------------------- _F_CASES = [ (r"\w+", "café"), (r"\w+", "naïve"), (r"\s+", "a \t b"), # U+00A0 (NBSP) is deliberately not probed here: CPython's \s # matches it (Unicode's White_Space property includes it) but # glibc's iswspace() under the C.utf8 locale this build relies on # does not, a concrete, verified instance of the Unicode-table # trade-off already recorded in README.md "Known deviations". # Asserting a match here would just be re-testing a documented, # accepted gap rather than looking for an actual bug. (r"[a-z]+", "CAFE"), (r"\d+", "123"), (r"\b\w+\b", "hello world"), (r".", "x"), (r"a.c", "abc"), (r"[^\d]+", "abc123"), (r"\w{3}", "abc"), (r"^\w+$", "word"), (r"\W+", "abc!@#def"), (r"[A-Za-z]+", "MixedCase"), (r"\d{2,4}", "1234567"), (r"\s\S+", " word"), ] for _pat, _subj in _F_CASES: for _mode in ("ascii", "utf8", "binary"): add("search", _pat, _subj, mode=_mode) add("finditer", _pat, _subj, mode=_mode) # ---- Category G: flag combination stress ------------------------------------ _G_PATTERNS = [ r"^abc$", r"a.b", r"[a-z]+", r"\bword\b", r"a{2,4}", r"(foo|bar)+", r"\d+\.\d+", r"x*y+z?", r"[^abc]+", r"(?:ab)+", ] _G_SUBJECTS = ["AbC\ndef", "line1\nLINE2\nline3", "aAbBcC", "xxyyzz"] _G_FLAG_COMBOS = [ [], ["IGNORECASE"], ["MULTILINE"], ["DOTALL"], ["VERBOSE"], ["IGNORECASE", "MULTILINE"], ["MULTILINE", "DOTALL"], ["IGNORECASE", "DOTALL"], ] for _pat in _G_PATTERNS: for _subj in _G_SUBJECTS: for _fl in _G_FLAG_COMBOS: add("search", _pat, _subj, flags=_fl) add("finditer", _pat, _subj, flags=_fl) # ============================================================================= # Category H: BINARY mode with genuinely arbitrary raw bytes (not just UTF-8 # encoded text). `subject` here is a real Python `bytes` object; gen.py's # encode() passes it through unchanged rather than UTF-8-encoding it, so # these exercise byte values and byte sequences that are not, and are not # meant to be, valid UTF-8 at all. # ============================================================================= _H_CASES = [ (r"a.c", b"a\x00c"), (r"a.c", b"a\xffc"), (r"[\x00-\x1f]+", b"hello\x01\x02\x03world"), (r"\x00", b"abc\x00def"), (r".", b"\xff"), (r".+", b"\x00\x01\x02\xfd\xfe\xff"), (r"a\x00+b", b"a\x00\x00\x00b"), (r"[\x80-\xff]+", b"abc\x80\x90\xa0\xffdef"), (r"[\x80-\xff]+", b"abcdef"), (r"\w+", b"abc\x80\x90def", "ascii"), (r"\w+", b"abc\x80\x90def", "ascii", ("LOCALE",)), (r"(.)\1", b"\x00\x00"), (r"(.)\1", b"\x00\x01"), (r"(\xff+)", b"\xff\xff\xff"), (r"[^\x00]+", b"abc\x00def"), (r"^\xff", b"\xffabc"), (r"\xff$", b"abc\xff"), (r"a", b"A", "binary", ("IGNORECASE",)), (r"[a-z]+", b"ABC\x80abc", "binary", ("IGNORECASE",)), ] for _row in _H_CASES: _pat, _subj = _row[0], _row[1] _mode = _row[2] if len(_row) > 2 else "binary" _fl = _row[3] if len(_row) > 3 else () add("search", _pat, _subj, mode=_mode, flags=_fl) add("finditer", _pat, _subj, mode=_mode, flags=_fl) # Raw-byte sub/split: replacement templates and split still need to work # correctly with embedded NUL and high bytes on both sides. _H_SUB_CASES = [ (r"\x00", b"a\x00b\x00c", "-"), (r"[\x80-\xff]", b"a\x80b\x90c", "?"), (r"(.)\x00", b"a\x00b\x00", r"[\1]"), ] for _pat, _subj, _repl in _H_SUB_CASES: add("sub", _pat, _subj, mode="binary", repl=_repl) _H_SPLIT_CASES = [ (r"\x00+", b"a\x00\x00b\x00c"), (r"[\x80-\xff]", b"a\x80b\x90c\xa0d"), ] for _pat, _subj in _H_SPLIT_CASES: add("split", _pat, _subj, mode="binary") # ============================================================================= # Category I: UTF-8 edge cases -- real multi-byte content beyond simple # accented Latin: 4-byte (astral plane) code points, combining marks, # right-to-left scripts, CJK, mixed-width strings, and offset correctness # for backreferences/groups/lookaround spanning multi-byte characters. # ============================================================================= _I_CASES = [ (r"\w+", "hello \U0001F600 world"), # emoji, a 4-byte code point (r".", "\U0001F600"), (r"\W", "\U0001F600"), (r"(.)", "\U0001F600\U0001F601"), (r"é", "é"), # 'e' + combining acute accent (r"\w+", "éclair"), (r"[؀-ۿ]+", "السلام"), # Arabic (r"\w+", "שלום"), # Hebrew (r"[一-鿿]+", "中文字符"), # CJK unified ideographs (r"\w+", "日本語"), # Japanese (r"(\w)(\w)\2\1", "文字字文"), (r"(.)\1+", "\U0001F600\U0001F600\U0001F600"), (r"^.{3}$", "aé\U0001F600"), # mixed 1/2/4-byte codepoints, fixed count (r"\bworld\b", "café world café"), (r"(?<=é)\w+", "cafémonde"), (r"(?=\U0001F600)", "x\U0001F600y"), (r"[^\x00-\x7f]+", "abcéèêdef"), # Fullwidth digits (U+FF10-FF19, Unicode category Nd) are deliberately # not probed here: CPython's \d matches them (it matches the Nd # category), but glibc's iswdigit() under the C.utf8 locale this # build relies on does not, the same class of Unicode-table gap as # the NBSP case in README.md "Known deviations", which now also # records this specific instance. (r"café", "CAFÉ", "utf8", ("IGNORECASE",)), (r"ß", "ß", "utf8", ("IGNORECASE",)), # German sharp s ] for _row in _I_CASES: _pat, _subj = _row[0], _row[1] _mode = _row[2] if len(_row) > 2 else "utf8" _fl = _row[3] if len(_row) > 3 else () add("search", _pat, _subj, mode=_mode, flags=_fl) add("fullmatch", _pat, _subj, mode=_mode, flags=_fl) add("finditer", _pat, _subj, mode=_mode, flags=_fl) _I_SUB_CASES = [ (r"\U0001F600", "hi \U0001F600 there", ":)"), (r"(\w)", "café", r"[\1]"), (r"[一-鿿]", "中文abc", "?"), ] for _pat, _subj, _repl in _I_SUB_CASES: add("sub", _pat, _subj, mode="utf8", repl=_repl) _I_SPLIT_CASES = [ (r"\s+", "café au lait 中文"), (r"(\U0001F600)", "a\U0001F600b\U0001F600c"), ] for _pat, _subj in _I_SPLIT_CASES: add("split", _pat, _subj, mode="utf8") # ============================================================================= # Category J: seeded random combinatorics. A fixed seed makes this a # deterministic, reproducible regression test, not a flaky fuzzer: the same # cases are generated every run, so a failure here is exactly as # reproducible and reportable as a hand-written one. Explores combinations # no one sat down and thought to write by hand, across all three modes. # ============================================================================= import random as _random # noqa: E402 _rng = _random.Random(0xC0FFEE) _J_ASCII_ATOMS = ["a", "b", "c", "1", "2", "_", " ", "[a-c]", "[0-9]", r"\d", r"\w", r"\s", "."] _J_UTF8_ATOMS = _J_ASCII_ATOMS + ["é", "è", "中", "ا", "\U0001F600"] _J_QUANTS = ["", "?", "*", "+", "{1,2}", "{0,2}", "*?", "+?", "??"] _J_WRAPS = ["{0}", "({0})", "(?:{0})", "(?P{0})"] _J_ASCII_SUBJ_CHARS = list("abc123_ ") _J_UTF8_SUBJ_CHARS = _J_ASCII_SUBJ_CHARS + ["é", "è", "中", "文", "ا", "\U0001F600", "\U0001F601"] _J_SPECIAL_ATOMS = {r"\d", r"\w", r"\s", ".", "[a-c]", "[0-9]"} def _j_random_fragment(mode, group_no): atoms = _J_UTF8_ATOMS if mode == "utf8" else _J_ASCII_ATOMS atom = _rng.choice(atoms) if atom not in _J_SPECIAL_ATOMS: atom = re.escape(atom) q = _rng.choice(_J_QUANTS) wrap = _rng.choice(_J_WRAPS) if "%d" in wrap: wrap = wrap % group_no return wrap.format(atom + q) def _j_random_pattern(mode): n = _rng.randint(1, 4) group_no = 1 parts = [] for _ in range(n): frag = _j_random_fragment(mode, group_no) if frag.startswith("(") and not frag.startswith("(?:"): group_no += 1 parts.append(frag) joiner = "|" if _rng.random() < 0.25 else "" return joiner.join(parts) def _j_random_subject(mode, length): chars = _J_UTF8_SUBJ_CHARS if mode == "utf8" else _J_ASCII_SUBJ_CHARS return "".join(_rng.choice(chars) for _ in range(length)) _J_FLAG_CHOICES = [[], ["IGNORECASE"], ["MULTILINE"], ["DOTALL"], ["IGNORECASE", "MULTILINE"]] _J_MODES = ["ascii", "utf8", "binary"] _J_OPS = ["search", "finditer", "fullmatch"] # Patterns/subjects for "binary" mode are still built from the ASCII-safe # atom pool (Category H already covers genuinely arbitrary, non-UTF-8 # binary content deliberately and by hand); here binary mode exercises # byte-for-byte matching of ordinary ASCII text through the BINARY # encoding path specifically, distinct from the same text through ASCII # or UTF8 mode, which Category F already probes with a fixed pattern set. _j_generated = 0 for _ in range(900): _mode = _rng.choice(_J_MODES) _atom_mode = "utf8" if _mode == "utf8" else "ascii" _pat = _j_random_pattern(_atom_mode) _subj = _j_random_subject(_atom_mode, _rng.randint(0, 10)) _fl = _rng.choice(_J_FLAG_CHOICES) _op = _rng.choice(_J_OPS) add(_op, _pat, _subj, mode=_mode, flags=_fl) _j_generated += 1