Files
retoorandClaude Sonnet 5 b0991814cb Expand BINARY/UTF8/ASCII edge-case testing to 3,252 cases, fix a real LOCALE bug
Three new categories added to tests/cases.py (TEST_PLAN.md records the
detail): Category H exercises BINARY mode with genuinely arbitrary raw
bytes (embedded NUL, high bytes, non-UTF-8 sequences) via real Python
`bytes` subjects, not just UTF-8-encoded text; Category I exercises
UTF8 mode edge cases (4-byte/astral code points, combining marks,
Arabic, Hebrew, CJK, offset correctness across multi-byte characters);
Category J is a fixed-seed (reproducible, not flaky) random generator
combining the existing atom/quantifier/grouping vocabulary across all
three modes.

This found and fixed a real bug, not just a test-generation one:
LOCALE, in BINARY/ASCII mode, treated bytes 0x80-0xFF as word
characters, based on an unverified assumption about what the "C"
locale does. Checked directly against both the C standard's own
guarantee for isalnum() under "C" and a real CPython interpreter with
re.LOCALE and the "C" locale explicitly set, neither treats anything
above 0x7f as a word character. Fixed in regexx.c's cls_is_word;
LOCALE is now documented as an accepted no-op in non-UTF8 mode,
matching verified reality instead of a prior assumption (concept.md
13.4, docs/API.md, README.md "Known deviations").

Two more findings were test-generation bugs, not regexx bugs: gen.py's
own ASCII-mode ground truth used Python str + re.ASCII (code-point
space) instead of a bytes pattern against a bytes subject (what
regexx's byte-oriented ASCII mode actually is), and LOCALE combined
with the (now removed as redundant) auto-added re.ASCII flag raised
ValueError in Python for being an incompatible combination. Both
fixed in gen.py.

Two further findings were concrete instances of an already-documented
category (glibc's wctype.h Unicode tables not matching CPython's own
exactly): U+00A0 and fullwidth digits U+FF10-FF19 are recognized by
CPython's \s/\d but not by glibc's iswspace()/iswdigit() under C.utf8.
Recorded in README.md, not patched, for the reason already given for
the first such instance (NBSP) in the previous commit.

After these fixes: all 3,252 committed cases pass, clean under
AddressSanitizer/UndefinedBehaviorSanitizer. The Category J generator
was additionally run against 5 more seeds at 3,000 iterations each
(26,760 further checks) as exploratory validation, all passing; not
committed, to keep the regular suite's size proportionate.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01EjuMk8kY9SDus1wWe2K9xY
2026-09-14 09:25:50 +00:00

520 lines
21 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""Test cases for regexx, checked against CPython's own `re` module
(concept.md Section 11). Each case records a pattern/subject pair and
an operation; gen.py computes the expected result with Python's `re`
and emits a C assertion that checks regexx against it.
"""
import re
CASES = []
def add(op, pattern, subject, mode="utf8", flags=(), **kw):
CASES.append(dict(op=op, pattern=pattern, subject=subject, mode=mode, flags=tuple(flags), **kw))
# ---- literals, concatenation, anchors --------------------------------
add("search", r"abc", "xxabcxx")
add("fullmatch", r"abc", "abc")
add("fullmatch", r"abc", "xabc")
add("match", r"abc", "abcdef")
add("match", r"abc", "xabcdef")
add("search", r"^abc$", "abc")
add("search", r"^abc$", "xabc")
add("search", r"^abc", "abc\nabc", flags=["MULTILINE"])
add("finditer", r"^abc", "abc\nabc", flags=["MULTILINE"])
add("search", r"abc$", "xxabc\nabc", flags=["MULTILINE"])
add("search", r"\Aabc", "abc")
add("search", r"abc\Z", "xxabc")
add("search", r"abc\Z", "xxabc\n")
# ---- character classes -------------------------------------------------
add("search", r"[a-z]+", "ABCdefGHI")
add("search", r"[^a-z]+", "abcDEFghi")
add("search", r"[a-z]+", "ABC", flags=["IGNORECASE"])
add("finditer", r"[abc]", "xaybzc")
add("search", r"[]a]", "]a") # ']' literal as first class member
add("search", r"[a\-z]", "-")
# ---- shorthand classes ---------------------------------------------------
add("finditer", r"\d+", "ab123cd45")
add("finditer", r"\D+", "ab123cd45")
add("search", r"\s+", "a b")
add("search", r"\S+", " ab ")
add("finditer", r"\w+", "hi there, friend!")
add("finditer", r"\W+", "hi there, friend!")
add("search", r"[\d\S]", " ")
add("search", r"[\d\S]", "5")
# ---- word boundaries ------------------------------------------------------
add("finditer", r"\bcat\b", "cat catalog cat")
add("finditer", r"\Bcat\B", "concatenate")
# ---- quantifiers ------------------------------------------------------------
add("search", r"a*", "aaab")
add("search", r"a+", "baaab")
add("search", r"a?b", "b")
add("search", r"a{2,4}", "aaaaa")
add("search", r"a{2,4}?", "aaaaa")
add("search", r"a{3}", "aa")
add("search", r"a.*b", "axxxbxxxb")
add("search", r"a.*?b", "axxxbxxxb")
add("finditer", r"a*", "baaab") # exercises empty-match advancement
# ---- alternation and groups -------------------------------------------------
add("search", r"(foo|bar)baz", "barbaz")
add("search", r"(?:foo|bar)baz", "foobaz")
add("finditer", r"(a)(b)?", "ab a")
add("search", r"(?P<year>\d{4})-(?P<month>\d{2})", "2024-09")
# ---- backreferences ---------------------------------------------------------
add("search", r"(\w+) \1", "hello hello")
add("search", r"(\w+) \1", "hello world")
add("search", r"(?P<w>\w+) (?P=w)", "abc abc")
add("search", r"(\w)\1", "aa")
add("search", r"(\w)\1", "ab")
# ---- lookaround --------------------------------------------------------------
add("search", r"foo(?=bar)", "foobar")
add("search", r"foo(?=bar)", "foobaz")
add("search", r"foo(?!bar)", "foobaz")
add("search", r"foo(?!bar)", "foobar")
add("search", r"(?<=foo)bar", "foobar")
add("search", r"(?<=foo)bar", "xxxbar")
add("search", r"(?<!foo)bar", "xxxbar")
add("search", r"(?<!foo)bar", "foobar")
add("finditer", r"(?<=\d)(?=(\d{3})+(?!\d))", "1234567") # thousands-separator style
# ---- atomic groups / possessive quantifiers -----------------------------------
add("search", r"a(?>bc|b)c", "abc")
add("search", r"a(bc|b)c", "abc")
add("search", r"a*+a", "aaaa")
add("search", r"a*a", "aaaa")
# ---- DOTALL / VERBOSE ----------------------------------------------------------
add("search", r"a.b", "a\nb")
add("search", r"a.b", "a\nb", flags=["DOTALL"])
add("search", r"""
\d+ # the integer part
\. # the dot
\d+ # the fractional part
""", "pi is 3.14 roughly", flags=["VERBOSE"])
# ---- escapes ---------------------------------------------------------------------
add("search", r"a\tb", "a\tb")
add("search", r"\x41\x42", "AB")
add("search", r"é", "café")
# ---- UTF-8 mode ------------------------------------------------------------------
add("search", r"\w+", "café au lait")
add("finditer", r"\w+", "café 中文 word")
add("search", r"[à-ÿ]+", "éèê")
# ---- sub / subn -------------------------------------------------------------------
add("sub", r"a", "banana", repl="o", count=0)
add("sub", r"a", "banana", repl="o", count=2)
add("sub", r"(\w+)@(\w+)", "user@host", repl=r"\2@\1")
add("sub", r"(?P<user>\w+)@(?P<host>\w+)", "user@host", repl=r"\g<host>@\g<user>")
add("sub", r"\s+", "a b c", repl=" ")
# ---- split ------------------------------------------------------------------------
add("split", r"[,;]\s*", "a, b;c , d")
add("split", r"(,)", "a,b,c")
add("split", r"\s*", "abc")
add("split", r",", "a,b,c", maxsplit=1)
# ---- ASCII vs UTF8 mode differences for \w ----------------------------------------
add("search", r"\w+", "café", mode="ascii")
# ---- BINARY mode --------------------------------------------------------------------
add("search", r"a.c", "a\x00c", mode="binary")
add("finditer", r"\d+", "12ab34", mode="binary")
# =============================================================================
# Combinatorial expansion. See tests/TEST_PLAN.md for the category
# breakdown and rationale; this is its implementation. Every combination
# is generated programmatically (never hand-typed) so the category sizes
# here match what TEST_PLAN.md describes, and so the whole expansion can
# be re-tuned in one place rather than case by case.
# =============================================================================
# ---- Category A: quantifier x grouping x flags -----------------------------
_A_ATOMS = [
(r"a", "aaaBaaa"),
(r"[a-c]", "abcXabc"),
(r"\d", "123abc456"),
(r"\w", "foo_bar 123"),
(r".", "ab\ncd"),
]
_A_QUANTS = ["*", "+", "?", "{2,3}", "{1,}", "*?", "+?", "??"]
_A_GROUPS = ["{0}", "({0})", "(?:{0})", "(?P<g>{0})"]
_A_FLAGS = [[], ["IGNORECASE"], ["MULTILINE", "DOTALL"]]
for _atom, _subj in _A_ATOMS:
for _q in _A_QUANTS:
for _g in _A_GROUPS:
_pat = _g.format(_atom + _q)
for _fl in _A_FLAGS:
add("search", _pat, _subj, flags=_fl)
add("finditer", _pat, _subj, flags=_fl)
# ---- Category B: alternation x backreference x groups -----------------------
_B_CASES = [
(r"(cat|dog|bird)\1", ["catcat", "catdog", "birdbird"]),
(r"(?P<x>foo|bar)-(?P=x)", ["foo-foo", "bar-baz", "foo-bar"]),
(r"(a|b){2,3}\1?", ["ababab", "bbb", "aab"]),
(r"(red|green|blue) \1", ["red red", "green blue", "blue blue"]),
(r"(?P<w>\w+)\s+(?P=w)", ["hello hello", "foo bar", "x x"]),
(r"(ab|cd|ef)+\1", ["ababab", "cdcd", "efabef"]),
(r"(a|ab)(c|bcd)(d*)", ["abcd", "ac", "abcdd"]),
(r"(.)(.)\2\1", ["abba", "xyzy", "aa"]),
(r"(\d{2})-(\d{2})-\1", ["12-34-12", "12-34-56", "99-01-99"]),
(r"((a)|(b))\2?\3?", ["aa", "bb", "ab", "a"]),
(r"(foo|foobar)bar", ["foobar", "foobarbar", "xfoobar"]),
(r"(a+)(b+)\1\2", ["aabbaabb", "aabbab", "ab"]),
(r"(?:(a)|(b))\1?\2?", ["aa", "bb", "a", "b"]),
(r"(x|y|z)\1{1,2}", ["xxx", "yy", "z", "xyz"]),
(r"(?P<first>\w+),\s*(?P<last>\w+)", ["Doe, John", "Smith,Jane", "noComma"]),
(r"(the|a|an) (\w+)", ["the cat", "a dog", "an apple", "xyz abc"]),
(r"(\w)(\w)(\w)\3\2\1", ["abccba", "xyzzyx", "abcabc"]),
(r"(ab)+(\1)?", ["ababab", "abab", "ab"]),
(r"(a{2}|b{3})\1", ["aaaa", "bbbbbb", "aabb"]),
(r"(?P<n>\d+)\.(?P<d>\d+)", ["3.14", "0.5", "no dot"]),
]
for _pat, _subjs in _B_CASES:
for _subj in _subjs:
for _fl in ([], ["IGNORECASE"]):
add("search", _pat, _subj, flags=_fl)
add("fullmatch", _pat, _subj, flags=_fl)
# ---- Category C: lookaround combinatorics -----------------------------------
_C_CASES = [
(r"\d+(?=px)", ["100px", "100em", "px100"]),
(r"\d+(?!px)", ["100px", "100em", "100"]),
(r"(?<=\$)\d+", ["$100", "100", "€100"]),
(r"(?<!\$)\d+", ["$100", "100", "x100"]),
(r"foo(?=bar|baz)", ["foobar", "foobaz", "fooqux"]),
(r"(?<=foo)(bar|baz)", ["foobar", "foobaz", "xbar"]),
(r"\w+(?=\s*:)", ["key: value", "key :value", "noColon"]),
(r"(?<=^)\w+", ["hello world", " leading", "x"]),
(r"(?=\d)\w+", ["123abc", "abc123", "999"]),
(r"(?!\d)\w+", ["abc123", "123abc", "_private"]),
(r"(?<=[A-Z])\w+", ["Hello", "hello", "AWord"]),
(r"(?<![A-Z])\w+", ["Hello", "hello", "xAWord"]),
(r"\b\w+(?=ing\b)", ["running fast", "sing a song", "no match"]),
(r"(?<=\bthe\s)\w+", ["the cat sat", "a cat sat", "these cats"]),
(r"(a(?=b))+", ["ababab", "aaab", "aab"]),
(r"((?<=a)b)+", ["abbb", "bbb", "aabb"]),
(r"(?=(\d+))\d{2}", ["12345", "1", "99x"]),
(r"(?<=\d{2})\w+", ["12abc", "1abc", "abc"]),
(r"(?<!\d)(?=\w)\w+", ["abc123", "1abc", "_x"]),
(r"\w+(?<!ing)", ["running", "walk", "sing"]),
]
for _pat, _subjs in _C_CASES:
for _subj in _subjs:
add("search", _pat, _subj)
add("fullmatch", _pat, _subj)
# ---- Category D: sub/split specific combinatorics ---------------------------
_D_SUB_CASES = [
(r"\d+", "a1b22c333", "N"),
(r"\d+", "a1b22c333", "N", 1),
(r"\d+", "a1b22c333", "N", 2),
(r"(\w+)@(\w+)", "user@host and admin@server", r"\2@\1"),
(r"(\w+)@(\w+)", "user@host and admin@server", r"\2@\1", 1),
(r"(?P<u>\w+)@(?P<h>\w+)", "user@host", r"\g<h>@\g<u>"),
(r"\s+", "a b\tc\nd", " "),
(r"[aeiou]", "hello world", "*"),
(r"[aeiou]", "hello world", "*", 3),
(r"(a)(b)?", "a ab a", r"[\1-\2]"),
(r"^", "line1\nline2\nline3", "> ", ),
(r"(\w)(\w*)", "hello world", r"\2\1"),
]
for _row in _D_SUB_CASES:
_pat, _subj, _repl = _row[0], _row[1], _row[2]
_count = _row[3] if len(_row) > 3 else 0
add("sub", _pat, _subj, repl=_repl, count=_count)
_D_SPLIT_CASES = [
(r"[,;]\s*", "a, b; c,d ; e"),
(r"(,)", "a,b,,c"),
(r"\s*,\s*", "a , b,c , d"),
(r"(\W+)", "Words, words, words."),
(r"", "abc"),
(r"x*", "abxxxcxd"),
(r"(\d)", "a1b2c3"),
(r":", "a:b:c:d:e"),
]
for _pat, _subj in _D_SPLIT_CASES:
add("split", _pat, _subj)
add("split", _pat, _subj, maxsplit=1)
add("split", _pat, _subj, maxsplit=2)
# ---- Category E: real-world combination patterns ----------------------------
_E_CASES = [
(r"[\w.+-]+@[\w-]+\.[\w.-]+", ["contact us at jane.doe+test@example.co.uk please",
"no email here"], ("search", "finditer")),
(r"https?://[\w.-]+(?:/[\w./?%&=-]*)?", ["visit https://example.com/path?x=1&y=2 now",
"no url"], ("search", "finditer")),
(r"\b(?:\d{1,3}\.){3}\d{1,3}\b", ["server at 192.168.1.1 responded",
"not an ip 999.999.999.999 either way",
"no ip here"], ("search", "finditer")),
(r"\d{4}-\d{2}-\d{2}", ["date: 2024-09-14 today", "no date"], ("search", "fullmatch")),
(r"([01]\d|2[0-3]):[0-5]\d:[0-5]\d", ["time is 23:59:59 now", "25:61:00 invalid"], ("search", "fullmatch")),
(r"\(?\d{3}\)?[-.\s]?\d{3}[-.\s]?\d{4}", ["call (555) 123-4567 today", "555.123.4567", "not a phone"], ("search", "finditer")),
(r"#[0-9a-fA-F]{6}\b", ["color is #1a2b3c bright", "#zzzzzz invalid"], ("search", "finditer")),
(r"\d+\.\d+\.\d+(?:-\w+)?", ["version 1.2.3-beta released", "version 1.2.3 released", "no version"], ("search", "finditer")),
(r"(\w+)=(\w+)", ["key1=val1;key2=val2", "noEquals"], ("finditer",)),
(r'"(?:[^"\\]|\\.)*"', ['say "hello \\"world\\"" now', 'no quotes'], ("search", "finditer")),
(r"(?P<host>[\w.]+) (?P<method>GET|POST) (?P<path>/\S*) (?P<status>\d{3})",
["10.0.0.1 GET /index.html 200", "malformed log line"], ("search", "fullmatch")),
(r"\*\*[^*]+\*\*|\*[^*]+\*", ["this is **bold** and *italic* text", "plain text"], ("search", "finditer")),
(r"\([^()]*\)", ["outer (inner) text", "(single) (double) groups", "no parens"], ("search", "finditer")),
(r"[À-ɏ\w]+", ["café résumé naïve", "plain ascii"], ("search", "finditer")),
(r"^\s*#.*$", [" # a comment\ncode here", "no comment"], ("search",)),
]
for _pat, _subjs, _ops in _E_CASES:
for _subj in _subjs:
for _op in _ops:
if _op == "search":
add("search", _pat, _subj)
elif _op == "finditer":
add("finditer", _pat, _subj)
elif _op == "fullmatch":
add("fullmatch", _pat, _subj)
# ---- Category F: mode cross-checks (ascii/utf8/binary) ----------------------
_F_CASES = [
(r"\w+", "café"),
(r"\w+", "naïve"),
(r"\s+", "a \t b"),
# U+00A0 (NBSP) is deliberately not probed here: CPython's \s
# matches it (Unicode's White_Space property includes it) but
# glibc's iswspace() under the C.utf8 locale this build relies on
# does not, a concrete, verified instance of the Unicode-table
# trade-off already recorded in README.md "Known deviations".
# Asserting a match here would just be re-testing a documented,
# accepted gap rather than looking for an actual bug.
(r"[a-z]+", "CAFE"),
(r"\d+", "123"),
(r"\b\w+\b", "hello world"),
(r".", "x"),
(r"a.c", "abc"),
(r"[^\d]+", "abc123"),
(r"\w{3}", "abc"),
(r"^\w+$", "word"),
(r"\W+", "abc!@#def"),
(r"[A-Za-z]+", "MixedCase"),
(r"\d{2,4}", "1234567"),
(r"\s\S+", " word"),
]
for _pat, _subj in _F_CASES:
for _mode in ("ascii", "utf8", "binary"):
add("search", _pat, _subj, mode=_mode)
add("finditer", _pat, _subj, mode=_mode)
# ---- Category G: flag combination stress ------------------------------------
_G_PATTERNS = [
r"^abc$",
r"a.b",
r"[a-z]+",
r"\bword\b",
r"a{2,4}",
r"(foo|bar)+",
r"\d+\.\d+",
r"x*y+z?",
r"[^abc]+",
r"(?:ab)+",
]
_G_SUBJECTS = ["AbC\ndef", "line1\nLINE2\nline3", "aAbBcC", "xxyyzz"]
_G_FLAG_COMBOS = [
[], ["IGNORECASE"], ["MULTILINE"], ["DOTALL"], ["VERBOSE"],
["IGNORECASE", "MULTILINE"], ["MULTILINE", "DOTALL"], ["IGNORECASE", "DOTALL"],
]
for _pat in _G_PATTERNS:
for _subj in _G_SUBJECTS:
for _fl in _G_FLAG_COMBOS:
add("search", _pat, _subj, flags=_fl)
add("finditer", _pat, _subj, flags=_fl)
# =============================================================================
# Category H: BINARY mode with genuinely arbitrary raw bytes (not just UTF-8
# encoded text). `subject` here is a real Python `bytes` object; gen.py's
# encode() passes it through unchanged rather than UTF-8-encoding it, so
# these exercise byte values and byte sequences that are not, and are not
# meant to be, valid UTF-8 at all.
# =============================================================================
_H_CASES = [
(r"a.c", b"a\x00c"),
(r"a.c", b"a\xffc"),
(r"[\x00-\x1f]+", b"hello\x01\x02\x03world"),
(r"\x00", b"abc\x00def"),
(r".", b"\xff"),
(r".+", b"\x00\x01\x02\xfd\xfe\xff"),
(r"a\x00+b", b"a\x00\x00\x00b"),
(r"[\x80-\xff]+", b"abc\x80\x90\xa0\xffdef"),
(r"[\x80-\xff]+", b"abcdef"),
(r"\w+", b"abc\x80\x90def", "ascii"),
(r"\w+", b"abc\x80\x90def", "ascii", ("LOCALE",)),
(r"(.)\1", b"\x00\x00"),
(r"(.)\1", b"\x00\x01"),
(r"(\xff+)", b"\xff\xff\xff"),
(r"[^\x00]+", b"abc\x00def"),
(r"^\xff", b"\xffabc"),
(r"\xff$", b"abc\xff"),
(r"a", b"A", "binary", ("IGNORECASE",)),
(r"[a-z]+", b"ABC\x80abc", "binary", ("IGNORECASE",)),
]
for _row in _H_CASES:
_pat, _subj = _row[0], _row[1]
_mode = _row[2] if len(_row) > 2 else "binary"
_fl = _row[3] if len(_row) > 3 else ()
add("search", _pat, _subj, mode=_mode, flags=_fl)
add("finditer", _pat, _subj, mode=_mode, flags=_fl)
# Raw-byte sub/split: replacement templates and split still need to work
# correctly with embedded NUL and high bytes on both sides.
_H_SUB_CASES = [
(r"\x00", b"a\x00b\x00c", "-"),
(r"[\x80-\xff]", b"a\x80b\x90c", "?"),
(r"(.)\x00", b"a\x00b\x00", r"[\1]"),
]
for _pat, _subj, _repl in _H_SUB_CASES:
add("sub", _pat, _subj, mode="binary", repl=_repl)
_H_SPLIT_CASES = [
(r"\x00+", b"a\x00\x00b\x00c"),
(r"[\x80-\xff]", b"a\x80b\x90c\xa0d"),
]
for _pat, _subj in _H_SPLIT_CASES:
add("split", _pat, _subj, mode="binary")
# =============================================================================
# Category I: UTF-8 edge cases -- real multi-byte content beyond simple
# accented Latin: 4-byte (astral plane) code points, combining marks,
# right-to-left scripts, CJK, mixed-width strings, and offset correctness
# for backreferences/groups/lookaround spanning multi-byte characters.
# =============================================================================
_I_CASES = [
(r"\w+", "hello \U0001F600 world"), # emoji, a 4-byte code point
(r".", "\U0001F600"),
(r"\W", "\U0001F600"),
(r"(.)", "\U0001F600\U0001F601"),
(r"é", "é"), # 'e' + combining acute accent
(r"\w+", "éclair"),
(r"[؀-ۿ]+", "السلام"), # Arabic
(r"\w+", "שלום"), # Hebrew
(r"[一-鿿]+", "中文字符"), # CJK unified ideographs
(r"\w+", "日本語"), # Japanese
(r"(\w)(\w)\2\1", "文字字文"),
(r"(.)\1+", "\U0001F600\U0001F600\U0001F600"),
(r"^.{3}$", "aé\U0001F600"), # mixed 1/2/4-byte codepoints, fixed count
(r"\bworld\b", "café world café"),
(r"(?<=é)\w+", "cafémonde"),
(r"(?=\U0001F600)", "x\U0001F600y"),
(r"[^\x00-\x7f]+", "abcéèêdef"),
# Fullwidth digits (U+FF10-FF19, Unicode category Nd) are deliberately
# not probed here: CPython's \d matches them (it matches the Nd
# category), but glibc's iswdigit() under the C.utf8 locale this
# build relies on does not, the same class of Unicode-table gap as
# the NBSP case in README.md "Known deviations", which now also
# records this specific instance.
(r"café", "CAFÉ", "utf8", ("IGNORECASE",)),
(r"ß", "ß", "utf8", ("IGNORECASE",)), # German sharp s
]
for _row in _I_CASES:
_pat, _subj = _row[0], _row[1]
_mode = _row[2] if len(_row) > 2 else "utf8"
_fl = _row[3] if len(_row) > 3 else ()
add("search", _pat, _subj, mode=_mode, flags=_fl)
add("fullmatch", _pat, _subj, mode=_mode, flags=_fl)
add("finditer", _pat, _subj, mode=_mode, flags=_fl)
_I_SUB_CASES = [
(r"\U0001F600", "hi \U0001F600 there", ":)"),
(r"(\w)", "café", r"[\1]"),
(r"[一-鿿]", "中文abc", "?"),
]
for _pat, _subj, _repl in _I_SUB_CASES:
add("sub", _pat, _subj, mode="utf8", repl=_repl)
_I_SPLIT_CASES = [
(r"\s+", "café au lait 中文"),
(r"(\U0001F600)", "a\U0001F600b\U0001F600c"),
]
for _pat, _subj in _I_SPLIT_CASES:
add("split", _pat, _subj, mode="utf8")
# =============================================================================
# Category J: seeded random combinatorics. A fixed seed makes this a
# deterministic, reproducible regression test, not a flaky fuzzer: the same
# cases are generated every run, so a failure here is exactly as
# reproducible and reportable as a hand-written one. Explores combinations
# no one sat down and thought to write by hand, across all three modes.
# =============================================================================
import random as _random # noqa: E402
_rng = _random.Random(0xC0FFEE)
_J_ASCII_ATOMS = ["a", "b", "c", "1", "2", "_", " ", "[a-c]", "[0-9]", r"\d", r"\w", r"\s", "."]
_J_UTF8_ATOMS = _J_ASCII_ATOMS + ["é", "è", "中", "ا", "\U0001F600"]
_J_QUANTS = ["", "?", "*", "+", "{1,2}", "{0,2}", "*?", "+?", "??"]
_J_WRAPS = ["{0}", "({0})", "(?:{0})", "(?P<g%d>{0})"]
_J_ASCII_SUBJ_CHARS = list("abc123_ ")
_J_UTF8_SUBJ_CHARS = _J_ASCII_SUBJ_CHARS + ["é", "è", "中", "文", "ا", "\U0001F600", "\U0001F601"]
_J_SPECIAL_ATOMS = {r"\d", r"\w", r"\s", ".", "[a-c]", "[0-9]"}
def _j_random_fragment(mode, group_no):
atoms = _J_UTF8_ATOMS if mode == "utf8" else _J_ASCII_ATOMS
atom = _rng.choice(atoms)
if atom not in _J_SPECIAL_ATOMS:
atom = re.escape(atom)
q = _rng.choice(_J_QUANTS)
wrap = _rng.choice(_J_WRAPS)
if "%d" in wrap:
wrap = wrap % group_no
return wrap.format(atom + q)
def _j_random_pattern(mode):
n = _rng.randint(1, 4)
group_no = 1
parts = []
for _ in range(n):
frag = _j_random_fragment(mode, group_no)
if frag.startswith("(") and not frag.startswith("(?:"):
group_no += 1
parts.append(frag)
joiner = "|" if _rng.random() < 0.25 else ""
return joiner.join(parts)
def _j_random_subject(mode, length):
chars = _J_UTF8_SUBJ_CHARS if mode == "utf8" else _J_ASCII_SUBJ_CHARS
return "".join(_rng.choice(chars) for _ in range(length))
_J_FLAG_CHOICES = [[], ["IGNORECASE"], ["MULTILINE"], ["DOTALL"], ["IGNORECASE", "MULTILINE"]]
_J_MODES = ["ascii", "utf8", "binary"]
_J_OPS = ["search", "finditer", "fullmatch"]
# Patterns/subjects for "binary" mode are still built from the ASCII-safe
# atom pool (Category H already covers genuinely arbitrary, non-UTF-8
# binary content deliberately and by hand); here binary mode exercises
# byte-for-byte matching of ordinary ASCII text through the BINARY
# encoding path specifically, distinct from the same text through ASCII
# or UTF8 mode, which Category F already probes with a fixed pattern set.
_j_generated = 0
for _ in range(900):
_mode = _rng.choice(_J_MODES)
_atom_mode = "utf8" if _mode == "utf8" else "ascii"
_pat = _j_random_pattern(_atom_mode)
_subj = _j_random_subject(_atom_mode, _rng.randint(0, 10))
_fl = _rng.choice(_J_FLAG_CHOICES)
_op = _rng.choice(_J_OPS)
add(_op, _pat, _subj, mode=_mode, flags=_fl)
_j_generated += 1