Three new categories added to tests/cases.py (TEST_PLAN.md records the detail): Category H exercises BINARY mode with genuinely arbitrary raw bytes (embedded NUL, high bytes, non-UTF-8 sequences) via real Python `bytes` subjects, not just UTF-8-encoded text; Category I exercises UTF8 mode edge cases (4-byte/astral code points, combining marks, Arabic, Hebrew, CJK, offset correctness across multi-byte characters); Category J is a fixed-seed (reproducible, not flaky) random generator combining the existing atom/quantifier/grouping vocabulary across all three modes. This found and fixed a real bug, not just a test-generation one: LOCALE, in BINARY/ASCII mode, treated bytes 0x80-0xFF as word characters, based on an unverified assumption about what the "C" locale does. Checked directly against both the C standard's own guarantee for isalnum() under "C" and a real CPython interpreter with re.LOCALE and the "C" locale explicitly set, neither treats anything above 0x7f as a word character. Fixed in regexx.c's cls_is_word; LOCALE is now documented as an accepted no-op in non-UTF8 mode, matching verified reality instead of a prior assumption (concept.md 13.4, docs/API.md, README.md "Known deviations"). Two more findings were test-generation bugs, not regexx bugs: gen.py's own ASCII-mode ground truth used Python str + re.ASCII (code-point space) instead of a bytes pattern against a bytes subject (what regexx's byte-oriented ASCII mode actually is), and LOCALE combined with the (now removed as redundant) auto-added re.ASCII flag raised ValueError in Python for being an incompatible combination. Both fixed in gen.py. Two further findings were concrete instances of an already-documented category (glibc's wctype.h Unicode tables not matching CPython's own exactly): U+00A0 and fullwidth digits U+FF10-FF19 are recognized by CPython's \s/\d but not by glibc's iswspace()/iswdigit() under C.utf8. Recorded in README.md, not patched, for the reason already given for the first such instance (NBSP) in the previous commit. After these fixes: all 3,252 committed cases pass, clean under AddressSanitizer/UndefinedBehaviorSanitizer. The Category J generator was additionally run against 5 more seeds at 3,000 iterations each (26,760 further checks) as exploratory validation, all passing; not committed, to keep the regular suite's size proportionate. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01EjuMk8kY9SDus1wWe2K9xY
520 lines
21 KiB
Python
520 lines
21 KiB
Python
"""Test cases for regexx, checked against CPython's own `re` module
|
||
(concept.md Section 11). Each case records a pattern/subject pair and
|
||
an operation; gen.py computes the expected result with Python's `re`
|
||
and emits a C assertion that checks regexx against it.
|
||
"""
|
||
import re
|
||
|
||
CASES = []
|
||
|
||
|
||
def add(op, pattern, subject, mode="utf8", flags=(), **kw):
|
||
CASES.append(dict(op=op, pattern=pattern, subject=subject, mode=mode, flags=tuple(flags), **kw))
|
||
|
||
|
||
# ---- literals, concatenation, anchors --------------------------------
|
||
add("search", r"abc", "xxabcxx")
|
||
add("fullmatch", r"abc", "abc")
|
||
add("fullmatch", r"abc", "xabc")
|
||
add("match", r"abc", "abcdef")
|
||
add("match", r"abc", "xabcdef")
|
||
add("search", r"^abc$", "abc")
|
||
add("search", r"^abc$", "xabc")
|
||
add("search", r"^abc", "abc\nabc", flags=["MULTILINE"])
|
||
add("finditer", r"^abc", "abc\nabc", flags=["MULTILINE"])
|
||
add("search", r"abc$", "xxabc\nabc", flags=["MULTILINE"])
|
||
add("search", r"\Aabc", "abc")
|
||
add("search", r"abc\Z", "xxabc")
|
||
add("search", r"abc\Z", "xxabc\n")
|
||
|
||
# ---- character classes -------------------------------------------------
|
||
add("search", r"[a-z]+", "ABCdefGHI")
|
||
add("search", r"[^a-z]+", "abcDEFghi")
|
||
add("search", r"[a-z]+", "ABC", flags=["IGNORECASE"])
|
||
add("finditer", r"[abc]", "xaybzc")
|
||
add("search", r"[]a]", "]a") # ']' literal as first class member
|
||
add("search", r"[a\-z]", "-")
|
||
|
||
# ---- shorthand classes ---------------------------------------------------
|
||
add("finditer", r"\d+", "ab123cd45")
|
||
add("finditer", r"\D+", "ab123cd45")
|
||
add("search", r"\s+", "a b")
|
||
add("search", r"\S+", " ab ")
|
||
add("finditer", r"\w+", "hi there, friend!")
|
||
add("finditer", r"\W+", "hi there, friend!")
|
||
add("search", r"[\d\S]", " ")
|
||
add("search", r"[\d\S]", "5")
|
||
|
||
# ---- word boundaries ------------------------------------------------------
|
||
add("finditer", r"\bcat\b", "cat catalog cat")
|
||
add("finditer", r"\Bcat\B", "concatenate")
|
||
|
||
# ---- quantifiers ------------------------------------------------------------
|
||
add("search", r"a*", "aaab")
|
||
add("search", r"a+", "baaab")
|
||
add("search", r"a?b", "b")
|
||
add("search", r"a{2,4}", "aaaaa")
|
||
add("search", r"a{2,4}?", "aaaaa")
|
||
add("search", r"a{3}", "aa")
|
||
add("search", r"a.*b", "axxxbxxxb")
|
||
add("search", r"a.*?b", "axxxbxxxb")
|
||
add("finditer", r"a*", "baaab") # exercises empty-match advancement
|
||
|
||
# ---- alternation and groups -------------------------------------------------
|
||
add("search", r"(foo|bar)baz", "barbaz")
|
||
add("search", r"(?:foo|bar)baz", "foobaz")
|
||
add("finditer", r"(a)(b)?", "ab a")
|
||
add("search", r"(?P<year>\d{4})-(?P<month>\d{2})", "2024-09")
|
||
|
||
# ---- backreferences ---------------------------------------------------------
|
||
add("search", r"(\w+) \1", "hello hello")
|
||
add("search", r"(\w+) \1", "hello world")
|
||
add("search", r"(?P<w>\w+) (?P=w)", "abc abc")
|
||
add("search", r"(\w)\1", "aa")
|
||
add("search", r"(\w)\1", "ab")
|
||
|
||
# ---- lookaround --------------------------------------------------------------
|
||
add("search", r"foo(?=bar)", "foobar")
|
||
add("search", r"foo(?=bar)", "foobaz")
|
||
add("search", r"foo(?!bar)", "foobaz")
|
||
add("search", r"foo(?!bar)", "foobar")
|
||
add("search", r"(?<=foo)bar", "foobar")
|
||
add("search", r"(?<=foo)bar", "xxxbar")
|
||
add("search", r"(?<!foo)bar", "xxxbar")
|
||
add("search", r"(?<!foo)bar", "foobar")
|
||
add("finditer", r"(?<=\d)(?=(\d{3})+(?!\d))", "1234567") # thousands-separator style
|
||
|
||
# ---- atomic groups / possessive quantifiers -----------------------------------
|
||
add("search", r"a(?>bc|b)c", "abc")
|
||
add("search", r"a(bc|b)c", "abc")
|
||
add("search", r"a*+a", "aaaa")
|
||
add("search", r"a*a", "aaaa")
|
||
|
||
# ---- DOTALL / VERBOSE ----------------------------------------------------------
|
||
add("search", r"a.b", "a\nb")
|
||
add("search", r"a.b", "a\nb", flags=["DOTALL"])
|
||
add("search", r"""
|
||
\d+ # the integer part
|
||
\. # the dot
|
||
\d+ # the fractional part
|
||
""", "pi is 3.14 roughly", flags=["VERBOSE"])
|
||
|
||
# ---- escapes ---------------------------------------------------------------------
|
||
add("search", r"a\tb", "a\tb")
|
||
add("search", r"\x41\x42", "AB")
|
||
add("search", r"é", "café")
|
||
|
||
# ---- UTF-8 mode ------------------------------------------------------------------
|
||
add("search", r"\w+", "café au lait")
|
||
add("finditer", r"\w+", "café 中文 word")
|
||
add("search", r"[à-ÿ]+", "éèê")
|
||
|
||
# ---- sub / subn -------------------------------------------------------------------
|
||
add("sub", r"a", "banana", repl="o", count=0)
|
||
add("sub", r"a", "banana", repl="o", count=2)
|
||
add("sub", r"(\w+)@(\w+)", "user@host", repl=r"\2@\1")
|
||
add("sub", r"(?P<user>\w+)@(?P<host>\w+)", "user@host", repl=r"\g<host>@\g<user>")
|
||
add("sub", r"\s+", "a b c", repl=" ")
|
||
|
||
# ---- split ------------------------------------------------------------------------
|
||
add("split", r"[,;]\s*", "a, b;c , d")
|
||
add("split", r"(,)", "a,b,c")
|
||
add("split", r"\s*", "abc")
|
||
add("split", r",", "a,b,c", maxsplit=1)
|
||
|
||
# ---- ASCII vs UTF8 mode differences for \w ----------------------------------------
|
||
add("search", r"\w+", "café", mode="ascii")
|
||
|
||
# ---- BINARY mode --------------------------------------------------------------------
|
||
add("search", r"a.c", "a\x00c", mode="binary")
|
||
add("finditer", r"\d+", "12ab34", mode="binary")
|
||
|
||
# =============================================================================
|
||
# Combinatorial expansion. See tests/TEST_PLAN.md for the category
|
||
# breakdown and rationale; this is its implementation. Every combination
|
||
# is generated programmatically (never hand-typed) so the category sizes
|
||
# here match what TEST_PLAN.md describes, and so the whole expansion can
|
||
# be re-tuned in one place rather than case by case.
|
||
# =============================================================================
|
||
|
||
# ---- Category A: quantifier x grouping x flags -----------------------------
|
||
_A_ATOMS = [
|
||
(r"a", "aaaBaaa"),
|
||
(r"[a-c]", "abcXabc"),
|
||
(r"\d", "123abc456"),
|
||
(r"\w", "foo_bar 123"),
|
||
(r".", "ab\ncd"),
|
||
]
|
||
_A_QUANTS = ["*", "+", "?", "{2,3}", "{1,}", "*?", "+?", "??"]
|
||
_A_GROUPS = ["{0}", "({0})", "(?:{0})", "(?P<g>{0})"]
|
||
_A_FLAGS = [[], ["IGNORECASE"], ["MULTILINE", "DOTALL"]]
|
||
|
||
for _atom, _subj in _A_ATOMS:
|
||
for _q in _A_QUANTS:
|
||
for _g in _A_GROUPS:
|
||
_pat = _g.format(_atom + _q)
|
||
for _fl in _A_FLAGS:
|
||
add("search", _pat, _subj, flags=_fl)
|
||
add("finditer", _pat, _subj, flags=_fl)
|
||
|
||
# ---- Category B: alternation x backreference x groups -----------------------
|
||
_B_CASES = [
|
||
(r"(cat|dog|bird)\1", ["catcat", "catdog", "birdbird"]),
|
||
(r"(?P<x>foo|bar)-(?P=x)", ["foo-foo", "bar-baz", "foo-bar"]),
|
||
(r"(a|b){2,3}\1?", ["ababab", "bbb", "aab"]),
|
||
(r"(red|green|blue) \1", ["red red", "green blue", "blue blue"]),
|
||
(r"(?P<w>\w+)\s+(?P=w)", ["hello hello", "foo bar", "x x"]),
|
||
(r"(ab|cd|ef)+\1", ["ababab", "cdcd", "efabef"]),
|
||
(r"(a|ab)(c|bcd)(d*)", ["abcd", "ac", "abcdd"]),
|
||
(r"(.)(.)\2\1", ["abba", "xyzy", "aa"]),
|
||
(r"(\d{2})-(\d{2})-\1", ["12-34-12", "12-34-56", "99-01-99"]),
|
||
(r"((a)|(b))\2?\3?", ["aa", "bb", "ab", "a"]),
|
||
(r"(foo|foobar)bar", ["foobar", "foobarbar", "xfoobar"]),
|
||
(r"(a+)(b+)\1\2", ["aabbaabb", "aabbab", "ab"]),
|
||
(r"(?:(a)|(b))\1?\2?", ["aa", "bb", "a", "b"]),
|
||
(r"(x|y|z)\1{1,2}", ["xxx", "yy", "z", "xyz"]),
|
||
(r"(?P<first>\w+),\s*(?P<last>\w+)", ["Doe, John", "Smith,Jane", "noComma"]),
|
||
(r"(the|a|an) (\w+)", ["the cat", "a dog", "an apple", "xyz abc"]),
|
||
(r"(\w)(\w)(\w)\3\2\1", ["abccba", "xyzzyx", "abcabc"]),
|
||
(r"(ab)+(\1)?", ["ababab", "abab", "ab"]),
|
||
(r"(a{2}|b{3})\1", ["aaaa", "bbbbbb", "aabb"]),
|
||
(r"(?P<n>\d+)\.(?P<d>\d+)", ["3.14", "0.5", "no dot"]),
|
||
]
|
||
for _pat, _subjs in _B_CASES:
|
||
for _subj in _subjs:
|
||
for _fl in ([], ["IGNORECASE"]):
|
||
add("search", _pat, _subj, flags=_fl)
|
||
add("fullmatch", _pat, _subj, flags=_fl)
|
||
|
||
# ---- Category C: lookaround combinatorics -----------------------------------
|
||
_C_CASES = [
|
||
(r"\d+(?=px)", ["100px", "100em", "px100"]),
|
||
(r"\d+(?!px)", ["100px", "100em", "100"]),
|
||
(r"(?<=\$)\d+", ["$100", "100", "€100"]),
|
||
(r"(?<!\$)\d+", ["$100", "100", "x100"]),
|
||
(r"foo(?=bar|baz)", ["foobar", "foobaz", "fooqux"]),
|
||
(r"(?<=foo)(bar|baz)", ["foobar", "foobaz", "xbar"]),
|
||
(r"\w+(?=\s*:)", ["key: value", "key :value", "noColon"]),
|
||
(r"(?<=^)\w+", ["hello world", " leading", "x"]),
|
||
(r"(?=\d)\w+", ["123abc", "abc123", "999"]),
|
||
(r"(?!\d)\w+", ["abc123", "123abc", "_private"]),
|
||
(r"(?<=[A-Z])\w+", ["Hello", "hello", "AWord"]),
|
||
(r"(?<![A-Z])\w+", ["Hello", "hello", "xAWord"]),
|
||
(r"\b\w+(?=ing\b)", ["running fast", "sing a song", "no match"]),
|
||
(r"(?<=\bthe\s)\w+", ["the cat sat", "a cat sat", "these cats"]),
|
||
(r"(a(?=b))+", ["ababab", "aaab", "aab"]),
|
||
(r"((?<=a)b)+", ["abbb", "bbb", "aabb"]),
|
||
(r"(?=(\d+))\d{2}", ["12345", "1", "99x"]),
|
||
(r"(?<=\d{2})\w+", ["12abc", "1abc", "abc"]),
|
||
(r"(?<!\d)(?=\w)\w+", ["abc123", "1abc", "_x"]),
|
||
(r"\w+(?<!ing)", ["running", "walk", "sing"]),
|
||
]
|
||
for _pat, _subjs in _C_CASES:
|
||
for _subj in _subjs:
|
||
add("search", _pat, _subj)
|
||
add("fullmatch", _pat, _subj)
|
||
|
||
# ---- Category D: sub/split specific combinatorics ---------------------------
|
||
_D_SUB_CASES = [
|
||
(r"\d+", "a1b22c333", "N"),
|
||
(r"\d+", "a1b22c333", "N", 1),
|
||
(r"\d+", "a1b22c333", "N", 2),
|
||
(r"(\w+)@(\w+)", "user@host and admin@server", r"\2@\1"),
|
||
(r"(\w+)@(\w+)", "user@host and admin@server", r"\2@\1", 1),
|
||
(r"(?P<u>\w+)@(?P<h>\w+)", "user@host", r"\g<h>@\g<u>"),
|
||
(r"\s+", "a b\tc\nd", " "),
|
||
(r"[aeiou]", "hello world", "*"),
|
||
(r"[aeiou]", "hello world", "*", 3),
|
||
(r"(a)(b)?", "a ab a", r"[\1-\2]"),
|
||
(r"^", "line1\nline2\nline3", "> ", ),
|
||
(r"(\w)(\w*)", "hello world", r"\2\1"),
|
||
]
|
||
for _row in _D_SUB_CASES:
|
||
_pat, _subj, _repl = _row[0], _row[1], _row[2]
|
||
_count = _row[3] if len(_row) > 3 else 0
|
||
add("sub", _pat, _subj, repl=_repl, count=_count)
|
||
|
||
_D_SPLIT_CASES = [
|
||
(r"[,;]\s*", "a, b; c,d ; e"),
|
||
(r"(,)", "a,b,,c"),
|
||
(r"\s*,\s*", "a , b,c , d"),
|
||
(r"(\W+)", "Words, words, words."),
|
||
(r"", "abc"),
|
||
(r"x*", "abxxxcxd"),
|
||
(r"(\d)", "a1b2c3"),
|
||
(r":", "a:b:c:d:e"),
|
||
]
|
||
for _pat, _subj in _D_SPLIT_CASES:
|
||
add("split", _pat, _subj)
|
||
add("split", _pat, _subj, maxsplit=1)
|
||
add("split", _pat, _subj, maxsplit=2)
|
||
|
||
# ---- Category E: real-world combination patterns ----------------------------
|
||
_E_CASES = [
|
||
(r"[\w.+-]+@[\w-]+\.[\w.-]+", ["contact us at jane.doe+test@example.co.uk please",
|
||
"no email here"], ("search", "finditer")),
|
||
(r"https?://[\w.-]+(?:/[\w./?%&=-]*)?", ["visit https://example.com/path?x=1&y=2 now",
|
||
"no url"], ("search", "finditer")),
|
||
(r"\b(?:\d{1,3}\.){3}\d{1,3}\b", ["server at 192.168.1.1 responded",
|
||
"not an ip 999.999.999.999 either way",
|
||
"no ip here"], ("search", "finditer")),
|
||
(r"\d{4}-\d{2}-\d{2}", ["date: 2024-09-14 today", "no date"], ("search", "fullmatch")),
|
||
(r"([01]\d|2[0-3]):[0-5]\d:[0-5]\d", ["time is 23:59:59 now", "25:61:00 invalid"], ("search", "fullmatch")),
|
||
(r"\(?\d{3}\)?[-.\s]?\d{3}[-.\s]?\d{4}", ["call (555) 123-4567 today", "555.123.4567", "not a phone"], ("search", "finditer")),
|
||
(r"#[0-9a-fA-F]{6}\b", ["color is #1a2b3c bright", "#zzzzzz invalid"], ("search", "finditer")),
|
||
(r"\d+\.\d+\.\d+(?:-\w+)?", ["version 1.2.3-beta released", "version 1.2.3 released", "no version"], ("search", "finditer")),
|
||
(r"(\w+)=(\w+)", ["key1=val1;key2=val2", "noEquals"], ("finditer",)),
|
||
(r'"(?:[^"\\]|\\.)*"', ['say "hello \\"world\\"" now', 'no quotes'], ("search", "finditer")),
|
||
(r"(?P<host>[\w.]+) (?P<method>GET|POST) (?P<path>/\S*) (?P<status>\d{3})",
|
||
["10.0.0.1 GET /index.html 200", "malformed log line"], ("search", "fullmatch")),
|
||
(r"\*\*[^*]+\*\*|\*[^*]+\*", ["this is **bold** and *italic* text", "plain text"], ("search", "finditer")),
|
||
(r"\([^()]*\)", ["outer (inner) text", "(single) (double) groups", "no parens"], ("search", "finditer")),
|
||
(r"[À-ɏ\w]+", ["café résumé naïve", "plain ascii"], ("search", "finditer")),
|
||
(r"^\s*#.*$", [" # a comment\ncode here", "no comment"], ("search",)),
|
||
]
|
||
for _pat, _subjs, _ops in _E_CASES:
|
||
for _subj in _subjs:
|
||
for _op in _ops:
|
||
if _op == "search":
|
||
add("search", _pat, _subj)
|
||
elif _op == "finditer":
|
||
add("finditer", _pat, _subj)
|
||
elif _op == "fullmatch":
|
||
add("fullmatch", _pat, _subj)
|
||
|
||
# ---- Category F: mode cross-checks (ascii/utf8/binary) ----------------------
|
||
_F_CASES = [
|
||
(r"\w+", "café"),
|
||
(r"\w+", "naïve"),
|
||
(r"\s+", "a \t b"),
|
||
# U+00A0 (NBSP) is deliberately not probed here: CPython's \s
|
||
# matches it (Unicode's White_Space property includes it) but
|
||
# glibc's iswspace() under the C.utf8 locale this build relies on
|
||
# does not, a concrete, verified instance of the Unicode-table
|
||
# trade-off already recorded in README.md "Known deviations".
|
||
# Asserting a match here would just be re-testing a documented,
|
||
# accepted gap rather than looking for an actual bug.
|
||
(r"[a-z]+", "CAFE"),
|
||
(r"\d+", "123"),
|
||
(r"\b\w+\b", "hello world"),
|
||
(r".", "x"),
|
||
(r"a.c", "abc"),
|
||
(r"[^\d]+", "abc123"),
|
||
(r"\w{3}", "abc"),
|
||
(r"^\w+$", "word"),
|
||
(r"\W+", "abc!@#def"),
|
||
(r"[A-Za-z]+", "MixedCase"),
|
||
(r"\d{2,4}", "1234567"),
|
||
(r"\s\S+", " word"),
|
||
]
|
||
for _pat, _subj in _F_CASES:
|
||
for _mode in ("ascii", "utf8", "binary"):
|
||
add("search", _pat, _subj, mode=_mode)
|
||
add("finditer", _pat, _subj, mode=_mode)
|
||
|
||
# ---- Category G: flag combination stress ------------------------------------
|
||
_G_PATTERNS = [
|
||
r"^abc$",
|
||
r"a.b",
|
||
r"[a-z]+",
|
||
r"\bword\b",
|
||
r"a{2,4}",
|
||
r"(foo|bar)+",
|
||
r"\d+\.\d+",
|
||
r"x*y+z?",
|
||
r"[^abc]+",
|
||
r"(?:ab)+",
|
||
]
|
||
_G_SUBJECTS = ["AbC\ndef", "line1\nLINE2\nline3", "aAbBcC", "xxyyzz"]
|
||
_G_FLAG_COMBOS = [
|
||
[], ["IGNORECASE"], ["MULTILINE"], ["DOTALL"], ["VERBOSE"],
|
||
["IGNORECASE", "MULTILINE"], ["MULTILINE", "DOTALL"], ["IGNORECASE", "DOTALL"],
|
||
]
|
||
for _pat in _G_PATTERNS:
|
||
for _subj in _G_SUBJECTS:
|
||
for _fl in _G_FLAG_COMBOS:
|
||
add("search", _pat, _subj, flags=_fl)
|
||
add("finditer", _pat, _subj, flags=_fl)
|
||
|
||
# =============================================================================
|
||
# Category H: BINARY mode with genuinely arbitrary raw bytes (not just UTF-8
|
||
# encoded text). `subject` here is a real Python `bytes` object; gen.py's
|
||
# encode() passes it through unchanged rather than UTF-8-encoding it, so
|
||
# these exercise byte values and byte sequences that are not, and are not
|
||
# meant to be, valid UTF-8 at all.
|
||
# =============================================================================
|
||
_H_CASES = [
|
||
(r"a.c", b"a\x00c"),
|
||
(r"a.c", b"a\xffc"),
|
||
(r"[\x00-\x1f]+", b"hello\x01\x02\x03world"),
|
||
(r"\x00", b"abc\x00def"),
|
||
(r".", b"\xff"),
|
||
(r".+", b"\x00\x01\x02\xfd\xfe\xff"),
|
||
(r"a\x00+b", b"a\x00\x00\x00b"),
|
||
(r"[\x80-\xff]+", b"abc\x80\x90\xa0\xffdef"),
|
||
(r"[\x80-\xff]+", b"abcdef"),
|
||
(r"\w+", b"abc\x80\x90def", "ascii"),
|
||
(r"\w+", b"abc\x80\x90def", "ascii", ("LOCALE",)),
|
||
(r"(.)\1", b"\x00\x00"),
|
||
(r"(.)\1", b"\x00\x01"),
|
||
(r"(\xff+)", b"\xff\xff\xff"),
|
||
(r"[^\x00]+", b"abc\x00def"),
|
||
(r"^\xff", b"\xffabc"),
|
||
(r"\xff$", b"abc\xff"),
|
||
(r"a", b"A", "binary", ("IGNORECASE",)),
|
||
(r"[a-z]+", b"ABC\x80abc", "binary", ("IGNORECASE",)),
|
||
]
|
||
for _row in _H_CASES:
|
||
_pat, _subj = _row[0], _row[1]
|
||
_mode = _row[2] if len(_row) > 2 else "binary"
|
||
_fl = _row[3] if len(_row) > 3 else ()
|
||
add("search", _pat, _subj, mode=_mode, flags=_fl)
|
||
add("finditer", _pat, _subj, mode=_mode, flags=_fl)
|
||
|
||
# Raw-byte sub/split: replacement templates and split still need to work
|
||
# correctly with embedded NUL and high bytes on both sides.
|
||
_H_SUB_CASES = [
|
||
(r"\x00", b"a\x00b\x00c", "-"),
|
||
(r"[\x80-\xff]", b"a\x80b\x90c", "?"),
|
||
(r"(.)\x00", b"a\x00b\x00", r"[\1]"),
|
||
]
|
||
for _pat, _subj, _repl in _H_SUB_CASES:
|
||
add("sub", _pat, _subj, mode="binary", repl=_repl)
|
||
|
||
_H_SPLIT_CASES = [
|
||
(r"\x00+", b"a\x00\x00b\x00c"),
|
||
(r"[\x80-\xff]", b"a\x80b\x90c\xa0d"),
|
||
]
|
||
for _pat, _subj in _H_SPLIT_CASES:
|
||
add("split", _pat, _subj, mode="binary")
|
||
|
||
# =============================================================================
|
||
# Category I: UTF-8 edge cases -- real multi-byte content beyond simple
|
||
# accented Latin: 4-byte (astral plane) code points, combining marks,
|
||
# right-to-left scripts, CJK, mixed-width strings, and offset correctness
|
||
# for backreferences/groups/lookaround spanning multi-byte characters.
|
||
# =============================================================================
|
||
_I_CASES = [
|
||
(r"\w+", "hello \U0001F600 world"), # emoji, a 4-byte code point
|
||
(r".", "\U0001F600"),
|
||
(r"\W", "\U0001F600"),
|
||
(r"(.)", "\U0001F600\U0001F601"),
|
||
(r"é", "é"), # 'e' + combining acute accent
|
||
(r"\w+", "éclair"),
|
||
(r"[-ۿ]+", "السلام"), # Arabic
|
||
(r"\w+", "שלום"), # Hebrew
|
||
(r"[一-鿿]+", "中文字符"), # CJK unified ideographs
|
||
(r"\w+", "日本語"), # Japanese
|
||
(r"(\w)(\w)\2\1", "文字字文"),
|
||
(r"(.)\1+", "\U0001F600\U0001F600\U0001F600"),
|
||
(r"^.{3}$", "aé\U0001F600"), # mixed 1/2/4-byte codepoints, fixed count
|
||
(r"\bworld\b", "café world café"),
|
||
(r"(?<=é)\w+", "cafémonde"),
|
||
(r"(?=\U0001F600)", "x\U0001F600y"),
|
||
(r"[^\x00-\x7f]+", "abcéèêdef"),
|
||
# Fullwidth digits (U+FF10-FF19, Unicode category Nd) are deliberately
|
||
# not probed here: CPython's \d matches them (it matches the Nd
|
||
# category), but glibc's iswdigit() under the C.utf8 locale this
|
||
# build relies on does not, the same class of Unicode-table gap as
|
||
# the NBSP case in README.md "Known deviations", which now also
|
||
# records this specific instance.
|
||
(r"café", "CAFÉ", "utf8", ("IGNORECASE",)),
|
||
(r"ß", "ß", "utf8", ("IGNORECASE",)), # German sharp s
|
||
]
|
||
for _row in _I_CASES:
|
||
_pat, _subj = _row[0], _row[1]
|
||
_mode = _row[2] if len(_row) > 2 else "utf8"
|
||
_fl = _row[3] if len(_row) > 3 else ()
|
||
add("search", _pat, _subj, mode=_mode, flags=_fl)
|
||
add("fullmatch", _pat, _subj, mode=_mode, flags=_fl)
|
||
add("finditer", _pat, _subj, mode=_mode, flags=_fl)
|
||
|
||
_I_SUB_CASES = [
|
||
(r"\U0001F600", "hi \U0001F600 there", ":)"),
|
||
(r"(\w)", "café", r"[\1]"),
|
||
(r"[一-鿿]", "中文abc", "?"),
|
||
]
|
||
for _pat, _subj, _repl in _I_SUB_CASES:
|
||
add("sub", _pat, _subj, mode="utf8", repl=_repl)
|
||
|
||
_I_SPLIT_CASES = [
|
||
(r"\s+", "café au lait 中文"),
|
||
(r"(\U0001F600)", "a\U0001F600b\U0001F600c"),
|
||
]
|
||
for _pat, _subj in _I_SPLIT_CASES:
|
||
add("split", _pat, _subj, mode="utf8")
|
||
|
||
# =============================================================================
|
||
# Category J: seeded random combinatorics. A fixed seed makes this a
|
||
# deterministic, reproducible regression test, not a flaky fuzzer: the same
|
||
# cases are generated every run, so a failure here is exactly as
|
||
# reproducible and reportable as a hand-written one. Explores combinations
|
||
# no one sat down and thought to write by hand, across all three modes.
|
||
# =============================================================================
|
||
import random as _random # noqa: E402
|
||
|
||
_rng = _random.Random(0xC0FFEE)
|
||
|
||
_J_ASCII_ATOMS = ["a", "b", "c", "1", "2", "_", " ", "[a-c]", "[0-9]", r"\d", r"\w", r"\s", "."]
|
||
_J_UTF8_ATOMS = _J_ASCII_ATOMS + ["é", "è", "中", "ا", "\U0001F600"]
|
||
_J_QUANTS = ["", "?", "*", "+", "{1,2}", "{0,2}", "*?", "+?", "??"]
|
||
_J_WRAPS = ["{0}", "({0})", "(?:{0})", "(?P<g%d>{0})"]
|
||
|
||
_J_ASCII_SUBJ_CHARS = list("abc123_ ")
|
||
_J_UTF8_SUBJ_CHARS = _J_ASCII_SUBJ_CHARS + ["é", "è", "中", "文", "ا", "\U0001F600", "\U0001F601"]
|
||
|
||
|
||
_J_SPECIAL_ATOMS = {r"\d", r"\w", r"\s", ".", "[a-c]", "[0-9]"}
|
||
|
||
|
||
def _j_random_fragment(mode, group_no):
|
||
atoms = _J_UTF8_ATOMS if mode == "utf8" else _J_ASCII_ATOMS
|
||
atom = _rng.choice(atoms)
|
||
if atom not in _J_SPECIAL_ATOMS:
|
||
atom = re.escape(atom)
|
||
q = _rng.choice(_J_QUANTS)
|
||
wrap = _rng.choice(_J_WRAPS)
|
||
if "%d" in wrap:
|
||
wrap = wrap % group_no
|
||
return wrap.format(atom + q)
|
||
|
||
|
||
def _j_random_pattern(mode):
|
||
n = _rng.randint(1, 4)
|
||
group_no = 1
|
||
parts = []
|
||
for _ in range(n):
|
||
frag = _j_random_fragment(mode, group_no)
|
||
if frag.startswith("(") and not frag.startswith("(?:"):
|
||
group_no += 1
|
||
parts.append(frag)
|
||
joiner = "|" if _rng.random() < 0.25 else ""
|
||
return joiner.join(parts)
|
||
|
||
|
||
def _j_random_subject(mode, length):
|
||
chars = _J_UTF8_SUBJ_CHARS if mode == "utf8" else _J_ASCII_SUBJ_CHARS
|
||
return "".join(_rng.choice(chars) for _ in range(length))
|
||
|
||
|
||
_J_FLAG_CHOICES = [[], ["IGNORECASE"], ["MULTILINE"], ["DOTALL"], ["IGNORECASE", "MULTILINE"]]
|
||
_J_MODES = ["ascii", "utf8", "binary"]
|
||
_J_OPS = ["search", "finditer", "fullmatch"]
|
||
|
||
# Patterns/subjects for "binary" mode are still built from the ASCII-safe
|
||
# atom pool (Category H already covers genuinely arbitrary, non-UTF-8
|
||
# binary content deliberately and by hand); here binary mode exercises
|
||
# byte-for-byte matching of ordinary ASCII text through the BINARY
|
||
# encoding path specifically, distinct from the same text through ASCII
|
||
# or UTF8 mode, which Category F already probes with a fixed pattern set.
|
||
_j_generated = 0
|
||
for _ in range(900):
|
||
_mode = _rng.choice(_J_MODES)
|
||
_atom_mode = "utf8" if _mode == "utf8" else "ascii"
|
||
_pat = _j_random_pattern(_atom_mode)
|
||
_subj = _j_random_subject(_atom_mode, _rng.randint(0, 10))
|
||
_fl = _rng.choice(_J_FLAG_CHOICES)
|
||
_op = _rng.choice(_J_OPS)
|
||
add(_op, _pat, _subj, mode=_mode, flags=_fl)
|
||
_j_generated += 1
|