README.md and docs/API.md both previously documented "no way to list every name a pattern defines without already knowing what to look for" as an accepted limitation of Pattern_groupindex_lookup being the only access to groupindex. It was not actually a hard constraint: the underlying GroupIndex struct (regexx.c) already stores every name and its group number in two parallel arrays, populated once at compile time; only a public accessor was missing. Pattern_groupindex_count returns the number of named groups; Pattern_groupindex_at(self, i, &name) for 0 <= i < count writes the i-th name and returns its 1-based group number, or returns -1 for an out-of-range i. Enumeration order is declaration order, verified against a real CPython 3.11 interpreter to match groupindex's own practical (insertion-order) iteration order, not just assumed. Verified directly (count/name/group-number correctness, matching the exact snippet now in USAGE.md's own output), full 3,252-case suite unaffected (3252/3252, this is a pure accessor addition touching no matching logic), clean AddressSanitizer/UndefinedBehaviorSanitizer. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01EjuMk8kY9SDus1wWe2K9xY
152 lines
7.8 KiB
C
152 lines
7.8 KiB
C
/* regexx.h - public API for regexx, a single-file C regex interpreter
|
|
* with Python `re` semantics. See concept.md for the design rationale
|
|
* and Section 9 in particular for the naming convention this header
|
|
* follows: every identifier with a direct Python `re` counterpart uses
|
|
* that counterpart's exact spelling.
|
|
*
|
|
* Implementation status relative to concept.md: this is the v1
|
|
* implementation. It provides the full public API and pattern syntax
|
|
* described below, executed by one of two engines over a fully
|
|
* materialized copy of the input, chosen automatically per compiled
|
|
* pattern (concept.md Section 7.2/7.7): a Pike VM (a Thompson-NFA
|
|
* simulation, genuinely linear time, no backtracking at all) for any
|
|
* pattern with no backreference, lookaround, or atomic group; a
|
|
* recursive backtracking engine (concept.md Section 7.3), augmented
|
|
* with memoization and precomputed skip-ahead tables (regexx.c:
|
|
* run_memo, compute_maxrun, compute_next_prevmatch) that make the
|
|
* search-family functions (Pattern_search/finditer/split/sub) run in
|
|
* linear time for the common, backreference-free case there too,
|
|
* otherwise. Neither engine yet implements the streaming, bounded-
|
|
* memory version of Section 7.2; see README.md "Implementation status"
|
|
* and "The Pike VM" for the complete, itemized, measured account of
|
|
* what v1 does and does not cover, including a real, documented
|
|
* performance trade-off between the two engines that is not yet
|
|
* resolved (concept.md 7.8).
|
|
*/
|
|
#ifndef REGEXX_H
|
|
#define REGEXX_H
|
|
|
|
#include <stddef.h>
|
|
#include <stdint.h>
|
|
|
|
#ifdef __cplusplus
|
|
extern "C" {
|
|
#endif
|
|
|
|
/* ---- Flags (re.compile flags; bitwise OR, exact Python names) ---- */
|
|
#define IGNORECASE 0x0001
|
|
#define MULTILINE 0x0002
|
|
#define DOTALL 0x0004
|
|
#define VERBOSE 0x0008
|
|
#define ASCII 0x0010
|
|
#define UNICODE 0x0020 /* default for UTF8 mode; explicit for clarity */
|
|
#define LOCALE 0x0040
|
|
#define DEBUG 0x0080
|
|
|
|
/* Encoding mode, not a Python flag (Section 9.3): exactly one required. */
|
|
#define BINARY 0x0100
|
|
#define UTF8 0x0200
|
|
/* ASCII (0x0010) doubles as the ASCII *encoding* mode when neither
|
|
* BINARY nor UTF8 is given; see regexx.c encoding-mode resolution. */
|
|
|
|
/* ---- Types ---- */
|
|
|
|
typedef struct Pattern Pattern; /* re.Pattern */
|
|
typedef struct Match Match; /* re.Match */
|
|
typedef struct Input Input; /* no Python counterpart, see concept.md 9.0/9.2 */
|
|
|
|
typedef void (*MatchIterCb)(void *ctx, const Match *m);
|
|
typedef void (*MatchSubCb)(void *ctx, const Match *m, char **out, size_t *outlen);
|
|
|
|
typedef struct PatternError PatternError; /* re.error / re.PatternError */
|
|
struct PatternError {
|
|
const char *msg; /* re.error.msg */
|
|
const char *pattern; /* re.error.pattern */
|
|
int64_t pos; /* re.error.pos */
|
|
int64_t lineno; /* re.error.lineno */
|
|
int64_t colno; /* re.error.colno */
|
|
};
|
|
/* msg and pattern are heap allocated by whichever call filled this
|
|
* struct in; call PatternError_free once you are done reading it
|
|
* (safe to call on a zero-initialized or already-freed PatternError). */
|
|
void PatternError_free(PatternError *err);
|
|
|
|
struct Pattern {
|
|
const char *pattern; /* re.Pattern.pattern */
|
|
int flags; /* re.Pattern.flags */
|
|
int groups; /* re.Pattern.groups */
|
|
void *groupindex; /* re.Pattern.groupindex, name -> group number */
|
|
void *program; /* compiled bytecode, private */
|
|
};
|
|
|
|
struct Match {
|
|
Pattern *re; /* re.Match.re */
|
|
Input *string; /* re.Match.string */
|
|
int64_t pos, endpos; /* re.Match.pos, re.Match.endpos */
|
|
int lastindex; /* re.Match.lastindex */
|
|
const char *lastgroup; /* re.Match.lastgroup */
|
|
void *slots; /* private */
|
|
};
|
|
|
|
/* ---- Input construction (no Python counterpart) ---- */
|
|
Input *Input_from_buffer(const uint8_t *buf, size_t len); /* does not copy or take ownership */
|
|
Input *Input_from_file(const char *path, PatternError *err); /* seekable or not (pipes, "-" via /dev/stdin, etc. all work) */
|
|
void Input_free(Input *in);
|
|
|
|
/* ---- Pattern methods (bound to an already compiled Pattern) ---- */
|
|
int Pattern_match(Pattern *self, Input *string, int64_t pos, int64_t endpos, Match *out);
|
|
int Pattern_fullmatch(Pattern *self, Input *string, int64_t pos, int64_t endpos, Match *out);
|
|
int Pattern_search(Pattern *self, Input *string, int64_t pos, int64_t endpos, Match *out);
|
|
int Pattern_finditer(Pattern *self, Input *string, int64_t pos, int64_t endpos, MatchIterCb cb, void *ctx);
|
|
int Pattern_findall(Pattern *self, Input *string, int64_t pos, int64_t endpos, MatchIterCb cb, void *ctx);
|
|
/* cb fires once per element of the list re.split() would return, in
|
|
* order; read each element's text via Match_group(m, NULL, 0, ...). */
|
|
int Pattern_split(Pattern *self, Input *string, int maxsplit, MatchIterCb cb, void *ctx);
|
|
int Pattern_sub(Pattern *self, Input *string, const char *repl, MatchSubCb cb, void *ctx, int count, char **out, size_t *outlen);
|
|
int Pattern_subn(Pattern *self, Input *string, const char *repl, MatchSubCb cb, void *ctx, int count, char **out, size_t *outlen, int *n);
|
|
void Pattern_free(Pattern *self);
|
|
|
|
/* Looks up a named group in re.Pattern.groupindex; returns its 1-based
|
|
* group number, or -1 if no group by that name exists. */
|
|
int Pattern_groupindex_lookup(Pattern *self, const char *name);
|
|
/* Enumerates re.Pattern.groupindex as a whole (Python: len(pattern.
|
|
* groupindex), dict(pattern.groupindex).items()), in declaration order
|
|
* (Python's own dict does not guarantee an order at the language level,
|
|
* but in practice preserves insertion order the same way this does).
|
|
* Pattern_groupindex_count returns the number of named groups.
|
|
* Pattern_groupindex_at(self, i, &name) for 0 <= i < count writes a
|
|
* borrowed pointer (valid as long as self is) to the i-th name into
|
|
* *name and returns its 1-based group number; out of range i returns
|
|
* -1 and leaves *name untouched. */
|
|
int Pattern_groupindex_count(Pattern *self);
|
|
int Pattern_groupindex_at(Pattern *self, int i, const char **name);
|
|
|
|
/* ---- Match accessors ---- */
|
|
int Match_group(Match *self, const char *name_or_null, int index, const char **out, size_t *outlen);
|
|
int64_t Match_start(Match *self, int group);
|
|
int64_t Match_end(Match *self, int group);
|
|
void Match_span(Match *self, int group, int64_t *start, int64_t *end);
|
|
int64_t Match_start_byte(Match *self, int group);
|
|
int64_t Match_end_byte(Match *self, int group);
|
|
void Match_span_byte(Match *self, int group, int64_t *start, int64_t *end);
|
|
void Match_free(Match *self);
|
|
|
|
/* ---- Module level functions ---- */
|
|
Pattern *re_compile(const char *pattern, size_t len, int flags, PatternError *err);
|
|
int re_match(const char *pattern, size_t len, int flags, Input *string, Match *out);
|
|
int re_fullmatch(const char *pattern, size_t len, int flags, Input *string, Match *out);
|
|
int re_search(const char *pattern, size_t len, int flags, Input *string, Match *out);
|
|
int re_finditer(const char *pattern, size_t len, int flags, Input *string, MatchIterCb cb, void *ctx);
|
|
int re_findall(const char *pattern, size_t len, int flags, Input *string, MatchIterCb cb, void *ctx);
|
|
int re_split(const char *pattern, size_t len, int flags, Input *string, int maxsplit, MatchIterCb cb, void *ctx);
|
|
int re_sub(const char *pattern, size_t len, int flags, Input *string, const char *repl, MatchSubCb cb, void *ctx, int count, char **out, size_t *outlen);
|
|
int re_subn(const char *pattern, size_t len, int flags, Input *string, const char *repl, MatchSubCb cb, void *ctx, int count, char **out, size_t *outlen, int *n);
|
|
void re_escape(const char *in, size_t len, char **out, size_t *outlen);
|
|
void re_purge(void);
|
|
|
|
#ifdef __cplusplus
|
|
}
|
|
#endif
|
|
|
|
#endif /* REGEXX_H */
|