Add regexx: a single-file C regex interpreter with Python re semantics
concept.md is the full design document: the objective (Python re parity plus binary/ASCII/UTF-8 modes, gigabyte-scale input, single C file), the automata-theory argument for why unrestricted backreferences/lookaround are incompatible with strict single-pass constant memory, the resulting two-engine architecture, the exact Python-mirroring naming convention, and a full comparison against POSIX regex.h for C-background readers. regexx.c/regexx.h are the v1 implementation: parser, compiler to a Pike/backtracking-style bytecode, and a single recursive backtracking engine covering the pattern syntax and operations listed in README.md, validated against CPython's own re module output (tests/), clean under AddressSanitizer/UBSan, and stress-tested (50MB simple-quantifier match, graceful failure rather than a crash on complex repeats over large input, clean rejection of every intentionally unsupported construct). Also included: examples/rxgrep.c (a small grep-like program exercising all three data modes and the substitution API), the Makefile, the MIT LICENSE, and docs/API.md, an exhaustive reference for every type, flag, and function's exact return-value and memory-ownership convention, checked against the current source and against a real CPython interpreter rather than against memory. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01EjuMk8kY9SDus1wWe2K9xY
This commit is contained in:
@@ -0,0 +1,159 @@
|
||||
/* rxgrep - a small grep-like example program built on regexx, demonstrating
|
||||
* the library across its three data modes (Section 9.3) and both its
|
||||
* matching and substitution API (Section 9.1). See README.md for usage.
|
||||
*
|
||||
* This is example code, not part of the library; it is not held to the
|
||||
* same naming-convention rule as regexx.c/regexx.h (concept.md 9.0),
|
||||
* since it is an ordinary application, not part of the `re`-mirroring
|
||||
* surface.
|
||||
*/
|
||||
#include "regexx.h"
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
|
||||
static void usage(const char *prog) {
|
||||
fprintf(stderr,
|
||||
"usage: %s [-i] [-n] [-c] [-o] [-v] [-m MODE] PATTERN FILE\n"
|
||||
" %s [-m MODE] --sub REPL PATTERN FILE\n"
|
||||
"\n"
|
||||
" -i ignore case (IGNORECASE)\n"
|
||||
" -n prefix each match with its 1-based line number\n"
|
||||
" -c print only a count of matching lines\n"
|
||||
" -o print only the matched text, not the whole line\n"
|
||||
" -v print lines that do NOT match\n"
|
||||
" -m MODE ascii (default), utf8, or binary\n"
|
||||
" --sub REPL replace every match with REPL and print the result\n",
|
||||
prog, prog);
|
||||
}
|
||||
|
||||
static int mode_flag(const char *m) {
|
||||
if (strcmp(m, "ascii") == 0) return ASCII;
|
||||
if (strcmp(m, "utf8") == 0) return UTF8;
|
||||
if (strcmp(m, "binary") == 0) return BINARY;
|
||||
fprintf(stderr, "unknown mode '%s' (expected ascii, utf8, or binary)\n", m);
|
||||
exit(2);
|
||||
}
|
||||
|
||||
typedef struct { long lineno; int with_lineno; } MatchLinePrint;
|
||||
static void print_one_match(void *ctx, const Match *m_const) {
|
||||
MatchLinePrint *p = ctx;
|
||||
Match *m = (Match *)m_const;
|
||||
const char *g; size_t glen;
|
||||
Match_group(m, NULL, 0, &g, &glen);
|
||||
if (p->with_lineno) printf("%ld:", p->lineno);
|
||||
fwrite(g, 1, glen, stdout);
|
||||
printf("\n");
|
||||
}
|
||||
|
||||
int main(int argc, char **argv) {
|
||||
int flags = 0;
|
||||
int opt_n = 0, opt_c = 0, opt_o = 0, opt_v = 0;
|
||||
const char *sub_repl = NULL;
|
||||
int mode = ASCII;
|
||||
int i = 1;
|
||||
for (; i < argc; i++) {
|
||||
const char *a = argv[i];
|
||||
if (strcmp(a, "-m") == 0 && i + 1 < argc) { mode = mode_flag(argv[++i]); continue; }
|
||||
if (strcmp(a, "--sub") == 0 && i + 1 < argc) { sub_repl = argv[++i]; continue; }
|
||||
if (strcmp(a, "-h") == 0 || strcmp(a, "--help") == 0) { usage(argv[0]); return 0; }
|
||||
if (a[0] == '-' && a[1] != '-' && a[1] != '\0') {
|
||||
int bad = 0;
|
||||
for (const char *c = a + 1; *c; c++) {
|
||||
switch (*c) {
|
||||
case 'i': flags |= IGNORECASE; break;
|
||||
case 'n': opt_n = 1; break;
|
||||
case 'c': opt_c = 1; break;
|
||||
case 'o': opt_o = 1; break;
|
||||
case 'v': opt_v = 1; break;
|
||||
default: bad = 1; break;
|
||||
}
|
||||
}
|
||||
if (!bad) continue;
|
||||
}
|
||||
break;
|
||||
}
|
||||
if (argc - i != 2) { usage(argv[0]); return 2; }
|
||||
const char *pattern = argv[i], *path = argv[i + 1];
|
||||
|
||||
PatternError err; memset(&err, 0, sizeof err);
|
||||
Pattern *pat = re_compile(pattern, strlen(pattern), flags | mode, &err);
|
||||
if (!pat) {
|
||||
fprintf(stderr, "%s: pattern error at position %lld: %s\n", argv[0], (long long)err.pos, err.msg);
|
||||
return 2;
|
||||
}
|
||||
|
||||
if (sub_repl) {
|
||||
Input *file = Input_from_file(path, &err);
|
||||
if (!file) {
|
||||
fprintf(stderr, "%s: cannot read '%s': %s\n", argv[0], path, err.msg ? err.msg : "unknown error");
|
||||
Pattern_free(pat);
|
||||
return 2;
|
||||
}
|
||||
char *out; size_t outlen;
|
||||
int n;
|
||||
if (Pattern_subn(pat, file, sub_repl, NULL, NULL, 0, &out, &outlen, &n) < 0) {
|
||||
fprintf(stderr, "%s: matching failed (input too large/complex for this pattern, see README.md)\n", argv[0]);
|
||||
Input_free(file); Pattern_free(pat);
|
||||
return 1;
|
||||
}
|
||||
fwrite(out, 1, outlen, stdout);
|
||||
fprintf(stderr, "%d replacement(s)\n", n);
|
||||
free(out);
|
||||
Input_free(file); Pattern_free(pat);
|
||||
return n > 0 ? 0 : 1;
|
||||
}
|
||||
|
||||
/* Line oriented search: split on raw 0x0A, which is what every one
|
||||
* of the three modes (BINARY, ASCII, UTF8) agrees is a line break
|
||||
* (concept.md 7.2 notes 0x0A never appears inside a multi-byte
|
||||
* UTF-8 sequence), then search within each line's [pos, endpos).
|
||||
* Input's internals are intentionally opaque outside regexx.c
|
||||
* (concept.md 9.0), so each line is handed to the library as its
|
||||
* own small Input built with Input_from_buffer. */
|
||||
FILE *fp = fopen(path, "rb");
|
||||
if (!fp) { fprintf(stderr, "%s: cannot reopen '%s'\n", argv[0], path); return 2; }
|
||||
fseek(fp, 0, SEEK_END);
|
||||
long fsize = ftell(fp);
|
||||
rewind(fp);
|
||||
char *filebuf = malloc((size_t)fsize > 0 ? (size_t)fsize : 1);
|
||||
size_t got = fread(filebuf, 1, (size_t)fsize, fp);
|
||||
fclose(fp);
|
||||
|
||||
long lineno = 0;
|
||||
size_t line_start = 0;
|
||||
int matched_lines = 0;
|
||||
for (size_t pos = 0; pos <= got; pos++) {
|
||||
if (pos == got || filebuf[pos] == '\n') {
|
||||
lineno++;
|
||||
size_t line_len = pos - line_start;
|
||||
Input *lin = Input_from_buffer((const uint8_t *)(filebuf + line_start), line_len);
|
||||
Match m; memset(&m, 0, sizeof m);
|
||||
int r = Pattern_search(pat, lin, 0, -1, &m);
|
||||
int is_match = (r == 1);
|
||||
if (is_match) Match_free(&m);
|
||||
if (is_match != opt_v) {
|
||||
matched_lines++;
|
||||
if (!opt_c) {
|
||||
if (opt_o && is_match) {
|
||||
/* -o: every match on the line gets its own
|
||||
* output line, demonstrating Pattern_finditer. */
|
||||
MatchLinePrint ctx = { lineno, opt_n };
|
||||
Pattern_finditer(pat, lin, 0, -1, print_one_match, &ctx);
|
||||
} else {
|
||||
if (opt_n) printf("%ld:", lineno);
|
||||
fwrite(filebuf + line_start, 1, line_len, stdout);
|
||||
printf("\n");
|
||||
}
|
||||
}
|
||||
}
|
||||
Input_free(lin);
|
||||
line_start = pos + 1;
|
||||
}
|
||||
}
|
||||
if (opt_c) printf("%d\n", matched_lines);
|
||||
|
||||
free(filebuf);
|
||||
Pattern_free(pat);
|
||||
return matched_lines > 0 ? 0 : 1;
|
||||
}
|
||||
Reference in New Issue
Block a user