Each new program under examples/ isolates one distinct feature rather than being a general purpose tool like the existing rxgrep.c: binary_scan.c (raw byte-range classes including an embedded NUL and an embedded 0x0A), utf8_scripts.c (\w across Latin/Greek/Cyrillic/CJK text, code point versus byte offsets), ascii_logparse.c (named groups against structured log text), redos_atomic.c (atomic groups and possessive quantifiers timed directly against the unprotected form of the textbook (a+)+b ReDoS shape), empty_match_rule.c (CPython's undocumented empty-match retry rule, verified: \d*? against "123abc456" gives 16 matches, not 9), and large_file_search.c (Input_from_file's mmap-backed reading on a generated 100MB file, with elapsed time and peak RSS printed). Every example was compiled and run while writing it; the claims in each file's top comment are checked against its own output, not written by hand and left unverified. Also adds examples/bench_vs_posix.c, a direct, honestly reported comparison against the C standard library's own <regex.h> (regcomp/regexec) on six scenarios at multi-megabyte or multi-hundred-thousand-line scale, using only pattern syntax valid for both engines so they run the identical pattern text. glibc's DFA-backed engine wins five of six scenarios by 2x-35x, which is the expected outcome of a roughly 2000-line backtracking interpreter built for Python `re` compatibility competing against a mature, heavily optimized engine with a much smaller feature set; the sixth scenario has no POSIX equivalent at all (an atomic group). Every scenario's match count is cross-checked between the two engines as an independent correctness signal beyond the existing CPython-derived test suite. Two real issues were found and fixed while building this benchmark, not left in: iterating regexec() over an advancing string pointer is quadratic in practice (no way to bound the search without an implicit NUL-scan on every call), fixed by using REG_STARTEND instead; and a signed integer overflow (undefined behavior, caught by UBSan) in the benchmark's own pseudo-random text generator, fixed by using an unsigned accumulator. README.md and USAGE.md gain pointers to examples/README.md (the new per-example index) and a "Benchmarks" section summarizing the POSIX comparison honestly, including where it loses. The Makefile gains a `make examples` target building all seven programs. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01EjuMk8kY9SDus1wWe2K9xY
72 lines
3.2 KiB
C
72 lines
3.2 KiB
C
/* ascii_logparse - demonstrates ASCII mode doing what it is actually for:
|
|
* parsing structured, line-oriented text with named groups, then reading
|
|
* fields back out by name rather than by remembering a numeric position.
|
|
* This is the ordinary, common case the other examples in this directory
|
|
* deliberately are not (they each isolate one less-common feature
|
|
* instead); it is included so the directory has one example that looks
|
|
* like typical real-world usage: log filtering and field extraction.
|
|
*/
|
|
#include "regexx.h"
|
|
#include <stdio.h>
|
|
#include <string.h>
|
|
|
|
static const char *LOG =
|
|
"2024-01-15 10:23:45 INFO [auth] user 'alice' logged in from 10.0.0.5\n"
|
|
"2024-01-15 10:24:02 ERROR [db] connection refused to 10.0.0.9:5432\n"
|
|
"2024-01-15 10:24:03 WARN [auth] repeated login failure for user 'bob'\n"
|
|
"2024-01-15 10:25:11 ERROR [auth] user 'bob' locked out after 5 failures\n"
|
|
"2024-01-15 10:26:00 INFO [db] connection pool resized to 32\n";
|
|
|
|
int main(void) {
|
|
const char *pattern =
|
|
"(?P<date>\\d{4}-\\d{2}-\\d{2}) (?P<time>\\d{2}:\\d{2}:\\d{2}) "
|
|
"(?P<level>\\w+)\\s+\\[(?P<component>\\w+)\\] (?P<message>.+)";
|
|
PatternError err; memset(&err, 0, sizeof err);
|
|
Pattern *pat = re_compile(pattern, strlen(pattern), ASCII, &err);
|
|
if (!pat) {
|
|
fprintf(stderr, "compile error: %s\n", err.msg);
|
|
PatternError_free(&err);
|
|
return 1;
|
|
}
|
|
|
|
printf("-- every ERROR line, fields read back by name --\n");
|
|
size_t linelen = strlen(LOG);
|
|
size_t start = 0;
|
|
int error_count = 0;
|
|
for (size_t pos = 0; pos <= linelen; pos++) {
|
|
if (pos == linelen || LOG[pos] == '\n') {
|
|
size_t len = pos - start;
|
|
if (len > 0) {
|
|
Input *line = Input_from_buffer((const uint8_t *)(LOG + start), len);
|
|
Match m; memset(&m, 0, sizeof m);
|
|
if (Pattern_fullmatch(pat, line, 0, -1, &m) == 1) {
|
|
const char *level; size_t levellen;
|
|
Match_group(&m, "level", 0, &level, &levellen);
|
|
if (levellen == 5 && strncmp(level, "ERROR", 5) == 0) {
|
|
const char *time, *component, *message;
|
|
size_t timelen, componentlen, messagelen;
|
|
Match_group(&m, "time", 0, &time, &timelen);
|
|
Match_group(&m, "component", 0, &component, &componentlen);
|
|
Match_group(&m, "message", 0, &message, &messagelen);
|
|
printf("[%.*s] %.*s: %.*s\n",
|
|
(int)timelen, time,
|
|
(int)componentlen, component,
|
|
(int)messagelen, message);
|
|
error_count++;
|
|
}
|
|
Match_free(&m);
|
|
}
|
|
Input_free(line);
|
|
}
|
|
start = pos + 1;
|
|
}
|
|
}
|
|
printf("\n%d error line(s) found (Pattern_groupindex_lookup(\"message\") = %d,\n"
|
|
"used internally by name lookup so this program never hardcodes a\n"
|
|
"capturing group's numeric position)\n",
|
|
error_count, Pattern_groupindex_lookup(pat, "message"));
|
|
|
|
Pattern_free(pat);
|
|
return 0;
|
|
}
|