65 lines
2.7 KiB
C
65 lines
2.7 KiB
C
/* binary_scan - demonstrates BINARY mode: matching against raw bytes with
|
|||
|
|
* no decoding step at all, where every value 0-255 is an ordinary input
|
||
|
|
* character, including 0x00 (NUL) and 0x0A (the byte '\n' means in text).
|
||
|
|
* This is the one thing ASCII and UTF8 mode structurally cannot do: ASCII
|
||
|
|
* mode still treats the subject as bytes but restricts \d/\w/\s to the
|
||
|
|
* ASCII subset, and UTF8 mode decodes the subject before matching at all,
|
||
|
|
* so an arbitrary byte class spanning 0x00-0xFF describes decoded code
|
||
|
|
* points there, not raw bytes on the wire.
|
||
|
|
*
|
||
|
|
* The scenario: a small binary record format concatenated in one buffer,
|
||
|
|
* each record a 2-byte magic number followed by 4 raw payload bytes that
|
||
|
|
* may be anything, including bytes a C string function would treat as a
|
||
|
|
* terminator or a line break. A pattern built from re.escape'd magic bytes
|
||
|
|
* and a [\x00-\xff]{4} class extracts every record correctly regardless.
|
||
|
|
*/
|
||
|
|
#include "regexx.h"
|
||
|
|
#include <stdio.h>
|
||
|
|
#include <string.h>
|
||
|
|
|
||
|
|
typedef struct { int n; } Ctx;
|
||
|
|
|
||
|
|
static void print_record(void *ctx, const Match *m_const) {
|
||
|
|
Ctx *c = ctx;
|
||
|
|
Match *m = (Match *)m_const;
|
||
|
|
const char *g; size_t glen;
|
||
|
|
Match_group(m, NULL, 0, &g, &glen);
|
||
|
|
printf("record %d: magic=%02x%02x payload=", c->n, (unsigned char)g[0], (unsigned char)g[1]);
|
||
|
|
for (size_t i = 2; i < glen; i++) printf("%02x ", (unsigned char)g[i]);
|
||
|
|
printf("\n");
|
||
|
|
c->n++;
|
||
|
|
}
|
||
|
|
|
||
|
|
int main(void) {
|
||
|
|
/* Three records, back to back, with no separator: a real length-value
|
||
|
|
* binary format would carry its own framing; this example only needs
|
||
|
|
* a fixed-size payload to keep the pattern simple. Payload bytes
|
||
|
|
* deliberately include 0x00 and 0x0a, the two values that would break
|
||
|
|
* a NUL-terminated-string or line-oriented approach to scanning this
|
||
|
|
* same buffer. */
|
||
|
|
uint8_t data[] = {
|
||
|
|
0xca, 0xfe, 0x01, 0x00, 0x00, 0x00, /* record 0: payload has an embedded NUL */
|
||
|
|
0xca, 0xfe, 0x0a, 0x0a, 0xff, 0x7f, /* record 1: payload has embedded 0x0a bytes */
|
||
|
|
0xca, 0xfe, 0xde, 0xad, 0xbe, 0xef, /* record 2: ordinary payload */
|
||
|
|
};
|
||
|
|
|
||
|
|
const char *pattern = "\\xca\\xfe[\\x00-\\xff]{4}";
|
||
|
|
PatternError err; memset(&err, 0, sizeof err);
|
||
|
|
Pattern *pat = re_compile(pattern, strlen(pattern), BINARY, &err);
|
||
|
|
if (!pat) {
|
||
|
|
fprintf(stderr, "compile error: %s\n", err.msg);
|
||
|
|
PatternError_free(&err);
|
||
|
|
return 1;
|
||
|
|
}
|
||
|
|
|
||
|
|
Input *in = Input_from_buffer(data, sizeof data);
|
||
|
|
Ctx ctx = { 0 };
|
||
|
|
int n = Pattern_finditer(pat, in, 0, -1, print_record, &ctx);
|
||
|
|
printf("total records found: %d (buffer is %zu bytes with %d embedded NUL, %d embedded 0x0a)\n",
|
||
|
|
n, sizeof data, 1, 2);
|
||
|
|
|
||
|
|
Input_free(in);
|
||
|
|
Pattern_free(pat);
|
||
|
|
return 0;
|
||
|
|
}
|