/* utf8_scripts - demonstrates UTF8 mode: the subject is decoded into * Unicode code points before matching (not just treated as opaque bytes, * the way ASCII/BINARY mode do), so \w and IGNORECASE work correctly * across scripts, not only ASCII letters, and match positions come in two * different units depending on which accessor is used. * * The subject deliberately mixes Latin (with a combining-free accented * letter), Greek, Cyrillic, and a CJK ideograph, each of which encodes to * a different number of UTF-8 bytes per code point (1, 2, 2, and 3 * respectively), to make the code-point/byte distinction concrete rather * than something that happens to not matter for the specific example * chosen. */ #include "regexx.h" #include #include typedef struct { int n; } Ctx; static void print_word(void *ctx, const Match *m_const) { Ctx *c = ctx; Match *m = (Match *)m_const; const char *g; size_t glen; Match_group(m, NULL, 0, &g, &glen); int64_t cs, ce, bs, be; Match_span(m, 0, &cs, &ce); Match_span_byte(m, 0, &bs, &be); printf("word %d: '%.*s' codepoints=[%lld,%lld) (len %lld) bytes=[%lld,%lld) (len %lld)\n", c->n, (int)glen, g, (long long)cs, (long long)ce, (long long)(ce - cs), (long long)bs, (long long)be, (long long)(be - bs)); c->n++; } int main(void) { /* "café" (Latin, e-acute is 2 bytes) + " " + "Ελλάδα" (Greek) + " " * + "Россия" (Cyrillic) + " " + "日本" (CJK, 3 bytes per character). */ const char *subject = "caf\xc3\xa9 " "\xce\x95\xce\xbb\xce\xbb\xce\xac\xce\xb4\xce\xb1 " "\xd0\xa0\xd0\xbe\xd1\x81\xd1\x81\xd0\xb8\xd1\x8f " "\xe6\x97\xa5\xe6\x9c\xac"; const char *pattern = "\\w+"; PatternError err; memset(&err, 0, sizeof err); Pattern *pat = re_compile(pattern, strlen(pattern), UTF8, &err); if (!pat) { fprintf(stderr, "compile error: %s\n", err.msg); PatternError_free(&err); return 1; } Input *in = Input_from_buffer((const uint8_t *)subject, strlen(subject)); printf("subject is %zu bytes\n\n", strlen(subject)); Ctx ctx = { 0 }; int n = Pattern_finditer(pat, in, 0, -1, print_word, &ctx); printf("\ntotal words: %d\n", n); printf("\\w under UTF8 mode consults the platform's Unicode tables (glibc\n" "wctype.h under the C.utf8 locale, docs/API.md Section 2), so it\n" "recognizes Greek, Cyrillic, and CJK letters, not only ASCII ones;\n" "the byte length of each word differs from its codepoint length\n" "exactly where the script needs more than one byte per character.\n"); Input_free(in); Pattern_free(pat); return 0; }