68 lines
2.7 KiB
C
68 lines
2.7 KiB
C
/* utf8_scripts - demonstrates UTF8 mode: the subject is decoded into
|
|||
|
|
* Unicode code points before matching (not just treated as opaque bytes,
|
||
|
|
* the way ASCII/BINARY mode do), so \w and IGNORECASE work correctly
|
||
|
|
* across scripts, not only ASCII letters, and match positions come in two
|
||
|
|
* different units depending on which accessor is used.
|
||
|
|
*
|
||
|
|
* The subject deliberately mixes Latin (with a combining-free accented
|
||
|
|
* letter), Greek, Cyrillic, and a CJK ideograph, each of which encodes to
|
||
|
|
* a different number of UTF-8 bytes per code point (1, 2, 2, and 3
|
||
|
|
* respectively), to make the code-point/byte distinction concrete rather
|
||
|
|
* than something that happens to not matter for the specific example
|
||
|
|
* chosen.
|
||
|
|
*/
|
||
|
|
#include "regexx.h"
|
||
|
|
#include <stdio.h>
|
||
|
|
#include <string.h>
|
||
|
|
|
||
|
|
typedef struct { int n; } Ctx;
|
||
|
|
|
||
|
|
static void print_word(void *ctx, const Match *m_const) {
|
||
|
|
Ctx *c = ctx;
|
||
|
|
Match *m = (Match *)m_const;
|
||
|
|
const char *g; size_t glen;
|
||
|
|
Match_group(m, NULL, 0, &g, &glen);
|
||
|
|
int64_t cs, ce, bs, be;
|
||
|
|
Match_span(m, 0, &cs, &ce);
|
||
|
|
Match_span_byte(m, 0, &bs, &be);
|
||
|
|
printf("word %d: '%.*s' codepoints=[%lld,%lld) (len %lld) bytes=[%lld,%lld) (len %lld)\n",
|
||
|
|
c->n, (int)glen, g,
|
||
|
|
(long long)cs, (long long)ce, (long long)(ce - cs),
|
||
|
|
(long long)bs, (long long)be, (long long)(be - bs));
|
||
|
|
c->n++;
|
||
|
|
}
|
||
|
|
|
||
|
|
int main(void) {
|
||
|
|
/* "café" (Latin, e-acute is 2 bytes) + " " + "Ελλάδα" (Greek) + " "
|
||
|
|
* + "Россия" (Cyrillic) + " " + "日本" (CJK, 3 bytes per character). */
|
||
|
|
const char *subject =
|
||
|
|
"caf\xc3\xa9 "
|
||
|
|
"\xce\x95\xce\xbb\xce\xbb\xce\xac\xce\xb4\xce\xb1 "
|
||
|
|
"\xd0\xa0\xd0\xbe\xd1\x81\xd1\x81\xd0\xb8\xd1\x8f "
|
||
|
|
"\xe6\x97\xa5\xe6\x9c\xac";
|
||
|
|
|
||
|
|
const char *pattern = "\\w+";
|
||
|
|
PatternError err; memset(&err, 0, sizeof err);
|
||
|
|
Pattern *pat = re_compile(pattern, strlen(pattern), UTF8, &err);
|
||
|
|
if (!pat) {
|
||
|
|
fprintf(stderr, "compile error: %s\n", err.msg);
|
||
|
|
PatternError_free(&err);
|
||
|
|
return 1;
|
||
|
|
}
|
||
|
|
|
||
|
|
Input *in = Input_from_buffer((const uint8_t *)subject, strlen(subject));
|
||
|
|
printf("subject is %zu bytes\n\n", strlen(subject));
|
||
|
|
Ctx ctx = { 0 };
|
||
|
|
int n = Pattern_finditer(pat, in, 0, -1, print_word, &ctx);
|
||
|
|
printf("\ntotal words: %d\n", n);
|
||
|
|
printf("\\w under UTF8 mode consults the platform's Unicode tables (glibc\n"
|
||
|
|
"wctype.h under the C.utf8 locale, docs/API.md Section 2), so it\n"
|
||
|
|
"recognizes Greek, Cyrillic, and CJK letters, not only ASCII ones;\n"
|
||
|
|
"the byte length of each word differs from its codepoint length\n"
|
||
|
|
"exactly where the script needs more than one byte per character.\n");
|
||
|
|
|
||
|
|
Input_free(in);
|
||
|
|
Pattern_free(pat);
|
||
|
|
return 0;
|
||
|
|
}
|