feat: add \U style 8-hex-digit unicode escape support in string literals
Extend the compiler's string parser to handle `\U` escapes with 8 hex digits for codepoints above U+FFFF, replacing the previous TODO comment. The `readUnicodeEscape` helper now accepts a variable digit count (4 or 8) and the `readString` function dispatches `\u` with length 4 and `\U` with length 8. New test cases verify correct encoding of emoji and ancient scripts, plus error handling for incomplete long escapes and values that fit in 4 digits.
This commit is contained in:
@@ -702,10 +702,10 @@ static int readHexEscape(Parser* parser, int digits, const char* description)
|
||||
return value;
|
||||
}
|
||||
|
||||
// Reads a four hex digit Unicode escape sequence in a string literal.
|
||||
static void readUnicodeEscape(Parser* parser, ByteBuffer* string)
|
||||
// Reads a hex digit Unicode escape sequence in a string literal.
|
||||
static void readUnicodeEscape(Parser* parser, ByteBuffer* string, int length)
|
||||
{
|
||||
int value = readHexEscape(parser, 4, "Unicode");
|
||||
int value = readHexEscape(parser, length, "Unicode");
|
||||
|
||||
// Grow the buffer enough for the encoded result.
|
||||
int numBytes = wrenUtf8EncodeNumBytes(value);
|
||||
@@ -750,8 +750,8 @@ static void readString(Parser* parser)
|
||||
case 'n': wrenByteBufferWrite(parser->vm, &string, '\n'); break;
|
||||
case 'r': wrenByteBufferWrite(parser->vm, &string, '\r'); break;
|
||||
case 't': wrenByteBufferWrite(parser->vm, &string, '\t'); break;
|
||||
case 'u': readUnicodeEscape(parser, &string); break;
|
||||
// TODO: 'U' for 8 octet Unicode escapes.
|
||||
case 'u': readUnicodeEscape(parser, &string, 4); break;
|
||||
case 'U': readUnicodeEscape(parser, &string, 8); break;
|
||||
case 'v': wrenByteBufferWrite(parser->vm, &string, '\v'); break;
|
||||
case 'x':
|
||||
wrenByteBufferWrite(parser->vm, &string,
|
||||
|
||||
Reference in New Issue
Block a user