feat: rename IO.write to IO.print and add Unicode escape support in strings
Rename the IO.write method to IO.print, which now prints without a trailing newline, and add a new IO.print method that appends a newline after output. Update all benchmark, example, and test files to use IO.print instead of IO.write. Additionally, implement Unicode escape sequence parsing in the compiler's string tokenizer, supporting \uXXXX and \u{...} syntax with proper UTF-8 encoding for code points up to U+10FFFF.
This commit is contained in:
+65
-3
@@ -514,10 +514,71 @@ static void readName(Parser* parser, TokenType type)
|
||||
makeToken(parser, type);
|
||||
}
|
||||
|
||||
// Adds [c] to the current string literal being tokenized.
|
||||
static void addStringChar(Parser* parser, char c)
|
||||
// Adds [c] to the current string literal being tokenized. If [c] is outside of
|
||||
// ASCII range, it will emit the UTF-8 encoded byte sequence for it.
|
||||
static void addStringChar(Parser* parser, uint32_t c)
|
||||
{
|
||||
wrenByteBufferWrite(parser->vm, &parser->string, c);
|
||||
ByteBuffer* buffer = &parser->string;
|
||||
|
||||
if (c <= 0x7f)
|
||||
{
|
||||
// Single byte (i.e. fits in ASCII).
|
||||
wrenByteBufferWrite(parser->vm, buffer, c);
|
||||
}
|
||||
else if (c <= 0x7ff)
|
||||
{
|
||||
// Two byte sequence: 110xxxxx 10xxxxxx.
|
||||
wrenByteBufferWrite(parser->vm, buffer, 0xc0 | ((c & 0x7c0) >> 6));
|
||||
wrenByteBufferWrite(parser->vm, buffer, 0x80 | (c & 0x3f));
|
||||
}
|
||||
else if (c <= 0xffff)
|
||||
{
|
||||
// Three byte sequence: 1110xxxx 10xxxxxx 10xxxxxx.
|
||||
wrenByteBufferWrite(parser->vm, buffer, 0xe0 | ((c & 0xf000) >> 12));
|
||||
wrenByteBufferWrite(parser->vm, buffer, 0x80 | ((c & 0xfc0) >> 6));
|
||||
wrenByteBufferWrite(parser->vm, buffer, 0x80 | (c & 0x3f));
|
||||
}
|
||||
else if (c <= 0x10ffff)
|
||||
{
|
||||
// Four byte sequence: 11110xxx 10xxxxxx 10xxxxxx 10xxxxxx.
|
||||
wrenByteBufferWrite(parser->vm, buffer, 0xf0 | ((c & 0x1c0000) >> 18));
|
||||
wrenByteBufferWrite(parser->vm, buffer, 0x80 | ((c & 0x3f000) >> 12));
|
||||
wrenByteBufferWrite(parser->vm, buffer, 0x80 | ((c & 0xfc0) >> 6));
|
||||
wrenByteBufferWrite(parser->vm, buffer, 0x80 | (c & 0x3f));
|
||||
}
|
||||
else
|
||||
{
|
||||
// Invalid Unicode value. See: http://tools.ietf.org/html/rfc3629
|
||||
// TODO: Error.
|
||||
}
|
||||
}
|
||||
|
||||
// Reads the next character, which should be a hex digit (0-9, a-f, or A-F) and
|
||||
// returns its numeric value. If the character isn't a hex digit, returns -1.
|
||||
static int readHexDigit(Parser* parser)
|
||||
{
|
||||
char c = nextChar(parser);
|
||||
if (c >= '0' && c <= '9') return c - '0';
|
||||
if (c >= 'a' && c <= 'f') return c - 'a' + 10;
|
||||
if (c >= 'A' && c <= 'F') return c - 'A' + 10;
|
||||
return -1;
|
||||
}
|
||||
|
||||
// Reads a four hex digit Unicode escape sequence in a string literal.
|
||||
static void readUnicodeEscape(Parser* parser)
|
||||
{
|
||||
int value = 0;
|
||||
for (int i = 0; i < 4; i++)
|
||||
{
|
||||
char digit = readHexDigit(parser);
|
||||
// TODO: Signal error.
|
||||
if (digit == -1) return;
|
||||
|
||||
value = (value * 16) | digit;
|
||||
}
|
||||
|
||||
// TODO: Handle encoding!
|
||||
addStringChar(parser, value);
|
||||
}
|
||||
|
||||
// Finishes lexing a string literal.
|
||||
@@ -538,6 +599,7 @@ static void readString(Parser* parser)
|
||||
case '\\': addStringChar(parser, '\\'); break;
|
||||
case 'n': addStringChar(parser, '\n'); break;
|
||||
case 't': addStringChar(parser, '\t'); break;
|
||||
case 'u': readUnicodeEscape(parser); break;
|
||||
default:
|
||||
// TODO: Emit error token.
|
||||
break;
|
||||
|
||||
+10
-4
@@ -43,8 +43,14 @@
|
||||
// make_corelib. Do not edit here.
|
||||
const char* coreLibSource =
|
||||
"class IO {\n"
|
||||
" static print(obj) {\n"
|
||||
" IO.writeString_(obj.toString)\n"
|
||||
" IO.writeString_(\"\n\")\n"
|
||||
" return obj\n"
|
||||
" }\n"
|
||||
"\n"
|
||||
" static write(obj) {\n"
|
||||
" IO.write__native__(obj.toString)\n"
|
||||
" IO.writeString_(obj.toString)\n"
|
||||
" return obj\n"
|
||||
" }\n"
|
||||
"}\n"
|
||||
@@ -471,9 +477,9 @@ DEF_NATIVE(string_subscript)
|
||||
|
||||
DEF_NATIVE(io_writeString)
|
||||
{
|
||||
if (!validateString(vm, args, 1, "Argument")) return PRIM_ERROR;
|
||||
wrenPrintValue(args[1]);
|
||||
printf("\n");
|
||||
RETURN_VAL(args[1]);
|
||||
RETURN_NULL;
|
||||
}
|
||||
|
||||
DEF_NATIVE(os_clock)
|
||||
@@ -621,5 +627,5 @@ void wrenInitializeCore(WrenVM* vm)
|
||||
NATIVE(vm->numClass, "!= ", num_bangeq);
|
||||
|
||||
ObjClass* ioClass = AS_CLASS(findGlobal(vm, "IO"));
|
||||
NATIVE(ioClass->metaclass, "write__native__ ", io_writeString);
|
||||
NATIVE(ioClass->metaclass, "writeString_ ", io_writeString);
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user