Clarify how string subscripting handles UTF-8.

This commit is contained in:
Bob Nystrom
2015-01-22 16:38:03 -08:00
parent a92e58c804
commit a5b00cebe7
8 changed files with 115 additions and 26 deletions
+2 -7
View File
@@ -1078,13 +1078,7 @@ DEF_NATIVE(string_subscript)
int index = validateIndex(vm, args, string->length, 1, "Subscript");
if (index == -1) return PRIM_ERROR;
// The result is a one-character string.
// TODO: Handle UTF-8.
Value value = wrenNewUninitializedString(vm, 1);
ObjString* result = AS_STRING(value);
result->value[0] = AS_CSTRING(args[0])[index];
result->value[1] = '\0';
RETURN_VAL(value);
RETURN_VAL(wrenStringCodePointAt(vm, string, index));
}
if (!IS_RANGE(args[1]))
@@ -1092,6 +1086,7 @@ DEF_NATIVE(string_subscript)
RETURN_ERROR("Subscript must be a number or a range.");
}
// TODO: Handle UTF-8 here.
int step;
int count = string->length;
int start = calculateRange(vm, args, AS_RANGE(args[1]), &count, &step);
+24
View File
@@ -362,6 +362,30 @@ ObjString* wrenStringConcat(WrenVM* vm, const char* left, const char* right)
return string;
}
Value wrenStringCodePointAt(WrenVM* vm, ObjString* string, int index)
{
ASSERT(index >= 0, "Index out of bounds.");
ASSERT(index < string->length, "Index out of bounds.");
char first = string->value[index];
// The first byte's high bits tell us how many bytes are in the UTF-8
// sequence. If the byte starts with 10xxxxx, it's the middle of a UTF-8
// sequence, so return an empty string.
int numBytes;
if ((first & 0xc0) == 0x80) numBytes = 0;
else if ((first & 0xf8) == 0xf0) numBytes = 4;
else if ((first & 0xf0) == 0xe0) numBytes = 3;
else if ((first & 0xe0) == 0xc0) numBytes = 2;
else numBytes = 1;
Value value = wrenNewUninitializedString(vm, numBytes);
ObjString* result = AS_STRING(value);
memcpy(result->value, string->value + index, numBytes);
result->value[numBytes] = '\0';
return value;
}
Upvalue* wrenNewUpvalue(WrenVM* vm, Value* value)
{
Upvalue* upvalue = ALLOCATE(vm, Upvalue);
+5
View File
@@ -610,6 +610,11 @@ Value wrenNewUninitializedString(WrenVM* vm, size_t length);
// Creates a new string that is the concatenation of [left] and [right].
ObjString* wrenStringConcat(WrenVM* vm, const char* left, const char* right);
// Creates a new string containing the code point in [string] starting at byte
// [index]. If [index] points into the middle of a UTF-8 sequence, returns an
// empty string.
Value wrenStringCodePointAt(WrenVM* vm, ObjString* string, int index);
// Creates a new open upvalue pointing to [value] on the stack.
Upvalue* wrenNewUpvalue(WrenVM* vm, Value* value);