Clarify how string subscripting handles UTF-8.
This commit is contained in:
+2
-7
@@ -1078,13 +1078,7 @@ DEF_NATIVE(string_subscript)
|
||||
int index = validateIndex(vm, args, string->length, 1, "Subscript");
|
||||
if (index == -1) return PRIM_ERROR;
|
||||
|
||||
// The result is a one-character string.
|
||||
// TODO: Handle UTF-8.
|
||||
Value value = wrenNewUninitializedString(vm, 1);
|
||||
ObjString* result = AS_STRING(value);
|
||||
result->value[0] = AS_CSTRING(args[0])[index];
|
||||
result->value[1] = '\0';
|
||||
RETURN_VAL(value);
|
||||
RETURN_VAL(wrenStringCodePointAt(vm, string, index));
|
||||
}
|
||||
|
||||
if (!IS_RANGE(args[1]))
|
||||
@@ -1092,6 +1086,7 @@ DEF_NATIVE(string_subscript)
|
||||
RETURN_ERROR("Subscript must be a number or a range.");
|
||||
}
|
||||
|
||||
// TODO: Handle UTF-8 here.
|
||||
int step;
|
||||
int count = string->length;
|
||||
int start = calculateRange(vm, args, AS_RANGE(args[1]), &count, &step);
|
||||
|
||||
@@ -362,6 +362,30 @@ ObjString* wrenStringConcat(WrenVM* vm, const char* left, const char* right)
|
||||
return string;
|
||||
}
|
||||
|
||||
Value wrenStringCodePointAt(WrenVM* vm, ObjString* string, int index)
|
||||
{
|
||||
ASSERT(index >= 0, "Index out of bounds.");
|
||||
ASSERT(index < string->length, "Index out of bounds.");
|
||||
|
||||
char first = string->value[index];
|
||||
|
||||
// The first byte's high bits tell us how many bytes are in the UTF-8
|
||||
// sequence. If the byte starts with 10xxxxx, it's the middle of a UTF-8
|
||||
// sequence, so return an empty string.
|
||||
int numBytes;
|
||||
if ((first & 0xc0) == 0x80) numBytes = 0;
|
||||
else if ((first & 0xf8) == 0xf0) numBytes = 4;
|
||||
else if ((first & 0xf0) == 0xe0) numBytes = 3;
|
||||
else if ((first & 0xe0) == 0xc0) numBytes = 2;
|
||||
else numBytes = 1;
|
||||
|
||||
Value value = wrenNewUninitializedString(vm, numBytes);
|
||||
ObjString* result = AS_STRING(value);
|
||||
memcpy(result->value, string->value + index, numBytes);
|
||||
result->value[numBytes] = '\0';
|
||||
return value;
|
||||
}
|
||||
|
||||
Upvalue* wrenNewUpvalue(WrenVM* vm, Value* value)
|
||||
{
|
||||
Upvalue* upvalue = ALLOCATE(vm, Upvalue);
|
||||
|
||||
@@ -610,6 +610,11 @@ Value wrenNewUninitializedString(WrenVM* vm, size_t length);
|
||||
// Creates a new string that is the concatenation of [left] and [right].
|
||||
ObjString* wrenStringConcat(WrenVM* vm, const char* left, const char* right);
|
||||
|
||||
// Creates a new string containing the code point in [string] starting at byte
|
||||
// [index]. If [index] points into the middle of a UTF-8 sequence, returns an
|
||||
// empty string.
|
||||
Value wrenStringCodePointAt(WrenVM* vm, ObjString* string, int index);
|
||||
|
||||
// Creates a new open upvalue pointing to [value] on the stack.
|
||||
Upvalue* wrenNewUpvalue(WrenVM* vm, Value* value);
|
||||
|
||||
|
||||
Reference in New Issue
Block a user