diff --git a/include/internal.h b/include/internal.h index 9de1697..2d46ca5 100644 --- a/include/internal.h +++ b/include/internal.h @@ -227,9 +227,16 @@ struct ant_isolate_t { typedef struct { ant_offset_t len; + uint8_t is_ascii; char bytes[]; } ant_flat_string_t; +enum { + STR_ASCII_UNKNOWN = 0, + STR_ASCII_YES = 1, + STR_ASCII_NO = 2, +}; + typedef struct { ant_offset_t len; uint8_t depth; @@ -438,6 +445,31 @@ static inline ant_value_t defmethod(ant_t *js, ant_value_t obj, const char *name return mkprop(js, obj, k, fn, ANT_PROP_ATTR_WRITABLE | ANT_PROP_ATTR_CONFIGURABLE); } +static inline ant_flat_string_t *str_flat_from_bytes(const char *str) { + return (ant_flat_string_t *)((char *)str - offsetof(ant_flat_string_t, bytes)); +} + +static inline uint8_t str_detect_ascii_bytes(const char *str, size_t len) { + const unsigned char *s = (const unsigned char *)str; + for (size_t i = 0; i < len; i++) { + if (s[i] >= 0x80) return STR_ASCII_NO; + } + return STR_ASCII_YES; +} + +static inline void str_set_ascii_state(const char *str, uint8_t state) { + ant_flat_string_t *flat = str_flat_from_bytes(str); + flat->is_ascii = state; +} + +static inline bool str_is_ascii(const char *str) { + ant_flat_string_t *flat = str_flat_from_bytes(str); + if (flat->is_ascii == STR_ASCII_UNKNOWN) { + flat->is_ascii = str_detect_ascii_bytes(flat->bytes, (size_t)flat->len); + } + return flat->is_ascii == STR_ASCII_YES; +} + static inline void js_set_module_default(ant_t *js, ant_value_t lib, ant_value_t ctor_fn, const char *name) { js_set(js, ctor_fn, name, ctor_fn); js_set(js, lib, name, ctor_fn); diff --git a/src/ant.c b/src/ant.c index 3948b40..c44bc73 100644 --- a/src/ant.c +++ b/src/ant.c @@ -2307,7 +2307,12 @@ ant_value_t js_mkstr(ant_t *js, const void *ptr, size_t len) { flat->len = (ant_offset_t)len; if (ptr && len > 0) memcpy(flat->bytes, ptr, len); + flat->bytes[len] = '\0'; + flat->is_ascii = (ptr || len == 0) + ? str_detect_ascii_bytes(flat->bytes, len) + : STR_ASCII_UNKNOWN; + return mkval(T_STR, (uintptr_t)flat); } @@ -2324,7 +2329,12 @@ ant_value_t js_mkstr_permanent(ant_t *js, const void *ptr, size_t len) { flat->len = (ant_offset_t)len; if (ptr && len > 0) memcpy(flat->bytes, ptr, len); + flat->bytes[len] = '\0'; + flat->is_ascii = (ptr || len == 0) + ? str_detect_ascii_bytes(flat->bytes, len) + : STR_ASCII_UNKNOWN; + return mkval(T_STR, (uintptr_t)flat); } @@ -8832,7 +8842,6 @@ static ant_value_t builtin_string_codePointAt(ant_t *js, ant_value_t *args, int } static ant_value_t builtin_string_toLowerCase(ant_t *js, ant_value_t *args, int nargs) { - (void) args; (void) nargs; ant_value_t str = to_string_val(js, js->this_val); if (vtype(str) != T_STR) return js_mkerr(js, "toLowerCase called on non-string"); @@ -8845,6 +8854,7 @@ static ant_value_t builtin_string_toLowerCase(ant_t *js, ant_value_t *args, int ant_offset_t out_len = 0; utf8proc_ssize_t pos = 0; + while (pos < src_len) { utf8proc_int32_t cp; utf8proc_ssize_t n = utf8_next(src + pos, src_len - pos, &cp); @@ -8856,24 +8866,37 @@ static ant_value_t builtin_string_toLowerCase(ant_t *js, ant_value_t *args, int ant_value_t result = js_mkstr(js, NULL, out_len); if (is_err(result)) return result; + ant_offset_t result_len, result_off = vstr(js, result, &result_len); char *result_ptr = (char *)(uintptr_t)(result_off); + uint8_t ascii_state = STR_ASCII_YES; pos = 0; ant_offset_t wpos = 0; + while (pos < src_len) { utf8proc_int32_t cp; utf8proc_ssize_t n = utf8_next(src + pos, src_len - pos, &cp); - if (cp < 0) { result_ptr[wpos++] = (char)src[pos]; pos++; continue; } - wpos += (ant_offset_t)utf8proc_encode_char(utf8proc_tolower(cp), (utf8proc_uint8_t *)(result_ptr + wpos)); + + if (cp < 0) { + unsigned char byte = src[pos]; + if (byte >= 0x80) ascii_state = STR_ASCII_NO; + result_ptr[wpos++] = (char)byte; + pos++; continue; + } + + utf8proc_int32_t mapped = utf8proc_tolower(cp); + if (mapped >= 0x80) ascii_state = STR_ASCII_NO; + + wpos += (ant_offset_t)utf8proc_encode_char(mapped, (utf8proc_uint8_t *)(result_ptr + wpos)); pos += n; } - + + str_set_ascii_state(result_ptr, ascii_state); return result; } static ant_value_t builtin_string_toUpperCase(ant_t *js, ant_value_t *args, int nargs) { - (void) args; (void) nargs; ant_value_t str = to_string_val(js, js->this_val); if (vtype(str) != T_STR) return js_mkerr(js, "toUpperCase called on non-string"); @@ -8886,6 +8909,7 @@ static ant_value_t builtin_string_toUpperCase(ant_t *js, ant_value_t *args, int ant_offset_t out_len = 0; utf8proc_ssize_t pos = 0; + while (pos < src_len) { utf8proc_int32_t cp; utf8proc_ssize_t n = utf8_next(src + pos, src_len - pos, &cp); @@ -8897,19 +8921,33 @@ static ant_value_t builtin_string_toUpperCase(ant_t *js, ant_value_t *args, int ant_value_t result = js_mkstr(js, NULL, out_len); if (is_err(result)) return result; + ant_offset_t result_len, result_off = vstr(js, result, &result_len); char *result_ptr = (char *)(uintptr_t)(result_off); + uint8_t ascii_state = STR_ASCII_YES; pos = 0; ant_offset_t wpos = 0; + while (pos < src_len) { utf8proc_int32_t cp; utf8proc_ssize_t n = utf8_next(src + pos, src_len - pos, &cp); - if (cp < 0) { result_ptr[wpos++] = (char)src[pos]; pos++; continue; } - wpos += (ant_offset_t)utf8proc_encode_char(utf8proc_toupper(cp), (utf8proc_uint8_t *)(result_ptr + wpos)); + + if (cp < 0) { + unsigned char byte = src[pos]; + if (byte >= 0x80) ascii_state = STR_ASCII_NO; + result_ptr[wpos++] = (char)byte; + pos++; continue; + } + + utf8proc_int32_t mapped = utf8proc_toupper(cp); + if (mapped >= 0x80) ascii_state = STR_ASCII_NO; + + wpos += (ant_offset_t)utf8proc_encode_char(mapped, (utf8proc_uint8_t *)(result_ptr + wpos)); pos += n; } - + + str_set_ascii_state(result_ptr, ascii_state); return result; } @@ -8979,6 +9017,10 @@ static ant_value_t builtin_string_repeat(ant_t *js, ant_value_t *args, int nargs for (ant_offset_t i = 0; i < count; i++) { memcpy(result_ptr + i * str_len, str_ptr, str_len); } + str_set_ascii_state( + result_ptr, + str_is_ascii(str_ptr) ? STR_ASCII_YES : STR_ASCII_NO + ); return result; } @@ -9035,6 +9077,10 @@ static ant_value_t builtin_string_padStart(ant_t *js, ant_value_t *args, int nar pos += rem_bytes; } memcpy(result_ptr + pos, str_ptr, (size_t)str_len); + str_set_ascii_state( + result_ptr, + (str_is_ascii(pad_str) && str_is_ascii(str_ptr)) ? STR_ASCII_YES : STR_ASCII_NO + ); return result; } @@ -9091,6 +9137,10 @@ static ant_value_t builtin_string_padEnd(ant_t *js, ant_value_t *args, int nargs memcpy(result_ptr + pos, pad_str, rem_bytes); pos += rem_bytes; } + str_set_ascii_state( + result_ptr, + (str_is_ascii(str_ptr) && str_is_ascii(pad_str)) ? STR_ASCII_YES : STR_ASCII_NO + ); return result; } diff --git a/src/utf8.c b/src/utf8.c index 2216187..5514ad3 100644 --- a/src/utf8.c +++ b/src/utf8.c @@ -1,8 +1,170 @@ #include "utf8.h" #include "utils.h" +#include "internal.h" +#include "gc/objects.h" + #include #include #include +#include + +typedef struct { + uint64_t epoch; + const char *str; + size_t byte_len; + size_t byte_pos; + size_t utf16_pos; +} utf16_scan_cache_t; + +typedef struct { + const char *str; + size_t byte_len; + const unsigned char *start; + const unsigned char *end; + const unsigned char *p; + size_t utf16_pos; +} utf16_scan_cursor_t; + +static _Thread_local utf16_scan_cache_t utf16_scan_cache = { 0 }; + +static inline void utf16_scan_cache_sync_epoch(void) { + uint64_t epoch = gc_get_epoch(); + if (utf16_scan_cache.epoch == epoch) return; + utf16_scan_cache = (utf16_scan_cache_t){ .epoch = epoch }; +} + +static inline void utf16_scan_cursor_init( + utf16_scan_cursor_t *cursor, + const char *str, + size_t byte_len +) { + utf16_scan_cache_sync_epoch(); + cursor->str = str; + cursor->byte_len = byte_len; + cursor->start = (const unsigned char *)str; + cursor->end = cursor->start + byte_len; + cursor->p = cursor->start; + cursor->utf16_pos = 0; +} + +static inline bool utf16_scan_cache_matches(const utf16_scan_cursor_t *cursor) { + return utf16_scan_cache.str == cursor->str + && utf16_scan_cache.byte_pos <= cursor->byte_len; +} + +static inline void utf16_scan_cursor_resume_cached(utf16_scan_cursor_t *cursor) { + if (!utf16_scan_cache_matches(cursor)) return; + cursor->p = cursor->start + utf16_scan_cache.byte_pos; + cursor->utf16_pos = utf16_scan_cache.utf16_pos; +} + +static inline void utf16_scan_cursor_resume_utf16( + utf16_scan_cursor_t *cursor, + size_t target_utf16 +) { + if (!utf16_scan_cache_matches(cursor)) return; + if (target_utf16 < utf16_scan_cache.utf16_pos) return; + cursor->p = cursor->start + utf16_scan_cache.byte_pos; + cursor->utf16_pos = utf16_scan_cache.utf16_pos; +} + +static inline void utf16_scan_cursor_resume_byte( + utf16_scan_cursor_t *cursor, + size_t target_byte +) { + if (!utf16_scan_cache_matches(cursor)) return; + if (target_byte < utf16_scan_cache.byte_pos) return; + cursor->p = cursor->start + utf16_scan_cache.byte_pos; + cursor->utf16_pos = utf16_scan_cache.utf16_pos; +} + +static inline void utf16_scan_cursor_store(const utf16_scan_cursor_t *cursor) { + utf16_scan_cache.str = cursor->str; + utf16_scan_cache.byte_len = cursor->byte_len; + utf16_scan_cache.byte_pos = (size_t)(cursor->p - cursor->start); + utf16_scan_cache.utf16_pos = cursor->utf16_pos; +} + +static inline void utf16_scan_decode( + const unsigned char *p, + const unsigned char *end, + size_t *slen_out, + size_t *units_out, + uint32_t *cp_out +) { + unsigned char c = *p; + if (c < 0x80) { + if (cp_out) *cp_out = c; + *slen_out = 1; + *units_out = 1; + return; + } + + if ((c & 0xE0) == 0xC0) { + if (cp_out && p + 1 < end) { + *cp_out = ((uint32_t)(c & 0x1F) << 6) | (uint32_t)(p[1] & 0x3F); + *slen_out = 2; + *units_out = 1; + return; + } + if (!cp_out) { + *slen_out = 2; + *units_out = 1; + return; + } + } else if ((c & 0xF0) == 0xE0) { + if (cp_out && p + 2 < end) { + *cp_out = ((uint32_t)(c & 0x0F) << 12) + | ((uint32_t)(p[1] & 0x3F) << 6) + | (uint32_t)(p[2] & 0x3F); + *slen_out = 3; + *units_out = 1; + return; + } + if (!cp_out) { + *slen_out = 3; + *units_out = 1; + return; + } + } else if ((c & 0xF8) == 0xF0) { + if (cp_out && p + 3 < end) { + *cp_out = ((uint32_t)(c & 0x07) << 18) + | ((uint32_t)(p[1] & 0x3F) << 12) + | ((uint32_t)(p[2] & 0x3F) << 6) + | (uint32_t)(p[3] & 0x3F); + *slen_out = 4; + *units_out = 2; + return; + } + if (!cp_out) { + *slen_out = 4; + *units_out = 2; + return; + } + } + + if (cp_out) *cp_out = c; + *slen_out = 1; + *units_out = 1; +} + +static inline bool utf16_scan_cursor_advance( + utf16_scan_cursor_t *cursor, + const unsigned char *bound_end +) { + size_t slen, units; + const unsigned char *next; + + utf16_scan_decode(cursor->p, cursor->end, &slen, &units, NULL); + next = cursor->p + slen; + cursor->utf16_pos += units; + if (next > bound_end) { + cursor->p = bound_end; + return false; + } + cursor->p = next; + return true; +} static uint32_t utf8_decode(const unsigned char *buf, size_t len, int *seq_len) { if (len == 0) { *seq_len = 0; return 0; } @@ -133,32 +295,18 @@ size_t utf8_strlen(const char *str, size_t byte_len) { } size_t utf16_strlen(const char *str, size_t byte_len) { - const unsigned char *p = (const unsigned char *)str; - const unsigned char *end = p + byte_len; - - size_t i = 0; - for (; i + 8 <= byte_len; i += 8) { - uint64_t chunk; - memcpy(&chunk, p + i, 8); - if (chunk & 0x8080808080808080ULL) goto slow_path; - } - for (; i < byte_len; i++) { - if (p[i] & 0x80) goto slow_path; - } - return byte_len; - -slow_path:; - size_t count = i; - p += i; - while (p < end) { - unsigned char c = *p; - if ((c & 0xC0) != 0x80) { - count++; - if ((c & 0xF8) == 0xF0) count++; - } - p++; + if (str_is_ascii(str)) return byte_len; + + utf16_scan_cursor_t cursor; + utf16_scan_cursor_init(&cursor, str, byte_len); + utf16_scan_cursor_resume_cached(&cursor); + + while (cursor.p < cursor.end) { + utf16_scan_cursor_advance(&cursor, cursor.end); } - return count; + + utf16_scan_cursor_store(&cursor); + return cursor.utf16_pos; } int utf16_index_to_byte_offset( @@ -167,36 +315,36 @@ int utf16_index_to_byte_offset( size_t utf16_idx, size_t *out_char_bytes ) { - const unsigned char *p = (const unsigned char *)str; - const unsigned char *end = p + byte_len; - size_t utf16_pos = 0; + if (str_is_ascii(str)) { + if (utf16_idx > byte_len) return -1; + if (out_char_bytes) *out_char_bytes = (utf16_idx < byte_len) ? 1 : 0; + return (int)utf16_idx; + } + + utf16_scan_cursor_t cursor; + utf16_scan_cursor_init(&cursor, str, byte_len); + utf16_scan_cursor_resume_utf16(&cursor, utf16_idx); - while (p < end && utf16_pos < utf16_idx) { - unsigned char c = *p; - if (c < 0x80) { p++; utf16_pos++; } - else if ((c & 0xE0) == 0xC0) { p += 2; utf16_pos++; } - else if ((c & 0xF0) == 0xE0) { p += 3; utf16_pos++; } - else if ((c & 0xF8) == 0xF0) { p += 4; utf16_pos += 2; } - else { p++; utf16_pos++; } - if (p > end) p = end; + while (cursor.p < cursor.end && cursor.utf16_pos < utf16_idx) { + utf16_scan_cursor_advance(&cursor, cursor.end); } - if (p >= end) { - if (utf16_pos == utf16_idx) { + if (cursor.p >= cursor.end) { + if (cursor.utf16_pos == utf16_idx) { if (out_char_bytes) *out_char_bytes = 0; + utf16_scan_cursor_store(&cursor); return (int)byte_len; - } return -1; + } + utf16_scan_cursor_store(&cursor); + return -1; } - unsigned char c = *p; - size_t slen = (c < 0x80) - ? 1 : ((c & 0xE0) == 0xC0) - ? 2 : ((c & 0xF0) == 0xE0) - ? 3 : ((c & 0xF8) == 0xF0) - ? 4 : 1; + size_t slen, units; + utf16_scan_decode(cursor.p, cursor.end, &slen, &units, NULL); if (out_char_bytes) *out_char_bytes = slen; - return (int)(p - (const unsigned char *)str); + utf16_scan_cursor_store(&cursor); + return (int)(cursor.p - cursor.start); } int utf16_range_to_byte_range( @@ -207,94 +355,94 @@ int utf16_range_to_byte_range( size_t *byte_start, size_t *byte_end ) { - const unsigned char *p = (const unsigned char *)str; - const unsigned char *end = p + byte_len; - - size_t utf16_pos = 0; + if (str_is_ascii(str)) { + *byte_start = (utf16_start <= byte_len) ? utf16_start : byte_len; + *byte_end = (utf16_end <= byte_len) ? utf16_end : byte_len; + return 0; + } + + utf16_scan_cursor_t cursor; + utf16_scan_cursor_init(&cursor, str, byte_len); + utf16_scan_cursor_resume_utf16(&cursor, utf16_start); + size_t b_start = 0, b_end = byte_len; int found_start = 0, found_end = 0; - while (p < end) { - if (utf16_pos == utf16_start) { b_start = p - (const unsigned char *)str; found_start = 1; } - if (utf16_pos == utf16_end) { b_end = p - (const unsigned char *)str; found_end = 1; break; } - - unsigned char c = *p; - if (c < 0x80) { p++; utf16_pos++; } - else if ((c & 0xE0) == 0xC0) { p += 2; utf16_pos++; } - else if ((c & 0xF0) == 0xE0) { p += 3; utf16_pos++; } - else if ((c & 0xF8) == 0xF0) { p += 4; utf16_pos += 2; } - else { p++; utf16_pos++; } - if (p > end) p = end; + while (cursor.p < cursor.end) { + if (cursor.utf16_pos == utf16_start) { + b_start = (size_t)(cursor.p - cursor.start); + found_start = 1; + } + if (cursor.utf16_pos == utf16_end) { + b_end = (size_t)(cursor.p - cursor.start); + found_end = 1; + break; + } + utf16_scan_cursor_advance(&cursor, cursor.end); } - if (!found_start && utf16_start >= utf16_pos) b_start = byte_len; - if (!found_end && utf16_end >= utf16_pos) b_end = byte_len; + if (!found_start && utf16_start >= cursor.utf16_pos) b_start = byte_len; + if (!found_end && utf16_end >= cursor.utf16_pos) b_end = byte_len; *byte_start = b_start; *byte_end = b_end; + utf16_scan_cursor_store(&cursor); return 0; } size_t byte_offset_to_utf16(const char *str, size_t byte_off) { - const unsigned char *p = (const unsigned char *)str; - const unsigned char *end = p + byte_off; - size_t utf16_pos = 0; + if (str_is_ascii(str)) return byte_off; - while (p < end) { - unsigned char c = *p; - if (c < 0x80) { p++; utf16_pos++; } - else if ((c & 0xE0) == 0xC0) { p += 2; utf16_pos++; } - else if ((c & 0xF0) == 0xE0) { p += 3; utf16_pos++; } - else if ((c & 0xF8) == 0xF0) { p += 4; utf16_pos += 2; } - else { p++; utf16_pos++; } - if (p > end) p = end; + utf16_scan_cursor_t cursor; + const unsigned char *bound_end; + bool ended_on_boundary = true; + + utf16_scan_cursor_init(&cursor, str, byte_off); + utf16_scan_cursor_resume_byte(&cursor, byte_off); + bound_end = cursor.start + byte_off; + + while (cursor.p < bound_end) { + if (!utf16_scan_cursor_advance(&cursor, bound_end)) { + ended_on_boundary = false; + break; + } } - return utf16_pos; + + if (ended_on_boundary) utf16_scan_cursor_store(&cursor); + return cursor.utf16_pos; } uint32_t utf16_code_unit_at(const char *str, size_t byte_len, size_t utf16_idx) { - const unsigned char *p = (const unsigned char *)str; - const unsigned char *end = p + byte_len; - size_t utf16_pos = 0; + if (str_is_ascii(str)) { + if (utf16_idx >= byte_len) return 0xFFFFFFFF; + return (unsigned char)str[utf16_idx]; + } + + utf16_scan_cursor_t cursor; + utf16_scan_cursor_init(&cursor, str, byte_len); + utf16_scan_cursor_resume_utf16(&cursor, utf16_idx); - while (p < end) { - unsigned char c = *p; - size_t units, slen; + while (cursor.p < cursor.end) { + size_t slen, units; uint32_t cp; - if (c < 0x80) { cp = c; slen = 1; units = 1; } - else if ((c & 0xE0) == 0xC0 && p + 1 < end) { - cp = ((c & 0x1F) << 6) - | (p[1] & 0x3F); - slen = 2; units = 1; - } - else if ((c & 0xF0) == 0xE0 && p + 2 < end) { - cp = ((c & 0x0F) << 12) - | ((p[1] & 0x3F) << 6) - | (p[2] & 0x3F); - slen = 3; units = 1; - } - else if ((c & 0xF8) == 0xF0 && p + 3 < end) { - cp = ((c & 0x07) << 18) - | ((p[1] & 0x3F) << 12) - | ((p[2] & 0x3F) << 6) - | (p[3] & 0x3F); - slen = 4; units = 2; - } - else { cp = c; slen = 1; units = 1; } + utf16_scan_decode(cursor.p, cursor.end, &slen, &units, &cp); - if (utf16_pos == utf16_idx) { + if (cursor.utf16_pos == utf16_idx) { + utf16_scan_cursor_store(&cursor); if (units == 2) return 0xD800 + ((cp - 0x10000) >> 10); return cp; } - if (units == 2 && utf16_pos + 1 == utf16_idx) { + if (units == 2 && cursor.utf16_pos + 1 == utf16_idx) { + utf16_scan_cursor_store(&cursor); return 0xDC00 + ((cp - 0x10000) & 0x3FF); } - p += slen; - utf16_pos += units; + cursor.p += slen; + cursor.utf16_pos += units; } + utf16_scan_cursor_store(&cursor); return 0xFFFFFFFF; } @@ -384,44 +532,34 @@ done: } uint32_t utf16_codepoint_at(const char *str, size_t byte_len, size_t utf16_idx) { - const unsigned char *p = (const unsigned char *)str; - const unsigned char *end = p + byte_len; - size_t utf16_pos = 0; + if (str_is_ascii(str)) { + if (utf16_idx >= byte_len) return 0xFFFFFFFF; + return (unsigned char)str[utf16_idx]; + } + + utf16_scan_cursor_t cursor; + utf16_scan_cursor_init(&cursor, str, byte_len); + utf16_scan_cursor_resume_utf16(&cursor, utf16_idx); - while (p < end) { - unsigned char c = *p; - size_t units, slen; + while (cursor.p < cursor.end) { + size_t slen, units; uint32_t cp; - if (c < 0x80) { cp = c; slen = 1; units = 1; } - else if ((c & 0xE0) == 0xC0 && p + 1 < end) { - cp = ((c & 0x1F) << 6) - | (p[1] & 0x3F); - slen = 2; units = 1; - } - else if ((c & 0xF0) == 0xE0 && p + 2 < end) { - cp = ((c & 0x0F) << 12) - | ((p[1] & 0x3F) << 6) - | (p[2] & 0x3F); - slen = 3; units = 1; - } - else if ((c & 0xF8) == 0xF0 && p + 3 < end) { - cp = ((c & 0x07) << 18) - | ((p[1] & 0x3F) << 12) - | ((p[2] & 0x3F) << 6) - | (p[3] & 0x3F); - slen = 4; units = 2; - } - else { cp = c; slen = 1; units = 1; } + utf16_scan_decode(cursor.p, cursor.end, &slen, &units, &cp); - if (utf16_pos == utf16_idx) return cp; - if (units == 2 && utf16_pos + 1 == utf16_idx) { + if (cursor.utf16_pos == utf16_idx) { + utf16_scan_cursor_store(&cursor); + return cp; + } + if (units == 2 && cursor.utf16_pos + 1 == utf16_idx) { + utf16_scan_cursor_store(&cursor); return 0xDC00 + ((cp - 0x10000) & 0x3FF); } - p += slen; - utf16_pos += units; + cursor.p += slen; + cursor.utf16_pos += units; } + utf16_scan_cursor_store(&cursor); return 0xFFFFFFFF; }