diff --git a/examples/spec/textcodec.js b/examples/spec/textcodec.js index 454b140..1e22c54 100644 --- a/examples/spec/textcodec.js +++ b/examples/spec/textcodec.js @@ -1,4 +1,4 @@ -import { test, summary } from './helpers.js'; +import { test, testDeep, testThrows, summary } from './helpers.js'; console.log('TextEncoder/TextDecoder Tests\n'); @@ -16,8 +16,7 @@ test('TextDecoder', decoded, 'hello'); const utf8 = encoder.encode('日本語'); test('UTF-8 encode length', utf8.length, 9); -const utf8Decoded = decoder.decode(utf8); -test('UTF-8 decode', utf8Decoded, '日本語'); +test('UTF-8 decode', decoder.decode(utf8), '日本語'); const emoji = encoder.encode('😀'); test('emoji encode length', emoji.length, 4); @@ -28,7 +27,134 @@ test('empty encode length', empty.length, 0); test('empty decode', decoder.decode(empty), ''); const roundtrip = 'Hello, 世界! 🎉'; -const rt = decoder.decode(encoder.encode(roundtrip)); -test('roundtrip', rt, roundtrip); +test('roundtrip', decoder.decode(encoder.encode(roundtrip)), roundtrip); + +test('TextEncoder.encoding', encoder.encoding, 'utf-8'); +test('TextEncoder requires new', typeof TextEncoder, 'function'); +testThrows('TextEncoder without new throws', () => TextEncoder()); + +testDeep('encode lone high surrogate', [...encoder.encode('\uD800')], [0xef, 0xbf, 0xbd]); +testDeep('encode lone low surrogate', [...encoder.encode('\uDC00')], [0xef, 0xbf, 0xbd]); +testDeep('encode surrogate in string', [...encoder.encode('a\uD800b')], [0x61, 0xef, 0xbf, 0xbd, 0x62]); +testDeep('encode reversed surrogates', [...encoder.encode('\uDC00\uD800')], [0xef, 0xbf, 0xbd, 0xef, 0xbf, 0xbd]); +test('encode valid surrogate pair', encoder.encode('\uD834\uDD1E').length, 4); // U+1D11E 𝄞 + +const dest = new Uint8Array(10); +const result = encoder.encodeInto('hello', dest); +test('encodeInto read', result.read, 5); +test('encodeInto written', result.written, 5); +test('encodeInto data', decoder.decode(dest.subarray(0, result.written)), 'hello'); + +const small = new Uint8Array(2); +const partial = encoder.encodeInto('hello', small); +test('encodeInto partial written', partial.written, 2); +test('encodeInto partial read', partial.read, 2); + +test('TextDecoder default encoding', new TextDecoder().encoding, 'utf-8'); +test('TextDecoder utf8 alias', new TextDecoder('utf8').encoding, 'utf-8'); +test('TextDecoder case insensitive', new TextDecoder('UTF-8').encoding, 'utf-8'); +test('TextDecoder utf-16le label', new TextDecoder('utf-16le').encoding, 'utf-16le'); +test('TextDecoder utf-16be label', new TextDecoder('utf-16be').encoding, 'utf-16be'); +test('TextDecoder utf-16 alias', new TextDecoder('utf-16').encoding, 'utf-16le'); +testThrows('TextDecoder invalid label', () => new TextDecoder('bogus')); +testThrows('TextDecoder without new throws', () => TextDecoder()); + +test('fatal defaults false', new TextDecoder().fatal, false); +test('fatal option true', new TextDecoder('utf-8', { fatal: true }).fatal, true); +test('ignoreBOM defaults false', new TextDecoder().ignoreBOM, false); +test('ignoreBOM option true', new TextDecoder('utf-8', { ignoreBOM: true }).ignoreBOM, true); + +testThrows('fatal on invalid UTF-8', () => { + new TextDecoder('utf-8', { fatal: true }).decode(new Uint8Array([0xff])); +}); +testThrows('fatal on truncated sequence', () => { + new TextDecoder('utf-8', { fatal: true }).decode(new Uint8Array([0xc0])); +}); +testThrows('fatal on overlong', () => { + new TextDecoder('utf-8', { fatal: true }).decode(new Uint8Array([0xc0, 0x80])); +}); +test('non-fatal replacement', new TextDecoder().decode(new Uint8Array([0xff])), '\uFFFD'); +test('non-fatal truncated', new TextDecoder().decode(new Uint8Array([0xc0])), '\uFFFD'); + +test('UTF-8 BOM stripped by default', new TextDecoder().decode(new Uint8Array([0xef, 0xbb, 0xbf, 0x41])), 'A'); +test('UTF-8 BOM kept with ignoreBOM', new TextDecoder('utf-8', { ignoreBOM: true }).decode(new Uint8Array([0xef, 0xbb, 0xbf, 0x41])), '\uFEFFA'); + +{ + const sd = new TextDecoder(); + let out = ''; + out += sd.decode(new Uint8Array([0xf0, 0x9f, 0x92]), { stream: true }); + out += sd.decode(new Uint8Array([0xa9])); + test('streaming UTF-8 multi-byte', out, '\u{1F4A9}'); +} + +{ + const sd = new TextDecoder(); + let out = ''; + out += sd.decode(new Uint8Array([0xf0]), { stream: true }); + out += sd.decode(new Uint8Array([0x9f]), { stream: true }); + out += sd.decode(new Uint8Array([0x92]), { stream: true }); + out += sd.decode(new Uint8Array([0xa9])); + test('streaming UTF-8 byte-at-a-time', out, '\u{1F4A9}'); +} + +{ + const sd = new TextDecoder(); + let out = ''; + out += sd.decode(new Uint8Array([0xf0, 0x9f]), { stream: true }); + out += sd.decode(); + test('streaming flush incomplete sequence', out, '\uFFFD'); +} + +test('UTF-16LE basic', new TextDecoder('utf-16le').decode(new Uint8Array([0x41, 0x00, 0x42, 0x00])), 'AB'); +test('UTF-16LE surrogate pair', new TextDecoder('utf-16le').decode(new Uint8Array([0x34, 0xd8, 0x1e, 0xdd])), '\uD834\uDD1E'); +test('UTF-16LE BOM stripped', new TextDecoder('utf-16le').decode(new Uint8Array([0xff, 0xfe, 0x41, 0x00])), 'A'); + +test( + 'UTF-16LE BOM kept with ignoreBOM', + new TextDecoder('utf-16le', { ignoreBOM: true }).decode(new Uint8Array([0xff, 0xfe, 0x41, 0x00])), + '\uFEFFA' +); + +testThrows('UTF-16LE fatal on odd byte', () => { + new TextDecoder('utf-16le', { fatal: true }).decode(new Uint8Array([0x00])); +}); + +test('UTF-16LE non-fatal odd byte', new TextDecoder('utf-16le').decode(new Uint8Array([0x00])), '\uFFFD'); + +{ + const sd = new TextDecoder('utf-16le'); + let out = ''; + out += sd.decode(new Uint8Array([0x41]), { stream: true }); + out += sd.decode(new Uint8Array([0x00])); + test('UTF-16LE streaming split code unit', out, 'A'); +} + +{ + const sd = new TextDecoder('utf-16le'); + let out = ''; + out += sd.decode(new Uint8Array([0x34, 0xd8]), { stream: true }); + out += sd.decode(new Uint8Array([0x1e, 0xdd])); + test('UTF-16LE streaming split surrogate pair', out, '\uD834\uDD1E'); +} + +test('UTF-16BE basic', new TextDecoder('utf-16be').decode(new Uint8Array([0x00, 0x41, 0x00, 0x42])), 'AB'); +test('UTF-16BE surrogate pair', new TextDecoder('utf-16be').decode(new Uint8Array([0xd8, 0x34, 0xdd, 0x1e])), '\uD834\uDD1E'); +test('UTF-16BE BOM stripped', new TextDecoder('utf-16be').decode(new Uint8Array([0xfe, 0xff, 0x00, 0x41])), 'A'); + +{ + const buf = new Uint8Array([0x68, 0x69]).buffer; + test('decode ArrayBuffer', new TextDecoder().decode(buf), 'hi'); +} + +{ + const d = new TextDecoder(); + d.decode(new Uint8Array([0xf0, 0x9f]), { stream: true }); + const fresh = d.decode(new Uint8Array([0x41])); + test('decoder reuse resets', fresh, '\uFFFDA'); +} + +testDeep('encode undefined', [...encoder.encode(undefined)], []); +testDeep('encode no args', [...encoder.encode()], []); +test('decode no args', decoder.decode(), ''); summary(); diff --git a/include/modules/textcodec.h b/include/modules/textcodec.h index 153dcd6..a2ba3fe 100644 --- a/include/modules/textcodec.h +++ b/include/modules/textcodec.h @@ -1,6 +1,32 @@ #ifndef TEXTCODEC_H #define TEXTCODEC_H +#include +#include +#include + +#include "types.h" + +typedef enum { + TD_ENC_UTF8 = 0, + TD_ENC_UTF16LE, + TD_ENC_UTF16BE, +} td_encoding_t; + +typedef struct { + td_encoding_t encoding; + uint8_t pending[4]; + int pending_len; + bool fatal; + bool ignore_bom; + bool bom_seen; +} td_state_t; + void init_textcodec_module(void); +td_state_t *td_state_new(td_encoding_t enc, bool fatal, bool ignore_bom); + +ant_value_t td_decode(ant_t *js, td_state_t *st, const uint8_t *input, size_t input_len, bool stream); +ant_value_t te_encode(ant_t *js, const char *str, size_t str_len); + #endif diff --git a/include/utf8.h b/include/utf8.h index 80930b3..7ceed72 100644 --- a/include/utf8.h +++ b/include/utf8.h @@ -3,8 +3,21 @@ #include #include +#include #include +typedef struct { + bool ignore_bom; + bool bom_seen; + uint8_t pend_buf[3]; + int pend_pos; +} utf8_dec_t; + +utf8proc_ssize_t utf8_whatwg_decode( + utf8_dec_t *dec, const uint8_t *src, size_t len, + char *out, bool fatal, bool stream +); + size_t utf8_char_len_at(const char *str, size_t byte_len, size_t pos); size_t utf8_strlen(const char *str, size_t byte_len); size_t utf16_strlen(const char *str, size_t byte_len); diff --git a/src/modules/buffer.c b/src/modules/buffer.c index fac7a11..c937be6 100644 --- a/src/modules/buffer.c +++ b/src/modules/buffer.c @@ -685,7 +685,10 @@ static ant_value_t js_typedarray_constructor(ant_t *js, ant_value_t *args, int n snprintf(idx_str, sizeof(idx_str), "%zu", i); elem = js_get(js, args[0], idx_str); } - double val = vtype(elem) == T_NUM ? js_getnum(elem) : 0; + + double val = vtype(elem) == T_NUM + ? js_getnum(elem) + : js_to_number(js, elem); if (type > TYPED_ARRAY_BIGUINT64) goto W_DONE; goto *write_dispatch[type]; diff --git a/src/modules/textcodec.c b/src/modules/textcodec.c index 1bfe02e..fb90cd3 100644 --- a/src/modules/textcodec.c +++ b/src/modules/textcodec.c @@ -1,170 +1,458 @@ #include -#include #include +#include -#include "runtime.h" +#include "ant.h" #include "errors.h" +#include "runtime.h" #include "internal.h" -#include "silver/engine.h" +#include "descriptors.h" +#include "utf8.h" #include "modules/textcodec.h" #include "modules/buffer.h" #include "modules/symbol.h" -// TextEncoder.prototype.encode(string) -static ant_value_t js_textencoder_encode(ant_t *js, ant_value_t *args, int nargs) { - size_t str_len = 0; - const char *str = ""; +static ant_value_t g_textencoder_proto = 0; +static ant_value_t g_textdecoder_proto = 0; + +td_state_t *td_state_new(td_encoding_t enc, bool fatal, bool ignore_bom) { + td_state_t *st = calloc(1, sizeof(td_state_t)); + if (!st) return NULL; + st->encoding = enc; + st->fatal = fatal; + st->ignore_bom = ignore_bom; + return st; +} + +static td_state_t *td_get_state(ant_value_t obj) { + ant_value_t s = js_get_slot(obj, SLOT_DATA); + if (vtype(s) != T_NUM) return NULL; + return (td_state_t *)(uintptr_t)(size_t)js_getnum(s); +} + +static void td_finalize(ant_t *js, ant_object_t *obj) { + if (!obj->extra_slots) return; + ant_extra_slot_t *entries = (ant_extra_slot_t *)obj->extra_slots; - if (nargs > 0 && vtype(args[0]) == T_STR) { - str = js_getstr(js, args[0], &str_len); - if (!str) { str = ""; str_len = 0; } + for (uint8_t i = 0; i < obj->extra_count; i++) { + if (entries[i].slot == SLOT_DATA && vtype(entries[i].value) == T_NUM) { + free((td_state_t *)(uintptr_t)(size_t)js_getnum(entries[i].value)); + return; + }} +} + +static int resolve_encoding(const char *s, size_t len) { + static const struct { const char *label; uint8_t len; td_encoding_t enc; } map[] = { + {"unicode-1-1-utf-8", 18, TD_ENC_UTF8}, {"unicode11utf8", 13, TD_ENC_UTF8}, + {"unicode20utf8", 13, TD_ENC_UTF8}, {"utf-8", 5, TD_ENC_UTF8}, + {"utf8", 4, TD_ENC_UTF8}, {"x-unicode20utf8",17, TD_ENC_UTF8}, + {"unicodefffe", 11, TD_ENC_UTF16BE}, {"utf-16be", 8, TD_ENC_UTF16BE}, + {"csunicode", 9, TD_ENC_UTF16LE}, {"iso-10646-ucs-2",16, TD_ENC_UTF16LE}, + {"ucs-2", 5, TD_ENC_UTF16LE}, {"unicode", 7, TD_ENC_UTF16LE}, + {"unicodefeff", 11, TD_ENC_UTF16LE}, {"utf-16", 6, TD_ENC_UTF16LE}, + {"utf-16le", 8, TD_ENC_UTF16LE}, + {NULL, 0, 0} + }; + for (int i = 0; map[i].label; i++) { + if (len == map[i].len && strncasecmp(s, map[i].label, len) == 0) return (int)map[i].enc; + } + return -1; +} + +static const char *encoding_name(td_encoding_t enc) { +switch (enc) { + case TD_ENC_UTF16LE: return "utf-16le"; + case TD_ENC_UTF16BE: return "utf-16be"; + default: return "utf-8"; +}} + +static const char *trim_label(const char *s, size_t len, size_t *out_len) { + while (len > 0 && (unsigned char)*s <= 0x20) { s++; len--; } + while (len > 0 && (unsigned char)s[len - 1] <= 0x20) { len--; } + *out_len = len; + return s; +} + +static bool get_buffer_source(ant_t *js, ant_value_t arg, const uint8_t **out, size_t *len) { + ant_value_t slot = js_get_slot(arg, SLOT_BUFFER); + TypedArrayData *ta = (TypedArrayData *)js_gettypedarray(slot); + if (ta) { + if (!ta->buffer || ta->buffer->is_detached) { *out = NULL; *len = 0; return true; } + *out = ta->buffer->data + ta->byte_offset; + *len = ta->byte_length; + return true; } - ant_value_t glob = js_glob(js); - ant_value_t uint8array_ctor = js_get(js, glob, "Uint8Array"); - - if (vtype(uint8array_ctor) != T_FUNC && vtype(uint8array_ctor) != T_CFUNC) { - return js_mkerr_typed(js, JS_ERR_TYPE, "Uint8Array constructor missing"); + if (vtype(slot) == T_NUM) { + ArrayBufferData *ab = (ArrayBufferData *)(uintptr_t)(size_t)js_getnum(slot); + if (!ab || ab->is_detached) { *out = NULL; *len = 0; return true; } + *out = ab->data; + *len = ab->length; + return true; } - ant_value_t len_arg = js_mknum((double)str_len); - ant_value_t saved_new_target = js->new_target; + ant_value_t buf_prop = js_get(js, arg, "buffer"); + if (is_object_type(buf_prop)) { + ant_value_t buf_slot = js_get_slot(buf_prop, SLOT_BUFFER); - js->new_target = uint8array_ctor; - ant_value_t arr = sv_vm_call(js->vm, js, uint8array_ctor, js_mkundef(), &len_arg, 1, NULL, true); + if (vtype(buf_slot) == T_NUM) { + ArrayBufferData *ab = (ArrayBufferData *)(uintptr_t)(size_t)js_getnum(buf_slot); + if (!ab || ab->is_detached) { *out = NULL; *len = 0; return true; } + ant_value_t off_v = js_get(js, arg, "byteOffset"); + ant_value_t len_v = js_get(js, arg, "byteLength"); + size_t off = (vtype(off_v) == T_NUM) ? (size_t)js_getnum(off_v) : 0; + size_t blen = (vtype(len_v) == T_NUM) ? (size_t)js_getnum(len_v) : ab->length - off; + *out = ab->data + off; + *len = blen; + return true; + }} - js->new_target = saved_new_target; - if (vtype(arr) == T_ERR) return arr; + return false; +} + +static ant_value_t make_ctor(ant_t *js, ant_cfunc_t fn, ant_value_t proto, const char *name, size_t nlen) { + ant_value_t obj = js_mkobj(js); + js_set_slot(obj, SLOT_CFUNC, js_mkfun(fn)); + js_mkprop_fast(js, obj, "prototype", 9, proto); + js_mkprop_fast(js, obj, "name", 4, js_mkstr(js, name, nlen)); + js_set_descriptor(js, obj, "name", 4, 0); + + ant_value_t fn_val = js_obj_to_func(obj); + js_set(js, proto, "constructor", fn_val); + js_set_descriptor(js, proto, "constructor", 11, JS_DESC_W | JS_DESC_C); + + return fn_val; +} + +static ant_value_t js_textencoder_get_encoding(ant_t *js, ant_value_t *args, int nargs) { + return js_mkstr(js, "utf-8", 5); +} + +ant_value_t te_encode(ant_t *js, const char *str, size_t str_len) { + ArrayBufferData *ab = create_array_buffer_data(str_len); + if (!ab) return js_mkerr(js, "out of memory"); if (str_len > 0) { - ant_value_t ta_data_val = js_get_slot(arr, SLOT_BUFFER); - TypedArrayData *ta_data = (TypedArrayData *)js_gettypedarray(ta_data_val); - if (ta_data && ta_data->buffer && ta_data->buffer->data) memcpy(ta_data->buffer->data, str, str_len); + const uint8_t *s = (const uint8_t *)str; + uint8_t *d = ab->data; size_t i = 0; + + while (i < str_len) { + if (s[i] == 0xED && i + 2 < str_len && s[i+1] >= 0xA0 && s[i+1] <= 0xBF) { + d[i] = 0xEF; d[i+1] = 0xBF; d[i+2] = 0xBD; + i += 3; + } else { d[i] = s[i]; i++; }} } - return arr; + return create_typed_array(js, TYPED_ARRAY_UINT8, ab, 0, str_len, "Uint8Array"); } -// TextEncoder.prototype.encodeInto(string, uint8array) -static ant_value_t js_textencoder_encodeInto(ant_t *js, ant_value_t *args, int nargs) { - if (nargs < 2) { - return js_mkerr(js, "encodeInto requires string and Uint8Array arguments"); +static ant_value_t js_textencoder_encode(ant_t *js, ant_value_t *args, int nargs) { + size_t str_len = 0; + const char *str = ""; + + if (nargs > 0 && vtype(args[0]) == T_STR) { + str = js_getstr(js, args[0], &str_len); + if (!str) { str = ""; str_len = 0; } + } else if (nargs > 0 && vtype(args[0]) != T_UNDEF) { + ant_value_t sv = js_tostring_val(js, args[0]); + if (is_err(sv)) return sv; + str = js_getstr(js, sv, &str_len); + if (!str) { str = ""; str_len = 0; } } + return te_encode(js, str, str_len); +} + +static ant_value_t js_textencoder_encode_into(ant_t *js, ant_value_t *args, int nargs) { + if (nargs < 2) return js_mkerr_typed(js, JS_ERR_TYPE, "encodeInto requires 2 arguments"); + size_t str_len = 0; const char *str = ""; - if (vtype(args[0]) == T_STR) { str = js_getstr(js, args[0], &str_len); - if (!str) { - str = ""; - str_len = 0; - } + if (!str) { str = ""; str_len = 0; } + } else if (vtype(args[0]) != T_UNDEF) { + ant_value_t sv = js_tostring_val(js, args[0]); + if (is_err(sv)) return sv; + str = js_getstr(js, sv, &str_len); + if (!str) { str = ""; str_len = 0; } } + + TypedArrayData *ta = (TypedArrayData *)js_gettypedarray(js_get_slot(args[1], SLOT_BUFFER)); + if (!ta) return js_mkerr_typed(js, JS_ERR_TYPE, "Second argument must be a Uint8Array"); + + uint8_t *dest = (ta->buffer && !ta->buffer->is_detached) + ? ta->buffer->data + ta->byte_offset : NULL; + size_t available = ta->byte_length; + + const utf8proc_uint8_t *src = (const utf8proc_uint8_t *)str; + utf8proc_ssize_t src_len = (utf8proc_ssize_t)str_len; + utf8proc_ssize_t pos = 0; - ant_value_t ta_data_val = js_get_slot(args[1], SLOT_BUFFER); - TypedArrayData *ta_data = (TypedArrayData *)js_gettypedarray(ta_data_val); - if (!ta_data) return js_mkerr(js, "Second argument must be a Uint8Array"); - - size_t available = ta_data->byte_length; - size_t to_write = str_len < available ? str_len : available; - - if (to_write > 0) { - memcpy(ta_data->buffer->data + ta_data->byte_offset, str, to_write); + size_t written = 0; + size_t read_units = 0; + + while (pos < src_len) { + utf8proc_int32_t cp; + utf8proc_ssize_t n = utf8_next(src + pos, src_len - pos, &cp); + utf8proc_uint8_t tmp[4]; + utf8proc_ssize_t enc_len; + + if (cp >= 0xD800 && cp <= 0xDFFF) { + tmp[0] = 0xEF; tmp[1] = 0xBF; tmp[2] = 0xBD; + enc_len = 3; + } else { + enc_len = (cp >= 0) ? utf8proc_encode_char(cp, tmp) : 0; + if (enc_len <= 0) { tmp[0] = 0xEF; tmp[1] = 0xBF; tmp[2] = 0xBD; enc_len = 3; } + } + + if (written + (size_t)enc_len > available) break; + if (dest) memcpy(dest + written, tmp, (size_t)enc_len); + + written += (size_t)enc_len; + pos += n; + read_units += (cp >= 0x10000 && cp <= 0x10FFFF) ? 2 : 1; } - + ant_value_t result = js_mkobj(js); - js_set(js, result, "read", js_mknum((double)to_write)); - js_set(js, result, "written", js_mknum((double)to_write)); + js_set(js, result, "read", js_mknum((double)read_units)); + js_set(js, result, "written", js_mknum((double)written)); return result; } -static ant_value_t js_textencoder_constructor(ant_t *js, ant_value_t *args, int nargs) { - (void)args; - (void)nargs; - +static ant_value_t js_textencoder_ctor(ant_t *js, ant_value_t *args, int nargs) { + if (vtype(js->new_target) == T_UNDEF) + return js_mkerr_typed(js, JS_ERR_TYPE, "TextEncoder constructor requires 'new'"); ant_value_t obj = js_mkobj(js); - js_set(js, obj, "encoding", js_mkstr(js, "utf-8", 5)); - js_set(js, obj, "encode", js_mkfun(js_textencoder_encode)); - js_set(js, obj, "encodeInto", js_mkfun(js_textencoder_encodeInto)); - js_set_sym(js, obj, get_toStringTag_sym(), js_mkstr(js, "TextEncoder", 11)); - + ant_value_t proto = js_instance_proto_from_new_target(js, g_textencoder_proto); + if (is_object_type(proto)) js_set_proto_init(obj, proto); return obj; } -// TextDecoder.prototype.decode(bufferSource) -static ant_value_t js_textdecoder_decode(ant_t *js, ant_value_t *args, int nargs) { - if (nargs < 1) { - return js_mkstr(js, "", 0); +static ant_value_t js_textdecoder_get_encoding(ant_t *js, ant_value_t *args, int nargs) { + td_state_t *st = td_get_state(js->this_val); + const char *name = encoding_name(st ? st->encoding : TD_ENC_UTF8); + return js_mkstr(js, name, strlen(name)); +} + +static ant_value_t js_textdecoder_get_fatal(ant_t *js, ant_value_t *args, int nargs) { + td_state_t *st = td_get_state(js->this_val); + return (st && st->fatal) ? js_true : js_false; +} + +static ant_value_t js_textdecoder_get_ignore_bom(ant_t *js, ant_value_t *args, int nargs) { + td_state_t *st = td_get_state(js->this_val); + return (st && st->ignore_bom) ? js_true : js_false; +} + +static inline uint16_t u16_read(const uint8_t *p, bool be) { + return be + ? (uint16_t)((uint16_t)p[0] << 8 | p[1]) + : (uint16_t)((uint16_t)p[1] << 8 | p[0]); +} + +static inline size_t u8_emit(char *out, size_t o, utf8proc_int32_t cp) { + utf8proc_ssize_t n = utf8proc_encode_char(cp, (utf8proc_uint8_t *)(out + o)); + return n > 0 ? o + (size_t)n : o; +} + +static inline size_t u8_fffd(char *out, size_t o) { + out[o] = (char)0xEF; out[o+1] = (char)0xBF; out[o+2] = (char)0xBD; + return o + 3; +} + +#define U16_IS_HIGH(cu) ((cu) >= 0xD800 && (cu) <= 0xDBFF) +#define U16_IS_LOW(cu) ((cu) >= 0xDC00 && (cu) <= 0xDFFF) +#define U16_PAIR(hi,lo) (0x10000 + ((uint32_t)((hi) - 0xD800) << 10) + ((lo) - 0xDC00)) + +static utf8proc_ssize_t utf16_decode(td_state_t *st, const uint8_t *src, size_t len, char *out, bool stream) { + bool be = (st->encoding == TD_ENC_UTF16BE); + size_t i = 0, o = 0; + size_t avail; + + if (!st->bom_seen && len >= 2 && u16_read(src, be) == 0xFEFF && !st->ignore_bom) i = 2; + st->bom_seen = true; + + while (i < len) { + avail = len - i; + + if (avail < 2) goto pend_tail; + uint16_t cu = u16_read(src + i, be); + i += 2; + + if (!U16_IS_HIGH(cu) && !U16_IS_LOW(cu)) { + o = u8_emit(out, o, (utf8proc_int32_t)cu); + continue; + } + + if (U16_IS_LOW(cu)) goto err; + + avail = len - i; + if (avail < 2) goto pend_hi; + + uint16_t lo = u16_read(src + i, be); + if (U16_IS_LOW(lo)) { i += 2; o = u8_emit(out, o, (utf8proc_int32_t)U16_PAIR(cu, lo)); continue; } + + goto err; + + pend_tail: + if (stream) { st->pending[0] = src[i]; st->pending_len = 1; } + else goto err; + break; + + pend_hi: + if (stream) { st->pending_len = (int)(len - (i - 2)); memcpy(st->pending, src + i - 2, (size_t)st->pending_len); } + else { if (st->fatal) return -1; o = u8_fffd(out, o); if (avail == 1) o = u8_fffd(out, o); } + break; + + err: + if (st->fatal) return -1; + o = u8_fffd(out, o); + continue; } - ant_value_t ta_data_val = js_get_slot(args[0], SLOT_BUFFER); - TypedArrayData *ta_data = (TypedArrayData *)js_gettypedarray(ta_data_val); - if (ta_data) { - if (!ta_data->buffer) return js_mkstr(js, "", 0); - uint8_t *data = ta_data->buffer->data + ta_data->byte_offset; - size_t len = ta_data->byte_length; - return js_mkstr(js, (const char *)data, len); + return (utf8proc_ssize_t)o; +} + +#undef U16_IS_HIGH +#undef U16_IS_LOW +#undef U16_PAIR + +ant_value_t td_decode(ant_t *js, td_state_t *st, const uint8_t *input, size_t input_len, bool stream_mode) { + size_t total = (size_t)st->pending_len + input_len; + if (total == 0) { + if (!stream_mode) st->bom_seen = false; + return js_mkstr(js, "", 0); } - - ant_value_t ab_data_val = js_get_slot(args[0], SLOT_BUFFER); - if (vtype(ab_data_val) == T_NUM) { - ArrayBufferData *ab_data = (ArrayBufferData *)(uintptr_t)js_getnum(ab_data_val); - if (!ab_data || !ab_data->data) return js_mkstr(js, "", 0); - return js_mkstr(js, (const char *)ab_data->data, ab_data->length); + + uint8_t *work = NULL; + const uint8_t *src; + if (st->pending_len > 0) { + work = malloc(total); + if (!work) return js_mkerr(js, "out of memory"); + memcpy(work, st->pending, (size_t)st->pending_len); + if (input && input_len > 0) memcpy(work + st->pending_len, input, input_len); + src = work; + } else src = input; + st->pending_len = 0; + + char *out = malloc(total * 3 + 1); + if (!out) { free(work); return js_mkerr(js, "out of memory"); } + + utf8proc_ssize_t n; + if (st->encoding == TD_ENC_UTF16LE || st->encoding == TD_ENC_UTF16BE) { + n = utf16_decode(st, src, total, out, stream_mode); + } else { + utf8_dec_t dec = { .ignore_bom = st->ignore_bom, .bom_seen = st->bom_seen }; + n = utf8_whatwg_decode(&dec, src, total, out, st->fatal, stream_mode); + st->pending_len = dec.pend_pos; + memcpy(st->pending, dec.pend_buf, (size_t)dec.pend_pos); + st->bom_seen = stream_mode ? dec.bom_seen : false; + } + + if (n < 0) { + free(work); free(out); + return js_mkerr_typed(js, JS_ERR_TYPE, "The encoded data was not valid."); } + + if (st->encoding != TD_ENC_UTF8) { + if (!stream_mode) st->bom_seen = false; + } + + ant_value_t result = js_mkstr(js, out, (size_t)n); + free(work); + free(out); - return js_mkstr(js, "", 0); + return result; } -static ant_value_t js_textdecoder_constructor(ant_t *js, ant_value_t *args, int nargs) { - const char *encoding = "utf-8"; - size_t encoding_len = 5; - - if (nargs > 0 && vtype(args[0]) == T_STR) { - encoding = js_getstr(js, args[0], &encoding_len); - if (encoding && ( - strcasecmp(encoding, "utf-8") == 0 || - strcasecmp(encoding, "utf8") == 0) - ) { encoding = "utf-8"; encoding_len = 5; } +static ant_value_t js_textdecoder_decode(ant_t *js, ant_value_t *args, int nargs) { + td_state_t *st = td_get_state(js->this_val); + if (!st) return js_mkerr_typed(js, JS_ERR_TYPE, "Invalid TextDecoder"); + + bool stream_mode = false; + if (nargs > 1 && is_object_type(args[1])) { + ant_value_t sv = js_get(js, args[1], "stream"); + stream_mode = js_truthy(js, sv); } + + const uint8_t *input = NULL; + size_t input_len = 0; + if (nargs > 0 && is_object_type(args[0])) + get_buffer_source(js, args[0], &input, &input_len); + + return td_decode(js, st, input, input_len, stream_mode); +} + +static ant_value_t js_textdecoder_ctor(ant_t *js, ant_value_t *args, int nargs) { + if (vtype(js->new_target) == T_UNDEF) + return js_mkerr_typed(js, JS_ERR_TYPE, "TextDecoder constructor requires 'new'"); + + td_encoding_t enc = TD_ENC_UTF8; + if (nargs > 0 && vtype(args[0]) == T_STR) { + size_t llen; + const char *raw = js_getstr(js, args[0], &llen); + + if (raw) { + size_t tlen; + const char *trimmed = trim_label(raw, llen, &tlen); + int resolved = resolve_encoding(trimmed, tlen); + + if (resolved < 0) return js_mkerr_typed( + js, JS_ERR_RANGE, "Failed to construct 'TextDecoder': The encoding label provided ('%.*s') is invalid.", + (int)tlen, trimmed + ); + + enc = (td_encoding_t)resolved; + }} + + bool fatal = false; + bool ignore_bom = false; + if (nargs > 1 && is_object_type(args[1])) { + ant_value_t fv = js_get(js, args[1], "fatal"); + if (vtype(fv) != T_UNDEF) fatal = js_truthy(js, fv); + ant_value_t bv = js_get(js, args[1], "ignoreBOM"); + if (vtype(bv) != T_UNDEF) ignore_bom = js_truthy(js, bv); + } + + td_state_t *st = td_state_new(enc, fatal, ignore_bom); + if (!st) return js_mkerr(js, "out of memory"); + ant_value_t obj = js_mkobj(js); - js_set(js, obj, "encoding", js_mkstr(js, encoding, encoding_len)); - js_set(js, obj, "fatal", js_false); - js_set(js, obj, "ignoreBOM", js_false); - js_set(js, obj, "decode", js_mkfun(js_textdecoder_decode)); - js_set_sym(js, obj, get_toStringTag_sym(), js_mkstr(js, "TextDecoder", 11)); + ant_value_t proto = js_instance_proto_from_new_target(js, g_textdecoder_proto); + + if (is_object_type(proto)) js_set_proto_init(obj, proto); + js_set_slot(obj, SLOT_DATA, ANT_PTR(st)); + js_set_finalizer(obj, td_finalize); return obj; } void init_textcodec_module(void) { ant_t *js = rt->js; - ant_value_t glob = js_glob(js); - - ant_value_t textencoder_ctor_obj = js_mkobj(js); - js_set_slot(textencoder_ctor_obj, SLOT_CFUNC, js_mkfun(js_textencoder_constructor)); - ant_value_t textencoder_proto = js_mkobj(js); - - js_set(js, textencoder_proto, "encode", js_mkfun(js_textencoder_encode)); - js_set(js, textencoder_proto, "encodeInto", js_mkfun(js_textencoder_encodeInto)); - js_set(js, textencoder_proto, "encoding", js_mkstr(js, "utf-8", 5)); - js_set(js, textencoder_ctor_obj, "prototype", textencoder_proto); - ant_value_t textencoder_constructor = js_obj_to_func(textencoder_ctor_obj); - js_set(js, glob, "TextEncoder", textencoder_constructor); - - ant_value_t textdecoder_ctor_obj = js_mkobj(js); - js_set_slot(textdecoder_ctor_obj, SLOT_CFUNC, js_mkfun(js_textdecoder_constructor)); - ant_value_t textdecoder_proto = js_mkobj(js); - - js_set(js, textdecoder_proto, "decode", js_mkfun(js_textdecoder_decode)); - js_set(js, textdecoder_proto, "encoding", js_mkstr(js, "utf-8", 5)); - js_set(js, textdecoder_proto, "fatal", js_false); - js_set(js, textdecoder_proto, "ignoreBOM", js_false); - js_set(js, textdecoder_ctor_obj, "prototype", textdecoder_proto); - ant_value_t textdecoder_constructor = js_obj_to_func(textdecoder_ctor_obj); - js_set(js, glob, "TextDecoder", textdecoder_constructor); + ant_value_t g = js_glob(js); + + g_textencoder_proto = js_mkobj(js); + js_set_getter_desc(js, g_textencoder_proto, "encoding", 8, js_mkfun(js_textencoder_get_encoding), JS_DESC_C); + js_set(js, g_textencoder_proto, "encode", js_mkfun(js_textencoder_encode)); + js_set(js, g_textencoder_proto, "encodeInto", js_mkfun(js_textencoder_encode_into)); + js_set_sym(js, g_textencoder_proto, get_toStringTag_sym(), js_mkstr(js, "TextEncoder", 11)); + + ant_value_t te_ctor = make_ctor(js, js_textencoder_ctor, g_textencoder_proto, "TextEncoder", 11); + js_set(js, g, "TextEncoder", te_ctor); + js_set_descriptor(js, g, "TextEncoder", 11, JS_DESC_W | JS_DESC_C); + + g_textdecoder_proto = js_mkobj(js); + js_set_getter_desc(js, g_textdecoder_proto, "encoding", 8, js_mkfun(js_textdecoder_get_encoding), JS_DESC_C); + js_set_getter_desc(js, g_textdecoder_proto, "fatal", 5, js_mkfun(js_textdecoder_get_fatal), JS_DESC_C); + js_set_getter_desc(js, g_textdecoder_proto, "ignoreBOM", 9, js_mkfun(js_textdecoder_get_ignore_bom), JS_DESC_C); + js_set(js, g_textdecoder_proto, "decode", js_mkfun(js_textdecoder_decode)); + js_set_sym(js, g_textdecoder_proto, get_toStringTag_sym(), js_mkstr(js, "TextDecoder", 11)); + + ant_value_t td_ctor = make_ctor(js, js_textdecoder_ctor, g_textdecoder_proto, "TextDecoder", 11); + js_set(js, g, "TextDecoder", td_ctor); + js_set_descriptor(js, g, "TextDecoder", 11, JS_DESC_W | JS_DESC_C); } diff --git a/src/utf8.c b/src/utf8.c index 25df978..487d4f6 100644 --- a/src/utf8.c +++ b/src/utf8.c @@ -1,5 +1,6 @@ #include "utf8.h" #include +#include static uint32_t utf8_decode(const unsigned char *buf, size_t len, int *seq_len) { if (len == 0) { *seq_len = 0; return 0; } @@ -178,6 +179,91 @@ uint32_t utf16_code_unit_at(const char *str, size_t byte_len, size_t utf16_idx) return 0xFFFFFFFF; } +utf8proc_ssize_t utf8_whatwg_decode( + utf8_dec_t *dec, const uint8_t *src, size_t len, + char *out, bool fatal, bool stream +) { + static const void *tbl[256] = { + [0x00 ... 0x7F] = &&L_ASCII, + [0x80 ... 0xBF] = &&L_LONE, + [0xC0 ... 0xC1] = &&L_BAD, + [0xC2 ... 0xDF] = &&L_2, + [0xE0] = &&L_E0, + [0xE1 ... 0xEC] = &&L_3, + [0xED] = &&L_ED, + [0xEE ... 0xEF] = &&L_3, + [0xF0] = &&L_F0, + [0xF1 ... 0xF3] = &&L_4, + [0xF4] = &&L_F4, + [0xF5 ... 0xFF] = &&L_BAD, + }; + + size_t i = 0, o = 0; + int bc = 0; + + uint8_t lo = 0x80, hi = 0xBF; + utf8proc_int32_t cp = 0; + uint8_t pb[4]; int pp = 0; + +#define FFFD() do { out[o++]=(char)0xEF; out[o++]=(char)0xBF; out[o++]=(char)0xBD; } while(0) +#define NEXT() do { i++; if (i < len) goto *tbl[src[i]]; goto done; } while(0) + + if (!len) goto done; + goto *tbl[src[0]]; + +L_ASCII: + dec->bom_seen = true; + out[o++] = (char)src[i]; + NEXT(); + +L_LONE: +L_BAD: + if (fatal) return -1; + FFFD(); dec->bom_seen = true; + NEXT(); + +L_E0: bc=2; lo=0xA0; hi=0xBF; cp=src[i]&0x0F; pb[0]=src[i]; pp=1; i++; goto cont; +L_ED: bc=2; lo=0x80; hi=0x9F; cp=src[i]&0x0F; pb[0]=src[i]; pp=1; i++; goto cont; +L_3: bc=2; lo=0x80; hi=0xBF; cp=src[i]&0x0F; pb[0]=src[i]; pp=1; i++; goto cont; +L_F0: bc=3; lo=0x90; hi=0xBF; cp=src[i]&0x07; pb[0]=src[i]; pp=1; i++; goto cont; +L_F4: bc=3; lo=0x80; hi=0x8F; cp=src[i]&0x07; pb[0]=src[i]; pp=1; i++; goto cont; +L_4: bc=3; lo=0x80; hi=0xBF; cp=src[i]&0x07; pb[0]=src[i]; pp=1; i++; goto cont; +L_2: bc=1; lo=0x80; hi=0xBF; cp=src[i]&0x1F; pb[0]=src[i]; pp=1; i++; goto cont; + +cont: + while (bc > 0) { + if (i >= len) { + if (stream) { dec->pend_pos = pp; memcpy(dec->pend_buf, pb, pp); } + else { if (fatal) return -1; FFFD(); } + goto done; + } + uint8_t b = src[i]; + if (b < lo || b > hi) { + bc = 0; cp = 0; pp = 0; + if (fatal) return -1; + FFFD(); dec->bom_seen = true; + goto *tbl[b]; + } + lo = 0x80; hi = 0xBF; + cp = (cp << 6) | (b & 0x3F); + pb[pp++] = b; bc--; i++; + } + pp = 0; + if (!dec->bom_seen && cp == 0xFEFF && !dec->ignore_bom) dec->bom_seen = true; + else { + dec->bom_seen = true; + utf8proc_ssize_t n = utf8proc_encode_char(cp, (utf8proc_uint8_t *)(out + o)); + if (n > 0) o += (size_t)n; + } + cp = 0; + if (i < len) goto *tbl[src[i]]; + +done: +#undef FFFD +#undef NEXT + return (utf8proc_ssize_t)o; +} + uint32_t utf16_codepoint_at(const char *str, size_t byte_len, size_t utf16_idx) { const unsigned char *p = (const unsigned char *)str; const unsigned char *end = p + byte_len;