diff --git a/include/modules/uri.h b/include/modules/uri.h new file mode 100644 index 0000000..5c7353e --- /dev/null +++ b/include/modules/uri.h @@ -0,0 +1,6 @@ +#ifndef URI_H +#define URI_H + +void init_uri_module(void); + +#endif diff --git a/meson.build b/meson.build index 7f70660..8b120e4 100644 --- a/meson.build +++ b/meson.build @@ -74,7 +74,7 @@ endif build_date = run_command('date', '+%Y-%m-%d', check: true).stdout().strip() version_conf = configuration_data() -version_conf.set('ANT_VERSION', '0.1.0.14') +version_conf.set('ANT_VERSION', '0.1.0.15') version_conf.set('ANT_GIT_HASH', git_hash) version_conf.set('ANT_BUILD_DATE', build_date) diff --git a/src/main.c b/src/main.c index 7f02443..e2f52f4 100644 --- a/src/main.c +++ b/src/main.c @@ -30,6 +30,7 @@ #include "modules/ffi.h" #include "modules/events.h" #include "modules/performance.h" +#include "modules/uri.h" int js_result = EXIT_SUCCESS; @@ -195,6 +196,7 @@ int main(int argc, char *argv[]) { init_process_module(); init_events_module(); init_performance_module(); + init_uri_module(); ant_register_library(shell_library, "ant:shell", NULL); ant_register_library(ffi_library, "ant:ffi", NULL); diff --git a/src/modules/uri.c b/src/modules/uri.c new file mode 100644 index 0000000..393354d --- /dev/null +++ b/src/modules/uri.c @@ -0,0 +1,277 @@ +#include +#include +#include + +#include "ant.h" +#include "runtime.h" +#include "modules/uri.h" + +static int hex_digit(char c) { + if (c >= '0' && c <= '9') return c - '0'; + if (c >= 'A' && c <= 'F') return c - 'A' + 10; + if (c >= 'a' && c <= 'f') return c - 'a' + 10; + return -1; +} + +static int is_uri_unreserved(unsigned char c) { + return (c >= 'A' && c <= 'Z') || + (c >= 'a' && c <= 'z') || + (c >= '0' && c <= '9') || + c == '-' || c == '_' || c == '.' || c == '!' || + c == '~' || c == '*' || c == '\'' || c == '(' || c == ')'; +} + +static int is_uri_reserved(unsigned char c) { + return c == ';' || c == '/' || c == '?' || c == ':' || + c == '@' || c == '&' || c == '=' || c == '+' || + c == '$' || c == ',' || c == '#'; +} + +static int utf8_sequence_length(unsigned char first_byte) { + if ((first_byte & 0x80) == 0) return 1; + if ((first_byte & 0xE0) == 0xC0) return 2; + if ((first_byte & 0xF0) == 0xE0) return 3; + if ((first_byte & 0xF8) == 0xF0) return 4; + return -1; +} + +static int is_valid_continuation(unsigned char c) { + return (c & 0xC0) == 0x80; +} + +static int decode_escape_sequence(const char *str, size_t len, size_t *pos, unsigned char *out_byte) { + if (*pos + 2 >= len) return -1; + if (str[*pos] != '%') return -1; + + int high = hex_digit(str[*pos + 1]); + int low = hex_digit(str[*pos + 2]); + if (high < 0 || low < 0) return -1; + + *out_byte = (unsigned char)((high << 4) | low); + *pos += 3; + return 0; +} + +// encodeURIComponent() +static jsval_t js_encodeURIComponent(struct js *js, jsval_t *args, int nargs) { + jsval_t result; + char *out = NULL; + + if (nargs < 1) return js_mkstr(js, "undefined", 9); + + char *str = js_getstr(js, args[0], NULL); + if (!str) return js_mkstr(js, "", 0); + + size_t len = strlen(str); + size_t out_cap = len * 12 + 1; + out = malloc(out_cap); + if (!out) return js_mkerr(js, "out of memory"); + + size_t out_len = 0; + size_t i = 0; + + while (i < len) { + unsigned char c = (unsigned char)str[i]; + + if (is_uri_unreserved(c)) { + out[out_len++] = (char)c; + i++; + continue; + } + + int seq_len = utf8_sequence_length(c); + if (seq_len < 0) goto malformed; + if (i + seq_len > len) goto malformed; + + for (int j = 1; j < seq_len; j++) { + if (!is_valid_continuation((unsigned char)str[i + j])) goto malformed; + } + + for (int j = 0; j < seq_len; j++) { + out_len += sprintf(out + out_len, "%%%02X", (unsigned char)str[i + j]); + } + i += seq_len; + } + + out[out_len] = '\0'; + result = js_mkstr(js, out, out_len); + free(out); + return result; + +malformed: + free(out); + return js_mkerr(js, "URIError: URI malformed"); +} + +// encodeURI() +static jsval_t js_encodeURI(struct js *js, jsval_t *args, int nargs) { + jsval_t result; + char *out = NULL; + + if (nargs < 1) return js_mkstr(js, "undefined", 9); + + char *str = js_getstr(js, args[0], NULL); + if (!str) return js_mkstr(js, "", 0); + + size_t len = strlen(str); + size_t out_cap = len * 12 + 1; + out = malloc(out_cap); + if (!out) return js_mkerr(js, "out of memory"); + + size_t out_len = 0; + size_t i = 0; + + while (i < len) { + unsigned char c = (unsigned char)str[i]; + + if (is_uri_unreserved(c) || is_uri_reserved(c)) { + out[out_len++] = (char)c; + i++; + continue; + } + + int seq_len = utf8_sequence_length(c); + if (seq_len < 0) goto malformed; + if (i + seq_len > len) goto malformed; + + for (int j = 1; j < seq_len; j++) { + if (!is_valid_continuation((unsigned char)str[i + j])) goto malformed; + } + + for (int j = 0; j < seq_len; j++) { + out_len += sprintf(out + out_len, "%%%02X", (unsigned char)str[i + j]); + } + i += seq_len; + } + + out[out_len] = '\0'; + result = js_mkstr(js, out, out_len); + free(out); + return result; + +malformed: + free(out); + return js_mkerr(js, "URIError: URI malformed"); +} + +// decodeURIComponent() +static jsval_t js_decodeURIComponent(struct js *js, jsval_t *args, int nargs) { + jsval_t result; + char *out = NULL; + + if (nargs < 1) return js_mkstr(js, "undefined", 9); + + char *str = js_getstr(js, args[0], NULL); + if (!str) return js_mkstr(js, "", 0); + + size_t len = strlen(str); + out = malloc(len + 1); + if (!out) return js_mkerr(js, "out of memory"); + + size_t out_len = 0; + size_t i = 0; + + while (i < len) { + if (str[i] != '%') { + out[out_len++] = str[i++]; + continue; + } + + unsigned char first_byte; + if (decode_escape_sequence(str, len, &i, &first_byte) < 0) goto malformed; + + int seq_len = utf8_sequence_length(first_byte); + if (seq_len < 0) goto malformed; + + out[out_len++] = (char)first_byte; + + for (int j = 1; j < seq_len; j++) { + unsigned char cont_byte; + if (decode_escape_sequence(str, len, &i, &cont_byte) < 0) goto malformed; + if (!is_valid_continuation(cont_byte)) goto malformed; + out[out_len++] = (char)cont_byte; + } + } + + out[out_len] = '\0'; + result = js_mkstr(js, out, out_len); + free(out); + return result; + +malformed: + free(out); + return js_mkerr(js, "URIError: URI malformed"); +} + +// decodeURI() +static jsval_t js_decodeURI(struct js *js, jsval_t *args, int nargs) { + jsval_t result; + char *out = NULL; + + if (nargs < 1) return js_mkstr(js, "undefined", 9); + + char *str = js_getstr(js, args[0], NULL); + if (!str) return js_mkstr(js, "", 0); + + size_t len = strlen(str); + out = malloc(len + 1); + if (!out) return js_mkerr(js, "out of memory"); + + size_t out_len = 0; + size_t i = 0; + + while (i < len) { + if (str[i] != '%') { + out[out_len++] = str[i++]; + continue; + } + + if (i + 2 >= len) goto malformed; + + int high = hex_digit(str[i + 1]); + int low = hex_digit(str[i + 2]); + if (high < 0 || low < 0) goto malformed; + + unsigned char first_byte = (unsigned char)((high << 4) | low); + + if (first_byte < 128 && is_uri_reserved((char)first_byte)) { + out[out_len++] = str[i++]; + out[out_len++] = str[i++]; + out[out_len++] = str[i++]; + continue; + } + + i += 3; + + int seq_len = utf8_sequence_length(first_byte); + if (seq_len < 0) goto malformed; + + out[out_len++] = (char)first_byte; + + for (int j = 1; j < seq_len; j++) { + unsigned char cont_byte; + if (decode_escape_sequence(str, len, &i, &cont_byte) < 0) goto malformed; + if (!is_valid_continuation(cont_byte)) goto malformed; + out[out_len++] = (char)cont_byte; + } + } + + out[out_len] = '\0'; + result = js_mkstr(js, out, out_len); + free(out); + return result; + +malformed: + free(out); + return js_mkerr(js, "URIError: URI malformed"); +} + +void init_uri_module(void) { + struct js *js = rt->js; + jsval_t glob = js_glob(js); + + js_set(js, glob, "encodeURI", js_mkfun(js_encodeURI)); + js_set(js, glob, "encodeURIComponent", js_mkfun(js_encodeURIComponent)); + js_set(js, glob, "decodeURI", js_mkfun(js_decodeURI)); + js_set(js, glob, "decodeURIComponent", js_mkfun(js_decodeURIComponent)); +} diff --git a/tests/test_uri.js b/tests/test_uri.js new file mode 100644 index 0000000..10a2355 --- /dev/null +++ b/tests/test_uri.js @@ -0,0 +1,101 @@ +console.log('=== URI Encoding/Decoding Tests ===\n'); + +let passed = 0; +let failed = 0; + +function test(name, actual, expected) { + if (actual === expected) { + console.log(`✓ ${name}`); + passed++; + } else { + console.log(`✗ ${name}`); + console.log(` Expected: ${expected}`); + console.log(` Actual: ${actual}`); + failed++; + } +} + +function testThrows(name, fn) { + try { + fn(); + console.log(`✗ ${name} (expected to throw)`); + failed++; + } catch (e) { + console.log(`✓ ${name} (threw)`); + passed++; + } +} + +// encodeURIComponent tests +console.log('\n--- encodeURIComponent ---'); +test('encodes space', encodeURIComponent(' '), '%20'); +test('encodes special chars', encodeURIComponent('hello world!'), 'hello%20world!'); +test('preserves unreserved', encodeURIComponent('abc123'), 'abc123'); +test('preserves unreserved marks', encodeURIComponent("-_.!~*'()"), "-_.!~*'()"); +test('encodes reserved chars', encodeURIComponent(';/?:@&=+$,#'), '%3B%2F%3F%3A%40%26%3D%2B%24%2C%23'); +test('encodes Cyrillic', encodeURIComponent('шеллы'), '%D1%88%D0%B5%D0%BB%D0%BB%D1%8B'); +test('encodes Chinese', encodeURIComponent('中文'), '%E4%B8%AD%E6%96%87'); +test('encodes emoji', encodeURIComponent('😀'), '%F0%9F%98%80'); +test('empty string', encodeURIComponent(''), ''); + +// encodeURI tests +console.log('\n--- encodeURI ---'); +test('preserves URI structure', encodeURI('https://example.com/path?q=hello world'), 'https://example.com/path?q=hello%20world'); +test('preserves reserved chars', encodeURI(';/?:@&=+$,#'), ';/?:@&=+$,#'); +test('encodes space', encodeURI('hello world'), 'hello%20world'); +test('encodes Cyrillic in URL', encodeURI('https://mozilla.org/?x=шеллы'), 'https://mozilla.org/?x=%D1%88%D0%B5%D0%BB%D0%BB%D1%8B'); +test('empty string', encodeURI(''), ''); + +// decodeURIComponent tests +console.log('\n--- decodeURIComponent ---'); +test('decodes space', decodeURIComponent('%20'), ' '); +test('decodes special chars', decodeURIComponent('hello%20world%21'), 'hello world!'); +test('decodes Cyrillic', decodeURIComponent('%D1%88%D0%B5%D0%BB%D0%BB%D1%8B'), 'шеллы'); +test('decodes Chinese', decodeURIComponent('%E4%B8%AD%E6%96%87'), '中文'); +test('decodes emoji', decodeURIComponent('%F0%9F%98%80'), '😀'); +test('decodes reserved chars', decodeURIComponent('%3B%2F%3F%3A%40%26%3D%2B%24%2C%23'), ';/?:@&=+$,#'); +test('passes through plain text', decodeURIComponent('hello'), 'hello'); +test('empty string', decodeURIComponent(''), ''); +test('mixed encoded/plain', decodeURIComponent('hello%20world'), 'hello world'); + +// decodeURI tests +console.log('\n--- decodeURI ---'); +test( + 'decodes URL with Cyrillic', + decodeURI('https://developer.mozilla.org/ru/docs/JavaScript_%D1%88%D0%B5%D0%BB%D0%BB%D1%8B'), + 'https://developer.mozilla.org/ru/docs/JavaScript_шеллы' +); +test('preserves encoded reserved', decodeURI('https://example.com/docs/JavaScript%3A%20test'), 'https://example.com/docs/JavaScript%3A test'); +test('decodes non-reserved', decodeURI('hello%20world'), 'hello world'); +test('empty string', decodeURI(''), ''); + +// decodeURI vs decodeURIComponent comparison +console.log('\n--- decodeURI vs decodeURIComponent ---'); +const encoded = 'https://developer.mozilla.org/docs/JavaScript%3A%20a_scripting_language'; +test('decodeURI preserves %3A', decodeURI(encoded), 'https://developer.mozilla.org/docs/JavaScript%3A a_scripting_language'); +test('decodeURIComponent decodes %3A', decodeURIComponent(encoded), 'https://developer.mozilla.org/docs/JavaScript: a_scripting_language'); + +// Error cases +console.log('\n--- Error cases ---'); +testThrows('decodeURIComponent invalid sequence', () => decodeURIComponent('%E0%A4%A')); +testThrows('decodeURI invalid sequence', () => decodeURI('%E0%A4%A')); +testThrows('decodeURIComponent incomplete %', () => decodeURIComponent('%')); +testThrows('decodeURIComponent incomplete %X', () => decodeURIComponent('%2')); +testThrows('decodeURIComponent invalid hex', () => decodeURIComponent('%GG')); + +// Round-trip tests +console.log('\n--- Round-trip tests ---'); +const testStrings = ['hello world', 'foo=bar&baz=qux', 'шеллы', '中文测试', 'emoji: 😀🎉', 'special: !@#$%^&*()', 'path/to/file.txt']; + +for (const str of testStrings) { + const encoded = encodeURIComponent(str); + const decoded = decodeURIComponent(encoded); + test(`round-trip: "${str}"`, decoded, str); +} + +// Summary +console.log('\n=== Summary ==='); +console.log(`Passed: ${passed}`); +console.log(`Failed: ${failed}`); + +if (failed > 0) process.exit(1);