From 35ca2d4c9d44c46e3217a5ba0478004217835161 Mon Sep 17 00:00:00 2001 From: Okiki Ojo Date: Fri, 22 May 2026 02:16:34 -0400 Subject: [PATCH] feat(unicode): Add Unicode-aware alignment helpers Add a configurable columnOffset hook to undent so alignment can follow something other than raw UTF-16 offsets. This introduces the @okikio/undent/unicode subpath with terminal-style width helpers, generated East Asian Width tables, and a sync script that keeps those tables auditable against Unicode upstream. Wire the new subpath into Deno and npm exports, document when to opt into Unicode visual alignment, and cover the new API with dedicated tests. While threading the new option through the pipeline, also preserve inherited trim settings, keep non-structural Unicode whitespace intact, and avoid mutating frozen aligned wrappers so the alignment path stays predictable for callers. --- _unicode_constants.ts | 317 +++++++++++++ deno.json | 11 +- deno.lock | 207 ++++++++- mod.ts | 181 ++++++-- mod_test.ts | 66 +++ readme.md | 42 +- scripts/build_npm.ts | 22 +- scripts/sync_unicode_east_asian_width.ts | 475 +++++++++++++++++++ unicode.ts | 564 +++++++++++++++++++++++ unicode_test.ts | 171 +++++++ 10 files changed, 2022 insertions(+), 34 deletions(-) create mode 100644 _unicode_constants.ts create mode 100644 scripts/sync_unicode_east_asian_width.ts create mode 100644 unicode.ts create mode 100644 unicode_test.ts diff --git a/_unicode_constants.ts b/_unicode_constants.ts new file mode 100644 index 0000000..bc572b1 --- /dev/null +++ b/_unicode_constants.ts @@ -0,0 +1,317 @@ +/** + * Internal East Asian Width lookup tables generated from Unicode upstream. + * + * `unicode.ts` re-exports these tables as part of the public API, but the + * generated data itself lives here so maintenance scripts can update one small + * internal file instead of rewriting the public module. + */ + +export const EAST_ASIAN_WIDE_RANGES: ReadonlyArray = [ + [0x1100, 0x115f], + [0x231a, 0x231b], + [0x2329, 0x232a], + [0x23e9, 0x23ec], + [0x23f0, 0x23f0], + [0x23f3, 0x23f3], + [0x25fd, 0x25fe], + [0x2614, 0x2615], + [0x2630, 0x2637], + [0x2648, 0x2653], + [0x267f, 0x267f], + [0x268a, 0x268f], + [0x2693, 0x2693], + [0x26a1, 0x26a1], + [0x26aa, 0x26ab], + [0x26bd, 0x26be], + [0x26c4, 0x26c5], + [0x26ce, 0x26ce], + [0x26d4, 0x26d4], + [0x26ea, 0x26ea], + [0x26f2, 0x26f3], + [0x26f5, 0x26f5], + [0x26fa, 0x26fa], + [0x26fd, 0x26fd], + [0x2705, 0x2705], + [0x270a, 0x270b], + [0x2728, 0x2728], + [0x274c, 0x274c], + [0x274e, 0x274e], + [0x2753, 0x2755], + [0x2757, 0x2757], + [0x2795, 0x2797], + [0x27b0, 0x27b0], + [0x27bf, 0x27bf], + [0x2b1b, 0x2b1c], + [0x2b50, 0x2b50], + [0x2b55, 0x2b55], + [0x2e80, 0x2e99], + [0x2e9b, 0x2ef3], + [0x2f00, 0x2fd5], + [0x2ff0, 0x303e], + [0x3041, 0x3096], + [0x3099, 0x30ff], + [0x3105, 0x312f], + [0x3131, 0x318e], + [0x3190, 0x31e5], + [0x31ef, 0x321e], + [0x3220, 0x3247], + [0x3250, 0xa48c], + [0xa490, 0xa4c6], + [0xa960, 0xa97c], + [0xac00, 0xd7a3], + [0xf900, 0xfaff], + [0xfe10, 0xfe19], + [0xfe30, 0xfe52], + [0xfe54, 0xfe66], + [0xfe68, 0xfe6b], + [0xff01, 0xff60], + [0xffe0, 0xffe6], + [0x16fe0, 0x16fe4], + [0x16ff0, 0x16ff6], + [0x17000, 0x18cd5], + [0x18cff, 0x18d1e], + [0x18d80, 0x18df2], + [0x1aff0, 0x1aff3], + [0x1aff5, 0x1affb], + [0x1affd, 0x1affe], + [0x1b000, 0x1b122], + [0x1b132, 0x1b132], + [0x1b150, 0x1b152], + [0x1b155, 0x1b155], + [0x1b164, 0x1b167], + [0x1b170, 0x1b2fb], + [0x1d300, 0x1d356], + [0x1d360, 0x1d376], + [0x1f004, 0x1f004], + [0x1f0cf, 0x1f0cf], + [0x1f18e, 0x1f18e], + [0x1f191, 0x1f19a], + [0x1f200, 0x1f202], + [0x1f210, 0x1f23b], + [0x1f240, 0x1f248], + [0x1f250, 0x1f251], + [0x1f260, 0x1f265], + [0x1f300, 0x1f320], + [0x1f32d, 0x1f335], + [0x1f337, 0x1f37c], + [0x1f37e, 0x1f393], + [0x1f3a0, 0x1f3ca], + [0x1f3cf, 0x1f3d3], + [0x1f3e0, 0x1f3f0], + [0x1f3f4, 0x1f3f4], + [0x1f3f8, 0x1f43e], + [0x1f440, 0x1f440], + [0x1f442, 0x1f4fc], + [0x1f4ff, 0x1f53d], + [0x1f54b, 0x1f54e], + [0x1f550, 0x1f567], + [0x1f57a, 0x1f57a], + [0x1f595, 0x1f596], + [0x1f5a4, 0x1f5a4], + [0x1f5fb, 0x1f64f], + [0x1f680, 0x1f6c5], + [0x1f6cc, 0x1f6cc], + [0x1f6d0, 0x1f6d2], + [0x1f6d5, 0x1f6d8], + [0x1f6dc, 0x1f6df], + [0x1f6eb, 0x1f6ec], + [0x1f6f4, 0x1f6fc], + [0x1f7e0, 0x1f7eb], + [0x1f7f0, 0x1f7f0], + [0x1f90c, 0x1f93a], + [0x1f93c, 0x1f945], + [0x1f947, 0x1f9ff], + [0x1fa70, 0x1fa7c], + [0x1fa80, 0x1fa8a], + [0x1fa8e, 0x1fac6], + [0x1fac8, 0x1fac8], + [0x1facd, 0x1fadc], + [0x1fadf, 0x1faea], + [0x1faef, 0x1faf8], + [0x20000, 0x2fffd], + [0x30000, 0x3fffd], +]; + +export const EAST_ASIAN_AMBIGUOUS_RANGES: ReadonlyArray = [ + [0xa1, 0xa1], + [0xa4, 0xa4], + [0xa7, 0xa8], + [0xaa, 0xaa], + [0xad, 0xae], + [0xb0, 0xb4], + [0xb6, 0xba], + [0xbc, 0xbf], + [0xc6, 0xc6], + [0xd0, 0xd0], + [0xd7, 0xd8], + [0xde, 0xe1], + [0xe6, 0xe6], + [0xe8, 0xea], + [0xec, 0xed], + [0xf0, 0xf0], + [0xf2, 0xf3], + [0xf7, 0xfa], + [0xfc, 0xfc], + [0xfe, 0xfe], + [0x101, 0x101], + [0x111, 0x111], + [0x113, 0x113], + [0x11b, 0x11b], + [0x126, 0x127], + [0x12b, 0x12b], + [0x131, 0x133], + [0x138, 0x138], + [0x13f, 0x142], + [0x144, 0x144], + [0x148, 0x14b], + [0x14d, 0x14d], + [0x152, 0x153], + [0x166, 0x167], + [0x16b, 0x16b], + [0x1ce, 0x1ce], + [0x1d0, 0x1d0], + [0x1d2, 0x1d2], + [0x1d4, 0x1d4], + [0x1d6, 0x1d6], + [0x1d8, 0x1d8], + [0x1da, 0x1da], + [0x1dc, 0x1dc], + [0x251, 0x251], + [0x261, 0x261], + [0x2c4, 0x2c4], + [0x2c7, 0x2c7], + [0x2c9, 0x2cb], + [0x2cd, 0x2cd], + [0x2d0, 0x2d0], + [0x2d8, 0x2db], + [0x2dd, 0x2dd], + [0x2df, 0x2df], + [0x300, 0x36f], + [0x391, 0x3a1], + [0x3a3, 0x3a9], + [0x3b1, 0x3c1], + [0x3c3, 0x3c9], + [0x401, 0x401], + [0x410, 0x44f], + [0x451, 0x451], + [0x2010, 0x2010], + [0x2013, 0x2016], + [0x2018, 0x2019], + [0x201c, 0x201d], + [0x2020, 0x2022], + [0x2024, 0x2027], + [0x2030, 0x2030], + [0x2032, 0x2033], + [0x2035, 0x2035], + [0x203b, 0x203b], + [0x203e, 0x203e], + [0x2074, 0x2074], + [0x207f, 0x207f], + [0x2081, 0x2084], + [0x20ac, 0x20ac], + [0x2103, 0x2103], + [0x2105, 0x2105], + [0x2109, 0x2109], + [0x2113, 0x2113], + [0x2116, 0x2116], + [0x2121, 0x2122], + [0x2126, 0x2126], + [0x212b, 0x212b], + [0x2153, 0x2154], + [0x215b, 0x215e], + [0x2160, 0x216b], + [0x2170, 0x2179], + [0x2189, 0x2189], + [0x2190, 0x2199], + [0x21b8, 0x21b9], + [0x21d2, 0x21d2], + [0x21d4, 0x21d4], + [0x21e7, 0x21e7], + [0x2200, 0x2200], + [0x2202, 0x2203], + [0x2207, 0x2208], + [0x220b, 0x220b], + [0x220f, 0x220f], + [0x2211, 0x2211], + [0x2215, 0x2215], + [0x221a, 0x221a], + [0x221d, 0x2220], + [0x2223, 0x2223], + [0x2225, 0x2225], + [0x2227, 0x222c], + [0x222e, 0x222e], + [0x2234, 0x2237], + [0x223c, 0x223d], + [0x2248, 0x2248], + [0x224c, 0x224c], + [0x2252, 0x2252], + [0x2260, 0x2261], + [0x2264, 0x2267], + [0x226a, 0x226b], + [0x226e, 0x226f], + [0x2282, 0x2283], + [0x2286, 0x2287], + [0x2295, 0x2295], + [0x2299, 0x2299], + [0x22a5, 0x22a5], + [0x22bf, 0x22bf], + [0x2312, 0x2312], + [0x2460, 0x24e9], + [0x24eb, 0x254b], + [0x2550, 0x2573], + [0x2580, 0x258f], + [0x2592, 0x2595], + [0x25a0, 0x25a1], + [0x25a3, 0x25a9], + [0x25b2, 0x25b3], + [0x25b6, 0x25b7], + [0x25bc, 0x25bd], + [0x25c0, 0x25c1], + [0x25c6, 0x25c8], + [0x25cb, 0x25cb], + [0x25ce, 0x25d1], + [0x25e2, 0x25e5], + [0x25ef, 0x25ef], + [0x2605, 0x2606], + [0x2609, 0x2609], + [0x260e, 0x260f], + [0x261c, 0x261c], + [0x261e, 0x261e], + [0x2640, 0x2640], + [0x2642, 0x2642], + [0x2660, 0x2661], + [0x2663, 0x2665], + [0x2667, 0x266a], + [0x266c, 0x266d], + [0x266f, 0x266f], + [0x269e, 0x269f], + [0x26bf, 0x26bf], + [0x26c6, 0x26cd], + [0x26cf, 0x26d3], + [0x26d5, 0x26e1], + [0x26e3, 0x26e3], + [0x26e8, 0x26e9], + [0x26eb, 0x26f1], + [0x26f4, 0x26f4], + [0x26f6, 0x26f9], + [0x26fb, 0x26fc], + [0x26fe, 0x26ff], + [0x273d, 0x273d], + [0x2776, 0x277f], + [0x2b56, 0x2b59], + [0x3248, 0x324f], + [0xe000, 0xf8ff], + [0xfe00, 0xfe0f], + [0xfffd, 0xfffd], + [0x1f100, 0x1f10a], + [0x1f110, 0x1f12d], + [0x1f130, 0x1f169], + [0x1f170, 0x1f18d], + [0x1f18f, 0x1f190], + [0x1f19b, 0x1f1ac], + [0xe0100, 0xe01ef], + [0xf0000, 0xffffd], + [0x100000, 0x10fffd], +]; + +// End of generated East Asian Width tables. \ No newline at end of file diff --git a/deno.json b/deno.json index 094538f..3a34224 100644 --- a/deno.json +++ b/deno.json @@ -1,12 +1,17 @@ { "name": "@okikio/undent", "version": "0.2.1", - "exports": "./mod.ts", + "exports": { + ".": "./mod.ts", + "./unicode": "./unicode.ts" + }, "tasks": { "test": "deno test --trace-leaks --v8-flags=--expose-gc", "bench": "deno bench --allow-env=NODE_DISABLE_COLORS --v8-flags=--expose-gc", "forge": "deno run -A jsr:@roka/forge", - "build:npm": "deno run -A scripts/build_npm.ts" + "build:npm": "deno run -A scripts/build_npm.ts", + "unicode:eaw:check": "deno run --allow-net=www.unicode.org --allow-env=NAPI_RS_NATIVE_LIBRARY_PATH,NAPI_RS_ENFORCE_VERSION_CHECK,NAPI_RS_FORCE_WASI --allow-ffi scripts/sync_unicode_east_asian_width.ts", + "unicode:eaw:update": "deno run --allow-net=www.unicode.org --allow-env=NAPI_RS_NATIVE_LIBRARY_PATH,NAPI_RS_ENFORCE_VERSION_CHECK,NAPI_RS_FORCE_WASI --allow-ffi --allow-read=_unicode_constants.ts --allow-write=_unicode_constants.ts scripts/sync_unicode_east_asian_width.ts --write" }, "fmt": { "proseWrap": "preserve" @@ -14,6 +19,8 @@ "publish": { "include": [ "mod.ts", + "_unicode_constants.ts", + "unicode.ts", "changelog.md", "license", "readme.md" diff --git a/deno.lock b/deno.lock index 4b4b3e5..f315298 100644 --- a/deno.lock +++ b/deno.lock @@ -3,6 +3,7 @@ "specifiers": { "jsr:@david/code-block-writer@^13.0.3": "13.0.3", "jsr:@deno/dnt@*": "0.42.3", + "jsr:@optique/core@*": "1.0.2", "jsr:@std/assert@^1.0.14": "1.0.18", "jsr:@std/assert@^1.0.17": "1.0.18", "jsr:@std/expect@*": "1.0.17", @@ -10,15 +11,21 @@ "jsr:@std/fs@1": "1.0.22", "jsr:@std/internal@^1.0.10": "1.0.12", "jsr:@std/internal@^1.0.12": "1.0.12", + "jsr:@std/path@*": "1.1.4", "jsr:@std/path@1": "1.1.4", "jsr:@std/path@^1.1.4": "1.1.4", "jsr:@std/testing@*": "1.0.17", "jsr:@ts-morph/bootstrap@0.27": "0.27.0", "jsr:@ts-morph/common@0.27": "0.27.0", + "npm:@babel/parser@7.29.3": "7.29.3", + "npm:@oxc-parser/binding-wasm32-wasi@0.132.0": "0.132.0", "npm:dedent@*": "1.7.1", "npm:fast-check@*": "4.5.3", "npm:mitata@*": "1.0.34", - "npm:outdent@*": "0.8.0" + "npm:outdent@*": "0.8.0", + "npm:oxc-parser@0.132.0": "0.132.0", + "npm:typescript@*": "6.0.3", + "npm:typescript@6.0.3": "6.0.3" }, "jsr": { "@david/code-block-writer@13.0.3": { @@ -34,6 +41,9 @@ "jsr:@ts-morph/bootstrap" ] }, + "@optique/core@1.0.2": { + "integrity": "3f90a2965286cd4d4d103f6605ed9a00951393dbab0748402576350ad3780b33" + }, "@std/assert@1.0.18": { "integrity": "270245e9c2c13b446286de475131dc688ca9abcd94fc5db41d43a219b34d1c78", "dependencies": [ @@ -88,6 +98,166 @@ } }, "npm": { + "@babel/helper-string-parser@7.27.1": { + "integrity": "sha512-qMlSxKbpRlAridDExk92nSobyDdpPijUq2DW6oDnUqd0iOGxmQjyqhMIihI9+zv4LPyZdRje2cavWPbCbWm3eA==" + }, + "@babel/helper-validator-identifier@7.28.5": { + "integrity": "sha512-qSs4ifwzKJSV39ucNjsvc6WVHs6b7S03sOh2OcHF9UHfVPqWWALUsNUVzhSBiItjRZoLHx7nIarVjqKVusUZ1Q==" + }, + "@babel/parser@7.29.3": { + "integrity": "sha512-b3ctpQwp+PROvU/cttc4OYl4MzfJUWy6FZg+PMXfzmt/+39iHVF0sDfqay8TQM3JA2EUOyKcFZt75jWriQijsA==", + "dependencies": [ + "@babel/types" + ], + "bin": true + }, + "@babel/types@7.29.0": { + "integrity": "sha512-LwdZHpScM4Qz8Xw2iKSzS+cfglZzJGvofQICy7W7v4caru4EaAmyUuO6BGrbyQ2mYV11W0U8j5mBhd14dd3B0A==", + "dependencies": [ + "@babel/helper-string-parser", + "@babel/helper-validator-identifier" + ] + }, + "@emnapi/core@1.10.0": { + "integrity": "sha512-yq6OkJ4p82CAfPl0u9mQebQHKPJkY7WrIuk205cTYnYe+k2Z8YBh11FrbRG/H6ihirqcacOgl2BIO8oyMQLeXw==", + "dependencies": [ + "@emnapi/wasi-threads", + "tslib" + ] + }, + "@emnapi/runtime@1.10.0": { + "integrity": "sha512-ewvYlk86xUoGI0zQRNq/mC+16R1QeDlKQy21Ki3oSYXNgLb45GV1P6A0M+/s6nyCuNDqe5VpaY84BzXGwVbwFA==", + "dependencies": [ + "tslib" + ] + }, + "@emnapi/wasi-threads@1.2.1": { + "integrity": "sha512-uTII7OYF+/Mes/MrcIOYp5yOtSMLBWSIoLPpcgwipoiKbli6k322tcoFsxoIIxPDqW01SQGAgko4EzZi2BNv2w==", + "dependencies": [ + "tslib" + ] + }, + "@napi-rs/wasm-runtime@1.1.4_@emnapi+core@1.10.0_@emnapi+runtime@1.10.0": { + "integrity": "sha512-3NQNNgA1YSlJb/kMH1ildASP9HW7/7kYnRI2szWJaofaS1hWmbGI4H+d3+22aGzXXN9IJ+n+GiFVcGipJP18ow==", + "dependencies": [ + "@emnapi/core", + "@emnapi/runtime", + "@tybys/wasm-util" + ] + }, + "@oxc-parser/binding-android-arm-eabi@0.132.0": { + "integrity": "sha512-KrLaPWa5c9Y7LkW+rKkaUE3y7DBDrQtaf7rlsSDfv6KAHUjgzAIRA761Lrrp6//Yd/Rlie/yEOt9YENCoJnOcw==", + "os": ["android"], + "cpu": ["arm"] + }, + "@oxc-parser/binding-android-arm64@0.132.0": { + "integrity": "sha512-SThDrSeamB/kG2+NxcJ5/wSLcV6dUqDknrPLqFYQ0ST/55mtBP4M7Q/f3QbubH6aAd11wpzZn/nwbVRSdobOpg==", + "os": ["android"], + "cpu": ["arm64"] + }, + "@oxc-parser/binding-darwin-arm64@0.132.0": { + "integrity": "sha512-Lc0f/TYoKBghE5/2Gsv7bLXk+TJZunx2Tf61X8hG4ARXdc8UYI26dCGccFSd1AyFbK3jfaNXtMnupggDbjPXdQ==", + "os": ["darwin"], + "cpu": ["arm64"] + }, + "@oxc-parser/binding-darwin-x64@0.132.0": { + "integrity": "sha512-RG2eJIpf7C21z9HSSXFw1bTArdpKe7Y4fwcJTwRq1yCSe1vSavaN9GA1sm9KqzemTLAGVktQ+7qBTGp0vQeUZg==", + "os": ["darwin"], + "cpu": ["x64"] + }, + "@oxc-parser/binding-freebsd-x64@0.132.0": { + "integrity": "sha512-wQIPntPLtJ8NcBpvKPbEv3NqzV6k8eP8tP/jE9Rg8HTg/j7urZGFSsTCPCW5k77Qfw2DM4vRvc9p3I4yq/Shvw==", + "os": ["freebsd"], + "cpu": ["x64"] + }, + "@oxc-parser/binding-linux-arm-gnueabihf@0.132.0": { + "integrity": "sha512-PixKEpeSe3yxQWqNyOCBALRYc72+Tj7ILDofUl3iXo25cVOzLA6jHUhmOINRtWIPh7dbUie3QNeabwaQpZTw6w==", + "os": ["linux"], + "cpu": ["arm"] + }, + "@oxc-parser/binding-linux-arm-musleabihf@0.132.0": { + "integrity": "sha512-sCR+DzGHlyHKnbA2z9zWjTUhIo8Sy0enJl4RDsBwPmkxYynPatpwOAWe8W5127SlW0boqUWHGtr1NWn5UwIhXQ==", + "os": ["linux"], + "cpu": ["arm"] + }, + "@oxc-parser/binding-linux-arm64-gnu@0.132.0": { + "integrity": "sha512-sQBix5P2cW+IpzTcCwYxnh9yALrKSIkKJThspBvMGcygSMnbzkSvhN7SfuX1hvBk8y1XEChsdkU3ET0V5DmzUw==", + "os": ["linux"], + "cpu": ["arm64"] + }, + "@oxc-parser/binding-linux-arm64-musl@0.132.0": { + "integrity": "sha512-WozHg3Kc//8Sk756HXXgMbEAvqtG+Lzb9JOojwQzIGDtN78Az2dLttkb71akWYUF/8IgYfDSlfKh4Uot8is5Vw==", + "os": ["linux"], + "cpu": ["arm64"] + }, + "@oxc-parser/binding-linux-ppc64-gnu@0.132.0": { + "integrity": "sha512-CmX/ulNBOEwWTyVRmcpYKAcAizW6+OjtLJgo7fXoL9OqQvjF4VER8tPomv44vwzfSCy1BHbsB0ZlZYzYJNj4cA==", + "os": ["linux"], + "cpu": ["ppc64"] + }, + "@oxc-parser/binding-linux-riscv64-gnu@0.132.0": { + "integrity": "sha512-j9oQS+hM90SdhviNGWbPgT4+Rlq+ac++q/zjgwPD1mVHgxHzATvoRGtDx0sXGmFOQ9J9YkwAhYGb5MAHL6TAsA==", + "os": ["linux"], + "cpu": ["riscv64"] + }, + "@oxc-parser/binding-linux-riscv64-musl@0.132.0": { + "integrity": "sha512-bLz+Xi+Agnfmd7kWPEsSVwCn2k4EyIalZkNBcQ0OGIv9rqn8VgCPLNd03tM9mKX/5TdlvDXalz0q71BIrOPNqg==", + "os": ["linux"], + "cpu": ["riscv64"] + }, + "@oxc-parser/binding-linux-s390x-gnu@0.132.0": { + "integrity": "sha512-U6t2qbJU0ypTfyj9QV3W1Y6mITDTL8ai/OR6NUn85vyHthOvobKWgXzU4tu0EskSzlpuVFz1g0jFGulDIUKHxQ==", + "os": ["linux"], + "cpu": ["s390x"] + }, + "@oxc-parser/binding-linux-x64-gnu@0.132.0": { + "integrity": "sha512-WcEaSNHFk8yz5YFlQQAlhq6jOFmZBB/RKE7uzhyCIf+pF1Lmv9gUH4221mle2Gd9iHyWT3ySNph8yZgb1xYdWg==", + "os": ["linux"], + "cpu": ["x64"] + }, + "@oxc-parser/binding-linux-x64-musl@0.132.0": { + "integrity": "sha512-iQrV4iJzQgRwK3BWRmQl1C3C6g3wYpXN2WLdQdyR+efoUnncdShZAVp9OgcojtlD3MDRbuOMGG3SjxF4fL4nlQ==", + "os": ["linux"], + "cpu": ["x64"] + }, + "@oxc-parser/binding-openharmony-arm64@0.132.0": { + "integrity": "sha512-FWzmUGrZ6GUby4U7WIwcCtab6tdmlTO3xTRRKyb5kjIJVEiaUAT8animUG/nK8ZCA8gkRkPOTId4rl6uTqUmJQ==", + "os": ["openharmony"], + "cpu": ["arm64"] + }, + "@oxc-parser/binding-wasm32-wasi@0.132.0": { + "integrity": "sha512-TlbMppxJI5CjWDes0QaP6G3aneVg1yikBu5QYI+DUShF9WDL66ccgKFNNGmi/Wybtszw6hxwAvv76T4DaPKnHw==", + "dependencies": [ + "@emnapi/core", + "@emnapi/runtime", + "@napi-rs/wasm-runtime" + ], + "cpu": ["wasm32"] + }, + "@oxc-parser/binding-win32-arm64-msvc@0.132.0": { + "integrity": "sha512-RH/NbFjGKqdUAUi7Oh3LQPxUk2hsWFEEQ38HSnbRQT8QjBZFKqL1fMbmsB3N4jy/KPh9iX94+9dmkEMBBbambw==", + "os": ["win32"], + "cpu": ["arm64"] + }, + "@oxc-parser/binding-win32-ia32-msvc@0.132.0": { + "integrity": "sha512-JUr4jQY9jxoIB/YTLXr6XofSi5xikj6p5/Ns1h0VOBDT0j1jKU+kMsv2xxv51RwnETcXpA1Yw/9oUAfcqfaqEA==", + "os": ["win32"], + "cpu": ["ia32"] + }, + "@oxc-parser/binding-win32-x64-msvc@0.132.0": { + "integrity": "sha512-2dapgHpA5X8DSXF4AU36hJWYf6zP0tKjMXFRAZFBD62pkevW/uhFDXoFH9Y/3Fd2EtDrw5ByNnR1wVE9X9y0SQ==", + "os": ["win32"], + "cpu": ["x64"] + }, + "@oxc-project/types@0.132.0": { + "integrity": "sha512-FESMOxil5Se014ui/Eq8fT5uHJo6nIRwH0PfJrZJXs6Gek3ZVFOrpUv3YIZT20m+extU98Hg1Ym72U58rlsxUQ==" + }, + "@tybys/wasm-util@0.10.2": { + "integrity": "sha512-RoBvJ2X0wuKlWFIjrwffGw1IqZHKQqzIchKaadZZfnNpsAYp2mM0h36JtPCjNDAHGgYez/15uMBpfGwchhiMgg==", + "dependencies": [ + "tslib" + ] + }, "dedent@1.7.1": { "integrity": "sha512-9JmrhGZpOlEgOLdQgSm0zxFaYoQon408V1v49aqTWuXENVlnCuY9JBZcXZiCsZQWDjTm5Qf/nIvAy77mXDAjEg==" }, @@ -103,8 +273,43 @@ "outdent@0.8.0": { "integrity": "sha512-KiOAIsdpUTcAXuykya5fnVVT+/5uS0Q1mrkRHcF89tpieSmY33O/tmc54CqwA+bfhbtEfZUNLHaPUiB9X3jt1A==" }, + "oxc-parser@0.132.0": { + "integrity": "sha512-+0LAPHaqtfQlvWdpaAa09SmOaZZgP8C552xosEkGJ4+ruEwP1Vgx+sqBgcBCNfR6KDCmagGOZTde8wmAvcI/Hg==", + "dependencies": [ + "@oxc-project/types" + ], + "optionalDependencies": [ + "@oxc-parser/binding-android-arm-eabi", + "@oxc-parser/binding-android-arm64", + "@oxc-parser/binding-darwin-arm64", + "@oxc-parser/binding-darwin-x64", + "@oxc-parser/binding-freebsd-x64", + "@oxc-parser/binding-linux-arm-gnueabihf", + "@oxc-parser/binding-linux-arm-musleabihf", + "@oxc-parser/binding-linux-arm64-gnu", + "@oxc-parser/binding-linux-arm64-musl", + "@oxc-parser/binding-linux-ppc64-gnu", + "@oxc-parser/binding-linux-riscv64-gnu", + "@oxc-parser/binding-linux-riscv64-musl", + "@oxc-parser/binding-linux-s390x-gnu", + "@oxc-parser/binding-linux-x64-gnu", + "@oxc-parser/binding-linux-x64-musl", + "@oxc-parser/binding-openharmony-arm64", + "@oxc-parser/binding-wasm32-wasi", + "@oxc-parser/binding-win32-arm64-msvc", + "@oxc-parser/binding-win32-ia32-msvc", + "@oxc-parser/binding-win32-x64-msvc" + ] + }, "pure-rand@7.0.1": { "integrity": "sha512-oTUZM/NAZS8p7ANR3SHh30kXB+zK2r2BPcEn/awJIbOvq82WoMN4p62AWWp3Hhw50G0xMsw1mhIBLqHw64EcNQ==" + }, + "tslib@2.8.1": { + "integrity": "sha512-oJFu94HQb+KVduSUQL7wnpmqnfmLsOA/nAh6b6EH0wCEoK0/mPeXU6c3wKDV83MkOuHPRHtSXKKU99IBazS/2w==" + }, + "typescript@6.0.3": { + "integrity": "sha512-y2TvuxSZPDyQakkFRPZHKFm+KKVqIisdg9/CZwm9ftvKXLP8NRWj38/ODjNbr43SsoXqNuAisEf1GdCxqWcdBw==", + "bin": true } } } diff --git a/mod.ts b/mod.ts index 7185382..373631f 100644 --- a/mod.ts +++ b/mod.ts @@ -46,7 +46,9 @@ * Both paths share the same guarantees: non-whitespace content is never * removed, newlines in interpolated values are never normalized, and * multi-line values can be aligned at their insertion column with - * {@link align} or {@link embed}. + * {@link align} or {@link embed}. By default that insertion column is measured + * with {@link columnOffset}, and callers can override that policy with the + * `columnOffset` option when they need Unicode-aware visual alignment. * * @module */ @@ -81,6 +83,18 @@ export interface TrimSides { trailing?: TrimMode; } +/** + * Measure the current insertion column for alignment. + * + * `undent` calls this with the output accumulated so far and expects the + * number of spaces to prepend before later lines of an aligned value. + * + * The default implementation is {@link columnOffset}, which counts UTF-16 + * code units after the last newline. Override it when you need alignment to + * follow a different visual-width policy. + */ +export type ColumnOffsetFunction = (text: string) => number; + /** * Options for configuring an `undent` instance. * @@ -142,6 +156,18 @@ export interface UndentOptions { * @default false */ alignValues?: boolean; + + /** + * Measure the insertion column used by {@link align}, {@link embed}, and + * {@link alignValues}. + * + * The default is {@link columnOffset}, which counts UTF-16 code units after + * the last newline. Override this when you need alignment to follow a custom + * display-width policy such as Unicode terminal columns. + * + * @default columnOffset + */ + columnOffset?: ColumnOffsetFunction; } /** @@ -261,6 +287,8 @@ export interface ResolvedOptions { newline: string | null; /** When `true`, every multi-line interpolated value is automatically aligned at its insertion column. */ alignValues: boolean; + /** How to measure the insertion column used for alignment padding. */ + columnOffset: ColumnOffsetFunction; } // ========================================================================== @@ -307,8 +335,14 @@ export const indent: unique symbol = Symbol("undent.indent"); */ export const ALIGNED: unique symbol = Symbol("undent.aligned"); -/** Internal symbol for per-value aligned-text memoization. */ -const ALIGNED_TEXT_CACHE: unique symbol = Symbol("undent.alignedTextCache"); +/** + * Per-wrapper memoization for aligned text. + * + * We keep this cache outside the public wrapper object so rendering can reuse + * aligned output without mutating values returned by {@link align} or + * {@link embed}. + */ +const ALIGNED_TEXT_CACHE = new WeakMap>(); // Character codes used in hot loops. // Hex is compact for low-level scanning, so we document each value: @@ -347,10 +381,6 @@ export interface AlignedValue { readonly value: string; } -interface InternalAlignedValue extends AlignedValue { - [ALIGNED_TEXT_CACHE]?: Map; -} - /** * Mark an interpolated value for column alignment. * @@ -494,6 +524,7 @@ export const DEFAULTS: ResolvedOptions = { trimTrailing: "all", newline: null, alignValues: false, + columnOffset, }; /** @@ -666,7 +697,7 @@ function undentTag( // Fast path: when alignValues is true, always use aligned join. if (state.opts.alignValues) { - return joinAligned(segments, effectiveValues, true); + return joinAligned(state.opts, segments, effectiveValues, true); } // Common path: try plain join, bail to aligned if we hit a wrapped value. @@ -676,7 +707,7 @@ function undentTag( const raw = effectiveValues[i]; if (typeof raw === "object" && raw !== null && ALIGNED in raw) { // Found an aligned value — switch to aligned join for entire template. - return joinAligned(segments, effectiveValues, false); + return joinAligned(state.opts, segments, effectiveValues, false); } out += String(raw) + (segments[i + 1] ?? ""); } @@ -722,6 +753,9 @@ export function resolveOptions( if (options.alignValues !== undefined) { resolved.alignValues = options.alignValues; } + if (options.columnOffset !== undefined) { + resolved.columnOffset = options.columnOffset; + } if (options.newline !== undefined) { if (options.newline === null) resolved.newline = null; @@ -735,8 +769,8 @@ export function resolveOptions( resolved.trimLeading = options.trim; resolved.trimTrailing = options.trim; } else { - resolved.trimLeading = options.trim.leading ?? "all"; - resolved.trimTrailing = options.trim.trailing ?? "all"; + resolved.trimLeading = options.trim.leading ?? base.trimLeading; + resolved.trimTrailing = options.trim.trailing ?? base.trimTrailing; } } @@ -778,7 +812,7 @@ function getProcessedSegments( if (cached) return cached; const effectiveStrings = anchored - ? Array.prototype.slice.call(strings, 1) as string[] + ? strings.slice(1) : strings; // When anchored, the anchor's column IS the indent level — content @@ -1125,7 +1159,7 @@ function processStrings( if ( i === 0 && i === last && opts.trimLeading === "all" && opts.trimTrailing === "all" && - s.length > 0 && s.trim().length === 0 + s.length > 0 && isStructuralWhitespaceOnly(s) ) { s = ""; } @@ -1204,6 +1238,28 @@ export function dedentString( ): string { const len = input.length; if (len === 0) return ""; + const mayTrimLeading = trimLeading !== "none" && hasLeadingBlankLine(input); + const mayTrimTrailing = trimTrailing !== "none" && hasTrailingBlankLine(input); + + // Fast path for the common hot-path string case: a single logical line. + // In that shape, dedenting reduces to stripping leading spaces/tabs from the + // only content line. We can answer that by scanning the prefix instead of the + // whole string, which especially helps already-clean strings and very large + // single-line inputs. + if (input.indexOf("\n") === -1 && input.indexOf("\r") === -1) { + let firstNonWs = 0; + while (firstNonWs < len) { + const c = input.charCodeAt(firstNonWs); + if (c !== CC_SPACE && c !== CC_TAB) break; + firstNonWs++; + } + + if (firstNonWs === len) { + return trimLeading === "all" && trimTrailing === "all" ? "" : input; + } + + return firstNonWs === 0 ? input : input.slice(firstNonWs); + } // Pass 1: find minimum indent across non-blank lines. // Blank lines do not influence minIndent; they are structural only. @@ -1223,6 +1279,13 @@ export function dedentString( const c = input.charCodeAt(i); if (c !== CC_LF && c !== CC_CR) { const ws = i - lineStart; + // Once any content line starts at column 0, the common indent is fixed + // at 0 for the whole string. If there are also no blank wrapper lines to + // trim, dedenting cannot change the input, so we can return it without a + // full second pass. + if (ws === 0 && !mayTrimLeading && !mayTrimTrailing) { + return input; + } if (ws < minIndent) { minIndent = ws; if (ws === 0) break; // Can't go lower. @@ -1302,6 +1365,40 @@ export function dedentString( return result; } +/** + * Return true when the string starts with a blank line. + * + * A blank leading line is optional spaces/tabs followed by a newline sequence. + * This helper exists so dedentString() can cheaply decide whether trim work is + * even possible before it commits to a full multi-pass transformation. + */ +function hasLeadingBlankLine(text: string): boolean { + for (let i = 0; i < text.length; i++) { + const c = text.charCodeAt(i); + if (c === CC_SPACE || c === CC_TAB) continue; + return c === CC_LF || c === CC_CR; + } + + return false; +} + +/** + * Return true when the string ends with a blank line. + * + * This is the trailing-edge mirror of hasLeadingBlankLine(). It scans backward + * through optional spaces/tabs and reports whether the first non-horizontal + * whitespace byte is a newline sequence. + */ +function hasTrailingBlankLine(text: string): boolean { + for (let i = text.length - 1; i >= 0; i--) { + const c = text.charCodeAt(i); + if (c === CC_SPACE || c === CC_TAB) continue; + return c === CC_LF || c === CC_CR; + } + + return false; +} + /** * Trim mode "one" for the leading edge. * @@ -1435,6 +1532,7 @@ function trimTrailingBlankLinesAll(text: string): number { * 4. Otherwise, stringify and concatenate directly. */ function joinAligned( + opts: ResolvedOptions, strings: ReadonlyArray, values: ReadonlyArray, alignAll: boolean, @@ -1449,10 +1547,10 @@ function joinAligned( if (wrapped) { // Wrapped values always align. For hot loops with repeated values, // this path memoizes alignment by pad width and reuses results. - const pad = " ".repeat(columnOffset(out)); + const pad = " ".repeat(opts.columnOffset(out)); out += getAlignedWrappedText(raw, pad); } else if (alignAll && hasNewline(text)) { - out += alignText(text, " ".repeat(columnOffset(out))); + out += alignText(text, " ".repeat(opts.columnOffset(out))); } else { out += text; } @@ -1564,12 +1662,34 @@ function hasNewline(text: string): boolean { } /** - * Return aligned text for a wrapped value, using a small per-value - * cache keyed by the pad string. + * Return true when text contains only structural whitespace. * - * Targets hot `embed(...)` loops where both the value and insertion - * column repeat across iterations. Cache is bounded to - * {@link ALIGNED_TEXT_CACHE_MAX} entries per wrapped value. + * This helper is intentionally narrower than `String.prototype.trim()`. The + * template pipeline treats only spaces, tabs, and newline bytes as formatting + * characters, so Unicode whitespace such as NBSP should remain content. + */ +function isStructuralWhitespaceOnly(text: string): boolean { + for (let i = 0; i < text.length; i++) { + const c = text.charCodeAt(i); + if (c !== CC_SPACE && c !== CC_TAB && c !== CC_LF && c !== CC_CR) { + return false; + } + } + + return true; +} + +/** + * Return aligned text for a wrapped value. + * + * Repeated `align(...)` and `embed(...)` calls often reuse the same wrapper at + * the same insertion column. This helper memoizes those padded results so hot + * loops can skip recomputing identical alignment work. + * + * > The cache is keyed by the pad string and capped at + * > {@link ALIGNED_TEXT_CACHE_MAX} entries per wrapped value. Keeping the + * > cache small preserves the common fast path without letting rarely repeated + * > columns grow memory use without bound. */ function getAlignedWrappedText(value: AlignedValue, pad: string): string { const text = value.value; @@ -1577,8 +1697,7 @@ function getAlignedWrappedText(value: AlignedValue, pad: string): string { return text; } - const internal = value as InternalAlignedValue; - let cache = internal[ALIGNED_TEXT_CACHE]; + let cache = ALIGNED_TEXT_CACHE.get(value); if (cache) { const hit = cache.get(pad); if (hit !== undefined) return hit; @@ -1588,7 +1707,7 @@ function getAlignedWrappedText(value: AlignedValue, pad: string): string { if (!cache) { cache = new Map(); - internal[ALIGNED_TEXT_CACHE] = cache; + ALIGNED_TEXT_CACHE.set(value, cache); } if (cache.size >= ALIGNED_TEXT_CACHE_MAX) { @@ -1715,18 +1834,22 @@ export function rejoinLines( } /** - * Count characters from the last newline to the end of the string. + * Count how far the output has advanced since the last newline. + * + * Alignment uses this insertion offset to decide how many spaces to add before + * later lines of a wrapped value. * - * This gives the "column offset" — the horizontal position where the - * next character would appear. Used internally by alignment to decide - * how many spaces to pad. + * > This is a UTF-16 code-unit offset, not display width. That keeps the + * > helper fast and deterministic for string processing, but editors may show + * > a different visual column for tabs, emoji, combining marks, or full-width + * > characters. * * Uses `lastIndexOf` (implemented in C++ by V8) instead of a charcode * loop for ~100x speedup on long strings. * * @param text - The string to measure. - * @returns The number of characters after the final newline, or the - * full string length if there are no newlines. + * @returns The number of UTF-16 code units after the final newline, + * or the full string length if there are no newlines. * * @example Measuring the insertion column * ```ts diff --git a/mod_test.ts b/mod_test.ts index 32b5945..1797026 100644 --- a/mod_test.ts +++ b/mod_test.ts @@ -21,6 +21,7 @@ import undent, { } from "./mod.ts"; import type { AlignedValue, + ColumnOffsetFunction, ResolvedOptions, TrimMode, TrimSides, @@ -151,6 +152,11 @@ World World`; expect(result).toBe("Hello\nWorld"); }); + + it("preserves non-structural Unicode whitespace", () => { + const result = undent`\u00A0`; + expect(result).toBe("\u00A0"); + }); }); // ------------------------------------------------------------------------- @@ -415,6 +421,17 @@ World `; expect(result).toBe("\r\nfirst\r\nsecond"); }); + + it("inherits unspecified trim sides across chained .with() calls", () => { + const keep = undent.with({ trim: "none" }); + const next = keep.with({ trim: { leading: "one" } }); + + const result = next` + Hello + `; + + expect(result).toBe("Hello\n"); + }); }); describe("alignValues option", () => { @@ -482,6 +499,21 @@ World "class Foo {\n greet() {\n console.log('hi');\n }\n\n bye() {\n console.log('bye');\n }\n}", ); }); + + it("uses a custom columnOffset function when aligning values", () => { + const doubleWidth: ColumnOffsetFunction = (text) => + columnOffset(text) * 2; + const ua = undent.with({ + alignValues: true, + columnOffset: doubleWidth, + }); + + const result = ua` + > ${"a\nb"} + `; + + expect(result).toBe("> a\n b"); + }); }); }); @@ -538,6 +570,16 @@ World const result = undent.string(" hello\n world\nfoo"); expect(result).toBe(" hello\n world\nfoo"); }); + + it("returns already-clean multi-line strings unchanged", () => { + const input = "alpha\nbeta\ngamma"; + expect(undent.string(input)).toBe(input); + }); + + it("returns mixed-indent strings unchanged when one line is already at column 0", () => { + const input = " hello\nworld\n again"; + expect(undent.string(input)).toBe(input); + }); }); // ------------------------------------------------------------------------- @@ -912,6 +954,17 @@ World expect(result).toBe("before after"); }); + it("supports frozen aligned values", () => { + const value = Object.freeze(align("a\nb")); + + const result = undent` + list: + ${value} + `; + + expect(result).toBe("list:\n a\n b"); + }); + it("handles value with only newlines", () => { const result = undent` before ${align("\n\n")} after @@ -1371,6 +1424,12 @@ World expect(result.strategy).toBe("first"); }); + it("merges columnOffset", () => { + const custom: ColumnOffsetFunction = (text) => columnOffset(text) + 1; + const result = resolveOptions(DEFAULTS, { columnOffset: custom }); + expect(result.columnOffset).toBe(custom); + }); + it("merges trim string", () => { const result = resolveOptions(DEFAULTS, { trim: "none" }); expect(result.trimLeading).toBe("none"); @@ -1417,6 +1476,7 @@ World expect(DEFAULTS.trimTrailing).toBe("all"); expect(DEFAULTS.newline).toBe(null); expect(DEFAULTS.alignValues).toBe(false); + expect(DEFAULTS.columnOffset).toBe(columnOffset); }); }); @@ -1473,6 +1533,11 @@ World const a: AlignedValue = align("x"); expect(isAligned(a)).toBe(true); }); + + it("ColumnOffsetFunction is usable", () => { + const fn: ColumnOffsetFunction = columnOffset; + expect(fn('x\ny')).toBe(1); + }); }); // ========================================================================= @@ -2232,6 +2297,7 @@ World trimTrailing: "none", newline: "\r\n", alignValues: true, + columnOffset, }; const result = resolveOptions(custom, {}); expect(result).toEqual(custom); diff --git a/readme.md b/readme.md index 307d0c7..577e0f5 100644 --- a/readme.md +++ b/readme.md @@ -207,6 +207,12 @@ undent` // end ``` +By default, that insertion column is measured with JavaScript string offsets +after the last newline. That is fast and stable for code generation, but it is +not the same thing as visual width in a terminal or editor. Tabs, combining +marks, emoji, and full-width characters can render at different visual columns +than their UTF-16 length suggests. + When the value itself carries baked-in indentation — a SQL snippet from another file, a code block from a constant — use `embed()`. It strips the value's own indentation first, then aligns it at the insertion column: @@ -252,6 +258,31 @@ u` > `align()` and `embed()` always align regardless of the `alignValues` setting — > they're the per-value opt-in. +### Unicode visual alignment + +If you need terminal-style Unicode alignment, opt into the separate Unicode +helpers subpath instead of changing the default behavior for every caller: + +```ts +import { undent } from "@okikio/undent"; +import { createUnicodeColumnOffset } from "@okikio/undent/unicode"; + +const terminalUndent = undent.with({ + alignValues: true, + columnOffset: createUnicodeColumnOffset({ tabWidth: 4 }), +}); + +terminalUndent` + label: 界 ${"alpha\nbeta"} +`; +// label: 界 alpha +// beta +``` + +This mode is still best-effort. Visual width depends on the renderer, font, and +surrounding context, so `@okikio/undent/unicode` aims at common terminal-style +output rather than browser-perfect layout. + ### Trimming By default, `undent` removes all blank lines at the start and end of the output @@ -438,12 +469,20 @@ preserved byte-for-byte. | `alignText(text, pad)` | Pad subsequent lines of text with a prefix string | | `splitLines(text)` | Split a string preserving exact newline sequences | | `rejoinLines(lines, seps)` | Reconstruct a string from `splitLines` output | -| `columnOffset(text)` | Count characters since the last newline (insertion column) | +| `columnOffset(text)` | Count UTF-16 code units since the last newline (default alignment policy) | | `newlineLengthAt(text, i)` | Length of the newline sequence at position `i` (0, 1, or 2) | | `resolveOptions(base, overrides)` | Merge option objects for custom pipelines | | `DEFAULTS` | The default resolved options constant | | `indent` | Symbol for indent anchors | +### Unicode subpath + +| Export | Description | +| --------------------------------------- | --------------------------------------------------------------------- | +| `createUnicodeColumnOffset(options?)` | Build a terminal-style Unicode-aware `columnOffset` function | +| `unicodeColumnOffset(text, options?)` | Measure the last line of a string in visual columns | +| `visualColumnWidth(text, options?)` | Measure a single line in terminal-style display columns | + ### Options ```ts @@ -452,6 +491,7 @@ interface UndentOptions { trim?: TrimMode | TrimSides; // How to trim wrapper lines (default: "all") newline?: string | null; // Normalize segment newlines (default: null) alignValues?: boolean; // Auto-align all multi-line values (default: false) + columnOffset?: (text: string) => number; // Measure alignment columns (default: columnOffset) } type TrimMode = "all" | "one" | "none"; diff --git a/scripts/build_npm.ts b/scripts/build_npm.ts index b54e0f5..65a0a55 100644 --- a/scripts/build_npm.ts +++ b/scripts/build_npm.ts @@ -14,7 +14,18 @@ import denoJson from "../deno.json" with { type: "json" }; await emptyDir("./npm"); await build({ - entryPoints: ["./mod.ts"], + entryPoints: [ + { + kind: "export", + name: ".", + path: "./mod.ts", + }, + { + kind: "export", + name: "./unicode", + path: "./unicode.ts", + }, + ], outDir: "./npm", // mod.ts uses no Deno-specific globals (no Deno.*, no std/ imports), @@ -34,6 +45,15 @@ await build({ // which is the correct runtime for these tests. test: false, + // Do not publish declaration source maps. + // + // They create long generated JSON strings in `*.d.ts.map` files. + // Socket flags those as "Long strings" because that pattern can also + // appear in packed or obfuscated malware. For this package, the maps + // are not needed because the npm package does not publish the original + // `mod.ts` source next to them anyway. + declarationMap: false, + package: { name: denoJson.name, // "@okikio/undent" version: denoJson.version, diff --git a/scripts/sync_unicode_east_asian_width.ts b/scripts/sync_unicode_east_asian_width.ts new file mode 100644 index 0000000..afa2e30 --- /dev/null +++ b/scripts/sync_unicode_east_asian_width.ts @@ -0,0 +1,475 @@ +/** + * Verify or refresh the East Asian Width tables used by `unicode.ts`. + * + * Run with: + * deno task unicode:eaw:check + * deno task unicode:eaw:update + * + * The script treats Unicode's `EastAsianWidth.txt` as the source of truth for + * the East Asian Width tables consumed by `unicode.ts`. Default mode only + * checks for drift. Pass `--write` to rewrite the internal constants file in + * place. + * + * The fetch path verifies more than "latest responded": it reads the version + * from `latest/ucd/ReadMe.txt`, fetches the matching immutable versioned file, + * and compares SHA-256 digests before trusting the payload. That keeps the + * mutable `latest` alias honest and gives failures a concrete release anchor. + */ +import { object } from 'jsr:@optique/core/constructs'; +import { runParserSync } from 'jsr:@optique/core/facade'; +import { message } from 'jsr:@optique/core/message'; +import { option } from 'jsr:@optique/core/primitives'; +import { fromFileUrl } from 'jsr:@std/path/from-file-url'; +import { parseSync } from 'npm:oxc-parser@0.132.0'; + +import { undent } from '../mod.ts'; + +import { + EAST_ASIAN_AMBIGUOUS_RANGES, + EAST_ASIAN_WIDE_RANGES, +} from '../_unicode_constants.ts'; + +type Range = readonly [start: number, end: number]; + +type SyncMode = 'check' | 'write'; + +type ParsedEastAsianWidth = { + wideRanges: Range[]; + ambiguousRanges: Range[]; +}; + +type DriftSummary = { + extras: number; + missing: number; + extraSamples: string[]; + missingSamples: string[]; +}; + +type VerifiedUnicodeSource = { + version: string; + date: string; + text: string; + latestSha256: string; + versionedSha256: string; +}; + +type RangeConstantName = + | 'EAST_ASIAN_WIDE_RANGES' + | 'EAST_ASIAN_AMBIGUOUS_RANGES'; + +const UNICODE_HOST = 'www.unicode.org'; +const LATEST_README_URL = 'https://www.unicode.org/Public/UCD/latest/ucd/ReadMe.txt'; +const LATEST_EAST_ASIAN_WIDTH_URL = + 'https://www.unicode.org/Public/UCD/latest/ucd/EastAsianWidth.txt'; +const UNICODE_CONSTANTS_URL = new URL('../_unicode_constants.ts', import.meta.url); +const UNICODE_CONSTANTS_PATH = fromFileUrl(UNICODE_CONSTANTS_URL); +const FETCH_TIMEOUT_MS = 30_000; +const VERSION_PATTERN = /Version\s+(\d+\.\d+\.\d+)/; +const DATE_PATTERN = /# Date:\s+([^\n]+)/; +const RANGE_CONSTANT_NAMES: readonly RangeConstantName[] = [ + 'EAST_ASIAN_WIDE_RANGES', + 'EAST_ASIAN_AMBIGUOUS_RANGES', +]; + +const cliParser = object({ + write: option('--write', { + description: message`Rewrite _unicode_constants.ts with the verified upstream East Asian Width ranges.`, + }), +}); + +const cliArgs = runParserSync(cliParser, 'sync_unicode_east_asian_width.ts', Deno.args, { + help: { + option: { names: ['-h', '--help'] }, + onShow: Deno.exit, + }, + onError: Deno.exit, + description: message`Verify or refresh the East Asian Width tables used by unicode.ts.`, + footer: message`Tasks: deno task unicode:eaw:check, deno task unicode:eaw:update`, +}); +const mode: SyncMode = cliArgs.write ? 'write' : 'check'; + +await ensureRequiredPermissions(mode); + +const upstream = await fetchVerifiedUnicodeSource(); +const parsed = parseEastAsianWidth(upstream.text); + +const wideDrift = summarizeDrift( + expandRanges(EAST_ASIAN_WIDE_RANGES), + expandRanges(parsed.wideRanges), +); +const ambiguousDrift = summarizeDrift( + expandRanges(EAST_ASIAN_AMBIGUOUS_RANGES), + expandRanges(parsed.ambiguousRanges), +); + +if (mode === 'check') { + reportDrift(upstream, wideDrift, ambiguousDrift); + if (wideDrift.extras !== 0 || wideDrift.missing !== 0) { + Deno.exit(1); + } + if (ambiguousDrift.extras !== 0 || ambiguousDrift.missing !== 0) { + Deno.exit(1); + } + + console.log('_unicode_constants.ts East Asian Width tables match Unicode upstream.'); + Deno.exit(0); +} + +const constantsSource = await Deno.readTextFile(UNICODE_CONSTANTS_URL); +validateConstantsModule(constantsSource); +const updatedSource = renderConstantsFile(parsed); + +if (updatedSource !== constantsSource) { + await Deno.writeTextFile(UNICODE_CONSTANTS_URL, updatedSource); + console.log( + `Updated _unicode_constants.ts East Asian Width tables from Unicode ${upstream.version} (${upstream.versionedSha256}).`, + ); + Deno.exit(0); +} + +console.log('_unicode_constants.ts East Asian Width tables were already up to date.'); + +async function ensureRequiredPermissions( + mode: SyncMode, +): Promise { + await ensurePermission( + { name: 'net', host: UNICODE_HOST }, + `network access to ${UNICODE_HOST}`, + ); + + if (mode === 'write') { + await ensurePermission( + { name: 'read', path: UNICODE_CONSTANTS_PATH }, + `read access to ${UNICODE_CONSTANTS_PATH}`, + ); + await ensurePermission( + { name: 'write', path: UNICODE_CONSTANTS_PATH }, + `write access to ${UNICODE_CONSTANTS_PATH}`, + ); + } +} + +async function ensurePermission( + descriptor: Deno.PermissionDescriptor, + label: string, +): Promise { + const current = await Deno.permissions.query(descriptor); + if (current.state === 'granted') { + return; + } + + const requested = current.state === 'prompt' + ? await Deno.permissions.request(descriptor) + : current; + if (requested.state !== 'granted') { + throw new Error( + `This script requires ${label}. Grant the matching Deno permission and try again.`, + ); + } +} + +async function fetchVerifiedUnicodeSource(): Promise { + const readmeText = await fetchText(LATEST_README_URL); + const version = parseReadmeValue(readmeText, VERSION_PATTERN, 'Unicode version'); + const date = parseReadmeValue(readmeText, DATE_PATTERN, 'UCD date'); + const versionedUrl = `https://www.unicode.org/Public/${version}/ucd/EastAsianWidth.txt`; + + const latestBytes = await fetchBytes(LATEST_EAST_ASIAN_WIDTH_URL); + const versionedBytes = await fetchBytes(versionedUrl); + const latestSha256 = await sha256Hex(latestBytes); + const versionedSha256 = await sha256Hex(versionedBytes); + if (latestSha256 !== versionedSha256) { + throw new Error( + [ + 'Unicode upstream integrity check failed.', + `latest URL: ${LATEST_EAST_ASIAN_WIDTH_URL}`, + `versioned URL: ${versionedUrl}`, + `latest sha256: ${latestSha256}`, + `versioned sha256: ${versionedSha256}`, + ].join('\n'), + ); + } + + return { + version, + date, + text: new TextDecoder().decode(versionedBytes), + latestSha256, + versionedSha256, + }; +} + +function parseReadmeValue(text: string, pattern: RegExp, label: string): string { + const match = text.match(pattern); + if (!match?.[1]) { + throw new Error(`Could not parse ${label} from Unicode ReadMe.txt`); + } + + return match[1].trim(); +} + +async function fetchText(url: string): Promise { + return new TextDecoder().decode(await fetchBytes(url)); +} + +async function fetchBytes(url: string): Promise> { + const response = await fetch(url, { + headers: { + accept: 'text/plain; charset=utf-8', + }, + signal: AbortSignal.timeout(FETCH_TIMEOUT_MS), + }); + if (!response.ok) { + throw new Error(`Failed to fetch ${url}: ${response.status} ${response.statusText}`); + } + + const arrayBuffer = await response.arrayBuffer(); + return new Uint8Array(arrayBuffer); +} + +async function sha256Hex(bytes: Uint8Array): Promise { + const digest = await crypto.subtle.digest('SHA-256', bytes); + return Array.from(new Uint8Array(digest)) + .map((byte) => byte.toString(16).padStart(2, '0')) + .join(''); +} + +function parseEastAsianWidth(text: string): ParsedEastAsianWidth { + const wide: number[] = []; + const ambiguous: number[] = []; + + for (const rawLine of text.split(/\r?\n/)) { + const line = rawLine.replace(/#.*/, '').trim(); + if (line.length === 0) continue; + + const [rangePart, property] = line.split(';').map((value) => value.trim()); + if (!rangePart || !property) continue; + + const target = property === 'W' || property === 'F' + ? wide + : property === 'A' + ? ambiguous + : null; + if (target === null) continue; + + const [startHex, endHex = startHex] = rangePart.split('..'); + const start = Number.parseInt(startHex, 16); + const end = Number.parseInt(endHex, 16); + for (let codePoint = start; codePoint <= end; codePoint++) { + target.push(codePoint); + } + } + + wide.sort((left, right) => left - right); + ambiguous.sort((left, right) => left - right); + + return { + wideRanges: compressCodePoints(wide), + ambiguousRanges: compressCodePoints(ambiguous), + }; +} + +function compressCodePoints(codePoints: number[]): Range[] { + if (codePoints.length === 0) return []; + + const ranges: Range[] = []; + let start = codePoints[0]!; + let end = start; + + for (let index = 1; index < codePoints.length; index++) { + const codePoint = codePoints[index]!; + if (codePoint === end + 1) { + end = codePoint; + continue; + } + + ranges.push([start, end]); + start = codePoint; + end = codePoint; + } + + ranges.push([start, end]); + return ranges; +} + +function expandRanges(ranges: readonly Range[]): Set { + const out = new Set(); + for (const [start, end] of ranges) { + for (let codePoint = start; codePoint <= end; codePoint++) { + out.add(codePoint); + } + } + + return out; +} + +function summarizeDrift(current: Set, upstreamSet: Set): DriftSummary { + const extras: number[] = []; + const missing: number[] = []; + + for (const codePoint of current) { + if (!upstreamSet.has(codePoint)) extras.push(codePoint); + } + for (const codePoint of upstreamSet) { + if (!current.has(codePoint)) missing.push(codePoint); + } + + extras.sort((left, right) => left - right); + missing.sort((left, right) => left - right); + + return { + extras: extras.length, + missing: missing.length, + extraSamples: extras.slice(0, 10).map(formatCodePoint), + missingSamples: missing.slice(0, 10).map(formatCodePoint), + }; +} + +function formatCodePoint(codePoint: number): string { + return `U+${codePoint.toString(16).toUpperCase().padStart(4, '0')}`; +} + +function reportDrift( + upstream: VerifiedUnicodeSource, + wideDrift: DriftSummary, + ambiguousDrift: DriftSummary, +): void { + console.log('East Asian Width audit against Unicode upstream'); + console.log(` source: ${LATEST_EAST_ASIAN_WIDTH_URL}`); + console.log(` release: Unicode ${upstream.version} (${upstream.date})`); + console.log(` sha256(latest): ${upstream.latestSha256}`); + console.log(` sha256(versioned): ${upstream.versionedSha256}`); + console.log(renderDriftLine('wide/fullwidth', wideDrift)); + console.log(renderDriftLine('ambiguous', ambiguousDrift)); + + if (wideDrift.extras !== 0 || wideDrift.missing !== 0) { + console.log(` wide/fullwidth extra samples: ${wideDrift.extraSamples.join(', ') || 'none'}`); + console.log(` wide/fullwidth missing samples: ${wideDrift.missingSamples.join(', ') || 'none'}`); + } + if (ambiguousDrift.extras !== 0 || ambiguousDrift.missing !== 0) { + console.log(` ambiguous extra samples: ${ambiguousDrift.extraSamples.join(', ') || 'none'}`); + console.log(` ambiguous missing samples: ${ambiguousDrift.missingSamples.join(', ') || 'none'}`); + } + if ( + wideDrift.extras !== 0 || + wideDrift.missing !== 0 || + ambiguousDrift.extras !== 0 || + ambiguousDrift.missing !== 0 + ) { + console.log('Run `deno task unicode:eaw:update` to refresh _unicode_constants.ts.'); + } +} + +function renderDriftLine(label: string, drift: DriftSummary): string { + return ` ${label}: ${drift.extras} extra, ${drift.missing} missing`; +} + +function validateConstantsModule(source: string): void { + const result = parseSync(UNICODE_CONSTANTS_PATH, source, { + lang: 'ts', + astType: 'ts', + sourceType: 'module', + }); + + if (result.errors.length > 0) { + throw new Error( + [ + `Could not parse ${UNICODE_CONSTANTS_PATH}.`, + formatParseError(result.errors[0]), + ].join('\n'), + ); + } + + const found = new Set(); + for (const statement of result.program.body) { + const variableDeclaration = getExportedVariableDeclaration(statement); + if (variableDeclaration === null) continue; + + for (const declaration of variableDeclaration.declarations) { + const identifier = getIdentifierName(declaration); + if (identifier === null || !isRangeConstantName(identifier)) continue; + found.add(identifier); + } + } + + for (const name of RANGE_CONSTANT_NAMES) { + if (!found.has(name)) { + throw new Error( + `${UNICODE_CONSTANTS_PATH} must export ${name} before this script can update it.`, + ); + } + } +} + +function isRangeConstantName(value: string): value is RangeConstantName { + return value === 'EAST_ASIAN_WIDE_RANGES' || value === 'EAST_ASIAN_AMBIGUOUS_RANGES'; +} + +function getExportedVariableDeclaration(statement: unknown): { + declarations: readonly unknown[]; +} | null { + if (!isRecord(statement) || statement.type !== 'ExportNamedDeclaration') return null; + const declaration = statement.declaration; + if (!isRecord(declaration) || declaration.type !== 'VariableDeclaration') return null; + if (!Array.isArray(declaration.declarations)) return null; + return { + declarations: declaration.declarations, + }; +} + +function getIdentifierName(declaration: unknown): string | null { + if (!isRecord(declaration)) return null; + const id = declaration.id; + if (!isRecord(id) || id.type !== 'Identifier') return null; + return typeof id.name === 'string' ? id.name : null; +} + +function formatParseError(error: unknown): string { + if (!isRecord(error)) return 'Unknown parser error'; + const message = typeof error.message === 'string' ? error.message : 'Unknown parser error'; + const labels = Array.isArray(error.labels) ? error.labels : []; + const firstLabel = labels[0]; + const loc = isRecord(firstLabel) ? firstLabel : null; + const line = loc !== null && typeof loc.line === 'number' ? loc.line : null; + const column = loc !== null && typeof loc.column === 'number' ? loc.column : null; + if (line === null || column === null) return message; + return `${line}:${column} ${message}`; +} + +function isRecord(value: unknown): value is Record { + return typeof value === 'object' && value !== null; +} + +function renderConstantsFile(parsed: ParsedEastAsianWidth): string { + return [ + undent.with({ trim: { trailing: "one" }})` + /** + * Internal East Asian Width lookup tables generated from Unicode upstream. + * + * \`unicode.ts\` re-exports these tables as part of the public API, but the + * generated data itself lives here so maintenance scripts can update one small + * internal file instead of rewriting the public module. + */ + + `, + renderTable('EAST_ASIAN_WIDE_RANGES', parsed.wideRanges), + '', + renderTable('EAST_ASIAN_AMBIGUOUS_RANGES', parsed.ambiguousRanges), + '', + '// End of generated East Asian Width tables.', + ].join('\n'); +} + +function renderTable(name: RangeConstantName, ranges: readonly Range[]): string { + const lines = ranges.map(([start, end]) => `\t[${toHex(start)}, ${toHex(end)}],`); + + return [ + `export const ${name}: ReadonlyArray = [`, + ...lines, + '];', + ].join('\n'); +} + +function toHex(codePoint: number): string { + return `0x${codePoint.toString(16)}`; +} diff --git a/unicode.ts b/unicode.ts new file mode 100644 index 0000000..ede111a --- /dev/null +++ b/unicode.ts @@ -0,0 +1,564 @@ +import type { ColumnOffsetFunction } from './mod.ts'; +import { + EAST_ASIAN_AMBIGUOUS_RANGES as INTERNAL_EAST_ASIAN_AMBIGUOUS_RANGES, + EAST_ASIAN_WIDE_RANGES as INTERNAL_EAST_ASIAN_WIDE_RANGES, +} from './_unicode_constants.ts'; + +/** + * # Unicode Alignment Helpers + * + * This module provides a renderer-aware companion to `columnOffset()` from the + * root module. The default `undent` path measures alignment columns as UTF-16 + * code units because that is fast, deterministic, and a good fit for code + * generation. The helpers here opt into a different trade-off: best-effort + * terminal-style visual width. + * + * The measurement pipeline is: + * + * ```text + * input string + * -> slice last line after the final newline + * -> segment into grapheme clusters when Intl.Segmenter exists + * -> let widthOf(...) override specific graphemes + * -> otherwise apply built-in width rules + * -> sum visual columns + * ``` + * + * The built-in width rules intentionally stay conservative: + * + * - tabs are configurable because their width depends on the current column, + * - control characters and combining-only fragments contribute zero columns, + * - emoji-style pictographs and regional-indicator flags count as two columns, + * - East Asian wide/fullwidth ranges count as two columns, + * - ambiguous-width ranges can be narrow or wide by option, + * - everything else counts as one column. + * + * This is still not a universal truth for browsers or proportional fonts. It + * is an opt-in approximation for common monospace terminal output. + * + * @module + */ + +/** + * The current visual column before measuring the next grapheme cluster. + * + * Width functions receive this so they can implement stateful rules such as + * tab stops, where the width of `"\t"` depends on the current column rather + * than on the grapheme alone. + */ +export interface UnicodeColumnWidthState { + /** The current visual column before the grapheme is measured. */ + readonly column: number; +} + +/** + * Measure the visual width of one grapheme cluster. + * + * Return `undefined` to fall back to the built-in best-effort width rules. + * Any defined return value must be a non-negative integer. + */ +export type UnicodeWidthFunction = ( + grapheme: string, + state: UnicodeColumnWidthState, +) => number | undefined; + +/** + * Options for the Unicode-aware column measurement helpers in this module. + * + * These helpers are designed for terminal-style monospace alignment. They are + * intentionally opt-in because visual width depends on the renderer, font, and + * surrounding context. + */ +export interface UnicodeColumnOffsetOptions { + /** + * How tabs advance. + * + * Set to `false` to treat `"\t"` as one column. Set to a positive integer to + * align tabs to tab stops. + * + * @default false + */ + tabWidth?: number | false; + + /** + * How to treat East Asian Width "ambiguous" code points. + * + * Unicode recommends treating them as narrow when the rendering context is + * not known. + * + * @default "narrow" + */ + ambiguous?: 'narrow' | 'wide'; + + /** + * Override the width of specific grapheme clusters. + * + * This is the escape hatch for callers who know their actual display target. + * Return `undefined` to fall back to the built-in best-effort rules. + */ + widthOf?: UnicodeWidthFunction; +} + +/** Inclusive `[start, end]` code-point range used by the lookup tables. */ +export type CodePointRange = readonly [start: number, end: number]; + +/** + * Default option values for the Unicode-aware column helpers. + * + * The defaults intentionally choose the conservative terminal policy: tabs are + * treated as a single column unless callers opt into tab stops, and ambiguous + * East Asian Width code points stay narrow unless the rendering context says + * otherwise. + */ +export const DEFAULT_UNICODE_COLUMN_OFFSET_OPTIONS: Required< + Pick +> = { + tabWidth: false, + ambiguous: 'narrow', +}; + +/** + * East Asian wide/fullwidth ranges from Unicode's `EastAsianWidth.txt`. + * + * The runtime stays dependency-free by keeping the verified ranges inline, but + * the generated table now lives in `./_unicode_constants.ts` so maintenance + * tooling can refresh it without rewriting this public module. + */ +export const EAST_ASIAN_WIDE_RANGES: ReadonlyArray = + INTERNAL_EAST_ASIAN_WIDE_RANGES; + +/** + * East Asian ambiguous-width ranges from Unicode's `EastAsianWidth.txt`. + * + * These are the upstream `A` property ranges compressed into inclusive + * `[start, end]` pairs. Callers still choose how to interpret them through the + * `ambiguous` option, while the generated data lives in + * `./_unicode_constants.ts`. + */ +export const EAST_ASIAN_AMBIGUOUS_RANGES: ReadonlyArray = + INTERNAL_EAST_ASIAN_AMBIGUOUS_RANGES; + +/** + * Detect graphemes that contain emoji-style pictographs. + * + * The default width rules treat these graphemes as occupying two columns, + * matching the common terminal convention for emoji presentation. + */ +export const EXTENDED_PICTOGRAPHIC_RE = /\p{Extended_Pictographic}/u; + +/** Minimal shape of an `Intl.Segmenter#segment()` item. */ +export interface GraphemeSegment { + /** The grapheme cluster text for one segmentation result. */ + readonly segment: string; +} + +/** Minimal grapheme-segmentation interface used by this module. */ +export interface GraphemeSegmenter { + /** Segment the input string into grapheme-cluster records. */ + segment(input: string): Iterable; +} + +/** + * Structural type for runtimes that expose `Intl.Segmenter`. + * + * dnt's Node type environment does not include that API, so the module uses a + * structural type instead of referencing `Intl.Segmenter` directly. + */ +export type IntlWithSegmenter = typeof Intl & { + Segmenter?: new ( + locales?: string | string[], + options?: { granularity: 'grapheme' }, + ) => GraphemeSegmenter; +}; + +/** + * Fully resolved Unicode column-measurement options. + * + * The resolver fills in default `tabWidth` and `ambiguous` values while still + * carrying through caller-provided overrides such as `widthOf`. + */ +export type ResolvedUnicodeColumnOffsetOptions = + & Required> + & UnicodeColumnOffsetOptions; + +/** Shared typed reference to the global `Intl` object. */ +const intlWithSegmenter = Intl as IntlWithSegmenter; + +/** Lazily initialized grapheme segmenter cache. */ +let graphemeSegmenter: GraphemeSegmenter | null | undefined; + +/** + * Create a `columnOffset` function that measures terminal-style Unicode width. + * + * Use this with `undent.with({ columnOffset })` when later lines of aligned + * values should follow visual columns instead of raw UTF-16 code units. + * + * @example Aligning with terminal-style Unicode columns + * ```ts + * import { undent } from '@okikio/undent'; + * import { createUnicodeColumnOffset } from '@okikio/undent/unicode'; + * + * const terminalUndent = undent.with({ + * alignValues: true, + * columnOffset: createUnicodeColumnOffset(), + * }); + * + * terminalUndent` + * label: 界 ${'a\nb'} + * `; + * // "label: 界 a\n b" + * ``` + * + * @example Configuring tabs and ambiguous-width handling + * ```ts + * import { createUnicodeColumnOffset } from '@okikio/undent/unicode'; + * + * const columnOffset = createUnicodeColumnOffset({ + * tabWidth: 4, + * ambiguous: 'wide', + * }); + * ``` + */ +export function createUnicodeColumnOffset( + options: UnicodeColumnOffsetOptions = {}, +): ColumnOffsetFunction { + const resolved = resolveUnicodeColumnOffsetOptions(options); + + return function unicodeColumnOffsetWithOptions(text: string): number { + return unicodeColumnOffset(text, resolved); + }; +} + +/** + * Measure the visual insertion column after the final newline in `text`. + * + * This is the Unicode-aware companion to `columnOffset()` from the root module. + * It measures the last line in terminal-style display columns instead of UTF-16 + * code units. + * + * @example Measuring the last line of a string with wide characters + * ```ts + * import { unicodeColumnOffset } from '@okikio/undent/unicode'; + * + * unicodeColumnOffset('x\n界 '); // 3 + * ``` + */ +export function unicodeColumnOffset( + text: string, + options: UnicodeColumnOffsetOptions = {}, +): number { + return visualColumnWidth( + sliceAfterLastNewline(text), + resolveUnicodeColumnOffsetOptions(options), + ); +} + +/** + * Measure the terminal-style visual width of a single line. + * + * The function walks grapheme clusters, not raw UTF-16 code units. That keeps + * combining sequences and emoji clusters together before width rules are + * applied. + * + * This is still a best-effort estimate. Browser layout and proportional fonts + * can render the same text with different visual widths. + * + * @example Measuring combining marks as one visible cell + * ```ts + * import { visualColumnWidth } from '@okikio/undent/unicode'; + * + * visualColumnWidth('e\u0301 '); // 2 + * ``` + */ +export function visualColumnWidth( + text: string, + options: UnicodeColumnOffsetOptions = {}, +): number { + const resolved = resolveUnicodeColumnOffsetOptions(options); + let column = 0; + + for (const grapheme of graphemes(text)) { + const customWidth = resolved.widthOf?.(grapheme, { column }); + if (customWidth !== undefined) { + column += validateWidth(customWidth, 'widthOf(...)'); + continue; + } + + column += defaultGraphemeWidth(grapheme, column, resolved); + } + + return column; +} + +/** + * Normalize caller options once so the measurement loop can stay simple. + * + * Validation happens here instead of in the hot path so repeated calls from a + * preconfigured `createUnicodeColumnOffset()` instance only pay the check once. + */ +export function resolveUnicodeColumnOffsetOptions( + options: UnicodeColumnOffsetOptions, +): ResolvedUnicodeColumnOffsetOptions { + if (options.tabWidth !== undefined && options.tabWidth !== false) { + validateTabWidth(options.tabWidth); + } + + return Object.assign({}, DEFAULT_UNICODE_COLUMN_OFFSET_OPTIONS, options); +} + +/** + * Return the last logical line of a string. + * + * Alignment cares only about the insertion column after the final newline, so + * this trims away earlier lines before visual-width measurement begins. + */ +export function sliceAfterLastNewline(text: string): string { + const lastLF = text.lastIndexOf('\n'); + const lastCR = text.lastIndexOf('\r'); + const lastNL = lastLF > lastCR ? lastLF : lastCR; + + return lastNL === -1 ? text : text.slice(lastNL + 1); +} + +/** + * Measure one grapheme cluster with the built-in width rules. + * + * The order matters because some rules intentionally short-circuit others: + * + * ```text + * grapheme + * -> tab? => tab-stop width + * -> control only? => 0 + * -> emoji pictograph? => 2 + * -> regional indicator? => 2 + * -> inspect code points + * -> skip zero-width modifiers and joiners + * -> wide/fullwidth => 2 + * -> ambiguous+wide => 2 + * -> otherwise mark a visible base code point + * -> visible base found? => 1 + * -> otherwise => 0 + * ``` + * + * This keeps combining marks, variation selectors, and ZWJ glue from adding + * width on top of the base grapheme they modify. + */ +export function defaultGraphemeWidth( + grapheme: string, + column: number, + options: ResolvedUnicodeColumnOffsetOptions, +): number { + if (grapheme.length === 0) return 0; + + if (grapheme === '\t') { + if (options.tabWidth === false) return 1; + + const remainder = column % options.tabWidth; + return remainder === 0 ? options.tabWidth : options.tabWidth - remainder; + } + + if (isControlOnly(grapheme)) return 0; + if (EXTENDED_PICTOGRAPHIC_RE.test(grapheme)) return 2; + + let sawBaseCodePoint = false; + + for (const char of grapheme) { + const codePoint = char.codePointAt(0); + if (codePoint === undefined) continue; + if (isControlCodePoint(codePoint) || isZeroWidthCodePoint(codePoint)) { + continue; + } + + sawBaseCodePoint = true; + + if (isRegionalIndicatorCodePoint(codePoint)) { + return 2; + } + + if (isWideCodePoint(codePoint)) { + return 2; + } + + if ( + options.ambiguous === 'wide' && + isAmbiguousWidthCodePoint(codePoint) + ) { + return 2; + } + } + + return sawBaseCodePoint ? 1 : 0; +} + +/** + * Iterate text as grapheme clusters when the runtime supports it. + * + * `Intl.Segmenter` is the strongest available platform primitive because it + * follows Unicode grapheme-boundary rules. The fallback uses `for...of`, which + * still iterates code points instead of UTF-16 code units, but it cannot keep + * multi-code-point emoji sequences together. + */ +export function* graphemes(text: string): Iterable { + const segmenter = getGraphemeSegmenter(); + + if (segmenter !== null) { + for (const segment of segmenter.segment(text)) { + yield segment.segment; + } + + return; + } + + yield* text; +} + +/** + * Lazily create and memoize the grapheme segmenter. + * + * Importing the module should stay side-effect free and cheap. The segmenter is + * therefore constructed on first use instead of at module initialization time. + */ +export function getGraphemeSegmenter(): GraphemeSegmenter | null { + if (graphemeSegmenter !== undefined) { + return graphemeSegmenter; + } + + graphemeSegmenter = typeof intlWithSegmenter.Segmenter === 'function' + ? new intlWithSegmenter.Segmenter(undefined, { granularity: 'grapheme' }) + : null; + + return graphemeSegmenter; +} + +/** + * Validate widths returned by `widthOf(...)`. + * + * Custom width hooks are part of the public escape hatch, so failures should be + * loud and specific instead of silently coercing odd values. + */ +export function validateWidth(width: number, source: string): number { + if (!Number.isInteger(width) || width < 0) { + throw new TypeError( + `undent/unicode: ${source} must return a non-negative integer`, + ); + } + + return width; +} + +/** + * Validate tab-stop configuration once during option resolution. + */ +export function validateTabWidth(tabWidth: number): void { + if (!Number.isInteger(tabWidth) || tabWidth < 1) { + throw new TypeError( + 'undent/unicode: "tabWidth" must be a positive integer or false', + ); + } +} + +/** + * Return true when every code point in the grapheme is a control character. + * + * We only return zero width for a whole grapheme when it contains no visible + * base character at all. A visible character followed by controls is handled by + * the wider default grapheme logic instead of being dropped here. + */ +export function isControlOnly(grapheme: string): boolean { + for (const char of grapheme) { + const codePoint = char.codePointAt(0); + if (codePoint === undefined) continue; + if (!isControlCodePoint(codePoint)) return false; + } + + return true; +} + +/** + * Return true when the code point is a regional indicator symbol. + * + * These code points form Unicode flag graphemes in pairs. Terminals commonly + * render both the pair and the standalone symbol as emoji-width cells, so the + * built-in width rules treat them as occupying two columns. + */ +function isRegionalIndicatorCodePoint(codePoint: number): boolean { + return codePoint >= 0x1f1e6 && codePoint <= 0x1f1ff; +} + +/** + * Control characters never consume terminal columns by themselves. + */ +export function isControlCodePoint(codePoint: number): boolean { + return codePoint === 0x00 || + (codePoint >= 0x01 && codePoint <= 0x1f) || + (codePoint >= 0x7f && codePoint <= 0x9f); +} + +/** + * Zero-width code points modify neighboring characters instead of occupying + * their own cell. + * + * This includes combining marks, variation selectors, and the zero-width + * joiner used to glue emoji into a single presented grapheme. + */ +export function isZeroWidthCodePoint(codePoint: number): boolean { + return (codePoint >= 0x0300 && codePoint <= 0x036f) || + (codePoint >= 0x1ab0 && codePoint <= 0x1aff) || + (codePoint >= 0x1dc0 && codePoint <= 0x1dff) || + (codePoint >= 0x20d0 && codePoint <= 0x20ff) || + codePoint === 0x200d || + (codePoint >= 0xfe00 && codePoint <= 0xfe0f) || + (codePoint >= 0xfe20 && codePoint <= 0xfe2f) || + (codePoint >= 0xe0100 && codePoint <= 0xe01ef); +} + +/** + * Check East Asian wide/fullwidth ranges with a binary search. + * + * A flat range table is much easier to audit and extend than a long boolean + * expression, and binary search keeps the lookup cost predictable. + */ +export function isWideCodePoint(codePoint: number): boolean { + return isCodePointInRanges(codePoint, EAST_ASIAN_WIDE_RANGES); +} + +/** + * Check East Asian ambiguous-width ranges with the same binary-search helper. + */ +export function isAmbiguousWidthCodePoint(codePoint: number): boolean { + return isCodePointInRanges(codePoint, EAST_ASIAN_AMBIGUOUS_RANGES); +} + +/** + * Return true when `codePoint` falls inside any `[start, end]` range. + * + * The tables are sorted by start position, so binary search finds matches in + * `O(log n)` time without allocating or scanning the full list on every code + * point. + */ +export function isCodePointInRanges( + codePoint: number, + ranges: ReadonlyArray, +): boolean { + let low = 0; + let high = ranges.length - 1; + + while (low <= high) { + const middle = (low + high) >> 1; + const [start, end] = ranges[middle]!; + + if (codePoint < start) { + high = middle - 1; + continue; + } + + if (codePoint > end) { + low = middle + 1; + continue; + } + + return true; + } + + return false; +} \ No newline at end of file diff --git a/unicode_test.ts b/unicode_test.ts new file mode 100644 index 0000000..30bdb17 --- /dev/null +++ b/unicode_test.ts @@ -0,0 +1,171 @@ +import { describe, it } from 'jsr:@std/testing/bdd'; +import { expect } from 'jsr:@std/expect'; + +import { align, undent } from './mod.ts'; +import { + createUnicodeColumnOffset, + defaultGraphemeWidth, + graphemes, + isAmbiguousWidthCodePoint, + isCodePointInRanges, + isControlCodePoint, + isWideCodePoint, + resolveUnicodeColumnOffsetOptions, + sliceAfterLastNewline, + unicodeColumnOffset, + visualColumnWidth, +} from './unicode.ts'; + +describe('unicode alignment helpers', () => { + it('measures the last line with wide characters', () => { + expect(unicodeColumnOffset('prefix\n界 ')).toBe(3); + }); + + it('exports a resolver that merges default unicode options', () => { + const result = resolveUnicodeColumnOffsetOptions({}); + + expect(result).toEqual({ + tabWidth: false, + ambiguous: 'narrow', + }); + }); + + it('exports a helper that slices after the last newline', () => { + expect(sliceAfterLastNewline('alpha\r\nbeta')).toBe('beta'); + }); + + it('measures the last line after CRLF as well as LF', () => { + expect(unicodeColumnOffset('prefix\r\n界 ')).toBe(3); + }); + + it('treats combining marks as part of the same visible grapheme', () => { + expect(visualColumnWidth('e\u0301 ')).toBe(2); + }); + + it('exports default grapheme width rules for direct use', () => { + const options = resolveUnicodeColumnOffsetOptions({ ambiguous: 'wide' }); + + expect(defaultGraphemeWidth('\t', 3, options)).toBe(1); + expect(defaultGraphemeWidth('Ω', 0, options)).toBe(2); + }); + + it('treats emoji ZWJ sequences as a single wide grapheme', () => { + expect(visualColumnWidth('👨‍👩‍👧‍👦')).toBe(2); + }); + + it('exports grapheme iteration for callers that need the same segmentation', () => { + expect(Array.from(graphemes('e\u0301😀'))).toEqual(['é', '😀']); + }); + + it('treats regional-indicator flag sequences as wide', () => { + expect(visualColumnWidth('🇯🇵')).toBe(2); + }); + + it('treats fullwidth forms as wide', () => { + expect(visualColumnWidth('Hello')).toBe(10); + }); + + it('treats Unicode wide trigrams as wide after the upstream table audit', () => { + expect(visualColumnWidth('☰')).toBe(2); + }); + + it('exports wide and ambiguous range predicates', () => { + expect(isWideCodePoint('界'.codePointAt(0)!)).toBe(true); + expect(isAmbiguousWidthCodePoint('Ω'.codePointAt(0)!)).toBe(true); + }); + + it('exports low-level code-point classification helpers', () => { + expect(isControlCodePoint(0x09)).toBe(true); + expect(isCodePointInRanges(0x03a9, [[0x0391, 0x03a9]])).toBe(true); + }); + + it('treats ambiguous-width characters as narrow by default', () => { + expect(visualColumnWidth('Ω')).toBe(1); + }); + + it('treats ambiguous-width characters as wide when configured', () => { + expect(visualColumnWidth('Ω', { ambiguous: 'wide' })).toBe(2); + }); + + it('lets widthOf override specific grapheme widths', () => { + expect( + visualColumnWidth('a·b', { + widthOf(grapheme) { + return grapheme === '·' ? 3 : undefined; + }, + }), + ).toBe(5); + }); + + it('throws when widthOf returns a negative width', () => { + expect(() => + visualColumnWidth('a', { + widthOf() { + return -1; + }, + }) + ).toThrow('widthOf(...) must return a non-negative integer'); + }); + + it('throws when tabWidth is not a positive integer', () => { + expect(() => createUnicodeColumnOffset({ tabWidth: 0 })).toThrow( + '"tabWidth" must be a positive integer or false', + ); + }); + + it('supports tab stops when configured', () => { + const columnOffset = createUnicodeColumnOffset({ tabWidth: 4 }); + + expect(columnOffset('ab\t')).toBe(4); + expect(columnOffset('abc\t')).toBe(4); + expect(columnOffset('abcd\t')).toBe(8); + }); + + it('can treat tabs as one column when tabWidth is false', () => { + const columnOffset = createUnicodeColumnOffset({ tabWidth: false }); + + expect(columnOffset('ab\t')).toBe(3); + }); + + it('integrates with undent through the root columnOffset option', () => { + const terminalUndent = undent.with({ + alignValues: true, + columnOffset: createUnicodeColumnOffset(), + }); + + const result = terminalUndent` + label: 界 ${'a\nb'} + `; + + expect(result).toBe('label: 界 a\n b'); + }); + + it('affects wrapped align() values as well as alignValues', () => { + const terminalUndent = undent.with({ + columnOffset: createUnicodeColumnOffset(), + }); + + const result = terminalUndent` + label: 界 ${align('alpha\nbeta')} + `; + + expect(result).toBe('label: 界 alpha\n beta'); + }); + + it('supports custom widthOf in end-to-end alignment', () => { + const terminalUndent = undent.with({ + alignValues: true, + columnOffset: createUnicodeColumnOffset({ + widthOf(grapheme) { + return grapheme === '·' ? 3 : undefined; + }, + }), + }); + + const result = terminalUndent` + key· ${'x\ny'} + `; + + expect(result).toBe('key· x\n y'); + }); +}); \ No newline at end of file -- 2.51.2