diff --git a/_unicode_constants.ts b/_unicode_constants.ts new file mode 100644 index 0000000..bc572b1 --- /dev/null +++ b/_unicode_constants.ts @@ -0,0 +1,317 @@ +/** + * Internal East Asian Width lookup tables generated from Unicode upstream. + * + * `unicode.ts` re-exports these tables as part of the public API, but the + * generated data itself lives here so maintenance scripts can update one small + * internal file instead of rewriting the public module. + */ + +export const EAST_ASIAN_WIDE_RANGES: ReadonlyArray = [ + [0x1100, 0x115f], + [0x231a, 0x231b], + [0x2329, 0x232a], + [0x23e9, 0x23ec], + [0x23f0, 0x23f0], + [0x23f3, 0x23f3], + [0x25fd, 0x25fe], + [0x2614, 0x2615], + [0x2630, 0x2637], + [0x2648, 0x2653], + [0x267f, 0x267f], + [0x268a, 0x268f], + [0x2693, 0x2693], + [0x26a1, 0x26a1], + [0x26aa, 0x26ab], + [0x26bd, 0x26be], + [0x26c4, 0x26c5], + [0x26ce, 0x26ce], + [0x26d4, 0x26d4], + [0x26ea, 0x26ea], + [0x26f2, 0x26f3], + [0x26f5, 0x26f5], + [0x26fa, 0x26fa], + [0x26fd, 0x26fd], + [0x2705, 0x2705], + [0x270a, 0x270b], + [0x2728, 0x2728], + [0x274c, 0x274c], + [0x274e, 0x274e], + [0x2753, 0x2755], + [0x2757, 0x2757], + [0x2795, 0x2797], + [0x27b0, 0x27b0], + [0x27bf, 0x27bf], + [0x2b1b, 0x2b1c], + [0x2b50, 0x2b50], + [0x2b55, 0x2b55], + [0x2e80, 0x2e99], + [0x2e9b, 0x2ef3], + [0x2f00, 0x2fd5], + [0x2ff0, 0x303e], + [0x3041, 0x3096], + [0x3099, 0x30ff], + [0x3105, 0x312f], + [0x3131, 0x318e], + [0x3190, 0x31e5], + [0x31ef, 0x321e], + [0x3220, 0x3247], + [0x3250, 0xa48c], + [0xa490, 0xa4c6], + [0xa960, 0xa97c], + [0xac00, 0xd7a3], + [0xf900, 0xfaff], + [0xfe10, 0xfe19], + [0xfe30, 0xfe52], + [0xfe54, 0xfe66], + [0xfe68, 0xfe6b], + [0xff01, 0xff60], + [0xffe0, 0xffe6], + [0x16fe0, 0x16fe4], + [0x16ff0, 0x16ff6], + [0x17000, 0x18cd5], + [0x18cff, 0x18d1e], + [0x18d80, 0x18df2], + [0x1aff0, 0x1aff3], + [0x1aff5, 0x1affb], + [0x1affd, 0x1affe], + [0x1b000, 0x1b122], + [0x1b132, 0x1b132], + [0x1b150, 0x1b152], + [0x1b155, 0x1b155], + [0x1b164, 0x1b167], + [0x1b170, 0x1b2fb], + [0x1d300, 0x1d356], + [0x1d360, 0x1d376], + [0x1f004, 0x1f004], + [0x1f0cf, 0x1f0cf], + [0x1f18e, 0x1f18e], + [0x1f191, 0x1f19a], + [0x1f200, 0x1f202], + [0x1f210, 0x1f23b], + [0x1f240, 0x1f248], + [0x1f250, 0x1f251], + [0x1f260, 0x1f265], + [0x1f300, 0x1f320], + [0x1f32d, 0x1f335], + [0x1f337, 0x1f37c], + [0x1f37e, 0x1f393], + [0x1f3a0, 0x1f3ca], + [0x1f3cf, 0x1f3d3], + [0x1f3e0, 0x1f3f0], + [0x1f3f4, 0x1f3f4], + [0x1f3f8, 0x1f43e], + [0x1f440, 0x1f440], + [0x1f442, 0x1f4fc], + [0x1f4ff, 0x1f53d], + [0x1f54b, 0x1f54e], + [0x1f550, 0x1f567], + [0x1f57a, 0x1f57a], + [0x1f595, 0x1f596], + [0x1f5a4, 0x1f5a4], + [0x1f5fb, 0x1f64f], + [0x1f680, 0x1f6c5], + [0x1f6cc, 0x1f6cc], + [0x1f6d0, 0x1f6d2], + [0x1f6d5, 0x1f6d8], + [0x1f6dc, 0x1f6df], + [0x1f6eb, 0x1f6ec], + [0x1f6f4, 0x1f6fc], + [0x1f7e0, 0x1f7eb], + [0x1f7f0, 0x1f7f0], + [0x1f90c, 0x1f93a], + [0x1f93c, 0x1f945], + [0x1f947, 0x1f9ff], + [0x1fa70, 0x1fa7c], + [0x1fa80, 0x1fa8a], + [0x1fa8e, 0x1fac6], + [0x1fac8, 0x1fac8], + [0x1facd, 0x1fadc], + [0x1fadf, 0x1faea], + [0x1faef, 0x1faf8], + [0x20000, 0x2fffd], + [0x30000, 0x3fffd], +]; + +export const EAST_ASIAN_AMBIGUOUS_RANGES: ReadonlyArray = [ + [0xa1, 0xa1], + [0xa4, 0xa4], + [0xa7, 0xa8], + [0xaa, 0xaa], + [0xad, 0xae], + [0xb0, 0xb4], + [0xb6, 0xba], + [0xbc, 0xbf], + [0xc6, 0xc6], + [0xd0, 0xd0], + [0xd7, 0xd8], + [0xde, 0xe1], + [0xe6, 0xe6], + [0xe8, 0xea], + [0xec, 0xed], + [0xf0, 0xf0], + [0xf2, 0xf3], + [0xf7, 0xfa], + [0xfc, 0xfc], + [0xfe, 0xfe], + [0x101, 0x101], + [0x111, 0x111], + [0x113, 0x113], + [0x11b, 0x11b], + [0x126, 0x127], + [0x12b, 0x12b], + [0x131, 0x133], + [0x138, 0x138], + [0x13f, 0x142], + [0x144, 0x144], + [0x148, 0x14b], + [0x14d, 0x14d], + [0x152, 0x153], + [0x166, 0x167], + [0x16b, 0x16b], + [0x1ce, 0x1ce], + [0x1d0, 0x1d0], + [0x1d2, 0x1d2], + [0x1d4, 0x1d4], + [0x1d6, 0x1d6], + [0x1d8, 0x1d8], + [0x1da, 0x1da], + [0x1dc, 0x1dc], + [0x251, 0x251], + [0x261, 0x261], + [0x2c4, 0x2c4], + [0x2c7, 0x2c7], + [0x2c9, 0x2cb], + [0x2cd, 0x2cd], + [0x2d0, 0x2d0], + [0x2d8, 0x2db], + [0x2dd, 0x2dd], + [0x2df, 0x2df], + [0x300, 0x36f], + [0x391, 0x3a1], + [0x3a3, 0x3a9], + [0x3b1, 0x3c1], + [0x3c3, 0x3c9], + [0x401, 0x401], + [0x410, 0x44f], + [0x451, 0x451], + [0x2010, 0x2010], + [0x2013, 0x2016], + [0x2018, 0x2019], + [0x201c, 0x201d], + [0x2020, 0x2022], + [0x2024, 0x2027], + [0x2030, 0x2030], + [0x2032, 0x2033], + [0x2035, 0x2035], + [0x203b, 0x203b], + [0x203e, 0x203e], + [0x2074, 0x2074], + [0x207f, 0x207f], + [0x2081, 0x2084], + [0x20ac, 0x20ac], + [0x2103, 0x2103], + [0x2105, 0x2105], + [0x2109, 0x2109], + [0x2113, 0x2113], + [0x2116, 0x2116], + [0x2121, 0x2122], + [0x2126, 0x2126], + [0x212b, 0x212b], + [0x2153, 0x2154], + [0x215b, 0x215e], + [0x2160, 0x216b], + [0x2170, 0x2179], + [0x2189, 0x2189], + [0x2190, 0x2199], + [0x21b8, 0x21b9], + [0x21d2, 0x21d2], + [0x21d4, 0x21d4], + [0x21e7, 0x21e7], + [0x2200, 0x2200], + [0x2202, 0x2203], + [0x2207, 0x2208], + [0x220b, 0x220b], + [0x220f, 0x220f], + [0x2211, 0x2211], + [0x2215, 0x2215], + [0x221a, 0x221a], + [0x221d, 0x2220], + [0x2223, 0x2223], + [0x2225, 0x2225], + [0x2227, 0x222c], + [0x222e, 0x222e], + [0x2234, 0x2237], + [0x223c, 0x223d], + [0x2248, 0x2248], + [0x224c, 0x224c], + [0x2252, 0x2252], + [0x2260, 0x2261], + [0x2264, 0x2267], + [0x226a, 0x226b], + [0x226e, 0x226f], + [0x2282, 0x2283], + [0x2286, 0x2287], + [0x2295, 0x2295], + [0x2299, 0x2299], + [0x22a5, 0x22a5], + [0x22bf, 0x22bf], + [0x2312, 0x2312], + [0x2460, 0x24e9], + [0x24eb, 0x254b], + [0x2550, 0x2573], + [0x2580, 0x258f], + [0x2592, 0x2595], + [0x25a0, 0x25a1], + [0x25a3, 0x25a9], + [0x25b2, 0x25b3], + [0x25b6, 0x25b7], + [0x25bc, 0x25bd], + [0x25c0, 0x25c1], + [0x25c6, 0x25c8], + [0x25cb, 0x25cb], + [0x25ce, 0x25d1], + [0x25e2, 0x25e5], + [0x25ef, 0x25ef], + [0x2605, 0x2606], + [0x2609, 0x2609], + [0x260e, 0x260f], + [0x261c, 0x261c], + [0x261e, 0x261e], + [0x2640, 0x2640], + [0x2642, 0x2642], + [0x2660, 0x2661], + [0x2663, 0x2665], + [0x2667, 0x266a], + [0x266c, 0x266d], + [0x266f, 0x266f], + [0x269e, 0x269f], + [0x26bf, 0x26bf], + [0x26c6, 0x26cd], + [0x26cf, 0x26d3], + [0x26d5, 0x26e1], + [0x26e3, 0x26e3], + [0x26e8, 0x26e9], + [0x26eb, 0x26f1], + [0x26f4, 0x26f4], + [0x26f6, 0x26f9], + [0x26fb, 0x26fc], + [0x26fe, 0x26ff], + [0x273d, 0x273d], + [0x2776, 0x277f], + [0x2b56, 0x2b59], + [0x3248, 0x324f], + [0xe000, 0xf8ff], + [0xfe00, 0xfe0f], + [0xfffd, 0xfffd], + [0x1f100, 0x1f10a], + [0x1f110, 0x1f12d], + [0x1f130, 0x1f169], + [0x1f170, 0x1f18d], + [0x1f18f, 0x1f190], + [0x1f19b, 0x1f1ac], + [0xe0100, 0xe01ef], + [0xf0000, 0xffffd], + [0x100000, 0x10fffd], +]; + +// End of generated East Asian Width tables. \ No newline at end of file diff --git a/deno.json b/deno.json index 094538f..3a34224 100644 --- a/deno.json +++ b/deno.json @@ -1,12 +1,17 @@ { "name": "@okikio/undent", "version": "0.2.1", - "exports": "./mod.ts", + "exports": { + ".": "./mod.ts", + "./unicode": "./unicode.ts" + }, "tasks": { "test": "deno test --trace-leaks --v8-flags=--expose-gc", "bench": "deno bench --allow-env=NODE_DISABLE_COLORS --v8-flags=--expose-gc", "forge": "deno run -A jsr:@roka/forge", - "build:npm": "deno run -A scripts/build_npm.ts" + "build:npm": "deno run -A scripts/build_npm.ts", + "unicode:eaw:check": "deno run --allow-net=www.unicode.org --allow-env=NAPI_RS_NATIVE_LIBRARY_PATH,NAPI_RS_ENFORCE_VERSION_CHECK,NAPI_RS_FORCE_WASI --allow-ffi scripts/sync_unicode_east_asian_width.ts", + "unicode:eaw:update": "deno run --allow-net=www.unicode.org --allow-env=NAPI_RS_NATIVE_LIBRARY_PATH,NAPI_RS_ENFORCE_VERSION_CHECK,NAPI_RS_FORCE_WASI --allow-ffi --allow-read=_unicode_constants.ts --allow-write=_unicode_constants.ts scripts/sync_unicode_east_asian_width.ts --write" }, "fmt": { "proseWrap": "preserve" @@ -14,6 +19,8 @@ "publish": { "include": [ "mod.ts", + "_unicode_constants.ts", + "unicode.ts", "changelog.md", "license", "readme.md" diff --git a/deno.lock b/deno.lock index 4b4b3e5..f315298 100644 --- a/deno.lock +++ b/deno.lock @@ -3,6 +3,7 @@ "specifiers": { "jsr:@david/code-block-writer@^13.0.3": "13.0.3", "jsr:@deno/dnt@*": "0.42.3", + "jsr:@optique/core@*": "1.0.2", "jsr:@std/assert@^1.0.14": "1.0.18", "jsr:@std/assert@^1.0.17": "1.0.18", "jsr:@std/expect@*": "1.0.17", @@ -10,15 +11,21 @@ "jsr:@std/fs@1": "1.0.22", "jsr:@std/internal@^1.0.10": "1.0.12", "jsr:@std/internal@^1.0.12": "1.0.12", + "jsr:@std/path@*": "1.1.4", "jsr:@std/path@1": "1.1.4", "jsr:@std/path@^1.1.4": "1.1.4", "jsr:@std/testing@*": "1.0.17", "jsr:@ts-morph/bootstrap@0.27": "0.27.0", "jsr:@ts-morph/common@0.27": "0.27.0", + "npm:@babel/parser@7.29.3": "7.29.3", + "npm:@oxc-parser/binding-wasm32-wasi@0.132.0": "0.132.0", "npm:dedent@*": "1.7.1", "npm:fast-check@*": "4.5.3", "npm:mitata@*": "1.0.34", - "npm:outdent@*": "0.8.0" + "npm:outdent@*": "0.8.0", + "npm:oxc-parser@0.132.0": "0.132.0", + "npm:typescript@*": "6.0.3", + "npm:typescript@6.0.3": "6.0.3" }, "jsr": { "@david/code-block-writer@13.0.3": { @@ -34,6 +41,9 @@ "jsr:@ts-morph/bootstrap" ] }, + "@optique/core@1.0.2": { + "integrity": "3f90a2965286cd4d4d103f6605ed9a00951393dbab0748402576350ad3780b33" + }, "@std/assert@1.0.18": { "integrity": "270245e9c2c13b446286de475131dc688ca9abcd94fc5db41d43a219b34d1c78", "dependencies": [ @@ -88,6 +98,166 @@ } }, "npm": { + "@babel/helper-string-parser@7.27.1": { + "integrity": "sha512-qMlSxKbpRlAridDExk92nSobyDdpPijUq2DW6oDnUqd0iOGxmQjyqhMIihI9+zv4LPyZdRje2cavWPbCbWm3eA==" + }, + "@babel/helper-validator-identifier@7.28.5": { + "integrity": "sha512-qSs4ifwzKJSV39ucNjsvc6WVHs6b7S03sOh2OcHF9UHfVPqWWALUsNUVzhSBiItjRZoLHx7nIarVjqKVusUZ1Q==" + }, + "@babel/parser@7.29.3": { + "integrity": "sha512-b3ctpQwp+PROvU/cttc4OYl4MzfJUWy6FZg+PMXfzmt/+39iHVF0sDfqay8TQM3JA2EUOyKcFZt75jWriQijsA==", + "dependencies": [ + "@babel/types" + ], + "bin": true + }, + "@babel/types@7.29.0": { + "integrity": "sha512-LwdZHpScM4Qz8Xw2iKSzS+cfglZzJGvofQICy7W7v4caru4EaAmyUuO6BGrbyQ2mYV11W0U8j5mBhd14dd3B0A==", + "dependencies": [ + "@babel/helper-string-parser", + "@babel/helper-validator-identifier" + ] + }, + "@emnapi/core@1.10.0": { + "integrity": "sha512-yq6OkJ4p82CAfPl0u9mQebQHKPJkY7WrIuk205cTYnYe+k2Z8YBh11FrbRG/H6ihirqcacOgl2BIO8oyMQLeXw==", + "dependencies": [ + "@emnapi/wasi-threads", + "tslib" + ] + }, + "@emnapi/runtime@1.10.0": { + "integrity": "sha512-ewvYlk86xUoGI0zQRNq/mC+16R1QeDlKQy21Ki3oSYXNgLb45GV1P6A0M+/s6nyCuNDqe5VpaY84BzXGwVbwFA==", + "dependencies": [ + "tslib" + ] + }, + "@emnapi/wasi-threads@1.2.1": { + "integrity": "sha512-uTII7OYF+/Mes/MrcIOYp5yOtSMLBWSIoLPpcgwipoiKbli6k322tcoFsxoIIxPDqW01SQGAgko4EzZi2BNv2w==", + "dependencies": [ + "tslib" + ] + }, + "@napi-rs/wasm-runtime@1.1.4_@emnapi+core@1.10.0_@emnapi+runtime@1.10.0": { + "integrity": "sha512-3NQNNgA1YSlJb/kMH1ildASP9HW7/7kYnRI2szWJaofaS1hWmbGI4H+d3+22aGzXXN9IJ+n+GiFVcGipJP18ow==", + "dependencies": [ + "@emnapi/core", + "@emnapi/runtime", + "@tybys/wasm-util" + ] + }, + "@oxc-parser/binding-android-arm-eabi@0.132.0": { + "integrity": "sha512-KrLaPWa5c9Y7LkW+rKkaUE3y7DBDrQtaf7rlsSDfv6KAHUjgzAIRA761Lrrp6//Yd/Rlie/yEOt9YENCoJnOcw==", + "os": ["android"], + "cpu": ["arm"] + }, + "@oxc-parser/binding-android-arm64@0.132.0": { + "integrity": "sha512-SThDrSeamB/kG2+NxcJ5/wSLcV6dUqDknrPLqFYQ0ST/55mtBP4M7Q/f3QbubH6aAd11wpzZn/nwbVRSdobOpg==", + "os": ["android"], + "cpu": ["arm64"] + }, + "@oxc-parser/binding-darwin-arm64@0.132.0": { + "integrity": "sha512-Lc0f/TYoKBghE5/2Gsv7bLXk+TJZunx2Tf61X8hG4ARXdc8UYI26dCGccFSd1AyFbK3jfaNXtMnupggDbjPXdQ==", + "os": ["darwin"], + "cpu": ["arm64"] + }, + "@oxc-parser/binding-darwin-x64@0.132.0": { + "integrity": "sha512-RG2eJIpf7C21z9HSSXFw1bTArdpKe7Y4fwcJTwRq1yCSe1vSavaN9GA1sm9KqzemTLAGVktQ+7qBTGp0vQeUZg==", + "os": ["darwin"], + "cpu": ["x64"] + }, + "@oxc-parser/binding-freebsd-x64@0.132.0": { + "integrity": "sha512-wQIPntPLtJ8NcBpvKPbEv3NqzV6k8eP8tP/jE9Rg8HTg/j7urZGFSsTCPCW5k77Qfw2DM4vRvc9p3I4yq/Shvw==", + "os": ["freebsd"], + "cpu": ["x64"] + }, + "@oxc-parser/binding-linux-arm-gnueabihf@0.132.0": { + "integrity": "sha512-PixKEpeSe3yxQWqNyOCBALRYc72+Tj7ILDofUl3iXo25cVOzLA6jHUhmOINRtWIPh7dbUie3QNeabwaQpZTw6w==", + "os": ["linux"], + "cpu": ["arm"] + }, + "@oxc-parser/binding-linux-arm-musleabihf@0.132.0": { + "integrity": "sha512-sCR+DzGHlyHKnbA2z9zWjTUhIo8Sy0enJl4RDsBwPmkxYynPatpwOAWe8W5127SlW0boqUWHGtr1NWn5UwIhXQ==", + "os": ["linux"], + "cpu": ["arm"] + }, + "@oxc-parser/binding-linux-arm64-gnu@0.132.0": { + "integrity": "sha512-sQBix5P2cW+IpzTcCwYxnh9yALrKSIkKJThspBvMGcygSMnbzkSvhN7SfuX1hvBk8y1XEChsdkU3ET0V5DmzUw==", + "os": ["linux"], + "cpu": ["arm64"] + }, + "@oxc-parser/binding-linux-arm64-musl@0.132.0": { + "integrity": "sha512-WozHg3Kc//8Sk756HXXgMbEAvqtG+Lzb9JOojwQzIGDtN78Az2dLttkb71akWYUF/8IgYfDSlfKh4Uot8is5Vw==", + "os": ["linux"], + "cpu": ["arm64"] + }, + "@oxc-parser/binding-linux-ppc64-gnu@0.132.0": { + "integrity": "sha512-CmX/ulNBOEwWTyVRmcpYKAcAizW6+OjtLJgo7fXoL9OqQvjF4VER8tPomv44vwzfSCy1BHbsB0ZlZYzYJNj4cA==", + "os": ["linux"], + "cpu": ["ppc64"] + }, + "@oxc-parser/binding-linux-riscv64-gnu@0.132.0": { + "integrity": "sha512-j9oQS+hM90SdhviNGWbPgT4+Rlq+ac++q/zjgwPD1mVHgxHzATvoRGtDx0sXGmFOQ9J9YkwAhYGb5MAHL6TAsA==", + "os": ["linux"], + "cpu": ["riscv64"] + }, + "@oxc-parser/binding-linux-riscv64-musl@0.132.0": { + "integrity": "sha512-bLz+Xi+Agnfmd7kWPEsSVwCn2k4EyIalZkNBcQ0OGIv9rqn8VgCPLNd03tM9mKX/5TdlvDXalz0q71BIrOPNqg==", + "os": ["linux"], + "cpu": ["riscv64"] + }, + "@oxc-parser/binding-linux-s390x-gnu@0.132.0": { + "integrity": "sha512-U6t2qbJU0ypTfyj9QV3W1Y6mITDTL8ai/OR6NUn85vyHthOvobKWgXzU4tu0EskSzlpuVFz1g0jFGulDIUKHxQ==", + "os": ["linux"], + "cpu": ["s390x"] + }, + "@oxc-parser/binding-linux-x64-gnu@0.132.0": { + "integrity": "sha512-WcEaSNHFk8yz5YFlQQAlhq6jOFmZBB/RKE7uzhyCIf+pF1Lmv9gUH4221mle2Gd9iHyWT3ySNph8yZgb1xYdWg==", + "os": ["linux"], + "cpu": ["x64"] + }, + "@oxc-parser/binding-linux-x64-musl@0.132.0": { + "integrity": "sha512-iQrV4iJzQgRwK3BWRmQl1C3C6g3wYpXN2WLdQdyR+efoUnncdShZAVp9OgcojtlD3MDRbuOMGG3SjxF4fL4nlQ==", + "os": ["linux"], + "cpu": ["x64"] + }, + "@oxc-parser/binding-openharmony-arm64@0.132.0": { + "integrity": "sha512-FWzmUGrZ6GUby4U7WIwcCtab6tdmlTO3xTRRKyb5kjIJVEiaUAT8animUG/nK8ZCA8gkRkPOTId4rl6uTqUmJQ==", + "os": ["openharmony"], + "cpu": ["arm64"] + }, + "@oxc-parser/binding-wasm32-wasi@0.132.0": { + "integrity": "sha512-TlbMppxJI5CjWDes0QaP6G3aneVg1yikBu5QYI+DUShF9WDL66ccgKFNNGmi/Wybtszw6hxwAvv76T4DaPKnHw==", + "dependencies": [ + "@emnapi/core", + "@emnapi/runtime", + "@napi-rs/wasm-runtime" + ], + "cpu": ["wasm32"] + }, + "@oxc-parser/binding-win32-arm64-msvc@0.132.0": { + "integrity": "sha512-RH/NbFjGKqdUAUi7Oh3LQPxUk2hsWFEEQ38HSnbRQT8QjBZFKqL1fMbmsB3N4jy/KPh9iX94+9dmkEMBBbambw==", + "os": ["win32"], + "cpu": ["arm64"] + }, + "@oxc-parser/binding-win32-ia32-msvc@0.132.0": { + "integrity": "sha512-JUr4jQY9jxoIB/YTLXr6XofSi5xikj6p5/Ns1h0VOBDT0j1jKU+kMsv2xxv51RwnETcXpA1Yw/9oUAfcqfaqEA==", + "os": ["win32"], + "cpu": ["ia32"] + }, + "@oxc-parser/binding-win32-x64-msvc@0.132.0": { + "integrity": "sha512-2dapgHpA5X8DSXF4AU36hJWYf6zP0tKjMXFRAZFBD62pkevW/uhFDXoFH9Y/3Fd2EtDrw5ByNnR1wVE9X9y0SQ==", + "os": ["win32"], + "cpu": ["x64"] + }, + "@oxc-project/types@0.132.0": { + "integrity": "sha512-FESMOxil5Se014ui/Eq8fT5uHJo6nIRwH0PfJrZJXs6Gek3ZVFOrpUv3YIZT20m+extU98Hg1Ym72U58rlsxUQ==" + }, + "@tybys/wasm-util@0.10.2": { + "integrity": "sha512-RoBvJ2X0wuKlWFIjrwffGw1IqZHKQqzIchKaadZZfnNpsAYp2mM0h36JtPCjNDAHGgYez/15uMBpfGwchhiMgg==", + "dependencies": [ + "tslib" + ] + }, "dedent@1.7.1": { "integrity": "sha512-9JmrhGZpOlEgOLdQgSm0zxFaYoQon408V1v49aqTWuXENVlnCuY9JBZcXZiCsZQWDjTm5Qf/nIvAy77mXDAjEg==" }, @@ -103,8 +273,43 @@ "outdent@0.8.0": { "integrity": "sha512-KiOAIsdpUTcAXuykya5fnVVT+/5uS0Q1mrkRHcF89tpieSmY33O/tmc54CqwA+bfhbtEfZUNLHaPUiB9X3jt1A==" }, + "oxc-parser@0.132.0": { + "integrity": "sha512-+0LAPHaqtfQlvWdpaAa09SmOaZZgP8C552xosEkGJ4+ruEwP1Vgx+sqBgcBCNfR6KDCmagGOZTde8wmAvcI/Hg==", + "dependencies": [ + "@oxc-project/types" + ], + "optionalDependencies": [ + "@oxc-parser/binding-android-arm-eabi", + "@oxc-parser/binding-android-arm64", + "@oxc-parser/binding-darwin-arm64", + "@oxc-parser/binding-darwin-x64", + "@oxc-parser/binding-freebsd-x64", + "@oxc-parser/binding-linux-arm-gnueabihf", + "@oxc-parser/binding-linux-arm-musleabihf", + "@oxc-parser/binding-linux-arm64-gnu", + "@oxc-parser/binding-linux-arm64-musl", + "@oxc-parser/binding-linux-ppc64-gnu", + "@oxc-parser/binding-linux-riscv64-gnu", + "@oxc-parser/binding-linux-riscv64-musl", + "@oxc-parser/binding-linux-s390x-gnu", + "@oxc-parser/binding-linux-x64-gnu", + "@oxc-parser/binding-linux-x64-musl", + "@oxc-parser/binding-openharmony-arm64", + "@oxc-parser/binding-wasm32-wasi", + "@oxc-parser/binding-win32-arm64-msvc", + "@oxc-parser/binding-win32-ia32-msvc", + "@oxc-parser/binding-win32-x64-msvc" + ] + }, "pure-rand@7.0.1": { "integrity": "sha512-oTUZM/NAZS8p7ANR3SHh30kXB+zK2r2BPcEn/awJIbOvq82WoMN4p62AWWp3Hhw50G0xMsw1mhIBLqHw64EcNQ==" + }, + "tslib@2.8.1": { + "integrity": "sha512-oJFu94HQb+KVduSUQL7wnpmqnfmLsOA/nAh6b6EH0wCEoK0/mPeXU6c3wKDV83MkOuHPRHtSXKKU99IBazS/2w==" + }, + "typescript@6.0.3": { + "integrity": "sha512-y2TvuxSZPDyQakkFRPZHKFm+KKVqIisdg9/CZwm9ftvKXLP8NRWj38/ODjNbr43SsoXqNuAisEf1GdCxqWcdBw==", + "bin": true } } } diff --git a/mod.ts b/mod.ts index 7185382..373631f 100644 --- a/mod.ts +++ b/mod.ts @@ -46,7 +46,9 @@ * Both paths share the same guarantees: non-whitespace content is never * removed, newlines in interpolated values are never normalized, and * multi-line values can be aligned at their insertion column with - * {@link align} or {@link embed}. + * {@link align} or {@link embed}. By default that insertion column is measured + * with {@link columnOffset}, and callers can override that policy with the + * `columnOffset` option when they need Unicode-aware visual alignment. * * @module */ @@ -81,6 +83,18 @@ export interface TrimSides { trailing?: TrimMode; } +/** + * Measure the current insertion column for alignment. + * + * `undent` calls this with the output accumulated so far and expects the + * number of spaces to prepend before later lines of an aligned value. + * + * The default implementation is {@link columnOffset}, which counts UTF-16 + * code units after the last newline. Override it when you need alignment to + * follow a different visual-width policy. + */ +export type ColumnOffsetFunction = (text: string) => number; + /** * Options for configuring an `undent` instance. * @@ -142,6 +156,18 @@ export interface UndentOptions { * @default false */ alignValues?: boolean; + + /** + * Measure the insertion column used by {@link align}, {@link embed}, and + * {@link alignValues}. + * + * The default is {@link columnOffset}, which counts UTF-16 code units after + * the last newline. Override this when you need alignment to follow a custom + * display-width policy such as Unicode terminal columns. + * + * @default columnOffset + */ + columnOffset?: ColumnOffsetFunction; } /** @@ -261,6 +287,8 @@ export interface ResolvedOptions { newline: string | null; /** When `true`, every multi-line interpolated value is automatically aligned at its insertion column. */ alignValues: boolean; + /** How to measure the insertion column used for alignment padding. */ + columnOffset: ColumnOffsetFunction; } // ========================================================================== @@ -307,8 +335,14 @@ export const indent: unique symbol = Symbol("undent.indent"); */ export const ALIGNED: unique symbol = Symbol("undent.aligned"); -/** Internal symbol for per-value aligned-text memoization. */ -const ALIGNED_TEXT_CACHE: unique symbol = Symbol("undent.alignedTextCache"); +/** + * Per-wrapper memoization for aligned text. + * + * We keep this cache outside the public wrapper object so rendering can reuse + * aligned output without mutating values returned by {@link align} or + * {@link embed}. + */ +const ALIGNED_TEXT_CACHE = new WeakMap>(); // Character codes used in hot loops. // Hex is compact for low-level scanning, so we document each value: @@ -347,10 +381,6 @@ export interface AlignedValue { readonly value: string; } -interface InternalAlignedValue extends AlignedValue { - [ALIGNED_TEXT_CACHE]?: Map; -} - /** * Mark an interpolated value for column alignment. * @@ -494,6 +524,7 @@ export const DEFAULTS: ResolvedOptions = { trimTrailing: "all", newline: null, alignValues: false, + columnOffset, }; /** @@ -666,7 +697,7 @@ function undentTag( // Fast path: when alignValues is true, always use aligned join. if (state.opts.alignValues) { - return joinAligned(segments, effectiveValues, true); + return joinAligned(state.opts, segments, effectiveValues, true); } // Common path: try plain join, bail to aligned if we hit a wrapped value. @@ -676,7 +707,7 @@ function undentTag( const raw = effectiveValues[i]; if (typeof raw === "object" && raw !== null && ALIGNED in raw) { // Found an aligned value — switch to aligned join for entire template. - return joinAligned(segments, effectiveValues, false); + return joinAligned(state.opts, segments, effectiveValues, false); } out += String(raw) + (segments[i + 1] ?? ""); } @@ -722,6 +753,9 @@ export function resolveOptions( if (options.alignValues !== undefined) { resolved.alignValues = options.alignValues; } + if (options.columnOffset !== undefined) { + resolved.columnOffset = options.columnOffset; + } if (options.newline !== undefined) { if (options.newline === null) resolved.newline = null; @@ -735,8 +769,8 @@ export function resolveOptions( resolved.trimLeading = options.trim; resolved.trimTrailing = options.trim; } else { - resolved.trimLeading = options.trim.leading ?? "all"; - resolved.trimTrailing = options.trim.trailing ?? "all"; + resolved.trimLeading = options.trim.leading ?? base.trimLeading; + resolved.trimTrailing = options.trim.trailing ?? base.trimTrailing; } } @@ -778,7 +812,7 @@ function getProcessedSegments( if (cached) return cached; const effectiveStrings = anchored - ? Array.prototype.slice.call(strings, 1) as string[] + ? strings.slice(1) : strings; // When anchored, the anchor's column IS the indent level — content @@ -1125,7 +1159,7 @@ function processStrings( if ( i === 0 && i === last && opts.trimLeading === "all" && opts.trimTrailing === "all" && - s.length > 0 && s.trim().length === 0 + s.length > 0 && isStructuralWhitespaceOnly(s) ) { s = ""; } @@ -1204,6 +1238,28 @@ export function dedentString( ): string { const len = input.length; if (len === 0) return ""; + const mayTrimLeading = trimLeading !== "none" && hasLeadingBlankLine(input); + const mayTrimTrailing = trimTrailing !== "none" && hasTrailingBlankLine(input); + + // Fast path for the common hot-path string case: a single logical line. + // In that shape, dedenting reduces to stripping leading spaces/tabs from the + // only content line. We can answer that by scanning the prefix instead of the + // whole string, which especially helps already-clean strings and very large + // single-line inputs. + if (input.indexOf("\n") === -1 && input.indexOf("\r") === -1) { + let firstNonWs = 0; + while (firstNonWs < len) { + const c = input.charCodeAt(firstNonWs); + if (c !== CC_SPACE && c !== CC_TAB) break; + firstNonWs++; + } + + if (firstNonWs === len) { + return trimLeading === "all" && trimTrailing === "all" ? "" : input; + } + + return firstNonWs === 0 ? input : input.slice(firstNonWs); + } // Pass 1: find minimum indent across non-blank lines. // Blank lines do not influence minIndent; they are structural only. @@ -1223,6 +1279,13 @@ export function dedentString( const c = input.charCodeAt(i); if (c !== CC_LF && c !== CC_CR) { const ws = i - lineStart; + // Once any content line starts at column 0, the common indent is fixed + // at 0 for the whole string. If there are also no blank wrapper lines to + // trim, dedenting cannot change the input, so we can return it without a + // full second pass. + if (ws === 0 && !mayTrimLeading && !mayTrimTrailing) { + return input; + } if (ws < minIndent) { minIndent = ws; if (ws === 0) break; // Can't go lower. @@ -1302,6 +1365,40 @@ export function dedentString( return result; } +/** + * Return true when the string starts with a blank line. + * + * A blank leading line is optional spaces/tabs followed by a newline sequence. + * This helper exists so dedentString() can cheaply decide whether trim work is + * even possible before it commits to a full multi-pass transformation. + */ +function hasLeadingBlankLine(text: string): boolean { + for (let i = 0; i < text.length; i++) { + const c = text.charCodeAt(i); + if (c === CC_SPACE || c === CC_TAB) continue; + return c === CC_LF || c === CC_CR; + } + + return false; +} + +/** + * Return true when the string ends with a blank line. + * + * This is the trailing-edge mirror of hasLeadingBlankLine(). It scans backward + * through optional spaces/tabs and reports whether the first non-horizontal + * whitespace byte is a newline sequence. + */ +function hasTrailingBlankLine(text: string): boolean { + for (let i = text.length - 1; i >= 0; i--) { + const c = text.charCodeAt(i); + if (c === CC_SPACE || c === CC_TAB) continue; + return c === CC_LF || c === CC_CR; + } + + return false; +} + /** * Trim mode "one" for the leading edge. * @@ -1435,6 +1532,7 @@ function trimTrailingBlankLinesAll(text: string): number { * 4. Otherwise, stringify and concatenate directly. */ function joinAligned( + opts: ResolvedOptions, strings: ReadonlyArray, values: ReadonlyArray, alignAll: boolean, @@ -1449,10 +1547,10 @@ function joinAligned( if (wrapped) { // Wrapped values always align. For hot loops with repeated values, // this path memoizes alignment by pad width and reuses results. - const pad = " ".repeat(columnOffset(out)); + const pad = " ".repeat(opts.columnOffset(out)); out += getAlignedWrappedText(raw, pad); } else if (alignAll && hasNewline(text)) { - out += alignText(text, " ".repeat(columnOffset(out))); + out += alignText(text, " ".repeat(opts.columnOffset(out))); } else { out += text; } @@ -1564,12 +1662,34 @@ function hasNewline(text: string): boolean { } /** - * Return aligned text for a wrapped value, using a small per-value - * cache keyed by the pad string. + * Return true when text contains only structural whitespace. * - * Targets hot `embed(...)` loops where both the value and insertion - * column repeat across iterations. Cache is bounded to - * {@link ALIGNED_TEXT_CACHE_MAX} entries per wrapped value. + * This helper is intentionally narrower than `String.prototype.trim()`. The + * template pipeline treats only spaces, tabs, and newline bytes as formatting + * characters, so Unicode whitespace such as NBSP should remain content. + */ +function isStructuralWhitespaceOnly(text: string): boolean { + for (let i = 0; i < text.length; i++) { + const c = text.charCodeAt(i); + if (c !== CC_SPACE && c !== CC_TAB && c !== CC_LF && c !== CC_CR) { + return false; + } + } + + return true; +} + +/** + * Return aligned text for a wrapped value. + * + * Repeated `align(...)` and `embed(...)` calls often reuse the same wrapper at + * the same insertion column. This helper memoizes those padded results so hot + * loops can skip recomputing identical alignment work. + * + * > The cache is keyed by the pad string and capped at + * > {@link ALIGNED_TEXT_CACHE_MAX} entries per wrapped value. Keeping the + * > cache small preserves the common fast path without letting rarely repeated + * > columns grow memory use without bound. */ function getAlignedWrappedText(value: AlignedValue, pad: string): string { const text = value.value; @@ -1577,8 +1697,7 @@ function getAlignedWrappedText(value: AlignedValue, pad: string): string { return text; } - const internal = value as InternalAlignedValue; - let cache = internal[ALIGNED_TEXT_CACHE]; + let cache = ALIGNED_TEXT_CACHE.get(value); if (cache) { const hit = cache.get(pad); if (hit !== undefined) return hit; @@ -1588,7 +1707,7 @@ function getAlignedWrappedText(value: AlignedValue, pad: string): string { if (!cache) { cache = new Map(); - internal[ALIGNED_TEXT_CACHE] = cache; + ALIGNED_TEXT_CACHE.set(value, cache); } if (cache.size >= ALIGNED_TEXT_CACHE_MAX) { @@ -1715,18 +1834,22 @@ export function rejoinLines( } /** - * Count characters from the last newline to the end of the string. + * Count how far the output has advanced since the last newline. + * + * Alignment uses this insertion offset to decide how many spaces to add before + * later lines of a wrapped value. * - * This gives the "column offset" — the horizontal position where the - * next character would appear. Used internally by alignment to decide - * how many spaces to pad. + * > This is a UTF-16 code-unit offset, not display width. That keeps the + * > helper fast and deterministic for string processing, but editors may show + * > a different visual column for tabs, emoji, combining marks, or full-width + * > characters. * * Uses `lastIndexOf` (implemented in C++ by V8) instead of a charcode * loop for ~100x speedup on long strings. * * @param text - The string to measure. - * @returns The number of characters after the final newline, or the - * full string length if there are no newlines. + * @returns The number of UTF-16 code units after the final newline, + * or the full string length if there are no newlines. * * @example Measuring the insertion column * ```ts diff --git a/mod_test.ts b/mod_test.ts index 32b5945..1797026 100644 --- a/mod_test.ts +++ b/mod_test.ts @@ -21,6 +21,7 @@ import undent, { } from "./mod.ts"; import type { AlignedValue, + ColumnOffsetFunction, ResolvedOptions, TrimMode, TrimSides, @@ -151,6 +152,11 @@ World World`; expect(result).toBe("Hello\nWorld"); }); + + it("preserves non-structural Unicode whitespace", () => { + const result = undent`\u00A0`; + expect(result).toBe("\u00A0"); + }); }); // ------------------------------------------------------------------------- @@ -415,6 +421,17 @@ World `; expect(result).toBe("\r\nfirst\r\nsecond"); }); + + it("inherits unspecified trim sides across chained .with() calls", () => { + const keep = undent.with({ trim: "none" }); + const next = keep.with({ trim: { leading: "one" } }); + + const result = next` + Hello + `; + + expect(result).toBe("Hello\n"); + }); }); describe("alignValues option", () => { @@ -482,6 +499,21 @@ World "class Foo {\n greet() {\n console.log('hi');\n }\n\n bye() {\n console.log('bye');\n }\n}", ); }); + + it("uses a custom columnOffset function when aligning values", () => { + const doubleWidth: ColumnOffsetFunction = (text) => + columnOffset(text) * 2; + const ua = undent.with({ + alignValues: true, + columnOffset: doubleWidth, + }); + + const result = ua` + > ${"a\nb"} + `; + + expect(result).toBe("> a\n b"); + }); }); }); @@ -538,6 +570,16 @@ World const result = undent.string(" hello\n world\nfoo"); expect(result).toBe(" hello\n world\nfoo"); }); + + it("returns already-clean multi-line strings unchanged", () => { + const input = "alpha\nbeta\ngamma"; + expect(undent.string(input)).toBe(input); + }); + + it("returns mixed-indent strings unchanged when one line is already at column 0", () => { + const input = " hello\nworld\n again"; + expect(undent.string(input)).toBe(input); + }); }); // ------------------------------------------------------------------------- @@ -912,6 +954,17 @@ World expect(result).toBe("before after"); }); + it("supports frozen aligned values", () => { + const value = Object.freeze(align("a\nb")); + + const result = undent` + list: + ${value} + `; + + expect(result).toBe("list:\n a\n b"); + }); + it("handles value with only newlines", () => { const result = undent` before ${align("\n\n")} after @@ -1371,6 +1424,12 @@ World expect(result.strategy).toBe("first"); }); + it("merges columnOffset", () => { + const custom: ColumnOffsetFunction = (text) => columnOffset(text) + 1; + const result = resolveOptions(DEFAULTS, { columnOffset: custom }); + expect(result.columnOffset).toBe(custom); + }); + it("merges trim string", () => { const result = resolveOptions(DEFAULTS, { trim: "none" }); expect(result.trimLeading).toBe("none"); @@ -1417,6 +1476,7 @@ World expect(DEFAULTS.trimTrailing).toBe("all"); expect(DEFAULTS.newline).toBe(null); expect(DEFAULTS.alignValues).toBe(false); + expect(DEFAULTS.columnOffset).toBe(columnOffset); }); }); @@ -1473,6 +1533,11 @@ World const a: AlignedValue = align("x"); expect(isAligned(a)).toBe(true); }); + + it("ColumnOffsetFunction is usable", () => { + const fn: ColumnOffsetFunction = columnOffset; + expect(fn('x\ny')).toBe(1); + }); }); // ========================================================================= @@ -2232,6 +2297,7 @@ World trimTrailing: "none", newline: "\r\n", alignValues: true, + columnOffset, }; const result = resolveOptions(custom, {}); expect(result).toEqual(custom); diff --git a/readme.md b/readme.md index 307d0c7..577e0f5 100644 --- a/readme.md +++ b/readme.md @@ -207,6 +207,12 @@ undent` // end ``` +By default, that insertion column is measured with JavaScript string offsets +after the last newline. That is fast and stable for code generation, but it is +not the same thing as visual width in a terminal or editor. Tabs, combining +marks, emoji, and full-width characters can render at different visual columns +than their UTF-16 length suggests. + When the value itself carries baked-in indentation — a SQL snippet from another file, a code block from a constant — use `embed()`. It strips the value's own indentation first, then aligns it at the insertion column: @@ -252,6 +258,31 @@ u` > `align()` and `embed()` always align regardless of the `alignValues` setting — > they're the per-value opt-in. +### Unicode visual alignment + +If you need terminal-style Unicode alignment, opt into the separate Unicode +helpers subpath instead of changing the default behavior for every caller: + +```ts +import { undent } from "@okikio/undent"; +import { createUnicodeColumnOffset } from "@okikio/undent/unicode"; + +const terminalUndent = undent.with({ + alignValues: true, + columnOffset: createUnicodeColumnOffset({ tabWidth: 4 }), +}); + +terminalUndent` + label: 界 ${"alpha\nbeta"} +`; +// label: 界 alpha +// beta +``` + +This mode is still best-effort. Visual width depends on the renderer, font, and +surrounding context, so `@okikio/undent/unicode` aims at common terminal-style +output rather than browser-perfect layout. + ### Trimming By default, `undent` removes all blank lines at the start and end of the output @@ -438,12 +469,20 @@ preserved byte-for-byte. | `alignText(text, pad)` | Pad subsequent lines of text with a prefix string | | `splitLines(text)` | Split a string preserving exact newline sequences | | `rejoinLines(lines, seps)` | Reconstruct a string from `splitLines` output | -| `columnOffset(text)` | Count characters since the last newline (insertion column) | +| `columnOffset(text)` | Count UTF-16 code units since the last newline (default alignment policy) | | `newlineLengthAt(text, i)` | Length of the newline sequence at position `i` (0, 1, or 2) | | `resolveOptions(base, overrides)` | Merge option objects for custom pipelines | | `DEFAULTS` | The default resolved options constant | | `indent` | Symbol for indent anchors | +### Unicode subpath + +| Export | Description | +| --------------------------------------- | --------------------------------------------------------------------- | +| `createUnicodeColumnOffset(options?)` | Build a terminal-style Unicode-aware `columnOffset` function | +| `unicodeColumnOffset(text, options?)` | Measure the last line of a string in visual columns | +| `visualColumnWidth(text, options?)` | Measure a single line in terminal-style display columns | + ### Options ```ts @@ -452,6 +491,7 @@ interface UndentOptions { trim?: TrimMode | TrimSides; // How to trim wrapper lines (default: "all") newline?: string | null; // Normalize segment newlines (default: null) alignValues?: boolean; // Auto-align all multi-line values (default: false) + columnOffset?: (text: string) => number; // Measure alignment columns (default: columnOffset) } type TrimMode = "all" | "one" | "none"; diff --git a/scripts/build_npm.ts b/scripts/build_npm.ts index b54e0f5..65a0a55 100644 --- a/scripts/build_npm.ts +++ b/scripts/build_npm.ts @@ -14,7 +14,18 @@ import denoJson from "../deno.json" with { type: "json" }; await emptyDir("./npm"); await build({ - entryPoints: ["./mod.ts"], + entryPoints: [ + { + kind: "export", + name: ".", + path: "./mod.ts", + }, + { + kind: "export", + name: "./unicode", + path: "./unicode.ts", + }, + ], outDir: "./npm", // mod.ts uses no Deno-specific globals (no Deno.*, no std/ imports), @@ -34,6 +45,15 @@ await build({ // which is the correct runtime for these tests. test: false, + // Do not publish declaration source maps. + // + // They create long generated JSON strings in `*.d.ts.map` files. + // Socket flags those as "Long strings" because that pattern can also + // appear in packed or obfuscated malware. For this package, the maps + // are not needed because the npm package does not publish the original + // `mod.ts` source next to them anyway. + declarationMap: false, + package: { name: denoJson.name, // "@okikio/undent" version: denoJson.version, diff --git a/scripts/sync_unicode_east_asian_width.ts b/scripts/sync_unicode_east_asian_width.ts new file mode 100644 index 0000000..afa2e30 --- /dev/null +++ b/scripts/sync_unicode_east_asian_width.ts @@ -0,0 +1,475 @@ +/** + * Verify or refresh the East Asian Width tables used by `unicode.ts`. + * + * Run with: + * deno task unicode:eaw:check + * deno task unicode:eaw:update + * + * The script treats Unicode's `EastAsianWidth.txt` as the source of truth for + * the East Asian Width tables consumed by `unicode.ts`. Default mode only + * checks for drift. Pass `--write` to rewrite the internal constants file in + * place. + * + * The fetch path verifies more than "latest responded": it reads the version + * from `latest/ucd/ReadMe.txt`, fetches the matching immutable versioned file, + * and compares SHA-256 digests before trusting the payload. That keeps the + * mutable `latest` alias honest and gives failures a concrete release anchor. + */ +import { object } from 'jsr:@optique/core/constructs'; +import { runParserSync } from 'jsr:@optique/core/facade'; +import { message } from 'jsr:@optique/core/message'; +import { option } from 'jsr:@optique/core/primitives'; +import { fromFileUrl } from 'jsr:@std/path/from-file-url'; +import { parseSync } from 'npm:oxc-parser@0.132.0'; + +import { undent } from '../mod.ts'; + +import { + EAST_ASIAN_AMBIGUOUS_RANGES, + EAST_ASIAN_WIDE_RANGES, +} from '../_unicode_constants.ts'; + +type Range = readonly [start: number, end: number]; + +type SyncMode = 'check' | 'write'; + +type ParsedEastAsianWidth = { + wideRanges: Range[]; + ambiguousRanges: Range[]; +}; + +type DriftSummary = { + extras: number; + missing: number; + extraSamples: string[]; + missingSamples: string[]; +}; + +type VerifiedUnicodeSource = { + version: string; + date: string; + text: string; + latestSha256: string; + versionedSha256: string; +}; + +type RangeConstantName = + | 'EAST_ASIAN_WIDE_RANGES' + | 'EAST_ASIAN_AMBIGUOUS_RANGES'; + +const UNICODE_HOST = 'www.unicode.org'; +const LATEST_README_URL = 'https://www.unicode.org/Public/UCD/latest/ucd/ReadMe.txt'; +const LATEST_EAST_ASIAN_WIDTH_URL = + 'https://www.unicode.org/Public/UCD/latest/ucd/EastAsianWidth.txt'; +const UNICODE_CONSTANTS_URL = new URL('../_unicode_constants.ts', import.meta.url); +const UNICODE_CONSTANTS_PATH = fromFileUrl(UNICODE_CONSTANTS_URL); +const FETCH_TIMEOUT_MS = 30_000; +const VERSION_PATTERN = /Version\s+(\d+\.\d+\.\d+)/; +const DATE_PATTERN = /# Date:\s+([^\n]+)/; +const RANGE_CONSTANT_NAMES: readonly RangeConstantName[] = [ + 'EAST_ASIAN_WIDE_RANGES', + 'EAST_ASIAN_AMBIGUOUS_RANGES', +]; + +const cliParser = object({ + write: option('--write', { + description: message`Rewrite _unicode_constants.ts with the verified upstream East Asian Width ranges.`, + }), +}); + +const cliArgs = runParserSync(cliParser, 'sync_unicode_east_asian_width.ts', Deno.args, { + help: { + option: { names: ['-h', '--help'] }, + onShow: Deno.exit, + }, + onError: Deno.exit, + description: message`Verify or refresh the East Asian Width tables used by unicode.ts.`, + footer: message`Tasks: deno task unicode:eaw:check, deno task unicode:eaw:update`, +}); +const mode: SyncMode = cliArgs.write ? 'write' : 'check'; + +await ensureRequiredPermissions(mode); + +const upstream = await fetchVerifiedUnicodeSource(); +const parsed = parseEastAsianWidth(upstream.text); + +const wideDrift = summarizeDrift( + expandRanges(EAST_ASIAN_WIDE_RANGES), + expandRanges(parsed.wideRanges), +); +const ambiguousDrift = summarizeDrift( + expandRanges(EAST_ASIAN_AMBIGUOUS_RANGES), + expandRanges(parsed.ambiguousRanges), +); + +if (mode === 'check') { + reportDrift(upstream, wideDrift, ambiguousDrift); + if (wideDrift.extras !== 0 || wideDrift.missing !== 0) { + Deno.exit(1); + } + if (ambiguousDrift.extras !== 0 || ambiguousDrift.missing !== 0) { + Deno.exit(1); + } + + console.log('_unicode_constants.ts East Asian Width tables match Unicode upstream.'); + Deno.exit(0); +} + +const constantsSource = await Deno.readTextFile(UNICODE_CONSTANTS_URL); +validateConstantsModule(constantsSource); +const updatedSource = renderConstantsFile(parsed); + +if (updatedSource !== constantsSource) { + await Deno.writeTextFile(UNICODE_CONSTANTS_URL, updatedSource); + console.log( + `Updated _unicode_constants.ts East Asian Width tables from Unicode ${upstream.version} (${upstream.versionedSha256}).`, + ); + Deno.exit(0); +} + +console.log('_unicode_constants.ts East Asian Width tables were already up to date.'); + +async function ensureRequiredPermissions( + mode: SyncMode, +): Promise { + await ensurePermission( + { name: 'net', host: UNICODE_HOST }, + `network access to ${UNICODE_HOST}`, + ); + + if (mode === 'write') { + await ensurePermission( + { name: 'read', path: UNICODE_CONSTANTS_PATH }, + `read access to ${UNICODE_CONSTANTS_PATH}`, + ); + await ensurePermission( + { name: 'write', path: UNICODE_CONSTANTS_PATH }, + `write access to ${UNICODE_CONSTANTS_PATH}`, + ); + } +} + +async function ensurePermission( + descriptor: Deno.PermissionDescriptor, + label: string, +): Promise { + const current = await Deno.permissions.query(descriptor); + if (current.state === 'granted') { + return; + } + + const requested = current.state === 'prompt' + ? await Deno.permissions.request(descriptor) + : current; + if (requested.state !== 'granted') { + throw new Error( + `This script requires ${label}. Grant the matching Deno permission and try again.`, + ); + } +} + +async function fetchVerifiedUnicodeSource(): Promise { + const readmeText = await fetchText(LATEST_README_URL); + const version = parseReadmeValue(readmeText, VERSION_PATTERN, 'Unicode version'); + const date = parseReadmeValue(readmeText, DATE_PATTERN, 'UCD date'); + const versionedUrl = `https://www.unicode.org/Public/${version}/ucd/EastAsianWidth.txt`; + + const latestBytes = await fetchBytes(LATEST_EAST_ASIAN_WIDTH_URL); + const versionedBytes = await fetchBytes(versionedUrl); + const latestSha256 = await sha256Hex(latestBytes); + const versionedSha256 = await sha256Hex(versionedBytes); + if (latestSha256 !== versionedSha256) { + throw new Error( + [ + 'Unicode upstream integrity check failed.', + `latest URL: ${LATEST_EAST_ASIAN_WIDTH_URL}`, + `versioned URL: ${versionedUrl}`, + `latest sha256: ${latestSha256}`, + `versioned sha256: ${versionedSha256}`, + ].join('\n'), + ); + } + + return { + version, + date, + text: new TextDecoder().decode(versionedBytes), + latestSha256, + versionedSha256, + }; +} + +function parseReadmeValue(text: string, pattern: RegExp, label: string): string { + const match = text.match(pattern); + if (!match?.[1]) { + throw new Error(`Could not parse ${label} from Unicode ReadMe.txt`); + } + + return match[1].trim(); +} + +async function fetchText(url: string): Promise { + return new TextDecoder().decode(await fetchBytes(url)); +} + +async function fetchBytes(url: string): Promise> { + const response = await fetch(url, { + headers: { + accept: 'text/plain; charset=utf-8', + }, + signal: AbortSignal.timeout(FETCH_TIMEOUT_MS), + }); + if (!response.ok) { + throw new Error(`Failed to fetch ${url}: ${response.status} ${response.statusText}`); + } + + const arrayBuffer = await response.arrayBuffer(); + return new Uint8Array(arrayBuffer); +} + +async function sha256Hex(bytes: Uint8Array): Promise { + const digest = await crypto.subtle.digest('SHA-256', bytes); + return Array.from(new Uint8Array(digest)) + .map((byte) => byte.toString(16).padStart(2, '0')) + .join(''); +} + +function parseEastAsianWidth(text: string): ParsedEastAsianWidth { + const wide: number[] = []; + const ambiguous: number[] = []; + + for (const rawLine of text.split(/\r?\n/)) { + const line = rawLine.replace(/#.*/, '').trim(); + if (line.length === 0) continue; + + const [rangePart, property] = line.split(';').map((value) => value.trim()); + if (!rangePart || !property) continue; + + const target = property === 'W' || property === 'F' + ? wide + : property === 'A' + ? ambiguous + : null; + if (target === null) continue; + + const [startHex, endHex = startHex] = rangePart.split('..'); + const start = Number.parseInt(startHex, 16); + const end = Number.parseInt(endHex, 16); + for (let codePoint = start; codePoint <= end; codePoint++) { + target.push(codePoint); + } + } + + wide.sort((left, right) => left - right); + ambiguous.sort((left, right) => left - right); + + return { + wideRanges: compressCodePoints(wide), + ambiguousRanges: compressCodePoints(ambiguous), + }; +} + +function compressCodePoints(codePoints: number[]): Range[] { + if (codePoints.length === 0) return []; + + const ranges: Range[] = []; + let start = codePoints[0]!; + let end = start; + + for (let index = 1; index < codePoints.length; index++) { + const codePoint = codePoints[index]!; + if (codePoint === end + 1) { + end = codePoint; + continue; + } + + ranges.push([start, end]); + start = codePoint; + end = codePoint; + } + + ranges.push([start, end]); + return ranges; +} + +function expandRanges(ranges: readonly Range[]): Set { + const out = new Set(); + for (const [start, end] of ranges) { + for (let codePoint = start; codePoint <= end; codePoint++) { + out.add(codePoint); + } + } + + return out; +} + +function summarizeDrift(current: Set, upstreamSet: Set): DriftSummary { + const extras: number[] = []; + const missing: number[] = []; + + for (const codePoint of current) { + if (!upstreamSet.has(codePoint)) extras.push(codePoint); + } + for (const codePoint of upstreamSet) { + if (!current.has(codePoint)) missing.push(codePoint); + } + + extras.sort((left, right) => left - right); + missing.sort((left, right) => left - right); + + return { + extras: extras.length, + missing: missing.length, + extraSamples: extras.slice(0, 10).map(formatCodePoint), + missingSamples: missing.slice(0, 10).map(formatCodePoint), + }; +} + +function formatCodePoint(codePoint: number): string { + return `U+${codePoint.toString(16).toUpperCase().padStart(4, '0')}`; +} + +function reportDrift( + upstream: VerifiedUnicodeSource, + wideDrift: DriftSummary, + ambiguousDrift: DriftSummary, +): void { + console.log('East Asian Width audit against Unicode upstream'); + console.log(` source: ${LATEST_EAST_ASIAN_WIDTH_URL}`); + console.log(` release: Unicode ${upstream.version} (${upstream.date})`); + console.log(` sha256(latest): ${upstream.latestSha256}`); + console.log(` sha256(versioned): ${upstream.versionedSha256}`); + console.log(renderDriftLine('wide/fullwidth', wideDrift)); + console.log(renderDriftLine('ambiguous', ambiguousDrift)); + + if (wideDrift.extras !== 0 || wideDrift.missing !== 0) { + console.log(` wide/fullwidth extra samples: ${wideDrift.extraSamples.join(', ') || 'none'}`); + console.log(` wide/fullwidth missing samples: ${wideDrift.missingSamples.join(', ') || 'none'}`); + } + if (ambiguousDrift.extras !== 0 || ambiguousDrift.missing !== 0) { + console.log(` ambiguous extra samples: ${ambiguousDrift.extraSamples.join(', ') || 'none'}`); + console.log(` ambiguous missing samples: ${ambiguousDrift.missingSamples.join(', ') || 'none'}`); + } + if ( + wideDrift.extras !== 0 || + wideDrift.missing !== 0 || + ambiguousDrift.extras !== 0 || + ambiguousDrift.missing !== 0 + ) { + console.log('Run `deno task unicode:eaw:update` to refresh _unicode_constants.ts.'); + } +} + +function renderDriftLine(label: string, drift: DriftSummary): string { + return ` ${label}: ${drift.extras} extra, ${drift.missing} missing`; +} + +function validateConstantsModule(source: string): void { + const result = parseSync(UNICODE_CONSTANTS_PATH, source, { + lang: 'ts', + astType: 'ts', + sourceType: 'module', + }); + + if (result.errors.length > 0) { + throw new Error( + [ + `Could not parse ${UNICODE_CONSTANTS_PATH}.`, + formatParseError(result.errors[0]), + ].join('\n'), + ); + } + + const found = new Set(); + for (const statement of result.program.body) { + const variableDeclaration = getExportedVariableDeclaration(statement); + if (variableDeclaration === null) continue; + + for (const declaration of variableDeclaration.declarations) { + const identifier = getIdentifierName(declaration); + if (identifier === null || !isRangeConstantName(identifier)) continue; + found.add(identifier); + } + } + + for (const name of RANGE_CONSTANT_NAMES) { + if (!found.has(name)) { + throw new Error( + `${UNICODE_CONSTANTS_PATH} must export ${name} before this script can update it.`, + ); + } + } +} + +function isRangeConstantName(value: string): value is RangeConstantName { + return value === 'EAST_ASIAN_WIDE_RANGES' || value === 'EAST_ASIAN_AMBIGUOUS_RANGES'; +} + +function getExportedVariableDeclaration(statement: unknown): { + declarations: readonly unknown[]; +} | null { + if (!isRecord(statement) || statement.type !== 'ExportNamedDeclaration') return null; + const declaration = statement.declaration; + if (!isRecord(declaration) || declaration.type !== 'VariableDeclaration') return null; + if (!Array.isArray(declaration.declarations)) return null; + return { + declarations: declaration.declarations, + }; +} + +function getIdentifierName(declaration: unknown): string | null { + if (!isRecord(declaration)) return null; + const id = declaration.id; + if (!isRecord(id) || id.type !== 'Identifier') return null; + return typeof id.name === 'string' ? id.name : null; +} + +function formatParseError(error: unknown): string { + if (!isRecord(error)) return 'Unknown parser error'; + const message = typeof error.message === 'string' ? error.message : 'Unknown parser error'; + const labels = Array.isArray(error.labels) ? error.labels : []; + const firstLabel = labels[0]; + const loc = isRecord(firstLabel) ? firstLabel : null; + const line = loc !== null && typeof loc.line === 'number' ? loc.line : null; + const column = loc !== null && typeof loc.column === 'number' ? loc.column : null; + if (line === null || column === null) return message; + return `${line}:${column} ${message}`; +} + +function isRecord(value: unknown): value is Record { + return typeof value === 'object' && value !== null; +} + +function renderConstantsFile(parsed: ParsedEastAsianWidth): string { + return [ + undent.with({ trim: { trailing: "one" }})` + /** + * Internal East Asian Width lookup tables generated from Unicode upstream. + * + * \`unicode.ts\` re-exports these tables as part of the public API, but the + * generated data itself lives here so maintenance scripts can update one small + * internal file instead of rewriting the public module. + */ + + `, + renderTable('EAST_ASIAN_WIDE_RANGES', parsed.wideRanges), + '', + renderTable('EAST_ASIAN_AMBIGUOUS_RANGES', parsed.ambiguousRanges), + '', + '// End of generated East Asian Width tables.', + ].join('\n'); +} + +function renderTable(name: RangeConstantName, ranges: readonly Range[]): string { + const lines = ranges.map(([start, end]) => `\t[${toHex(start)}, ${toHex(end)}],`); + + return [ + `export const ${name}: ReadonlyArray = [`, + ...lines, + '];', + ].join('\n'); +} + +function toHex(codePoint: number): string { + return `0x${codePoint.toString(16)}`; +} diff --git a/unicode.ts b/unicode.ts new file mode 100644 index 0000000..ede111a --- /dev/null +++ b/unicode.ts @@ -0,0 +1,564 @@ +import type { ColumnOffsetFunction } from './mod.ts'; +import { + EAST_ASIAN_AMBIGUOUS_RANGES as INTERNAL_EAST_ASIAN_AMBIGUOUS_RANGES, + EAST_ASIAN_WIDE_RANGES as INTERNAL_EAST_ASIAN_WIDE_RANGES, +} from './_unicode_constants.ts'; + +/** + * # Unicode Alignment Helpers + * + * This module provides a renderer-aware companion to `columnOffset()` from the + * root module. The default `undent` path measures alignment columns as UTF-16 + * code units because that is fast, deterministic, and a good fit for code + * generation. The helpers here opt into a different trade-off: best-effort + * terminal-style visual width. + * + * The measurement pipeline is: + * + * ```text + * input string + * -> slice last line after the final newline + * -> segment into grapheme clusters when Intl.Segmenter exists + * -> let widthOf(...) override specific graphemes + * -> otherwise apply built-in width rules + * -> sum visual columns + * ``` + * + * The built-in width rules intentionally stay conservative: + * + * - tabs are configurable because their width depends on the current column, + * - control characters and combining-only fragments contribute zero columns, + * - emoji-style pictographs and regional-indicator flags count as two columns, + * - East Asian wide/fullwidth ranges count as two columns, + * - ambiguous-width ranges can be narrow or wide by option, + * - everything else counts as one column. + * + * This is still not a universal truth for browsers or proportional fonts. It + * is an opt-in approximation for common monospace terminal output. + * + * @module + */ + +/** + * The current visual column before measuring the next grapheme cluster. + * + * Width functions receive this so they can implement stateful rules such as + * tab stops, where the width of `"\t"` depends on the current column rather + * than on the grapheme alone. + */ +export interface UnicodeColumnWidthState { + /** The current visual column before the grapheme is measured. */ + readonly column: number; +} + +/** + * Measure the visual width of one grapheme cluster. + * + * Return `undefined` to fall back to the built-in best-effort width rules. + * Any defined return value must be a non-negative integer. + */ +export type UnicodeWidthFunction = ( + grapheme: string, + state: UnicodeColumnWidthState, +) => number | undefined; + +/** + * Options for the Unicode-aware column measurement helpers in this module. + * + * These helpers are designed for terminal-style monospace alignment. They are + * intentionally opt-in because visual width depends on the renderer, font, and + * surrounding context. + */ +export interface UnicodeColumnOffsetOptions { + /** + * How tabs advance. + * + * Set to `false` to treat `"\t"` as one column. Set to a positive integer to + * align tabs to tab stops. + * + * @default false + */ + tabWidth?: number | false; + + /** + * How to treat East Asian Width "ambiguous" code points. + * + * Unicode recommends treating them as narrow when the rendering context is + * not known. + * + * @default "narrow" + */ + ambiguous?: 'narrow' | 'wide'; + + /** + * Override the width of specific grapheme clusters. + * + * This is the escape hatch for callers who know their actual display target. + * Return `undefined` to fall back to the built-in best-effort rules. + */ + widthOf?: UnicodeWidthFunction; +} + +/** Inclusive `[start, end]` code-point range used by the lookup tables. */ +export type CodePointRange = readonly [start: number, end: number]; + +/** + * Default option values for the Unicode-aware column helpers. + * + * The defaults intentionally choose the conservative terminal policy: tabs are + * treated as a single column unless callers opt into tab stops, and ambiguous + * East Asian Width code points stay narrow unless the rendering context says + * otherwise. + */ +export const DEFAULT_UNICODE_COLUMN_OFFSET_OPTIONS: Required< + Pick +> = { + tabWidth: false, + ambiguous: 'narrow', +}; + +/** + * East Asian wide/fullwidth ranges from Unicode's `EastAsianWidth.txt`. + * + * The runtime stays dependency-free by keeping the verified ranges inline, but + * the generated table now lives in `./_unicode_constants.ts` so maintenance + * tooling can refresh it without rewriting this public module. + */ +export const EAST_ASIAN_WIDE_RANGES: ReadonlyArray = + INTERNAL_EAST_ASIAN_WIDE_RANGES; + +/** + * East Asian ambiguous-width ranges from Unicode's `EastAsianWidth.txt`. + * + * These are the upstream `A` property ranges compressed into inclusive + * `[start, end]` pairs. Callers still choose how to interpret them through the + * `ambiguous` option, while the generated data lives in + * `./_unicode_constants.ts`. + */ +export const EAST_ASIAN_AMBIGUOUS_RANGES: ReadonlyArray = + INTERNAL_EAST_ASIAN_AMBIGUOUS_RANGES; + +/** + * Detect graphemes that contain emoji-style pictographs. + * + * The default width rules treat these graphemes as occupying two columns, + * matching the common terminal convention for emoji presentation. + */ +export const EXTENDED_PICTOGRAPHIC_RE = /\p{Extended_Pictographic}/u; + +/** Minimal shape of an `Intl.Segmenter#segment()` item. */ +export interface GraphemeSegment { + /** The grapheme cluster text for one segmentation result. */ + readonly segment: string; +} + +/** Minimal grapheme-segmentation interface used by this module. */ +export interface GraphemeSegmenter { + /** Segment the input string into grapheme-cluster records. */ + segment(input: string): Iterable; +} + +/** + * Structural type for runtimes that expose `Intl.Segmenter`. + * + * dnt's Node type environment does not include that API, so the module uses a + * structural type instead of referencing `Intl.Segmenter` directly. + */ +export type IntlWithSegmenter = typeof Intl & { + Segmenter?: new ( + locales?: string | string[], + options?: { granularity: 'grapheme' }, + ) => GraphemeSegmenter; +}; + +/** + * Fully resolved Unicode column-measurement options. + * + * The resolver fills in default `tabWidth` and `ambiguous` values while still + * carrying through caller-provided overrides such as `widthOf`. + */ +export type ResolvedUnicodeColumnOffsetOptions = + & Required> + & UnicodeColumnOffsetOptions; + +/** Shared typed reference to the global `Intl` object. */ +const intlWithSegmenter = Intl as IntlWithSegmenter; + +/** Lazily initialized grapheme segmenter cache. */ +let graphemeSegmenter: GraphemeSegmenter | null | undefined; + +/** + * Create a `columnOffset` function that measures terminal-style Unicode width. + * + * Use this with `undent.with({ columnOffset })` when later lines of aligned + * values should follow visual columns instead of raw UTF-16 code units. + * + * @example Aligning with terminal-style Unicode columns + * ```ts + * import { undent } from '@okikio/undent'; + * import { createUnicodeColumnOffset } from '@okikio/undent/unicode'; + * + * const terminalUndent = undent.with({ + * alignValues: true, + * columnOffset: createUnicodeColumnOffset(), + * }); + * + * terminalUndent` + * label: 界 ${'a\nb'} + * `; + * // "label: 界 a\n b" + * ``` + * + * @example Configuring tabs and ambiguous-width handling + * ```ts + * import { createUnicodeColumnOffset } from '@okikio/undent/unicode'; + * + * const columnOffset = createUnicodeColumnOffset({ + * tabWidth: 4, + * ambiguous: 'wide', + * }); + * ``` + */ +export function createUnicodeColumnOffset( + options: UnicodeColumnOffsetOptions = {}, +): ColumnOffsetFunction { + const resolved = resolveUnicodeColumnOffsetOptions(options); + + return function unicodeColumnOffsetWithOptions(text: string): number { + return unicodeColumnOffset(text, resolved); + }; +} + +/** + * Measure the visual insertion column after the final newline in `text`. + * + * This is the Unicode-aware companion to `columnOffset()` from the root module. + * It measures the last line in terminal-style display columns instead of UTF-16 + * code units. + * + * @example Measuring the last line of a string with wide characters + * ```ts + * import { unicodeColumnOffset } from '@okikio/undent/unicode'; + * + * unicodeColumnOffset('x\n界 '); // 3 + * ``` + */ +export function unicodeColumnOffset( + text: string, + options: UnicodeColumnOffsetOptions = {}, +): number { + return visualColumnWidth( + sliceAfterLastNewline(text), + resolveUnicodeColumnOffsetOptions(options), + ); +} + +/** + * Measure the terminal-style visual width of a single line. + * + * The function walks grapheme clusters, not raw UTF-16 code units. That keeps + * combining sequences and emoji clusters together before width rules are + * applied. + * + * This is still a best-effort estimate. Browser layout and proportional fonts + * can render the same text with different visual widths. + * + * @example Measuring combining marks as one visible cell + * ```ts + * import { visualColumnWidth } from '@okikio/undent/unicode'; + * + * visualColumnWidth('e\u0301 '); // 2 + * ``` + */ +export function visualColumnWidth( + text: string, + options: UnicodeColumnOffsetOptions = {}, +): number { + const resolved = resolveUnicodeColumnOffsetOptions(options); + let column = 0; + + for (const grapheme of graphemes(text)) { + const customWidth = resolved.widthOf?.(grapheme, { column }); + if (customWidth !== undefined) { + column += validateWidth(customWidth, 'widthOf(...)'); + continue; + } + + column += defaultGraphemeWidth(grapheme, column, resolved); + } + + return column; +} + +/** + * Normalize caller options once so the measurement loop can stay simple. + * + * Validation happens here instead of in the hot path so repeated calls from a + * preconfigured `createUnicodeColumnOffset()` instance only pay the check once. + */ +export function resolveUnicodeColumnOffsetOptions( + options: UnicodeColumnOffsetOptions, +): ResolvedUnicodeColumnOffsetOptions { + if (options.tabWidth !== undefined && options.tabWidth !== false) { + validateTabWidth(options.tabWidth); + } + + return Object.assign({}, DEFAULT_UNICODE_COLUMN_OFFSET_OPTIONS, options); +} + +/** + * Return the last logical line of a string. + * + * Alignment cares only about the insertion column after the final newline, so + * this trims away earlier lines before visual-width measurement begins. + */ +export function sliceAfterLastNewline(text: string): string { + const lastLF = text.lastIndexOf('\n'); + const lastCR = text.lastIndexOf('\r'); + const lastNL = lastLF > lastCR ? lastLF : lastCR; + + return lastNL === -1 ? text : text.slice(lastNL + 1); +} + +/** + * Measure one grapheme cluster with the built-in width rules. + * + * The order matters because some rules intentionally short-circuit others: + * + * ```text + * grapheme + * -> tab? => tab-stop width + * -> control only? => 0 + * -> emoji pictograph? => 2 + * -> regional indicator? => 2 + * -> inspect code points + * -> skip zero-width modifiers and joiners + * -> wide/fullwidth => 2 + * -> ambiguous+wide => 2 + * -> otherwise mark a visible base code point + * -> visible base found? => 1 + * -> otherwise => 0 + * ``` + * + * This keeps combining marks, variation selectors, and ZWJ glue from adding + * width on top of the base grapheme they modify. + */ +export function defaultGraphemeWidth( + grapheme: string, + column: number, + options: ResolvedUnicodeColumnOffsetOptions, +): number { + if (grapheme.length === 0) return 0; + + if (grapheme === '\t') { + if (options.tabWidth === false) return 1; + + const remainder = column % options.tabWidth; + return remainder === 0 ? options.tabWidth : options.tabWidth - remainder; + } + + if (isControlOnly(grapheme)) return 0; + if (EXTENDED_PICTOGRAPHIC_RE.test(grapheme)) return 2; + + let sawBaseCodePoint = false; + + for (const char of grapheme) { + const codePoint = char.codePointAt(0); + if (codePoint === undefined) continue; + if (isControlCodePoint(codePoint) || isZeroWidthCodePoint(codePoint)) { + continue; + } + + sawBaseCodePoint = true; + + if (isRegionalIndicatorCodePoint(codePoint)) { + return 2; + } + + if (isWideCodePoint(codePoint)) { + return 2; + } + + if ( + options.ambiguous === 'wide' && + isAmbiguousWidthCodePoint(codePoint) + ) { + return 2; + } + } + + return sawBaseCodePoint ? 1 : 0; +} + +/** + * Iterate text as grapheme clusters when the runtime supports it. + * + * `Intl.Segmenter` is the strongest available platform primitive because it + * follows Unicode grapheme-boundary rules. The fallback uses `for...of`, which + * still iterates code points instead of UTF-16 code units, but it cannot keep + * multi-code-point emoji sequences together. + */ +export function* graphemes(text: string): Iterable { + const segmenter = getGraphemeSegmenter(); + + if (segmenter !== null) { + for (const segment of segmenter.segment(text)) { + yield segment.segment; + } + + return; + } + + yield* text; +} + +/** + * Lazily create and memoize the grapheme segmenter. + * + * Importing the module should stay side-effect free and cheap. The segmenter is + * therefore constructed on first use instead of at module initialization time. + */ +export function getGraphemeSegmenter(): GraphemeSegmenter | null { + if (graphemeSegmenter !== undefined) { + return graphemeSegmenter; + } + + graphemeSegmenter = typeof intlWithSegmenter.Segmenter === 'function' + ? new intlWithSegmenter.Segmenter(undefined, { granularity: 'grapheme' }) + : null; + + return graphemeSegmenter; +} + +/** + * Validate widths returned by `widthOf(...)`. + * + * Custom width hooks are part of the public escape hatch, so failures should be + * loud and specific instead of silently coercing odd values. + */ +export function validateWidth(width: number, source: string): number { + if (!Number.isInteger(width) || width < 0) { + throw new TypeError( + `undent/unicode: ${source} must return a non-negative integer`, + ); + } + + return width; +} + +/** + * Validate tab-stop configuration once during option resolution. + */ +export function validateTabWidth(tabWidth: number): void { + if (!Number.isInteger(tabWidth) || tabWidth < 1) { + throw new TypeError( + 'undent/unicode: "tabWidth" must be a positive integer or false', + ); + } +} + +/** + * Return true when every code point in the grapheme is a control character. + * + * We only return zero width for a whole grapheme when it contains no visible + * base character at all. A visible character followed by controls is handled by + * the wider default grapheme logic instead of being dropped here. + */ +export function isControlOnly(grapheme: string): boolean { + for (const char of grapheme) { + const codePoint = char.codePointAt(0); + if (codePoint === undefined) continue; + if (!isControlCodePoint(codePoint)) return false; + } + + return true; +} + +/** + * Return true when the code point is a regional indicator symbol. + * + * These code points form Unicode flag graphemes in pairs. Terminals commonly + * render both the pair and the standalone symbol as emoji-width cells, so the + * built-in width rules treat them as occupying two columns. + */ +function isRegionalIndicatorCodePoint(codePoint: number): boolean { + return codePoint >= 0x1f1e6 && codePoint <= 0x1f1ff; +} + +/** + * Control characters never consume terminal columns by themselves. + */ +export function isControlCodePoint(codePoint: number): boolean { + return codePoint === 0x00 || + (codePoint >= 0x01 && codePoint <= 0x1f) || + (codePoint >= 0x7f && codePoint <= 0x9f); +} + +/** + * Zero-width code points modify neighboring characters instead of occupying + * their own cell. + * + * This includes combining marks, variation selectors, and the zero-width + * joiner used to glue emoji into a single presented grapheme. + */ +export function isZeroWidthCodePoint(codePoint: number): boolean { + return (codePoint >= 0x0300 && codePoint <= 0x036f) || + (codePoint >= 0x1ab0 && codePoint <= 0x1aff) || + (codePoint >= 0x1dc0 && codePoint <= 0x1dff) || + (codePoint >= 0x20d0 && codePoint <= 0x20ff) || + codePoint === 0x200d || + (codePoint >= 0xfe00 && codePoint <= 0xfe0f) || + (codePoint >= 0xfe20 && codePoint <= 0xfe2f) || + (codePoint >= 0xe0100 && codePoint <= 0xe01ef); +} + +/** + * Check East Asian wide/fullwidth ranges with a binary search. + * + * A flat range table is much easier to audit and extend than a long boolean + * expression, and binary search keeps the lookup cost predictable. + */ +export function isWideCodePoint(codePoint: number): boolean { + return isCodePointInRanges(codePoint, EAST_ASIAN_WIDE_RANGES); +} + +/** + * Check East Asian ambiguous-width ranges with the same binary-search helper. + */ +export function isAmbiguousWidthCodePoint(codePoint: number): boolean { + return isCodePointInRanges(codePoint, EAST_ASIAN_AMBIGUOUS_RANGES); +} + +/** + * Return true when `codePoint` falls inside any `[start, end]` range. + * + * The tables are sorted by start position, so binary search finds matches in + * `O(log n)` time without allocating or scanning the full list on every code + * point. + */ +export function isCodePointInRanges( + codePoint: number, + ranges: ReadonlyArray, +): boolean { + let low = 0; + let high = ranges.length - 1; + + while (low <= high) { + const middle = (low + high) >> 1; + const [start, end] = ranges[middle]!; + + if (codePoint < start) { + high = middle - 1; + continue; + } + + if (codePoint > end) { + low = middle + 1; + continue; + } + + return true; + } + + return false; +} \ No newline at end of file diff --git a/unicode_test.ts b/unicode_test.ts new file mode 100644 index 0000000..30bdb17 --- /dev/null +++ b/unicode_test.ts @@ -0,0 +1,171 @@ +import { describe, it } from 'jsr:@std/testing/bdd'; +import { expect } from 'jsr:@std/expect'; + +import { align, undent } from './mod.ts'; +import { + createUnicodeColumnOffset, + defaultGraphemeWidth, + graphemes, + isAmbiguousWidthCodePoint, + isCodePointInRanges, + isControlCodePoint, + isWideCodePoint, + resolveUnicodeColumnOffsetOptions, + sliceAfterLastNewline, + unicodeColumnOffset, + visualColumnWidth, +} from './unicode.ts'; + +describe('unicode alignment helpers', () => { + it('measures the last line with wide characters', () => { + expect(unicodeColumnOffset('prefix\n界 ')).toBe(3); + }); + + it('exports a resolver that merges default unicode options', () => { + const result = resolveUnicodeColumnOffsetOptions({}); + + expect(result).toEqual({ + tabWidth: false, + ambiguous: 'narrow', + }); + }); + + it('exports a helper that slices after the last newline', () => { + expect(sliceAfterLastNewline('alpha\r\nbeta')).toBe('beta'); + }); + + it('measures the last line after CRLF as well as LF', () => { + expect(unicodeColumnOffset('prefix\r\n界 ')).toBe(3); + }); + + it('treats combining marks as part of the same visible grapheme', () => { + expect(visualColumnWidth('e\u0301 ')).toBe(2); + }); + + it('exports default grapheme width rules for direct use', () => { + const options = resolveUnicodeColumnOffsetOptions({ ambiguous: 'wide' }); + + expect(defaultGraphemeWidth('\t', 3, options)).toBe(1); + expect(defaultGraphemeWidth('Ω', 0, options)).toBe(2); + }); + + it('treats emoji ZWJ sequences as a single wide grapheme', () => { + expect(visualColumnWidth('👨‍👩‍👧‍👦')).toBe(2); + }); + + it('exports grapheme iteration for callers that need the same segmentation', () => { + expect(Array.from(graphemes('e\u0301😀'))).toEqual(['é', '😀']); + }); + + it('treats regional-indicator flag sequences as wide', () => { + expect(visualColumnWidth('🇯🇵')).toBe(2); + }); + + it('treats fullwidth forms as wide', () => { + expect(visualColumnWidth('Hello')).toBe(10); + }); + + it('treats Unicode wide trigrams as wide after the upstream table audit', () => { + expect(visualColumnWidth('☰')).toBe(2); + }); + + it('exports wide and ambiguous range predicates', () => { + expect(isWideCodePoint('界'.codePointAt(0)!)).toBe(true); + expect(isAmbiguousWidthCodePoint('Ω'.codePointAt(0)!)).toBe(true); + }); + + it('exports low-level code-point classification helpers', () => { + expect(isControlCodePoint(0x09)).toBe(true); + expect(isCodePointInRanges(0x03a9, [[0x0391, 0x03a9]])).toBe(true); + }); + + it('treats ambiguous-width characters as narrow by default', () => { + expect(visualColumnWidth('Ω')).toBe(1); + }); + + it('treats ambiguous-width characters as wide when configured', () => { + expect(visualColumnWidth('Ω', { ambiguous: 'wide' })).toBe(2); + }); + + it('lets widthOf override specific grapheme widths', () => { + expect( + visualColumnWidth('a·b', { + widthOf(grapheme) { + return grapheme === '·' ? 3 : undefined; + }, + }), + ).toBe(5); + }); + + it('throws when widthOf returns a negative width', () => { + expect(() => + visualColumnWidth('a', { + widthOf() { + return -1; + }, + }) + ).toThrow('widthOf(...) must return a non-negative integer'); + }); + + it('throws when tabWidth is not a positive integer', () => { + expect(() => createUnicodeColumnOffset({ tabWidth: 0 })).toThrow( + '"tabWidth" must be a positive integer or false', + ); + }); + + it('supports tab stops when configured', () => { + const columnOffset = createUnicodeColumnOffset({ tabWidth: 4 }); + + expect(columnOffset('ab\t')).toBe(4); + expect(columnOffset('abc\t')).toBe(4); + expect(columnOffset('abcd\t')).toBe(8); + }); + + it('can treat tabs as one column when tabWidth is false', () => { + const columnOffset = createUnicodeColumnOffset({ tabWidth: false }); + + expect(columnOffset('ab\t')).toBe(3); + }); + + it('integrates with undent through the root columnOffset option', () => { + const terminalUndent = undent.with({ + alignValues: true, + columnOffset: createUnicodeColumnOffset(), + }); + + const result = terminalUndent` + label: 界 ${'a\nb'} + `; + + expect(result).toBe('label: 界 a\n b'); + }); + + it('affects wrapped align() values as well as alignValues', () => { + const terminalUndent = undent.with({ + columnOffset: createUnicodeColumnOffset(), + }); + + const result = terminalUndent` + label: 界 ${align('alpha\nbeta')} + `; + + expect(result).toBe('label: 界 alpha\n beta'); + }); + + it('supports custom widthOf in end-to-end alignment', () => { + const terminalUndent = undent.with({ + alignValues: true, + columnOffset: createUnicodeColumnOffset({ + widthOf(grapheme) { + return grapheme === '·' ? 3 : undefined; + }, + }), + }); + + const result = terminalUndent` + key· ${'x\ny'} + `; + + expect(result).toBe('key· x\n y'); + }); +}); \ No newline at end of file