From 096fb617bc8d7811025b17e8ffde66d9cb34b635 Mon Sep 17 00:00:00 2001 From: Joshua Reusch Date: Mon, 30 Sep 2024 21:00:28 +0200 Subject: [PATCH] builder pattern, remove codepoint/grapheme_cluster --- src/string_width.gleam | 169 ++++++++++---------- test/benchmark.gleam | 7 +- test/despair_test.gleam | 28 ++-- test/string_width_test.gleam | 293 ++++++++++++++++++----------------- 4 files changed, 262 insertions(+), 235 deletions(-) diff --git a/src/string_width.gleam b/src/string_width.gleam index c6ecd9c..7bc6ed8 100644 --- a/src/string_width.gleam +++ b/src/string_width.gleam @@ -16,47 +16,16 @@ import string_width/internal/tables // https://gitlab.freedesktop.org/terminal-wg/specifications/-/issues/36 // v3: -// - we can optimise the top-level non-foldy things by stripping out things of known length beforehand -// see the fstw regex magic -// - options builder, _with api // - handle tabs // - implement line wrapping? truncation? pad/alignment? // how much would this break optimisations? // I should provide ways to implement layout, not do layout myself -/// Options to change the default behaviour of the functions in this library. -/// If you are unsure what to do here, passing an empty array is almost always the right call! -pub type Option { - /// Most terminal emulators do not handle grapheme clusters well and will - /// instead show their decomposition. To make sure a given string always fits - /// even on those terminals, the functions in this package will copy this - /// behaviour as well. - /// - /// If you pass this option, the returned numbers will more accurately represent - /// the width of a string on a website or in an editor. - /// - /// **NOTE:** Grapheme support in terminals is highly experimental and subject - /// to ongoing discussion! When working on CLIs or TUIs, you do not want to use - /// this option most of the time. - /// - /// See also [Grapheme Clusters and Terminal Emulators](https://mitchellh.com/writing/grapheme-clusters-in-terminals) - /// for a better explanation on how terminals behave. - HandleGraphemeClusters - /// Do not ignore ansi escape sequences, and count them as regular characters. - /// - /// You can pass this option as an optimisation if you are sure that your - /// string doesn't contain any ansi escape codes. - CountAnsiEscapeCodes - /// Some characters are marked by Unicode as "ambiguous", meaning they may - /// occupy 1 or 2 cells, depending on the context, current language, selected - /// font, surrounding text, and more. - /// - /// Unicode recommends treating these characters as narrow by default, - /// but you can change this behaviour using this option. - AmbiguousAsWide -} +// -- OPTIONS ------------------------------------------------------------------ -type Options { +/// Options to change the default behaviour of the functions in this library. +/// If you are unsure what to do here, the defaults should work great! +pub opaque type Options { Options( count_ansi_escape_codes: Bool, ambiguous_width: Int, @@ -64,16 +33,55 @@ type Options { ) } -fn parse_options(options: List(Option)) -> Options { - let defaults = Options(False, 1, False) - use record, option <- list.fold(options, defaults) - case option { - CountAnsiEscapeCodes -> Options(..record, count_ansi_escape_codes: True) - AmbiguousAsWide -> Options(..record, ambiguous_width: 2) - HandleGraphemeClusters -> Options(..record, handle_grapheme_clusters: True) - } +const default_options = Options( + count_ansi_escape_codes: False, + ambiguous_width: 1, + handle_grapheme_clusters: False, +) + +/// Start building up new options. +pub fn new() -> Options { + default_options +} + +/// Most terminal emulators do not handle grapheme clusters well and will +/// instead show their decomposition. To make sure a given string always fits +/// even on those terminals, the functions in this package will copy this +/// behaviour as well. +/// +/// If you enable this option, the returned numbers will more accurately represent +/// the width of a string on a website or in an editor. +/// +/// **NOTE:** Grapheme support in terminals is highly experimental and subject +/// to ongoing discussion! When working on CLIs or TUIs, you do not want to use +/// this option most of the time. +/// +/// See also [Grapheme Clusters and Terminal Emulators](https://mitchellh.com/writing/grapheme-clusters-in-terminals) +/// for a better explanation on how terminals behave. +pub fn handle_grapheme_clusters(options: Options) -> Options { + Options(..options, handle_grapheme_clusters: True) } +/// Some characters are marked by Unicode as "ambiguous", meaning they may +/// occupy 1 or 2 cells, depending on the context, current language, selected +/// font, surrounding text, and more. +/// +/// Unicode recommends treating these characters as narrow by default, +/// but you can change this behaviour using this option. +pub fn ambiguous_as_wide(options: Options) -> Options { + Options(..options, ambiguous_width: 2) +} + +/// Do not ignore ansi escape sequences, and count them as regular characters. +/// +/// You can enable this option as an optimisation if you are sure that your +/// string doesn't contain any ansi escape codes. +pub fn count_ansi_escape_codes(options: Options) -> Options { + Options(..options, count_ansi_escape_codes: True) +} + +// -- MEASURING STRINGS -------------------------------------------------------- + /// Get the number of columns required to print a line in a terminal. /// /// Line breaks are ignored. If the given string contains newlines, @@ -82,26 +90,44 @@ fn parse_options(options: List(Option)) -> Options { /// ### Examples /// /// ```gleam -/// line("äöüè", []) +/// line("äöüè") /// // --> 4 /// -/// line("안녕하세요", []) +/// line("안녕하세요") /// // --> 10 /// -/// line("👩‍👩‍👦‍👦", []) +/// line("👩‍👩‍👦‍👦") /// // --> 8 /// /// line("👩‍👩‍👦‍👦", [HandleGraphemeClusters]) /// // --> 2 /// -/// line("\u{1B}[31mhello\u{1B}[39m", []) +/// line("\u{1B}[31mhello\u{1B}[39m") /// // --> 5 /// ``` -pub fn line(str: String, options: List(Option)) -> Int { - let options = parse_options(options) +pub fn line(str: String) -> Int { + do_line(default_options, str) +} + +/// Like `line`, but use custom options. +/// +/// ### Example +/// +/// ```gleam +/// let options = +/// new() +/// |> handle_grapheme_clusters +/// +/// line_with("👩‍👩‍👦‍👦", options) +/// // --> 2 +/// ``` +pub fn line_with(str: String, options: Options) -> Int { do_line(options, str) } +// this is ugly, and done as a perfomance optimization. Hopefully nothing breaks +// too soon. + @target(erlang) fn do_line(options: Options, str: String) -> Int { let ansi_ranges = case options.count_ansi_escape_codes { @@ -138,15 +164,18 @@ fn do_line(options: Options, str: String) -> Int { /// ### Examples /// /// ```gleam -/// dimensions("안녕하세요", []) +/// dimensions("안녕하세요") /// // --> #(1, 10) /// -/// dimensions("hello,\n안녕하세요", []) +/// dimensions("hello,\n안녕하세요") /// // --> #(2, 10) /// ``` -pub fn dimensions(str: String, options: List(Option)) -> #(Int, Int) { - let options = parse_options(options) +pub fn dimensions(str: String) -> #(Int, Int) { + dimensions_with(str, default_options) +} +/// Like `dimensions`, but use custom options. +pub fn dimensions_with(str: String, options: Options) -> #(Int, Int) { let lines = string.split(str, on: "\n") use #(rows, cols), line <- list.fold(lines, #(0, 0)) @@ -154,6 +183,8 @@ pub fn dimensions(str: String, options: List(Option)) -> #(Int, Int) { #(rows + 1, int.max(cols, line_width)) } +// -- FOLD --------------------------------------------------------------------- + /// Iterate over the measured components of a string. Components are either /// graphemes or codepoints, depending on the `HandleGraphemeClusters` option, /// or other undivisible sequences, like ANSI escape codes. @@ -165,17 +196,17 @@ pub fn dimensions(str: String, options: List(Option)) -> #(Int, Int) { /// /// Concatenating all components is guaranteed to produce the original string. /// -/// ## Examples +/// ### Examples /// /// ```gleam /// // A slower string_width.line implementation that doesn't handle tabs -/// fold("hello", [], from: 0, with: fn(so_far, _chr, width) { width + so_far }) +/// fold("hello", new(), from: 0, with: fn(so_far, _chr, width) { width + so_far }) /// // --> 5 /// ``` /// /// ```gleam /// // truncate a string after 50 characters, but keep all ansi sequences. -/// use #(total, acc), chr, width <- fold(input, [], from: #(0, "")) +/// use #(total, acc), chr, width <- fold(input, new(), from: #(0, "")) /// case total >= 50 { /// True -> case chr { /// "\u{1b}" <> _ -> #(total, acc <> chr) @@ -187,11 +218,10 @@ pub fn dimensions(str: String, options: List(Option)) -> #(Int, Int) { /// ``` pub fn fold( over string: String, - using options: List(Option), + using options: Options, from state: state, with fun: fn(state, String, Int) -> state, ) -> state { - let options = parse_options(options) let ansi_ranges = case options.count_ansi_escape_codes { True -> [] False -> ansi.match(string) @@ -258,16 +288,6 @@ fn do_fold_graphemes( fun(state, grapheme, width) } -/// Estimate the required width for a single grapheme cluster. -/// -/// **Note:** Grapheme cluster handling in terminals is highly experimental and -/// subject to ongoing discussions, and very inconsistent across different -/// terminal emulators. The exact width is context-dependent, so this value may -/// may exactly match what you might expect! -pub fn grapheme_cluster(grapheme: String, options: List(Option)) -> Int { - do_grapheme_cluster(parse_options(options), grapheme) -} - fn do_grapheme_cluster(options: Options, string: String) -> Int { // The Unicode Core spec (https://github.com/contour-terminal/terminal-unicode-core) // proposing Mode 2027 says the following: @@ -306,19 +326,6 @@ fn do_grapheme_cluster(options: Options, string: String) -> Int { } } -/// Estimate the required width of a single unicode code point. -/// -/// If you encounter a mismatch that is consistent across multiple terminal -/// emulators, please open an issue or ping me on Discord! -pub fn codepoint(chr: UtfCodepoint, options: List(Option)) -> Int { - wcwidth(parse_options(options), string.utf_codepoint_to_int(chr)) -} - -/// -pub fn int_codepoint(cp: Int, options: List(Option)) -> Int { - wcwidth(parse_options(options), cp) -} - fn wcwidth(options: Options, cp: Int) -> Int { // see: https://git.musl-libc.org/cgit/musl/tree/src/ctype/wcwidth.c case cp { diff --git a/test/benchmark.gleam b/test/benchmark.gleam index f8297ca..158e9a7 100644 --- a/test/benchmark.gleam +++ b/test/benchmark.gleam @@ -2,18 +2,17 @@ import gleam/io import string_width pub fn main() { - io.debug(loop([], 1_000_000, 0)) + io.debug(loop(1_000_000, 0)) } -fn loop(options, i, sum) { +fn loop(i, sum) { case i > 0 { True -> { let width = string_width.line( "\u{1b}[1;95m· \u{1b}[0m\u{1b}[K\u{1b}[33mdeno\u{1b}[0m\u{1b}[K, \u{1b}[33mtypescript\u{1b}[0m\u{1b}[K, \u{1b}[33mts\u{1b}[0m\u{1b}[K", - options, ) - loop(options, i - 1, width + sum) + loop(i - 1, width + sum) } False -> sum } diff --git a/test/despair_test.gleam b/test/despair_test.gleam index b6e748d..be94b97 100644 --- a/test/despair_test.gleam +++ b/test/despair_test.gleam @@ -1,17 +1,25 @@ import gleeunit/should -import string_width.{HandleGraphemeClusters} +import string_width + +fn sw(str: String) { + let options = + string_width.new() + |> string_width.handle_grapheme_clusters + + string_width.line_with(str, options) +} @target(erlang) pub fn despair1_test() { // 2 grapheme clusters in Unicode 15 - string_width.line("র্য", [HandleGraphemeClusters]) |> should.equal(2) + sw("র্য") |> should.equal(2) } @target(javascript) pub fn despair1_test() { // 1 grapheme cluster in unicode 15.1 // We return 2 because of our custom extension. - string_width.line("র্য", [HandleGraphemeClusters]) |> should.equal(2) + sw("র্য") |> should.equal(2) } @target(erlang) @@ -19,7 +27,7 @@ pub fn despair2_test() { // 2 grapheme clusters in Unicde 15. // NOTE: this string is different from the one above in that it includes // an additional ZWJ between the letters! - string_width.line("র‍্য", [HandleGraphemeClusters]) |> should.equal(2) + sw("র‍্য") |> should.equal(2) } @target(javascript) @@ -31,14 +39,14 @@ pub fn despair2_test() { // VSCode treats it as 2 separate characters though, and everyone renders it wide. // // We return 2 because of our custom extension. - string_width.line("র‍্য", [HandleGraphemeClusters]) |> should.equal(2) + sw("র‍্য") |> should.equal(2) } @target(erlang) pub fn despair3_test() { // 5 grapheme clusters un Unicode 15 // in particular, "च्छे" is split into 2 clusters - string_width.line("अनुच्छेद", [HandleGraphemeClusters]) + sw("अनुच्छेद") |> should.equal(5) } @@ -50,21 +58,21 @@ pub fn despair3_test() { // in the database that I know of that would tell it to. // // We return 5 instead of 4 because of our custom extension. - string_width.line("अनुच्छेद", [HandleGraphemeClusters]) + sw("अनुच्छेद") |> should.equal(5) } @target(erlang) pub fn despair4_test() { // 2 grapheme clusters in Unicode 15 - string_width.line("क्षि", [HandleGraphemeClusters]) |> should.equal(2) + sw("क्षि") |> should.equal(2) } @target(javascript) pub fn despair4_test() { // 1 grapheme cluster in Unicode 15.1 // We return 2 because of our custom extension. - string_width.line("क्षि", [HandleGraphemeClusters]) |> should.equal(2) + sw("क्षि") |> should.equal(2) } pub fn despair5_test() { @@ -72,5 +80,5 @@ pub fn despair5_test() { // still render it wide. // Our "2 codepoints with width force it wide rule" makes it wide, but other // implementations disagree. - string_width.line("バ", [HandleGraphemeClusters]) |> should.equal(2) + sw("バ") |> should.equal(2) } diff --git a/test/string_width_test.gleam b/test/string_width_test.gleam index 994bd5c..bbccbfb 100644 --- a/test/string_width_test.gleam +++ b/test/string_width_test.gleam @@ -1,37 +1,49 @@ import gleeunit import gleeunit/should -import string_width.{ - AmbiguousAsWide, CountAnsiEscapeCodes, HandleGraphemeClusters, dimensions, - line as sw, -} +import string_width.{dimensions, line as sw} pub fn main() { gleeunit.main() } +fn swg(str: String) { + let options = string_width.new() |> string_width.handle_grapheme_clusters + string_width.line_with(str, options) +} + +fn sww(str: String) { + let options = string_width.new() |> string_width.ambiguous_as_wide + string_width.line_with(str, options) +} + +fn swc(str: String) { + let options = string_width.new() |> string_width.count_ansi_escape_codes + string_width.line_with(str, options) +} + pub fn main_test() { - sw("abcde", []) |> should.equal(5) - sw("古池や", []) |> should.equal(6) - sw("あいうabc", []) |> should.equal(9) - sw("±", []) |> should.equal(1) - sw("ノード.js", []) |> should.equal(9) - sw("你好", []) |> should.equal(4) - sw("안녕하세요", []) |> should.equal(10) + sw("abcde") |> should.equal(5) + sw("古池や") |> should.equal(6) + sw("あいうabc") |> should.equal(9) + sw("±") |> should.equal(1) + sw("ノード.js") |> should.equal(9) + sw("你好") |> should.equal(4) + sw("안녕하세요") |> should.equal(10) } pub fn docs_test() { - sw("äöüè", []) |> should.equal(4) - sw("안녕하세요", []) |> should.equal(10) - sw("123|", []) - sw("🏳️‍⚧️", []) |> should.equal(2) - sw("🏳️‍⚧️", [HandleGraphemeClusters]) |> should.equal(2) - sw("👩‍👩‍👦‍👦", []) |> should.equal(8) - sw("👩‍👩‍👦‍👦", [HandleGraphemeClusters]) |> should.equal(2) - sw("\u{1B}[31mhello\u{1B}[39m", []) |> should.equal(5) - - dimensions("안녕하세요", []) |> should.equal(#(1, 10)) - - dimensions("hello,\n안녕하세요", []) |> should.equal(#(2, 10)) + sw("äöüè") |> should.equal(4) + sw("안녕하세요") |> should.equal(10) + sw("123|") + sw("🏳️‍⚧️") |> should.equal(2) + swg("🏳️‍⚧️") |> should.equal(2) + sw("👩‍👩‍👦‍👦") |> should.equal(8) + swg("👩‍👩‍👦‍👦") |> should.equal(2) + sw("\u{1B}[31mhello\u{1B}[39m") |> should.equal(5) + + dimensions("안녕하세요") |> should.equal(#(1, 10)) + + dimensions("hello,\n안녕하세요") |> should.equal(#(2, 10)) } pub fn pedantic_test() { @@ -43,228 +55,226 @@ pub fn pedantic_test() { // let sw = fn(str, opts) { sw(io.debug(str), opts) } // 🏳️‍⚧️ 2 2 | 2 2 4 3 2 2 - sw("🏳️‍⚧️", []) |> should.equal(2) - sw("🏳️‍⚧️", [HandleGraphemeClusters]) |> should.equal(2) + sw("🏳️‍⚧️") |> should.equal(2) + swg("🏳️‍⚧️") |> should.equal(2) // 👩‍👩‍👧‍👦 2 2 | 2 8 8 8 2 2 - sw("👩‍👩‍👦‍👦", []) |> should.equal(8) - sw("👩‍👩‍👦‍👦", [HandleGraphemeClusters]) |> should.equal(2) + sw("👩‍👩‍👦‍👦") |> should.equal(8) + swg("👩‍👩‍👦‍👦") |> should.equal(2) // র্য 1 2 | 1 2 2 2 2 1 - sw("র্য", []) |> should.equal(2) + sw("র্য") |> should.equal(2) // despair1 // র‍্য 1 3 | 2 2 2 2 2 2 - sw("র‍্য", []) |> should.equal(2) + sw("র‍্য") |> should.equal(2) // despair2 // अनुच्छेद 4 5 | 5 5 5 5 5 5 - sw("अनुच्छेद", []) |> should.equal(5) + sw("अनुच्छेद") |> should.equal(5) // despair3 // ශ්ර 2 2 | 2.5 2 2 2 2 2 - sw("ශ්ර", []) |> should.equal(2) - sw("ශ්ර", [HandleGraphemeClusters]) |> should.equal(2) + sw("ශ්ර") |> should.equal(2) + swg("ශ්ර") |> should.equal(2) // ශ්‍ර 2 3 | 1.5 2 2 2 2 1 - sw("ශ්‍ර", []) |> should.equal(2) - sw("ශ්‍ර", [HandleGraphemeClusters]) |> should.equal(2) + sw("ශ්‍ර") |> should.equal(2) + swg("ශ්‍ර") |> should.equal(2) // क्षि 1 2 | 2 3 2 3 2 1 - sw("क्षि", []) |> should.equal(3) + sw("क्षि") |> should.equal(3) // despair4 // ↔️ 2 2 | 1 1 2 1 1 1 - sw("↔️", []) |> should.equal(1) - sw("↔️", [HandleGraphemeClusters]) |> should.equal(2) + sw("↔️") |> should.equal(1) + swg("↔️") |> should.equal(2) // VS 16 forces double-width // 葛󠄀 2 2 | 2 2 2 2 2 2 - sw("葛󠄀", []) |> should.equal(2) - sw("葛󠄀", [HandleGraphemeClusters]) |> should.equal(2) + sw("葛󠄀") |> should.equal(2) + swg("葛󠄀") |> should.equal(2) // Z͑ͫ̓ͪ̂ͫ̽͏̴̙̤̞͉͚̯̞̠͍A̴̵̜̰͔ͫ͗͢L̠ͨͧͩ͘G̴̻͈͍͔̹̑͗̎̅͛́Ǫ̵̹̻̝̳͂̌̌͘!͖̬̰̙̗̿̋ͥͥ̂ͣ̐́́͜͞ 6 6 | 6 6 6 6 6 6 sw( "Z͑ͫ̓ͪ̂ͫ̽͏̴̙̤̞͉͚̯̞̠͍A̴̵̜̰͔ͫ͗͢L̠ͨͧͩ͘G̴̻͈͍͔̹̑͗̎̅͛́Ǫ̵̹̻̝̳͂̌̌͘!͖̬̰̙̗̿̋ͥͥ̂ͣ̐́́͜͞", - [], ) |> should.equal(6) - sw( + swg( "Z͑ͫ̓ͪ̂ͫ̽͏̴̙̤̞͉͚̯̞̠͍A̴̵̜̰͔ͫ͗͢L̠ͨͧͩ͘G̴̻͈͍͔̹̑͗̎̅͛́Ǫ̵̹̻̝̳͂̌̌͘!͖̬̰̙̗̿̋ͥͥ̂ͣ̐́́͜͞", - [HandleGraphemeClusters], ) |> should.equal(6) // バ| :L 1 2 | 2 2 2 2 1 - sw("バ", []) |> should.equal(2) + sw("バ") |> should.equal(2) // despair5 } pub fn ambigous_test() { - sw("⛣", []) |> should.equal(1) - sw("⛣", [AmbiguousAsWide]) |> should.equal(2) - sw("★", []) |> should.equal(1) - sw("★", [AmbiguousAsWide]) |> should.equal(2) - sw("“", []) |> should.equal(1) - sw("“", [AmbiguousAsWide]) |> should.equal(2) - sw("あいう★", []) |> should.equal(7) - sw("あいう★", [AmbiguousAsWide]) |> should.equal(8) - sw("★あいう", []) |> should.equal(7) - sw("★あいう", [AmbiguousAsWide]) |> should.equal(8) + sw("⛣") |> should.equal(1) + sww("⛣") |> should.equal(2) + sw("★") |> should.equal(1) + sww("★") |> should.equal(2) + sw("“") |> should.equal(1) + sww("“") |> should.equal(2) + sw("あいう★") |> should.equal(7) + sww("あいう★") |> should.equal(8) + sw("★あいう") |> should.equal(7) + sww("★あいう") |> should.equal(8) } pub fn v5v7_test() { // these are examples in https://github.com/sindresorhus/string-width/issues/56 - sw("◼", []) |> should.equal(1) - sw("⚠", []) |> should.equal(1) - sw("✔", []) |> should.equal(1) - sw("♥", []) |> should.equal(1) - sw("♀", []) |> should.equal(1) - sw("♂", []) |> should.equal(1) - sw("\u{21a9}", []) |> should.equal(1) - sw("\u{2194}", []) |> should.equal(1) - sw("\u{2197}", []) |> should.equal(1) - sw("✔", []) |> should.equal(1) - sw("\u{2709}", []) |> should.equal(1) - sw("\u{26a0}", []) |> should.equal(1) + sw("◼") |> should.equal(1) + sw("⚠") |> should.equal(1) + sw("✔") |> should.equal(1) + sw("♥") |> should.equal(1) + sw("♀") |> should.equal(1) + sw("♂") |> should.equal(1) + sw("\u{21a9}") |> should.equal(1) + sw("\u{2194}") |> should.equal(1) + sw("\u{2197}") |> should.equal(1) + sw("✔") |> should.equal(1) + sw("\u{2709}") |> should.equal(1) + sw("\u{26a0}") |> should.equal(1) } pub fn dakuten_test() { // https://github.com/sindresorhus/string-width/issues/55 - sw("バ", []) |> should.equal(2) - sw("パ", []) |> should.equal(2) + sw("バ") |> should.equal(2) + sw("パ") |> should.equal(2) } pub fn ansi_test() { - sw("\u{1B}[31m\u{1B}[39m", []) |> should.equal(0) - sw("\u{1B}[31m\u{1B}[39m", [CountAnsiEscapeCodes]) + sw("\u{1B}[31m\u{1B}[39m") |> should.equal(0) + swc("\u{1B}[31m\u{1B}[39m") |> should.equal(8) - sw("\u{1B}]8;;https://github.com\u{0007}Click\u{1B}]8;;\u{0007}", []) + sw("\u{1B}]8;;https://github.com\u{0007}Click\u{1B}]8;;\u{0007}") |> should.equal(5) } pub fn default_emoji_presentation_test() { // ⌚ - sw("\u{231A}", []) |> should.equal(2) + sw("\u{231A}") |> should.equal(2) } pub fn default_text_presentation_emoji_test() { // ↔️ default text presentation rendered as emoji sw:2 - sw("\u{2194}", []) |> should.equal(1) + sw("\u{2194}") |> should.equal(1) // I do not know currently how to handle this properly. - // sw("\u{2194}\u{FE0F}", []) |> should.equal(2) + // sw("\u{2194}\u{FE0F}") |> should.equal(2) } pub fn emoji_modifier_base_test() { // 👩 - sw("\u{1F469}", []) |> should.equal(2) + sw("\u{1F469}") |> should.equal(2) } pub fn emoji_with_modifier_test() { - sw("\u{1F469}\u{1F3FF}", []) |> should.equal(4) + sw("\u{1F469}\u{1F3FF}") |> should.equal(4) } pub fn variation_selector_test() { - sw("\u{845B}\u{E0100}", []) |> should.equal(2) - sw("A\u{FE0F}", []) |> should.equal(1) - sw("\u{FE0F}", []) |> should.equal(0) + sw("\u{845B}\u{E0100}") |> should.equal(2) + sw("A\u{FE0F}") |> should.equal(1) + sw("\u{FE0F}") |> should.equal(0) } pub fn thai_test() { - sw("_\u{0E34}", []) |> should.equal(1) - sw("ปฏัก", []) |> should.equal(3) + sw("_\u{0E34}") |> should.equal(1) + sw("ปฏัก") |> should.equal(3) } pub fn control_characters_test() { - sw("\u{0}", []) |> should.equal(0) - sw("\u{1f}", []) |> should.equal(0) - sw("\u{7f}", []) |> should.equal(0) - sw("\u{86}", []) |> should.equal(0) - sw("\u{9f}", []) |> should.equal(0) - sw("\u{1b}", []) |> should.equal(0) + sw("\u{0}") |> should.equal(0) + sw("\u{1f}") |> should.equal(0) + sw("\u{7f}") |> should.equal(0) + sw("\u{86}") |> should.equal(0) + sw("\u{9f}") |> should.equal(0) + sw("\u{1b}") |> should.equal(0) } pub fn combining_characters_test() { - sw("x\u{0300}", []) |> should.equal(1) - sw("\u{0300}\u{0301}", []) |> should.equal(0) - sw("e\u{0301}e", []) |> should.equal(2) - sw("x\u{036f}", []) |> should.equal(1) - sw("\u{036f}\u{036f}", []) |> should.equal(0) + sw("x\u{0300}") |> should.equal(1) + sw("\u{0300}\u{0301}") |> should.equal(0) + sw("e\u{0301}e") |> should.equal(2) + sw("x\u{036f}") |> should.equal(1) + sw("\u{036f}\u{036f}") |> should.equal(0) } pub fn zwj_test() { - sw("👶", []) |> should.equal(2) - sw("👶🏽", []) |> should.equal(4) - sw("👩‍👩‍👦‍👦", []) |> should.equal(8) - sw("👨‍❤️‍💋‍👨", []) |> should.equal(7) + sw("👶") |> should.equal(2) + sw("👶🏽") |> should.equal(4) + sw("👩‍👩‍👦‍👦") |> should.equal(8) + sw("👨‍❤️‍💋‍👨") |> should.equal(7) } pub fn zwj_cluster_test() { - sw("👶", [HandleGraphemeClusters]) |> should.equal(2) - sw("👶🏽", [HandleGraphemeClusters]) |> should.equal(2) - sw("👩‍👩‍👦‍👦", [HandleGraphemeClusters]) + swg("👶") |> should.equal(2) + swg("👶🏽") |> should.equal(2) + swg("👩‍👩‍👦‍👦") |> should.equal(2) - sw("👨‍❤️‍💋‍👨", [HandleGraphemeClusters]) + swg("👨‍❤️‍💋‍👨") |> should.equal(2) - sw("\u{1F469}\u{1F3FF}", [HandleGraphemeClusters]) + swg("\u{1F469}\u{1F3FF}") |> should.equal(2) - sw("👩‍🎓", [HandleGraphemeClusters]) |> should.equal(2) + swg("👩‍🎓") |> should.equal(2) } pub fn zero_width_test() { - sw("\u{200b}", []) |> should.equal(0) - sw("x\u{200b}x", []) |> should.equal(2) - sw("\u{200c}", []) |> should.equal(0) - sw("x\u{200c}x", []) |> should.equal(2) - sw("\u{200d}", []) |> should.equal(0) - sw("x\u{200d}x", []) |> should.equal(2) - sw("\u{feff}", []) |> should.equal(0) - sw("x\u{feff}x", []) |> should.equal(2) + sw("\u{200b}") |> should.equal(0) + sw("x\u{200b}x") |> should.equal(2) + sw("\u{200c}") |> should.equal(0) + sw("x\u{200c}x") |> should.equal(2) + sw("\u{200d}") |> should.equal(0) + sw("x\u{200d}x") |> should.equal(2) + sw("\u{feff}") |> should.equal(0) + sw("x\u{feff}x") |> should.equal(2) } pub fn variation_selectors_test() { // regional indicator symbol A with variation selector - sw("\u{1f1e6}\u{fe0f}", []) |> should.equal(1) - sw("A\u{fe0f}", []) |> should.equal(1) - sw("\u{fe0f}", []) |> should.equal(0) + sw("\u{1f1e6}\u{fe0f}") |> should.equal(1) + sw("A\u{fe0f}") |> should.equal(1) + sw("\u{fe0f}") |> should.equal(0) } pub fn edge_cases_test() { - sw("", []) |> should.equal(0) - sw("\u{200b}\u{200b}", []) |> should.equal(0) - sw("x\u{200b}x\u{200b}", []) |> should.equal(2) - sw("x\u{0300}x\u{0300}", []) |> should.equal(2) + sw("") |> should.equal(0) + sw("\u{200b}\u{200b}") |> should.equal(0) + sw("x\u{200b}x\u{200b}") |> should.equal(2) + sw("x\u{0300}x\u{0300}") |> should.equal(2) // 😀 with variation selector - sw("😀\u{fe0f}", []) |> should.equal(2) - sw("👩‍🎓", []) |> should.equal(4) + sw("😀\u{fe0f}") |> should.equal(2) + sw("👩‍🎓") |> should.equal(4) // combining diacritical marks extended - sw("x\u{1ab0}x\u{1ab0}", []) |> should.equal(2) + sw("x\u{1ab0}x\u{1ab0}") |> should.equal(2) // combining diacritical marks supplement - sw("x\u{1dc0}x\u{1dc0}", []) |> should.equal(2) + sw("x\u{1dc0}x\u{1dc0}") |> should.equal(2) // combining diacritical marks for symbols - sw("x\u{20d0}x\u{20d0}", []) |> should.equal(2) + sw("x\u{20d0}x\u{20d0}") |> should.equal(2) // combining half marks - sw("x\u{fe20}x\u{fe20}", []) |> should.equal(2) + sw("x\u{fe20}x\u{fe20}") |> should.equal(2) } pub fn ignores_default_ignorable_test() { // word joiner - sw("\u{2060}", []) |> should.equal(0) + sw("\u{2060}") |> should.equal(0) // function application - sw("\u{2061}", []) |> should.equal(0) + sw("\u{2061}") |> should.equal(0) // invisible times - sw("\u{2062}", []) |> should.equal(0) + sw("\u{2062}") |> should.equal(0) // invisible separator - sw("\u{2063}", []) |> should.equal(0) + sw("\u{2063}") |> should.equal(0) // invisible plus - sw("\u{2064}", []) |> should.equal(0) + sw("\u{2064}") |> should.equal(0) // zero-width no-break space - sw("\u{feff}", []) |> should.equal(0) - sw("x\u{2060}x", []) |> should.equal(2) - sw("x\u{2061}x", []) |> should.equal(2) - sw("x\u{2062}x", []) |> should.equal(2) - sw("x\u{2063}x", []) |> should.equal(2) - sw("x\u{2064}x", []) |> should.equal(2) - sw("x\u{feff}x", []) |> should.equal(2) + sw("\u{feff}") |> should.equal(0) + sw("x\u{2060}x") |> should.equal(2) + sw("x\u{2061}x") |> should.equal(2) + sw("x\u{2062}x") |> should.equal(2) + sw("x\u{2063}x") |> should.equal(2) + sw("x\u{2064}x") |> should.equal(2) + sw("x\u{feff}x") |> should.equal(2) } pub fn fold_identity_test() { @@ -274,10 +284,13 @@ pub fn fold_identity_test() { |> should.equal(input) } - do_test("äöüè", []) - do_test("안녕하세요", []) - do_test("🏳️‍⚧️", []) - do_test("🏳️‍⚧️", [HandleGraphemeClusters]) - do_test("👩‍👩‍👦‍👦", []) - do_test("wobble\u{1B}[31mhello\u{1B}[39mwibble", []) + let default = string_width.new() + let clusters = default |> string_width.handle_grapheme_clusters + + do_test("äöüè", default) + do_test("안녕하세요", default) + do_test("🏳️‍⚧️", default) + do_test("🏳️‍⚧️", clusters) + do_test("👩‍👩‍👦‍👦", default) + do_test("wobble\u{1B}[31mhello\u{1B}[39mwibble", default) } -- 2.51.2