diff --git a/src/string_width.gleam b/src/string_width.gleam index 833a405..244d1a0 100644 --- a/src/string_width.gleam +++ b/src/string_width.gleam @@ -22,7 +22,7 @@ import string_width/internal/tables pub opaque type Options { Options( count_ansi_codes: Bool, - ambiguous_width: Int, + ambiguous_as_wide: Bool, mode_2027: Bool, tab_width: Int, tab_offset: Int, @@ -31,7 +31,7 @@ pub opaque type Options { const default_options = Options( count_ansi_codes: False, - ambiguous_width: 1, + ambiguous_as_wide: False, mode_2027: False, tab_width: 8, tab_offset: 0, @@ -69,7 +69,7 @@ pub fn mode_2027(options: Options) -> Options { /// Unicode recommends treating these characters as narrow by default, /// but you can change this behaviour using this option. pub fn ambiguous_as_wide(options: Options) -> Options { - Options(..options, ambiguous_width: 2) + Options(..options, ambiguous_as_wide: True) } /// Do not ignore ansi escape sequences, and count them as regular characters. @@ -162,36 +162,39 @@ pub fn dimensions(str: String) -> #(Int, Int) { pub fn dimensions_with(str: String, options: Options) -> #(Int, Int) { let #(str, ranges, range_width) = prepare_measure(options, str) - let #(rows, cols_max, cols_curr) = { - use #(rows, cols_max, cols_curr), chr, width <- do_fold( - over: str, - at: 0, - using: options, - from: #(0, 0, 0), - ranges: ranges, - range_width: range_width, - ) + let fun = fn(state, chr, width) { + let #(rows, cols_max, cols_curr) = state case chr { - "\n" -> #(rows + 1, int.max(cols_max, cols_curr), 0) - "\t" -> #( - rows, - cols_max, - // round up to next tab_width boundary. - { { cols_curr + options.tab_offset } / options.tab_width + 1 } - * options.tab_width - - options.tab_offset, - ) + "\n" -> #(rows + 1, int.max(cols_max, cols_curr), options.tab_offset) + "\t" -> #(rows, cols_max, tab(options, cols_curr)) _ -> #(rows, cols_max, cols_curr + width) } } + let on_range = fn(state, ansi, length) { + fun(state, ansi, range_width * length) + } + + let on_chars = case options.mode_2027 { + True -> fn(state, str) { fold_chars_2027(options, str, state, fun) } + False -> fn(state, str) { fold_chars_raw(options, str, state, fun) } + } + + let #(rows, cols_max, cols_curr) = + fold_parts(str, 0, ranges, #(0, 0, options.tab_offset), on_chars, on_range) + case cols_curr > 0 { - True -> #(rows + 1, int.max(cols_max, cols_curr)) - False -> #(rows, cols_max) + True -> #(rows + 1, int.max(cols_max, cols_curr) - options.tab_offset) + False -> #(rows, cols_max - options.tab_offset) } } +/// Round up to the next tab boundary. +fn tab(options: Options, col: Int) -> Int { + { col / options.tab_width + 1 } * options.tab_width +} + // magic hook that allows us to not be stupid with FFI and still get // _most_ of the performance (we are now like 50% slower). @external(javascript, "./string_width_ffi.mjs", "prepare_measure") @@ -206,8 +209,102 @@ fn prepare_measure( #(str, ansi_ranges, 0) } +// -- TRUNCATE / PAD ----------------------------------------------------------- + +// "make lines be width" +// line to short? align left/right/center/stretch +// line to long? truncate/wrap +// break at? shys/whitespace/unicode word boudnaries/grahpeme boundaries? this is hard +// widows when wrapping +// margins +// + +// if one component is followed by another zero-width component treat them as one? +// maybe even in fold? +// mode_2027_ext +// mode_wcwidth + +// mixed: pass grapheme clusters but measure codepoints +// individual_codepoints (default see above) / for grapheme clusters, pass width at the first cp +// passing codepoints is only useful for measure? I don't think a user ever wants that + +// fold over words? (handle shy hyphens/soft breaks etc) +// I do not want to implement full unicode word segmentation here though +// line segmentation might be easy enough though +// can also FFI in javascript for the segmenters + // -- FOLD --------------------------------------------------------------------- +/// Returns true if a given component string recieved in `fold` is an ANSI +/// escape sequence. +pub fn is_ansi_component(str: String, options: Options) -> Bool { + case options.count_ansi_codes { + True -> False + False -> + case str { + "\u{1b}" <> _ | "\u{9b}" <> _ | "\u{9d}" <> _ -> True + _ -> False + } + } +} + +/// A higher-level `fold` that keeps track of the position inside of the string. +/// +/// Handles tabs and newlines, and always passes full grapheme clusters, +/// regardless of options. Concatenating the graphemes produces the original string. +/// +/// The `with` function is called with `(state, grapheme, width, row, col)`. +pub fn fold( + over str: String, + from state: state, + with fun: fn(state, String, Int, Int, Int) -> state, +) -> state { + fold_with(str, default_options, state, fun) +} + +/// A higher-level `fold` that keeps track of the position inside of the string +/// +/// Handles tabs and newlines, and always passes full grapheme clusters, +/// regardless of options. Concatenating the graphemes produces the original string. +/// +/// The `with` function is called with `(state, grapheme, width, row, col)`. +pub fn fold_with( + over str: String, + using options: Options, + from state: state, + with fun: fn(state, String, Int, Int, Int) -> state, +) -> state { + let fun = fn(state, chr, width) { + let #(state, row, col) = state + + let state = fun(state, chr, width, row, col) + + case chr { + "\n" -> #(state, row + 1, options.tab_offset) + "\t" -> #(state, row, tab(options, col)) + _ -> #(state, row, col + width) + } + } + + let on_chars = case options.mode_2027 { + True -> fn(state, str) { fold_chars_2027(options, str, state, fun) } + False -> fn(state, str) { fold_chars_wcwidth(options, str, state, fun) } + } + + let ansi_ranges = case options.count_ansi_codes { + True -> [] + False -> ansi.match(str) + } + + let on_range = fn(state, ansi, _) { fun(state, ansi, 0) } + + let initial = #(state, 0, options.tab_offset) + let #(state, _, _) = + fold_parts(str, 0, ansi_ranges, initial, on_chars, on_range) + + state +} + /// Iterate over the measured components of a string. Components are either /// graphemes or codepoints, depending on the `mode_2027` option, /// or other undivisible sequences, like ANSI escape codes. @@ -223,13 +320,13 @@ fn prepare_measure( /// /// ```gleam /// // A slower string_width.line implementation that doesn't handle tabs -/// fold("hello", new(), from: 0, with: fn(so_far, _chr, width) { width + so_far }) +/// fold_raw("hello", new(), from: 0, with: fn(so_far, _chr, width) { width + so_far }) /// // --> 5 /// ``` /// /// ```gleam /// // truncate a string after 50 characters, but keep all ansi sequences. -/// use #(total, acc), chr, width <- fold(input, new(), from: #(0, "")) +/// use #(total, acc), chr, width <- fold_raw(input, new(), from: #(0, "")) /// case total >= 50 { /// True -> case chr { /// "\u{1b}" <> _ -> #(total, acc <> chr) @@ -239,7 +336,7 @@ fn prepare_measure( /// False -> #(total + width, acc <> chr) /// } /// ``` -pub fn fold( +pub fn fold_raw( over string: String, using options: Options, from state: state, @@ -249,67 +346,79 @@ pub fn fold( True -> [] False -> ansi.match(string) } - do_fold( - using: options, - over: string, - at: 0, - ranges: ansi_ranges, - range_width: 0, - from: state, - with: fun, - ) + + let on_chars = case options.mode_2027 { + True -> fn(state, str) { fold_chars_2027(options, str, state, fun) } + False -> fn(state, str) { fold_chars_raw(options, str, state, fun) } + } + + let on_range = fn(state, ansi, _) { fun(state, ansi, 0) } + + fold_parts(string, 0, ansi_ranges, state, on_chars, on_range) } -fn do_fold( - using options: Options, +fn fold_parts( over string: String, at offset: Int, ranges ranges: List(#(Int, Int)), - range_width range_width: Int, from state: state, - with fun: fn(state, String, Int) -> state, + on_chars on_chars: fn(state, String) -> state, + on_range on_range: fn(state, String, Int) -> state, ) -> state { case ranges { + [#(start, length), ..ranges] if start == offset -> { + let #(range_slice, string) = unsafe_split(string, length) + + let state = on_range(state, range_slice, length) + + let offset = start + length + fold_parts(string, offset, ranges, state, on_chars, on_range) + } [#(start, length), ..ranges] -> { let #(before, at_range) = unsafe_split(string, start - offset) let #(range_slice, string) = unsafe_split(at_range, length) let state = state - |> do_fold_characters(options, before, _, fun) - |> fun(range_slice, range_width * length) + |> on_chars(before) + |> on_range(range_slice, length) let offset = start + length - do_fold(options, string, offset, ranges, range_width, state, fun) + fold_parts(string, offset, ranges, state, on_chars, on_range) } - [] -> do_fold_characters(options, string, state, fun) + [] -> + case string { + "" -> state + _ -> on_chars(state, string) + } } } -fn do_fold_characters( +fn fold_chars_raw( options: Options, string: String, state: state, fun: fn(state, String, Int) -> state, ) -> state { - case options.mode_2027 { - True -> do_fold_graphemes(options, string, state, fun) - False -> do_fold_codepoints(options, string, state, fun) - } + use state, cp <- fold_codepoints(string, state) + fun(state, utf_codepoint_to_string(cp), wcwidth(options, cp)) } -fn do_fold_codepoints( +fn fold_chars_wcwidth( options: Options, string: String, state: state, fun: fn(state, String, Int) -> state, ) -> state { - use state, cp <- fold_codepoints(string, state) - let width = wcwidth(options, cp) - fun(state, utf_codepoint_to_string(cp), width) + use state, grapheme <- fold_graphemes(string, state) + let width = { + use acc, cp <- fold_codepoints(grapheme, 0) + acc + wcwidth(options, cp) + } + fun(state, grapheme, width) } -fn do_fold_graphemes( +fn fold_chars_2027( options: Options, string: String, state: state, @@ -368,13 +477,11 @@ fn wcwidth(options: Options, cp: Int) -> Int { case is_ignorable(cp) { True -> 0 False -> - case is_ambiguous(cp) { - True -> options.ambiguous_width - False -> - case is_wide(cp) { - True -> 2 - False -> 1 - } + case + is_wide(cp) || { options.ambiguous_as_wide && is_ambiguous(cp) } + { + True -> 2 + False -> 1 } } } @@ -412,10 +519,7 @@ fn is_ambiguous(cp: Int) -> Bool { fn is_wide(cp: Int) -> Bool { case cp <= 0x1ffff { True -> table_lookup(tables.wide, cp) - False -> { - // NOTE: this assumes we checked ambiguous first! - cp >= 0x20000 && cp <= 0x40000 - } + False -> cp >= 0x20000 && cp <= 0x40000 } } diff --git a/test/string_width_test.gleam b/test/string_width_test.gleam index 4ea9796..689c029 100644 --- a/test/string_width_test.gleam +++ b/test/string_width_test.gleam @@ -287,7 +287,7 @@ pub fn ignores_default_ignorable_test() { pub fn fold_identity_test() { let reducer = fn(acc, chr, _) { acc <> chr } let do_test = fn(input, options) { - string_width.fold(input, options, "", reducer) + string_width.fold_raw(input, options, "", reducer) |> should.equal(input) }