Something went wrong. Try again.
A charm-like tui library
Something went wrong. Try again.
123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294//! Terminal cell width of grapheme clusters and strings.//!//! Port of ultraviolet's width model. ultraviolet delegates to//! `x/ansi.StringWidth` (grapheme clusters via rivo/uniseg) and//! `x/ansi.StringWidthWc` (per-codepoint wcwidth). `codepointWidth` is the//! shared primitive: a per-code-point wcwidth with static//! East-Asian-Width / combining-mark range tables (≈Unicode 15).//!//! - `wcWidth` sums the width of each code point.//! - `graphemeWidth` folds zero-width code points (combining marks, variation//! selectors, ZWJ, bidi controls) into the preceding base, which the sum//! does naturally since they contribute 0. ZWJ emoji sequences are measured//! per component initially (see `Brain.md` design decision 5).//!//! Neither method is ANSI-aware: callers that may hold styled strings should//! strip escape sequences first (see `Styled.scan`).//!//! This is a per-code-point approximation: it is correct for CJK, fullwidth//! forms, and standalone emoji, but is **not** grapheme-cluster aware — ZWJ//! emoji sequences (e.g. 👨👩👧) are measured per component.const std = @import("std");
const t = std.testing;
const width = @This();
const Range = struct { lo: u21, hi: u21 };
/// Code points rendered two cells wide: East Asian Wide + Fullwidth, CJK,/// Hangul syllables, and emoji blocks. Sorted, non-overlapping.const wide = [_]Range{ .{ .lo = 0x1100, .hi = 0x115F }, // Hangul Jamo .{ .lo = 0x2329, .hi = 0x232A }, // angle brackets .{ .lo = 0x2E80, .hi = 0x303E }, // CJK radicals … CJK symbols .{ .lo = 0x3041, .hi = 0x33FF }, // Hiragana … CJK compat .{ .lo = 0x3400, .hi = 0x4DBF }, // CJK Extension A .{ .lo = 0x4E00, .hi = 0x9FFF }, // CJK Unified Ideographs .{ .lo = 0xA000, .hi = 0xA4CF }, // Yi .{ .lo = 0xA960, .hi = 0xA97F }, // Hangul Jamo Extended-A .{ .lo = 0xAC00, .hi = 0xD7A3 }, // Hangul Syllables .{ .lo = 0xF900, .hi = 0xFAFF }, // CJK Compatibility Ideographs .{ .lo = 0xFE10, .hi = 0xFE19 }, // Vertical Forms .{ .lo = 0xFE30, .hi = 0xFE6F }, // CJK Compatibility / Small Forms .{ .lo = 0xFF00, .hi = 0xFF60 }, // Fullwidth Forms .{ .lo = 0xFFE0, .hi = 0xFFE6 }, // Fullwidth Signs .{ .lo = 0x16FE0, .hi = 0x16FE4 }, // Tangut/Khitan marks .{ .lo = 0x17000, .hi = 0x18CFF }, // Tangut, Khitan .{ .lo = 0x1AFF0, .hi = 0x1B2FF }, // Kana Extended/Supplement .{ .lo = 0x1F004, .hi = 0x1F004 }, // Mahjong red dragon .{ .lo = 0x1F0CF, .hi = 0x1F0CF }, // playing card black joker .{ .lo = 0x1F18E, .hi = 0x1F18E }, // negative squared AB .{ .lo = 0x1F191, .hi = 0x1F19A }, // squared CL … VS .{ .lo = 0x1F200, .hi = 0x1F2FF }, // Enclosed Ideographic Supplement .{ .lo = 0x1F300, .hi = 0x1F64F }, // Misc Symbols & Pictographs, Emoticons .{ .lo = 0x1F680, .hi = 0x1F6FF }, // Transport & Map Symbols .{ .lo = 0x1F900, .hi = 0x1F9FF }, // Supplemental Symbols & Pictographs .{ .lo = 0x1FA70, .hi = 0x1FAFF }, // Symbols & Pictographs Extended-A .{ .lo = 0x20000, .hi = 0x3FFFD }, // CJK Extension B … G};
/// Code points rendered zero cells wide: combining marks, variation/// selectors, and zero-width formatting characters. Partial coverage — the/// common blocks plus the larger Arabic/Indic/Thai mark ranges. Sorted.const zero_width = [_]Range{ .{ .lo = 0x0300, .hi = 0x036F }, // Combining Diacritical Marks .{ .lo = 0x0483, .hi = 0x0489 }, .{ .lo = 0x0591, .hi = 0x05BD }, .{ .lo = 0x05BF, .hi = 0x05BF }, .{ .lo = 0x05C1, .hi = 0x05C2 }, .{ .lo = 0x05C4, .hi = 0x05C5 }, .{ .lo = 0x05C7, .hi = 0x05C7 }, .{ .lo = 0x0610, .hi = 0x061A }, .{ .lo = 0x064B, .hi = 0x065F }, .{ .lo = 0x0670, .hi = 0x0670 }, .{ .lo = 0x06D6, .hi = 0x06DC }, .{ .lo = 0x06DF, .hi = 0x06E4 }, .{ .lo = 0x06E7, .hi = 0x06E8 }, .{ .lo = 0x06EA, .hi = 0x06ED }, .{ .lo = 0x0711, .hi = 0x0711 }, .{ .lo = 0x0730, .hi = 0x074A }, .{ .lo = 0x07A6, .hi = 0x07B0 }, .{ .lo = 0x07EB, .hi = 0x07F3 }, .{ .lo = 0x0816, .hi = 0x0819 }, .{ .lo = 0x081B, .hi = 0x0823 }, .{ .lo = 0x0825, .hi = 0x0827 }, .{ .lo = 0x0829, .hi = 0x082D }, .{ .lo = 0x0859, .hi = 0x085B }, .{ .lo = 0x08E3, .hi = 0x0902 }, .{ .lo = 0x093A, .hi = 0x093A }, .{ .lo = 0x093C, .hi = 0x093C }, .{ .lo = 0x0941, .hi = 0x0948 }, .{ .lo = 0x094D, .hi = 0x094D }, .{ .lo = 0x0951, .hi = 0x0957 }, .{ .lo = 0x0962, .hi = 0x0963 }, .{ .lo = 0x0E31, .hi = 0x0E31 }, .{ .lo = 0x0E34, .hi = 0x0E3A }, .{ .lo = 0x0E47, .hi = 0x0E4E }, .{ .lo = 0x135D, .hi = 0x135F }, .{ .lo = 0x1AB0, .hi = 0x1AFF }, // Combining Diacritical Marks Extended .{ .lo = 0x1DC0, .hi = 0x1DFF }, // Combining Diacritical Marks Supplement .{ .lo = 0x200B, .hi = 0x200F }, // ZWSP, ZWNJ, ZWJ, LRM, RLM .{ .lo = 0x202A, .hi = 0x202E }, // bidi embeddings .{ .lo = 0x2060, .hi = 0x2064 }, .{ .lo = 0x2066, .hi = 0x206F }, .{ .lo = 0x20D0, .hi = 0x20FF }, // Combining Marks for Symbols .{ .lo = 0xFE00, .hi = 0xFE0F }, // Variation Selectors .{ .lo = 0xFE20, .hi = 0xFE2F }, // Combining Half Marks .{ .lo = 0xE0100, .hi = 0xE01EF }, // Variation Selectors Supplement};
fn inRanges(ranges: []const Range, cp: u21) bool { var lo: usize = 0; var hi: usize = ranges.len; while (lo < hi) { const mid = lo + (hi - lo) / 2; const r = ranges[mid]; if (cp < r.lo) { hi = mid; } else if (cp > r.hi) { lo = mid + 1; } else { return true; } } return false;}
/// Returns the terminal cell width of `cp`: 0 for control characters,/// combining marks, and zero-width formatting; 2 for wide/fullwidth code/// points; 1 otherwise.pub fn codepointWidth(cp: u21) u8 { if (cp == 0) return 0; // C0 controls, DEL, and C1 controls. if (cp < 0x20 or (cp >= 0x7f and cp < 0xa0)) return 0; if (cp < 0x300) return 1; // fast path: Latin/punctuation if (inRanges(&zero_width, cp)) return 0; if (inRanges(&wide, cp)) return 2; return 1;}
/// How to measure the terminal width of a grapheme cluster or string.pub const Method = enum { /// Per-code-point wcwidth (default; mirrors `ansi.WcWidth`). wcwidth, /// Grapheme-cluster aware (mirrors `ansi.GraphemeWidth`). Best effort — /// folded on top of wcwidth. grapheme,
pub fn stringWidth(method: Method, s: []const u8) usize { return switch (method) { .wcwidth => wcWidth(s), .grapheme => graphemeWidth(s), }; }};
/// WcWidth is the per-code-point width of `cp` as measured by/// [codepointWidth].pub const wcRuneWidth = codepointWidth;
/// Returns the width of a single code point.pub fn runeWidth(cp: u21) usize { return codepointWidth(cp);}
/// Returns the display width of a single grapheme cluster (best effort)./// Zero-width code points (combining marks, VS, ZWJ, …) contribute nothing,/// so the cluster keeps the width of its base code point.pub fn graphemeWidth(s: []const u8) usize { return forEachCodepoint(s, 0);}
/// Returns the sum of the code point widths of `s`.pub fn wcWidth(s: []const u8) usize { return forEachCodepoint(s, 0);}
/// Sums `codepointWidth` across the code points of `s`.////// The input is not assumed to be valid UTF-8: a run whose lead byte is/// invalid, that is truncated, or whose continuation bytes are malformed is/// counted as a single width-1 replacement character, mirroring how the/// terminal renders undecodable bytes.fn forEachCodepoint(s: []const u8, init: usize) usize { var total = init; var i: usize = 0; while (i < s.len) { const len = std.unicode.utf8ByteSequenceLength(s[i]) catch { total += 1; i += 1; continue; }; if (i + @as(usize, len) > s.len) { // Truncated sequence: the remaining tail is its continuation // bytes, so it renders as a single replacement character. total += 1; i = s.len; continue; } const cp = std.unicode.utf8Decode(s[i .. i + len]) catch { total += 1; i += 1; continue; }; total += codepointWidth(cp); i += len; } return total;}
/// Returns the width of `s` using the given method, ignoring any escape/// sequences in `s` is NOT supported — use [Styled.scan]'s plain cells for/// styled strings.pub fn stringWidth(method: Method, s: []const u8) usize { return method.stringWidth(s);}
test "codepointWidth ascii" { try t.expectEqual(1, codepointWidth('A')); try t.expectEqual(0, codepointWidth(0x07)); // BEL try t.expectEqual(0, codepointWidth(0x7f)); // DEL}
test "codepointWidth wide" { try t.expectEqual(2, codepointWidth('世')); // U+4E16 try t.expectEqual(2, codepointWidth(0x1F389)); // 🎉 try t.expectEqual(2, codepointWidth(0xFF21)); // fullwidth A}
test "codepointWidth zero" { try t.expectEqual(0, codepointWidth(0x0301)); // combining acute try t.expectEqual(0, codepointWidth(0x200D)); // ZWJ try t.expectEqual(0, codepointWidth(0xFE0F)); // VS-16}
test "wcWidth ascii" { try t.expectEqual(3, wcWidth("abc")); try t.expectEqual(0, wcWidth(""));}
test "wcWidth wide" { try t.expectEqual(2, wcWidth("世")); try t.expectEqual(4, wcWidth("世界"));}
test "graphemeWidth folds combining marks" { // e + combining acute try t.expectEqual(1, graphemeWidth("e\u{0301}")); // ZWJ family emoji measured per component (Brain decision #5): try t.expectEqual(4, graphemeWidth("\u{1F468}\u{200D}\u{1F468}"));}
test "width tolerates invalid utf-8" { // Invalid lead byte, continuation byte, truncated sequence, overlong ... // each counts as a width-1 replacement character. try t.expectEqual(1, wcWidth("\x80")); try t.expectEqual(1, graphemeWidth("\x80")); try t.expectEqual(1, wcWidth("\xff")); try t.expectEqual(2, wcWidth("a\x80")); try t.expectEqual(1, wcWidth("\xc3")); // truncated 2-byte lead try t.expectEqual(1, wcWidth("\xe3\x81")); // truncated 3-byte lead try t.expectEqual(1, wcWidth("\xe3\x82")); // bad continuation try t.expectEqual(2, wcWidth("\xe3\x81\x82")); // valid あ}
test "grapheme widest component keeps width" { try t.expectEqual(1, graphemeWidth("a\u{0301}")); try t.expectEqual(2, graphemeWidth("\u{1F389}")); // 🎉}
test "fuzz width" { try t.fuzz({}, fuzzWidth, .{ .corpus = &.{ "a", "世", "e\u{301}", "🏳️\u{fe0f}\u{200d}🌈", "\x1b[31m", "\x00\x07\x7f\x80\xc0\xff", }, });}
fn fuzzWidth(_: void, smith: *t.Smith) anyerror!void { var buf: [128]u8 = undefined; const len = smith.slice(&buf); _ = width.stringWidth(.wcwidth, buf[0..len]); _ = width.stringWidth(.grapheme, buf[0..len]);
const cp = smith.value(u21); _ = width.codepointWidth(cp); _ = width.runeWidth(cp);}