//! Terminal cell width of grapheme clusters and strings. //! //! Port of ultraviolet's width model. ultraviolet delegates to //! `x/ansi.StringWidth` (grapheme clusters via rivo/uniseg) and //! `x/ansi.StringWidthWc` (per-codepoint wcwidth). `codepointWidth` is the //! shared primitive: a per-code-point wcwidth with static //! East-Asian-Width / combining-mark range tables (β‰ˆUnicode 15). //! //! - `wcWidth` sums the width of each code point. //! - `graphemeWidth` folds zero-width code points (combining marks, variation //! selectors, ZWJ, bidi controls) into the preceding base, which the sum //! does naturally since they contribute 0. ZWJ emoji sequences are measured //! per component initially (see `Brain.md` design decision 5). //! //! Neither method is ANSI-aware: callers that may hold styled strings should //! strip escape sequences first (see `Styled.scan`). //! //! This is a per-code-point approximation: it is correct for CJK, fullwidth //! forms, and standalone emoji, but is **not** grapheme-cluster aware β€” ZWJ //! emoji sequences (e.g. πŸ‘¨β€πŸ‘©β€πŸ‘§) are measured per component. const std = @import("std"); const t = std.testing; const width = @This(); const Range = struct { lo: u21, hi: u21 }; /// Code points rendered two cells wide: East Asian Wide + Fullwidth, CJK, /// Hangul syllables, and emoji blocks. Sorted, non-overlapping. const wide = [_]Range{ .{ .lo = 0x1100, .hi = 0x115F }, // Hangul Jamo .{ .lo = 0x2329, .hi = 0x232A }, // angle brackets .{ .lo = 0x2E80, .hi = 0x303E }, // CJK radicals … CJK symbols .{ .lo = 0x3041, .hi = 0x33FF }, // Hiragana … CJK compat .{ .lo = 0x3400, .hi = 0x4DBF }, // CJK Extension A .{ .lo = 0x4E00, .hi = 0x9FFF }, // CJK Unified Ideographs .{ .lo = 0xA000, .hi = 0xA4CF }, // Yi .{ .lo = 0xA960, .hi = 0xA97F }, // Hangul Jamo Extended-A .{ .lo = 0xAC00, .hi = 0xD7A3 }, // Hangul Syllables .{ .lo = 0xF900, .hi = 0xFAFF }, // CJK Compatibility Ideographs .{ .lo = 0xFE10, .hi = 0xFE19 }, // Vertical Forms .{ .lo = 0xFE30, .hi = 0xFE6F }, // CJK Compatibility / Small Forms .{ .lo = 0xFF00, .hi = 0xFF60 }, // Fullwidth Forms .{ .lo = 0xFFE0, .hi = 0xFFE6 }, // Fullwidth Signs .{ .lo = 0x16FE0, .hi = 0x16FE4 }, // Tangut/Khitan marks .{ .lo = 0x17000, .hi = 0x18CFF }, // Tangut, Khitan .{ .lo = 0x1AFF0, .hi = 0x1B2FF }, // Kana Extended/Supplement .{ .lo = 0x1F004, .hi = 0x1F004 }, // Mahjong red dragon .{ .lo = 0x1F0CF, .hi = 0x1F0CF }, // playing card black joker .{ .lo = 0x1F18E, .hi = 0x1F18E }, // negative squared AB .{ .lo = 0x1F191, .hi = 0x1F19A }, // squared CL … VS .{ .lo = 0x1F200, .hi = 0x1F2FF }, // Enclosed Ideographic Supplement .{ .lo = 0x1F300, .hi = 0x1F64F }, // Misc Symbols & Pictographs, Emoticons .{ .lo = 0x1F680, .hi = 0x1F6FF }, // Transport & Map Symbols .{ .lo = 0x1F900, .hi = 0x1F9FF }, // Supplemental Symbols & Pictographs .{ .lo = 0x1FA70, .hi = 0x1FAFF }, // Symbols & Pictographs Extended-A .{ .lo = 0x20000, .hi = 0x3FFFD }, // CJK Extension B … G }; /// Code points rendered zero cells wide: combining marks, variation /// selectors, and zero-width formatting characters. Partial coverage β€” the /// common blocks plus the larger Arabic/Indic/Thai mark ranges. Sorted. const zero_width = [_]Range{ .{ .lo = 0x0300, .hi = 0x036F }, // Combining Diacritical Marks .{ .lo = 0x0483, .hi = 0x0489 }, .{ .lo = 0x0591, .hi = 0x05BD }, .{ .lo = 0x05BF, .hi = 0x05BF }, .{ .lo = 0x05C1, .hi = 0x05C2 }, .{ .lo = 0x05C4, .hi = 0x05C5 }, .{ .lo = 0x05C7, .hi = 0x05C7 }, .{ .lo = 0x0610, .hi = 0x061A }, .{ .lo = 0x064B, .hi = 0x065F }, .{ .lo = 0x0670, .hi = 0x0670 }, .{ .lo = 0x06D6, .hi = 0x06DC }, .{ .lo = 0x06DF, .hi = 0x06E4 }, .{ .lo = 0x06E7, .hi = 0x06E8 }, .{ .lo = 0x06EA, .hi = 0x06ED }, .{ .lo = 0x0711, .hi = 0x0711 }, .{ .lo = 0x0730, .hi = 0x074A }, .{ .lo = 0x07A6, .hi = 0x07B0 }, .{ .lo = 0x07EB, .hi = 0x07F3 }, .{ .lo = 0x0816, .hi = 0x0819 }, .{ .lo = 0x081B, .hi = 0x0823 }, .{ .lo = 0x0825, .hi = 0x0827 }, .{ .lo = 0x0829, .hi = 0x082D }, .{ .lo = 0x0859, .hi = 0x085B }, .{ .lo = 0x08E3, .hi = 0x0902 }, .{ .lo = 0x093A, .hi = 0x093A }, .{ .lo = 0x093C, .hi = 0x093C }, .{ .lo = 0x0941, .hi = 0x0948 }, .{ .lo = 0x094D, .hi = 0x094D }, .{ .lo = 0x0951, .hi = 0x0957 }, .{ .lo = 0x0962, .hi = 0x0963 }, .{ .lo = 0x0E31, .hi = 0x0E31 }, .{ .lo = 0x0E34, .hi = 0x0E3A }, .{ .lo = 0x0E47, .hi = 0x0E4E }, .{ .lo = 0x135D, .hi = 0x135F }, .{ .lo = 0x1AB0, .hi = 0x1AFF }, // Combining Diacritical Marks Extended .{ .lo = 0x1DC0, .hi = 0x1DFF }, // Combining Diacritical Marks Supplement .{ .lo = 0x200B, .hi = 0x200F }, // ZWSP, ZWNJ, ZWJ, LRM, RLM .{ .lo = 0x202A, .hi = 0x202E }, // bidi embeddings .{ .lo = 0x2060, .hi = 0x2064 }, .{ .lo = 0x2066, .hi = 0x206F }, .{ .lo = 0x20D0, .hi = 0x20FF }, // Combining Marks for Symbols .{ .lo = 0xFE00, .hi = 0xFE0F }, // Variation Selectors .{ .lo = 0xFE20, .hi = 0xFE2F }, // Combining Half Marks .{ .lo = 0xE0100, .hi = 0xE01EF }, // Variation Selectors Supplement }; fn inRanges(ranges: []const Range, cp: u21) bool { var lo: usize = 0; var hi: usize = ranges.len; while (lo < hi) { const mid = lo + (hi - lo) / 2; const r = ranges[mid]; if (cp < r.lo) { hi = mid; } else if (cp > r.hi) { lo = mid + 1; } else { return true; } } return false; } /// Returns the terminal cell width of `cp`: 0 for control characters, /// combining marks, and zero-width formatting; 2 for wide/fullwidth code /// points; 1 otherwise. pub fn codepointWidth(cp: u21) u8 { if (cp == 0) return 0; // C0 controls, DEL, and C1 controls. if (cp < 0x20 or (cp >= 0x7f and cp < 0xa0)) return 0; if (cp < 0x300) return 1; // fast path: Latin/punctuation if (inRanges(&zero_width, cp)) return 0; if (inRanges(&wide, cp)) return 2; return 1; } /// How to measure the terminal width of a grapheme cluster or string. pub const Method = enum { /// Per-code-point wcwidth (default; mirrors `ansi.WcWidth`). wcwidth, /// Grapheme-cluster aware (mirrors `ansi.GraphemeWidth`). Best effort β€” /// folded on top of wcwidth. grapheme, pub fn stringWidth(method: Method, s: []const u8) usize { return switch (method) { .wcwidth => wcWidth(s), .grapheme => graphemeWidth(s), }; } }; /// WcWidth is the per-code-point width of `cp` as measured by /// [codepointWidth]. pub const wcRuneWidth = codepointWidth; /// Returns the width of a single code point. pub fn runeWidth(cp: u21) usize { return codepointWidth(cp); } /// Returns the display width of a single grapheme cluster (best effort). /// Zero-width code points (combining marks, VS, ZWJ, …) contribute nothing, /// so the cluster keeps the width of its base code point. pub fn graphemeWidth(s: []const u8) usize { return forEachCodepoint(s, 0); } /// Returns the sum of the code point widths of `s`. pub fn wcWidth(s: []const u8) usize { return forEachCodepoint(s, 0); } /// Sums `codepointWidth` across the code points of `s`. /// /// The input is not assumed to be valid UTF-8: a run whose lead byte is /// invalid, that is truncated, or whose continuation bytes are malformed is /// counted as a single width-1 replacement character, mirroring how the /// terminal renders undecodable bytes. fn forEachCodepoint(s: []const u8, init: usize) usize { var total = init; var i: usize = 0; while (i < s.len) { const len = std.unicode.utf8ByteSequenceLength(s[i]) catch { total += 1; i += 1; continue; }; if (i + @as(usize, len) > s.len) { // Truncated sequence: the remaining tail is its continuation // bytes, so it renders as a single replacement character. total += 1; i = s.len; continue; } const cp = std.unicode.utf8Decode(s[i .. i + len]) catch { total += 1; i += 1; continue; }; total += codepointWidth(cp); i += len; } return total; } /// Returns the width of `s` using the given method, ignoring any escape /// sequences in `s` is NOT supported β€” use [Styled.scan]'s plain cells for /// styled strings. pub fn stringWidth(method: Method, s: []const u8) usize { return method.stringWidth(s); } test "codepointWidth ascii" { try t.expectEqual(1, codepointWidth('A')); try t.expectEqual(0, codepointWidth(0x07)); // BEL try t.expectEqual(0, codepointWidth(0x7f)); // DEL } test "codepointWidth wide" { try t.expectEqual(2, codepointWidth('δΈ–')); // U+4E16 try t.expectEqual(2, codepointWidth(0x1F389)); // πŸŽ‰ try t.expectEqual(2, codepointWidth(0xFF21)); // fullwidth A } test "codepointWidth zero" { try t.expectEqual(0, codepointWidth(0x0301)); // combining acute try t.expectEqual(0, codepointWidth(0x200D)); // ZWJ try t.expectEqual(0, codepointWidth(0xFE0F)); // VS-16 } test "wcWidth ascii" { try t.expectEqual(3, wcWidth("abc")); try t.expectEqual(0, wcWidth("")); } test "wcWidth wide" { try t.expectEqual(2, wcWidth("δΈ–")); try t.expectEqual(4, wcWidth("δΈ–η•Œ")); } test "graphemeWidth folds combining marks" { // e + combining acute try t.expectEqual(1, graphemeWidth("e\u{0301}")); // ZWJ family emoji measured per component (Brain decision #5): try t.expectEqual(4, graphemeWidth("\u{1F468}\u{200D}\u{1F468}")); } test "width tolerates invalid utf-8" { // Invalid lead byte, continuation byte, truncated sequence, overlong ... // each counts as a width-1 replacement character. try t.expectEqual(1, wcWidth("\x80")); try t.expectEqual(1, graphemeWidth("\x80")); try t.expectEqual(1, wcWidth("\xff")); try t.expectEqual(2, wcWidth("a\x80")); try t.expectEqual(1, wcWidth("\xc3")); // truncated 2-byte lead try t.expectEqual(1, wcWidth("\xe3\x81")); // truncated 3-byte lead try t.expectEqual(1, wcWidth("\xe3\x82")); // bad continuation try t.expectEqual(2, wcWidth("\xe3\x81\x82")); // valid あ } test "grapheme widest component keeps width" { try t.expectEqual(1, graphemeWidth("a\u{0301}")); try t.expectEqual(2, graphemeWidth("\u{1F389}")); // πŸŽ‰ } test "fuzz width" { try t.fuzz({}, fuzzWidth, .{ .corpus = &.{ "a", "δΈ–", "e\u{301}", "🏳️\u{fe0f}\u{200d}🌈", "\x1b[31m", "\x00\x07\x7f\x80\xc0\xff", }, }); } fn fuzzWidth(_: void, smith: *t.Smith) anyerror!void { var buf: [128]u8 = undefined; const len = smith.slice(&buf); _ = width.stringWidth(.wcwidth, buf[0..len]); _ = width.stringWidth(.grapheme, buf[0..len]); const cp = smith.value(u21); _ = width.codepointWidth(cp); _ = width.runeWidth(cp); }