// Lexicon limits come in two units — maxGraphemes (what a reader calls a // character) and maxLength (UTF-8 bytes). String.length is neither. const segmenter = new Intl.Segmenter("en", { granularity: "grapheme" }); const encoder = new TextEncoder(); export function graphemeLength(text: string): number { let n = 0; for (const _ of segmenter.segment(text)) n++; return n; } export function utf8Length(text: string): number { return encoder.encode(text).length; } export interface TextLimit { graphemes: number; bytes: number; } export function withinLimit(text: string, limit: TextLimit): boolean { return ( graphemeLength(text) <= limit.graphemes && utf8Length(text) <= limit.bytes ); } // Grapheme boundaries, not code units: a code-unit slice splits surrogate pairs // and detaches combining marks. The byte pass only binds on emoji-dense text. export function truncateToLimit(text: string, limit: TextLimit): string { let graphemes = [...segmenter.segment(text)].map((s) => s.segment); if (graphemes.length > limit.graphemes) { graphemes = graphemes.slice(0, limit.graphemes); } let out = graphemes.join(""); while (graphemes.length > 0 && utf8Length(out) > limit.bytes) { graphemes.pop(); out = graphemes.join(""); } return out; }