Something went wrong. Try again.
🌿 Collaborative wiki on ATProto
Something went wrong. Try again.
1.3 kB · 41 lines
TypeScript
at main
123456789101112131415161718192021222324252627282930313233343536373839404142// Lexicon limits come in two units — maxGraphemes (what a reader calls a// character) and maxLength (UTF-8 bytes). String.length is neither.
const segmenter = new Intl.Segmenter("en", { granularity: "grapheme" });const encoder = new TextEncoder();
export function graphemeLength(text: string): number { let n = 0; for (const _ of segmenter.segment(text)) n++; return n;}
export function utf8Length(text: string): number { return encoder.encode(text).length;}
export interface TextLimit { graphemes: number; bytes: number;}
export function withinLimit(text: string, limit: TextLimit): boolean { return ( graphemeLength(text) <= limit.graphemes && utf8Length(text) <= limit.bytes );}
// Grapheme boundaries, not code units: a code-unit slice splits surrogate pairs// and detaches combining marks. The byte pass only binds on emoji-dense text.export function truncateToLimit(text: string, limit: TextLimit): string { let graphemes = [...segmenter.segment(text)].map((s) => s.segment); if (graphemes.length > limit.graphemes) { graphemes = graphemes.slice(0, limit.graphemes); } let out = graphemes.join(""); while (graphemes.length > 0 && utf8Length(out) > limit.bytes) { graphemes.pop(); out = graphemes.join(""); } return out;}