From 58e54caf695dfc38a46ebb74bee806220f7f77db Mon Sep 17 00:00:00 2001 From: dawn <90008@gaze.systems> Date: Fri, 26 Dec 2025 04:07:48 +0300 Subject: [PATCH] use facets, better-er rich text --- deno.lock | 18 ++ package.json | 2 + src/components/BskyPost.svelte | 27 +-- src/components/RichText.svelte | 71 +++++++ src/lib/richtext/index.ts | 77 ++++++++ src/lib/richtext/parser.ts | 349 +++++++++++++++++++++++++++++++++ 6 files changed, 521 insertions(+), 23 deletions(-) create mode 100644 src/components/RichText.svelte create mode 100644 src/lib/richtext/index.ts create mode 100644 src/lib/richtext/parser.ts diff --git a/deno.lock b/deno.lock index 5e1d8f1..0121c80 100644 --- a/deno.lock +++ b/deno.lock @@ -2,6 +2,8 @@ "version": "5", "specifiers": { "npm:@atcute/atproto@^3.1.9": "3.1.9", + "npm:@atcute/bluesky-richtext-builder@^2.0.4": "2.0.4", + "npm:@atcute/bluesky-richtext-segmenter@^2.0.4": "2.0.4", "npm:@atcute/bluesky@^3.2.14": "3.2.14", "npm:@atcute/client@^4.1.1": "4.1.1", "npm:@atcute/identity-resolver@^1.2.1": "1.2.1_@atcute+identity@1.1.3", @@ -48,6 +50,20 @@ "@atcute/lexicons" ] }, + "@atcute/bluesky-richtext-builder@2.0.4": { + "integrity": "sha512-ydA9VWBPsBE/gbu1vYbmh7AZ8FLfxp+LE4eH5GgOTCOxwhs7Mgy1oHrHY+Er6gu6PfdoUoGso0uI3Wl3ZF/Mxg==", + "dependencies": [ + "@atcute/bluesky", + "@atcute/lexicons" + ] + }, + "@atcute/bluesky-richtext-segmenter@2.0.4": { + "integrity": "sha512-6m5QEAv4lU3qTy5MeJXJRRG33acipYJnMW1T7W/KrMyThGhQ7jSTTh8Z48quElgivgX7MDj6o/ow1oLUsjsCKw==", + "dependencies": [ + "@atcute/bluesky", + "@atcute/lexicons" + ] + }, "@atcute/bluesky@3.2.14": { "integrity": "sha512-XlVuF55AYIyplmKvlGLlj+cUvk9ggxNRPczkTPIY991xJ4qDxDHpBJ39ekAV4dWcuBoRo2o9JynzpafPu2ljDA==", "dependencies": [ @@ -1721,6 +1737,8 @@ "packageJson": { "dependencies": [ "npm:@atcute/atproto@^3.1.9", + "npm:@atcute/bluesky-richtext-builder@^2.0.4", + "npm:@atcute/bluesky-richtext-segmenter@^2.0.4", "npm:@atcute/bluesky@^3.2.14", "npm:@atcute/client@^4.1.1", "npm:@atcute/identity-resolver@^1.2.1", diff --git a/package.json b/package.json index a67c992..dcade0d 100644 --- a/package.json +++ b/package.json @@ -16,6 +16,8 @@ "dependencies": { "@atcute/atproto": "^3.1.9", "@atcute/bluesky": "^3.2.14", + "@atcute/bluesky-richtext-builder": "^2.0.4", + "@atcute/bluesky-richtext-segmenter": "^2.0.4", "@atcute/client": "^4.1.1", "@atcute/identity": "^1.1.3", "@atcute/identity-resolver": "^1.2.1", diff --git a/src/components/BskyPost.svelte b/src/components/BskyPost.svelte index 29fd0f7..afbf892 100644 --- a/src/components/BskyPost.svelte +++ b/src/components/BskyPost.svelte @@ -28,12 +28,13 @@ import * as TID from '@atcute/tid'; import type { PostWithUri } from '$lib/at/fetch'; import { onMount } from 'svelte'; - import { isActorIdentifier, type AtprotoDid } from '@atcute/lexicons/syntax'; + import { type AtprotoDid } from '@atcute/lexicons/syntax'; import { derived } from 'svelte/store'; import Device from 'svelte-device-info'; import Dropdown from './Dropdown.svelte'; import { type AppBskyEmbeds } from '$lib/at/types'; import { settings } from '$lib/settings'; + import RichText from './RichText.svelte'; interface Props { client: AtpClient; @@ -347,27 +348,7 @@ {#if profileDesc.length > 0}

- {#each profileDesc.split(/(\s)/) as line, idx (idx)} - {#if line === '\n'} -
- {:else if isActorIdentifier(line.replace(/^@/, ''))} - {line} - {:else if line.startsWith('https://')} - {line.replace(/https?:\/\//, '')} - {:else} - {line} - {/if} - {/each} +

{/if} @@ -446,7 +427,7 @@

- {record.text} + {#if isOnPostComposer && record.embed} {@render embedBadge(record.embed)} {/if} diff --git a/src/components/RichText.svelte b/src/components/RichText.svelte new file mode 100644 index 0000000..f4dbe25 --- /dev/null +++ b/src/components/RichText.svelte @@ -0,0 +1,71 @@ + + +{#snippet plainText(text: string)} + {#each text.split(/(\s)/) as line, idx (idx)} + {#if line === '\n'} +
+ {:else} + {line} + {/if} + {/each} +{/snippet} + +{#snippet segments(segments: RichtextSegment[])} + {#each segments as segment, idx ([segment, idx])} + {@const { text, features: _features } = segment} + {@const features = _features ?? []} + {#if features.length > 0} + {#each features as feature, idx ([feature, idx])} + {#if feature.$type === 'app.bsky.richtext.facet#mention'} + {@render plainText(text)} + {:else if feature.$type === 'app.bsky.richtext.facet#link'} + {@const uri = new URL(feature.uri)} + {@render plainText(uri.href.replace(`${uri.protocol}//`, ''))} + {:else if feature.$type === 'app.bsky.richtext.facet#tag'} + {@render plainText(text)} + {:else} + {@render plainText(text)} + {/if} + {/each} + {:else} + {@render plainText(text)} + {/if} + {/each} +{/snippet} + +{#await richtext} + {@render plainText(text)} +{:then richtext} + {@render segments(segmentize(richtext.text, richtext.facets))} +{/await} diff --git a/src/lib/richtext/index.ts b/src/lib/richtext/index.ts new file mode 100644 index 0000000..4f24b9b --- /dev/null +++ b/src/lib/richtext/index.ts @@ -0,0 +1,77 @@ +import RichtextBuilder, { type BakedRichtext } from '@atcute/bluesky-richtext-builder'; +import { tokenize, type Token } from '$lib/richtext/parser'; +import type { Did, GenericUri, Handle } from '@atcute/lexicons'; +import type { AtpClient } from '$lib/at/client'; + +export const parseToRichText = ( + client: AtpClient, + text: string +): ReturnType => { + const tokens = tokenize(text); + return processTokens(client, tokens); +}; + +const processTokens = async (client: AtpClient, tokens: Token[]): Promise => { + const rt = new RichtextBuilder(); + + for (const token of tokens) { + switch (token.type) { + case 'text': + rt.addText(token.content); + break; + case 'mention': { + let did: Did | undefined = token.did as Did | undefined; + if (!did) { + const handle = token.handle as Handle; + const result = await client.resolveHandle(handle); + if (result.ok) did = result.value; + } + if (did) rt.addMention(token.raw, did); + else rt.addText(token.raw); + break; + } + case 'topic': + rt.addTag(token.name); + break; + case 'autolink': + rt.addLink(token.url, token.url as GenericUri); + break; + case 'link': { + // flatten children to text + const text = flattenToText(token.children); + rt.addLink(text, token.url as GenericUri); + break; + } + case 'escape': + rt.addText(token.escaped); + break; + // formatting tokens (strong, emphasis, etc.) don't map to facets + // so just extract their text content + case 'strong': + case 'emphasis': + case 'underline': + case 'delete': + rt.addText(flattenToText(token.children)); + break; + case 'code': + rt.addText(token.content); + break; + case 'emote': + // handle emotes as needed + rt.addText(token.raw); + break; + } + } + + return rt.build(); +}; + +const flattenToText = (tokens: Token[]): string => { + return tokens + .map((t) => { + if ('content' in t) return t.content; + if ('children' in t) return flattenToText(t.children); + return t.raw; + }) + .join(''); +}; diff --git a/src/lib/richtext/parser.ts b/src/lib/richtext/parser.ts new file mode 100644 index 0000000..9a58ff6 --- /dev/null +++ b/src/lib/richtext/parser.ts @@ -0,0 +1,349 @@ +// taken and modified from: https://github.com/mary-ext/atcute/blob/trunk/packages/bluesky/richtext-parser/lib/index.ts + +const ESCAPE_RE = /^\\([^0-9A-Za-z\s])/; + +const MENTION_RE = /^[@@]([a-zA-Z0-9-]+(?:\.[a-zA-Z0-9-]+)*(?:\.[a-zA-Z]{2,}))($|\s|\p{P})/u; + +const DID_RE = /^(did:([a-z0-9]+):([A-Za-z0-9.\-_%:]+))($|\s|\p{P})/u; + +const TOPIC_RE = + /^(?:#(?!\ufe0f|\u20e3)|#)([\p{N}]*[\p{L}\p{M}\p{Pc}][\p{L}\p{M}\p{Pc}\p{N}]*)($|\s|\p{P})/u; + +const EMOTE_RE = /^:([\w-]+):/; + +const AUTOLINK_RE = /^https?:\/\/[\S]+/; +const AUTOLINK_BACKPEDAL_RE = /(?:(??(?:\s+['"]([^]*?)['"])?\s*\)/; +const UNESCAPE_URL_RE = /\\([^0-9A-Za-z\s])/g; + +const EMPHASIS_RE = + /^\b_((?:__|\\[^]|[^\\_])+?)_\b|^\*(?=\S)((?:\*\*|\\[^]|\s+(?:\\[^]|[^\s*\\]|\*\*)|[^\s*\\])+?)\*(?!\*)/; + +const STRONG_RE = /^\*\*((?:\\[^]|[^\\])+?)\*\*(?!\*)/; + +const UNDERLINE_RE = /^__((?:\\[^]|~(?!~)|[^~\\]|\s(?!~~))+?)__(?!_)/; + +const DELETE_RE = /^~~((?:\\[^]|~(?!~)|[^~\\]|\s(?!~~))+?)~~/; + +const CODE_RE = /^(`+)([^]*?[^`])\1(?!`)/; +const CODE_ESCAPE_BACKTICKS_RE = /^ (?= *`)|(` *) $/g; + +const TEXT_RE = + /^[^]+?(?:(?=$|[~*_`:\\[]|https?:\/\/)|(?<=\s|[(){}/\\[\]\-|:;'".,=+])(?=[@@##]|did:[a-z0-9]+:))/; + +export interface EscapeToken { + type: 'escape'; + raw: string; + escaped: string; +} + +export interface MentionToken { + type: 'mention'; + raw: string; + handle?: string; + did?: string; +} + +export interface TopicToken { + type: 'topic'; + raw: string; + name: string; +} + +export interface EmoteToken { + type: 'emote'; + raw: string; + name: string; +} + +export interface AutolinkToken { + type: 'autolink'; + raw: string; + url: string; +} + +export interface LinkToken { + type: 'link'; + raw: string; + url: string; + children: Token[]; +} + +export interface UnderlineToken { + type: 'underline'; + raw: string; + children: Token[]; +} + +export interface StrongToken { + type: 'strong'; + raw: string; + children: Token[]; +} + +export interface EmphasisToken { + type: 'emphasis'; + raw: string; + children: Token[]; +} + +export interface DeleteToken { + type: 'delete'; + raw: string; + children: Token[]; +} + +export interface CodeToken { + type: 'code'; + raw: string; + content: string; +} + +export interface TextToken { + type: 'text'; + raw: string; + content: string; +} + +export type Token = + | EscapeToken + | MentionToken + | TopicToken + | EmoteToken + | AutolinkToken + | LinkToken + | StrongToken + | EmphasisToken + | UnderlineToken + | DeleteToken + | CodeToken + | TextToken; + +const tokenizeEscape = (src: string): EscapeToken | undefined => { + const match = ESCAPE_RE.exec(src); + if (match) { + return { + type: 'escape', + raw: match[0], + escaped: match[1] + }; + } +}; + +const tokenizeMention = (src: string): MentionToken | undefined => { + const match = MENTION_RE.exec(src); + if (match && match[2] !== '@') { + const suffix = match[2].length; + + return { + type: 'mention', + raw: suffix > 0 ? match[0].slice(0, -suffix) : match[0], + handle: match[1] + }; + } + + const didMatch = DID_RE.exec(src); + if (didMatch) { + const suffix = didMatch[4].length; + + return { + type: 'mention', + raw: suffix > 0 ? didMatch[0].slice(0, -suffix) : didMatch[0], + did: didMatch[1] + }; + } +}; + +const tokenizeTopic = (src: string): TopicToken | undefined => { + const match = TOPIC_RE.exec(src); + if (match && match[2] !== '#') { + const suffix = match[2].length; + + return { + type: 'topic', + raw: suffix > 0 ? match[0].slice(0, -suffix) : match[0], + name: match[1] + }; + } +}; + +const tokenizeEmote = (src: string): EmoteToken | undefined => { + const match = EMOTE_RE.exec(src); + if (match) { + return { + type: 'emote', + raw: match[0], + name: match[1] + }; + } +}; + +const tokenizeAutolink = (src: string): AutolinkToken | undefined => { + const match = AUTOLINK_RE.exec(src); + if (match) { + const url = match[0].replace(AUTOLINK_BACKPEDAL_RE, ''); + + return { + type: 'autolink', + raw: url, + url: url + }; + } +}; + +const tokenizeLink = (src: string): LinkToken | undefined => { + const match = LINK_RE.exec(src); + if (match) { + return { + type: 'link', + raw: match[0], + url: match[2].replace(UNESCAPE_URL_RE, '$1'), + children: tokenize(match[1]) + }; + } +}; + +const _tokenizeEmphasis = (src: string): EmphasisToken | undefined => { + const match = EMPHASIS_RE.exec(src); + if (match) { + return { + type: 'emphasis', + raw: match[0], + children: tokenize(match[2] || match[1]) + }; + } +}; + +const _tokenizeStrong = (src: string): StrongToken | undefined => { + const match = STRONG_RE.exec(src); + if (match) { + return { + type: 'strong', + raw: match[0], + children: tokenize(match[1]) + }; + } +}; + +const _tokenizeUnderline = (src: string): UnderlineToken | undefined => { + const match = UNDERLINE_RE.exec(src); + if (match) { + return { + type: 'underline', + raw: match[0], + children: tokenize(match[1]) + }; + } +}; + +const tokenizeEmStrongU = ( + src: string +): EmphasisToken | StrongToken | UnderlineToken | undefined => { + let token: EmphasisToken | StrongToken | UnderlineToken | undefined; + + { + const match = _tokenizeEmphasis(src); + if (match && (!token || match.raw.length > token.raw.length)) { + token = match; + } + } + + { + const match = _tokenizeStrong(src); + if (match && (!token || match.raw.length > token.raw.length)) { + token = match; + } + } + + { + const match = _tokenizeUnderline(src); + if (match && (!token || match.raw.length > token.raw.length)) { + token = match; + } + } + + return token; +}; + +const tokenizeDelete = (src: string): DeleteToken | undefined => { + const match = DELETE_RE.exec(src); + if (match) { + return { + type: 'delete', + raw: match[0], + children: tokenize(match[1]) + }; + } +}; + +const tokenizeCode = (src: string): CodeToken | undefined => { + const match = CODE_RE.exec(src); + if (match) { + return { + type: 'code', + raw: match[0], + content: match[2].replace(CODE_ESCAPE_BACKTICKS_RE, '$1') + }; + } +}; + +const tokenizeText = (src: string): TextToken | undefined => { + const match = TEXT_RE.exec(src); + if (match) { + return { + type: 'text', + raw: match[0], + content: match[0] + }; + } +}; + +export const tokenize = (src: string): Token[] => { + const tokens: Token[] = []; + + let last: Token | undefined; + let token: Token | undefined; + + while (src) { + last = token; + + if ( + (token = + tokenizeEscape(src) || + tokenizeMention(src) || + tokenizeAutolink(src) || + tokenizeTopic(src) || + tokenizeEmote(src) || + tokenizeLink(src) || + tokenizeEmStrongU(src) || + tokenizeDelete(src) || + tokenizeCode(src)) + ) { + src = src.slice(token.raw.length); + tokens.push(token); + continue; + } + + if ((token = tokenizeText(src))) { + src = src.slice(token.raw.length); + + if (last && last.type === 'text') { + last.raw += token.raw; + last.content += token.content; + token = last; + } else { + tokens.push(token); + } + + continue; + } + + if (src) { + throw new Error(`infinite loop encountered`); + } + } + + return tokens; +}; -- 2.51.2