// The front half of the prose renderer: markdown in, tokens out. // // Artifact and message bodies are agent-authored — untrusted text arriving from another member's // PDS. So `marked` is used for its *tokenizer only*: `Lexer` gives back a token tree and the // components walk that tree with ordinary Svelte interpolation. `marked.parse()`, the half that // produces HTML, is never called. There is no `{@html}` anywhere on the path and no sanitizer to // get wrong, because no markup is ever produced from a body — the worst a hostile body can do is // render as its own source. // // Two things a tokenizer will not decide for us, decided here: an entity is decoded to the // character it names (the renderer interpolates, so `&` would otherwise reach the screen // literally), and a link is only a link if its scheme is one we are willing to send someone to. import { parseImageLocator } from '@radial/core' import { Lexer, type Token, type Tokens } from 'marked' export type { Token, Tokens } const OPTIONS = { gfm: true, breaks: false, pedantic: false } /** The block tokens of a body. `space` is dropped: it is separation, not content. */ export function prose(body: string): Token[] { return new Lexer(OPTIONS).lex(body).filter((token) => token.type !== 'space') } /** The inline tokens of one run of text — a thread message, a question, a table cell. */ export function inline(text: string): Token[] { return Lexer.lexInline(text, OPTIONS) } const NAMED: Record = { amp: '&', lt: '<', gt: '>', quot: '"', apos: "'", nbsp: ' ', copy: '©', reg: '®', trade: '™', deg: '°', middot: '·', bull: '•', hellip: '…', mdash: '—', ndash: '–', lsquo: '‘', rsquo: '’', ldquo: '“', rdquo: '”', laquo: '«', raquo: '»', times: '×', divide: '÷', plusmn: '±', larr: '←', rarr: '→', harr: '↔', darr: '↓', uarr: '↑', } /** * Resolve the entities in a run of text. Unknown names are left alone — a body that says `&foo;` * meant to say `&foo;`, and CommonMark leaves it standing too. */ export function decode(text: string): string { return text.replace(/&(#\d{1,7}|#[xX][0-9a-fA-F]{1,6}|[a-zA-Z][a-zA-Z0-9]{1,31});/g, (whole, name: string) => { if (name.startsWith('#')) { const point = name[1] === 'x' || name[1] === 'X' ? parseInt(name.slice(2), 16) : Number(name.slice(1)) if (!Number.isInteger(point) || point <= 0 || point > 0x10ffff) return whole try { return String.fromCodePoint(point) } catch { return whole } } return NAMED[name] ?? whole }) } /** * The address a link may navigate to or an image may load from, or null for one that stays text. * * A body comes from another member's repo, so the scheme is an allowlist rather than a blocklist: * `javascript:` and `data:` are the obvious ones, but a relative link is refused too — it would * resolve against this app's origin, which is not where the body's author was pointing. */ export function safeHref(href: string): string | null { try { const url = new URL(href.trim()) return url.protocol === 'http:' || url.protocol === 'https:' || url.protocol === 'mailto:' ? url.href : null } catch { return null } } /** * The address an image may load from. Narrower than a link's: `mailto:` is somewhere to send a * reader, not somewhere to fetch from, and a body that names it meant something other than a * picture. */ export function safeImage(href: string): string | null { const url = safeHref(href) return url !== null && /^https?:\/\//.test(url) ? url : null } /** An `at://` URI, which the app shows as a chip rather than as somewhere to navigate. */ export const isAtUri = (href: string): boolean => /^at:\/\//i.test(href.trim()) /** * A picture stored as a `com.disnetdev.radial.image` record in the author's own repo, rather than at * an address. * * `safeImage` would refuse it — it is not an `https:` URL and is not meant to be one — so this is * checked first, and what it means is "resolve this", not "fetch this". The resolution is in * `images.svelte.ts` and the rule about what a locator may name is in `@radial/core`; both are * deliberately somewhere other than here, because this module decides what a body SAYS and neither * of those decides that. */ export const isImageLocator = (href: string): boolean => parseImageLocator(href) !== null // ── reading a token ─────────────────────────────────────────────────────────────────────────── // `Token` is a union with a `Generic` member, so narrowing it by `type` inside a template gives // back something with `any` on it. These accessors are where that is dealt with once: the parser's // shapes stop here, and the components below get plain, typed values. /** The children of a token — the runs inside a heading, the blocks inside a quote. */ export const kids = (token: Token): Token[] => 'tokens' in token && Array.isArray(token.tokens) ? (token.tokens as Token[]) : [] /** The literal text of a token, entities resolved. Code carries its own text, never decoded. */ export function textOf(token: Token): string { const text = 'text' in token && typeof token.text === 'string' ? token.text : '' return token.type === 'code' || token.type === 'codespan' || token.type === 'escape' ? text : decode(text) } export const hrefOf = (token: Token): string => 'href' in token && typeof token.href === 'string' ? token.href : '' export const titleOf = (token: Token): string | null => 'title' in token && typeof token.title === 'string' ? token.title : null export const depthOf = (token: Token): number => 'depth' in token && typeof token.depth === 'number' ? token.depth : 1 export const langOf = (token: Token): string => 'lang' in token && typeof token.lang === 'string' ? token.lang : '' export interface ProseItem { blocks: Token[] /** A GFM task item, which carries its own box rather than a bullet. */ task: boolean checked: boolean } export interface ProseList { ordered: boolean start: number items: ProseItem[] } export function listOf(token: Token): ProseList { const list = token as Tokens.List return { ordered: list.ordered === true, start: typeof list.start === 'number' && list.start > 0 ? list.start : 1, items: (list.items ?? []).map((item) => ({ // The checkbox is a token of its own; the box is drawn from `task`, so drop it from the run. blocks: (item.tokens ?? []).filter((child) => child.type !== 'checkbox'), task: item.task === true, checked: item.checked === true, })), } } export interface ProseCell { runs: Token[] align: 'center' | 'left' | 'right' | null } export interface ProseTable { header: ProseCell[] rows: ProseCell[][] } export function tableOf(token: Token): ProseTable { const table = token as Tokens.Table const cell = (source: Tokens.TableCell): ProseCell => ({ runs: source.tokens ?? [], align: source.align ?? null, }) return { header: (table.header ?? []).map(cell), rows: (table.rows ?? []).map((row) => row.map(cell)), } } /** The plain text of a token tree — what a row's one line and a title attribute want. */ export function plain(tokens: Token[]): string { return tokens .map((token) => { if (token.type === 'br') return ' ' if (token.type === 'list') return listOf(token).items.map((item) => plain(item.blocks)).join(' ') // An image stands for its alt text; a link for its label, not its address. if (token.type === 'image') return textOf(token) const children = kids(token) return children.length > 0 ? plain(children) : textOf(token) }) .join('') }