copyediting for your websites made visual and easy.
Something went wrong. Try again.
123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294// copypasta — source parser/extractor//// Tokenizes HTML-ish sources (plain HTML, Astro, Svelte) and extracts the// *text content* of elements as editable segments with exact source offsets,// so edits can be spliced back into the original file without touching markup.//// Design notes:// * Offsets are UTF-16 code-unit indices into the decoded JS string. They are// consistent because read + write both round-trip through the same decode.// * "{...}" runs are treated as opaque template expressions only for// .astro/.svelte. They surface as read-only `expr` segments so the editor// sees the full sentence without being able to break dynamic parts.// * Frontmatter (leading `---` block) is skipped as meta.
const VOID_TAGS = new Set([ 'area', 'base', 'br', 'col', 'embed', 'hr', 'img', 'input', 'link', 'meta', 'param', 'source', 'track', 'wbr',]);
// Content under these tags is never shown/edited (scripts, styles, embeds).const SKIP_TAGS = new Set([ 'script', 'style', 'noscript', 'template', 'iframe', 'object', 'svg', 'math', 'canvas', 'video', 'audio',]);
// Structural wrappers whose *direct* text is not meaningful copy.const NON_CONTENT_PARENTS = new Set(['html', 'head', 'body']);
// Phrase-level elements that should be rendered *inside* their parent block// rather than as standalone rows.const INLINE_TAGS = new Set([ 'a', 'span', 'strong', 'em', 'b', 'i', 'u', 's', 'small', 'mark', 'sub', 'sup', 'time', 'abbr', 'cite', 'q', 'ins', 'del', 'code', 'kbd', 'samp', 'var', 'bdi', 'bdo', 'data', 'dfn', 'rp', 'rt', 'ruby',]);
export const EXTS = new Set(['.html', '.htm', '.astro', '.svelte']);
// Elements whose text whitespace is significant (collapsed to nothing).const PRESERVE_WS_TAGS = new Set(['pre', 'textarea']);
// Collapse whitespace runs to single spaces (matching HTML rendering).// Block-edge trimming happens in the post-pass so single spaces are kept// between adjacent inline segments (e.g. "Drawn by " + "@goose.art").function normalizeWs(t) { return t.replace(/\s+/g, ' ');}
function isTagStart(s, i) { if (s[i] !== '<') return false; const n = s[i + 1]; if (n === undefined) return false; if (n === '!' || n === '?') return true; if (n === '/') return /[a-zA-Z]/.test(s[i + 2] || ''); return /[a-zA-Z]/.test(n);}
// Match a balanced "{...}" run starting at `start` (which is `{`).// Returns the index just after the matching `}` (or source length if unbalanced).// Skips // and /* */ comments so prose apostrophes inside JSX comments// (e.g. "isn't") aren't mistaken for string quotes.function matchBraces(s, start) { let depth = 0; let quote = null; let i = start; while (i < s.length) { const c = s[i]; const n = s[i + 1]; if (!quote && c === '/' && n === '/') { // line comment i += 2; while (i < s.length && s[i] !== '\n') i++; continue; } if (!quote && c === '/' && n === '*') { // block comment i += 2; while (i < s.length && !(s[i] === '*' && s[i + 1] === '/')) i++; i += 2; continue; } if (quote) { if (c === quote && s[i - 1] !== '\\') quote = null; i++; continue; } if (c === '"' || c === "'" || c === '`') { quote = c; i++; continue; } if (c === '{') depth++; else if (c === '}') { depth--; if (depth === 0) return i + 1; } i++; } return s.length;}
// Scan a tag starting at `start` (source[start] === '<').// Returns { name, closing, selfClosing, end } or null if not a tag.function scanTag(s, start) { let i = start + 1; let closing = false; if (s[i] === '/') { closing = true; i++; }
const nameStart = i; while (i < s.length && /[a-zA-Z0-9:._-]/.test(s[i])) i++; const name = s.slice(nameStart, i).toLowerCase(); if (!name) return null;
while (i < s.length) { const c = s[i]; if (c === '>') return { name, closing, selfClosing: false, end: i + 1 }; if (c === '/' && s[i + 1] === '>') { return { name, closing, selfClosing: !closing, end: i + 2 }; } if (/\s/.test(c)) { i++; continue; } if (c === '{') { i = matchBraces(s, i); continue; } // {value} shorthand attr // attribute name while (i < s.length && !/[\s=/>]/.test(s[i])) i++; while (i < s.length && /\s/.test(s[i])) i++; if (s[i] === '=') { i++; while (i < s.length && /\s/.test(s[i])) i++; if (s[i] === '"' || s[i] === "'") { const q = s[i++]; while (i < s.length && s[i] !== q) i++; if (i < s.length) i++; } else { // unquoted value; skip balanced braces so `attr={a ? b : c}` works while (i < s.length && !/[\s>]/.test(s[i])) { if (s[i] === '{') i = matchBraces(s, i); else i++; } } } } return null; // unterminated tag}
// Tokenize source into a flat list.// Recurses into "{...}" template expressions that contain markup, so copy// wrapped in JSX conditionals (e.g. `{cond && (<p>text</p>)}`) stays editable.// Loose JS text outside any element within a recursion is skipped (not copy).function tokenize(source, isTemplate, tokens, start, end) { let depth = 0; // element nesting depth within this tokenize range let i = start;
while (i < end) { const ch = source[i];
if (ch === '<') { if (source.startsWith('<!--', i)) { const ce = source.indexOf('-->', i + 4); if (ce === -1 || ce + 3 > end) break; tokens.push({ type: 'comment', start: i, end: ce + 3 }); i = ce + 3; continue; } if (source.startsWith('<!', i) || source.startsWith('<?', i)) { const ce = source.indexOf('>', i); if (ce === -1 || ce + 1 > end) break; tokens.push({ type: 'comment', start: i, end: ce + 1 }); i = ce + 1; continue; } const tag = scanTag(source, i); if (tag && tag.end <= end) { tokens.push({ type: tag.closing ? 'close' : 'open', name: tag.name, selfClosing: tag.selfClosing, start: i, end: tag.end, }); if (tag.closing) depth = Math.max(0, depth - 1); else if (!tag.selfClosing && !VOID_TAGS.has(tag.name)) depth++; i = tag.end; continue; } // fall through: '<' is literal text } else if (isTemplate && ch === '{') { let ce = matchBraces(source, i); if (ce > end) ce = end; // Skip JSX/Astro comments ({ /* ... */ }) — developer notes, not copy. let k = i + 1; while (k < ce && /\s/.test(source[k])) k++; if (source[k] === '/' && source[k + 1] === '*') { i = ce; continue; } if (ce > i + 1 && containsTag(source, i + 1, ce - 1)) { tokenize(source, isTemplate, tokens, i + 1, ce - 1); } else { tokens.push({ type: 'expr', start: i, end: ce, text: source.slice(i, ce) }); } i = ce; continue; }
// text run let j = i; while (j < end) { const c = source[j]; if (c === '<' && isTagStart(source, j)) break; if (isTemplate && c === '{') break; j++; } if (j === i) { i++; continue; } // safety against infinite loop if (depth > 0) tokens.push({ type: 'text', start: i, end: j, text: source.slice(i, j) }); i = j; }}
// True if [a, b) contains an element start (`<` followed by a letter).function containsTag(s, a, b) { for (let i = a; i < b; i++) { if (s[i] === '<' && /[a-zA-Z]/.test(s[i + 1] || '')) return true; } return false;}
// Offset of the end of a leading `---` frontmatter block, or 0.function frontmatterEnd(source) { if (!source.startsWith('---')) return 0; const nl = source.indexOf('\n'); if (nl === -1) return 0; const first = source.slice(0, nl); if (first.trim() !== '---') return 0; const re = /\r?\n---[ \t]*\r?\n/; const m = re.exec(source.slice(nl)); if (!m) return source.length; // unterminated frontmatter: skip everything return nl + m.index + m[0].length;}
export function parseSource(source, ext) { const isTemplate = ext === '.astro' || ext === '.svelte'; const startAt = isTemplate ? frontmatterEnd(source) : 0; const tokens = []; tokenize(source, isTemplate, tokens, startAt, source.length);
const segments = []; const stack = []; // open elements {name, id} (innermost last) let nextId = 1;
// nearest ancestor that is not an inline (phrase-level) tag const blockOf = () => { for (let i = stack.length - 1; i >= 0; i--) { if (!INLINE_TAGS.has(stack[i].name)) return stack[i]; } return stack.length ? stack[stack.length - 1] : { name: '(root)', id: 0 }; };
for (const t of tokens) { if (t.type === 'open') { if (!t.selfClosing && !VOID_TAGS.has(t.name)) stack.push({ name: t.name, id: nextId++ }); } else if (t.type === 'close') { const idx = stack.findLastIndex((e) => e.name === t.name); if (idx !== -1) stack.length = idx; } else if (t.type === 'text' || t.type === 'expr') { const ws = t.type === 'text' && t.text.trim() === ''; const parent = stack.length ? stack[stack.length - 1].name : '(root)'; const block = blockOf(); if (block.name === '(root)' || NON_CONTENT_PARENTS.has(block.name)) continue; if (stack.some((e) => SKIP_TAGS.has(e.name))) continue; const isText = t.type === 'text'; const preserve = isText && stack.some((e) => PRESERVE_WS_TAGS.has(e.name)); segments.push({ id: 's' + segments.length, kind: t.type, tag: parent, block: block.name, blockId: block.id, ws: ws || undefined, preserve: preserve || undefined, start: t.start, end: t.end, text: t.text, norm: isText && !preserve ? normalizeWs(t.text) : t.text, }); } }
// Trim block-edge whitespace while keeping single-space separators between // adjacent inline segments (e.g. "Drawn by " + <a>@goose.art</a> + "."). const edges = new Map(); // blockId -> { first, last } index of non-ws text for (let i = 0; i < segments.length; i++) { const s = segments[i]; if (s.kind !== 'text' || s.ws || s.preserve) continue; const e = edges.get(s.blockId); if (e) e.last = i; else edges.set(s.blockId, { first: i, last: i }); } for (const e of edges.values()) { const f = segments[e.first]; f.norm = f.norm.replace(/^\s+/, ''); if (e.last !== e.first) segments[e.last].norm = segments[e.last].norm.replace(/\s+$/, ''); else f.norm = f.norm.replace(/\s+$/, ''); }
return segments;}