diff --git a/scripts/content.js b/scripts/content.js index fde202a..05757d2 100644 --- a/scripts/content.js +++ b/scripts/content.js @@ -78,6 +78,9 @@ "style", "noscript", "iframe", + "embed", + "object", + "frame", "nav", "aside", "form", @@ -85,6 +88,37 @@ "input", ]; + /** + * Subtrees to omit from plain-text collection. `textContent` includes script/style/template + * bodies, which pulls in Astro/React/Vite inline bundles when walking `main div` etc. + * Also skip embed-like tags (iframe/object/embed) so news players do not dump URLs or attrs. + */ + const TEXT_SUBTREE_EXCLUDE_TAGS = new Set([ + ...EXCLUDE_TAGS, + "template", + ]); + + function shouldExcludeTextSubtree(tagName) { + return TEXT_SUBTREE_EXCLUDE_TAGS.has(tagName.toLowerCase()); + } + + /** + * Strip embed markup that appears as literal text in article bodies (CMS/oEmbed fallbacks). + * Complements subtree skipping — some sites still surface tags as visible copy. + */ + function sanitizeLiteralEmbedMarkup(text) { + if (!text || typeof text !== "string") { + return text; + } + let t = text; + t = t.replace(//gi, "\n"); + t = t.replace(/]{0,8000}\/?>/gi, "\n"); + t = t.replace(/<\/iframe>/gi, ""); + t = t.replace(/]{0,8000}\/?>/gi, "\n"); + t = t.replace(//gi, "\n"); + return t; + } + function extractWithReadability() { const documentClone = document.cloneNode(true); const reader = new Readability(documentClone); @@ -126,6 +160,7 @@ } let content = article.textContent || ""; + content = sanitizeLiteralEmbedMarkup(content); content = content .replace(/[^\S\n]+/g, " ") @@ -134,7 +169,7 @@ extractedText += content; - const bodyTextLen = (article.textContent || "").trim().length; + const bodyTextLen = content.trim().length; let wasTruncated = false; if (extractedText.length > MAX_LENGTH) { @@ -177,7 +212,7 @@ try { const elements = document.querySelectorAll(selector); for (const el of elements) { - const content = el.textContent.trim(); + const content = getTextContent(el).trim(); if (content.length < 20 || seen.has(content.substring(0, 100))) continue; const style = window.getComputedStyle(el); @@ -200,7 +235,7 @@ if (text.length < 500 && !wasTruncated) { const allParagraphs = document.querySelectorAll("p"); for (const p of allParagraphs) { - const content = p.textContent.trim(); + const content = getTextContent(p).trim(); if (content.length > 30 && !seen.has(content.substring(0, 100))) { const style = window.getComputedStyle(p); if (style.display === "none" || style.visibility === "hidden") continue; @@ -404,6 +439,10 @@ } else if (node.nodeType === Node.ELEMENT_NODE) { const tagName = node.tagName.toLowerCase(); + if (shouldExcludeTextSubtree(tagName)) { + continue; + } + if (["br", "p", "div", "li"].includes(tagName)) { text += " " + getTextContent(node) + " "; } else { @@ -416,7 +455,8 @@ } function cleanExtractedText(text, shouldTruncate = true) { - let cleaned = text + let cleaned = sanitizeLiteralEmbedMarkup(text); + cleaned = cleaned .replace(/[^\S\n]+/g, " ") .replace(/\n{3,}/g, "\n\n") .replace(/^\s+|\s+$/g, "");