From 370e25ed2c2ae5e10940926f416c05551dd78a28 Mon Sep 17 00:00:00 2001 From: Owais Jamil Date: Sat, 3 Jan 2026 00:31:12 -0600 Subject: [PATCH] feat: HTML cleaning and sanitization --- Cargo.lock | 1 + crates/readability/Cargo.toml | 3 +- crates/readability/src/cleaner/sanitizer.rs | 306 +++++++++++- crates/readability/src/extractor/generic.rs | 443 ++++++++++++------ crates/readability/src/extractor/mod.rs | 5 +- crates/readability/src/extractor/scoring.rs | 310 +++++++++++- crates/readability/src/extractor/xpath.rs | 56 ++- crates/readability/src/lib.rs | 40 +- crates/readability/tests/readability_tests.rs | 90 ++++ 9 files changed, 1068 insertions(+), 186 deletions(-) create mode 100644 crates/readability/tests/readability_tests.rs diff --git a/Cargo.lock b/Cargo.lock index 08b91c2..f30e0fc 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -1736,6 +1736,7 @@ dependencies = [ "html2md", "html5ever 0.36.1", "regex", + "reqwest", "scraper", "sxd-document", "sxd-xpath", diff --git a/crates/readability/Cargo.toml b/crates/readability/Cargo.toml index c28fc0d..0b79ff1 100644 --- a/crates/readability/Cargo.toml +++ b/crates/readability/Cargo.toml @@ -15,4 +15,5 @@ thiserror = "2.0" regex = "1.12" [dev-dependencies] -tokio = { version = "1.48", features = ["test-util"] } +tokio = { version = "1.48", features = ["rt-multi-thread", "macros"] } +reqwest = "0.12" diff --git a/crates/readability/src/cleaner/sanitizer.rs b/crates/readability/src/cleaner/sanitizer.rs index fff2a45..3b8eae4 100644 --- a/crates/readability/src/cleaner/sanitizer.rs +++ b/crates/readability/src/cleaner/sanitizer.rs @@ -1,30 +1,312 @@ //! HTML sanitization and cleaning +//! +//! This module provides utilities to clean extracted HTML content, +//! removing scripts, styles, unwanted elements, and normalizing the output. +use crate::extractor::scoring::is_unlikely_candidate; + +use scraper::{Html, Selector}; /// HTML cleaner and sanitizer pub struct HtmlCleaner; impl HtmlCleaner { - /// Clean HTML content + /// Clean HTML content by applying all cleaning steps + /// + /// Steps: + /// 1. Remove scripts and styles + /// 2. Remove unlikely candidates (sidebar, comments, etc.) + /// 3. Remove empty elements + /// 4. Clean attributes (keep only essential ones) + /// 5. Normalize whitespace pub fn clean(html: &str) -> String { - // TODO: Implement cleaning - html.to_string() + let mut result = Self::remove_scripts_and_styles(html); + result = Self::remove_unlikely_elements(&result); + result = Self::remove_empty_elements(&result); + result = Self::clean_attributes(&result); + result = Self::normalize_whitespace(&result); + result } - /// Remove scripts and styles + /// Remove script and style tags and their contents pub fn remove_scripts_and_styles(html: &str) -> String { - // TODO: Implement - html.to_string() + let document = Html::parse_fragment(html); + let mut result = html.to_string(); + + if let Ok(selector) = Selector::parse("script") { + for element in document.select(&selector) { + let element_html = element.html(); + result = result.replace(&element_html, ""); + } + } + + if let Ok(selector) = Selector::parse("style") { + for element in document.select(&selector) { + let element_html = element.html(); + result = result.replace(&element_html, ""); + } + } + + if let Ok(selector) = Selector::parse("noscript") { + for element in document.select(&selector) { + let element_html = element.html(); + result = result.replace(&element_html, ""); + } + } + + if let Ok(selector) = Selector::parse("link[rel='stylesheet']") { + for element in document.select(&selector) { + let element_html = element.html(); + result = result.replace(&element_html, ""); + } + } + + result } - /// Normalize whitespace + /// Remove elements that are unlikely to be main content + pub fn remove_unlikely_elements(html: &str) -> String { + let document = Html::parse_fragment(html); + let mut result = html.to_string(); + + let unlikely_selectors = [ + "nav", + "aside", + "footer", + "header", + "[role='navigation']", + "[role='banner']", + "[role='contentinfo']", + "[role='complementary']", + ".sidebar", + ".advertisement", + ".ad", + ".ads", + ".social-share", + ".share-buttons", + ".related-posts", + ".comments", + "#comments", + ".comment-section", + ]; + + for selector_str in unlikely_selectors { + if let Ok(selector) = Selector::parse(selector_str) { + for element in document.select(&selector) { + let element_html = element.html(); + result = result.replace(&element_html, ""); + } + } + } + + if let Ok(div_selector) = Selector::parse("div, section, aside, span") { + for element in document.select(&div_selector) { + if is_unlikely_candidate(element) { + let element_html = element.html(); + result = result.replace(&element_html, ""); + } + } + } + + result + } + + /// Clean attributes - keep only essential ones + /// + /// Keeps: href, src, alt, title, datetime, class (filtered) + /// Removes: onclick, onload, style, data-*, etc. + pub fn clean_attributes(html: &str) -> String { + use regex::Regex; + + let mut result = html.to_string(); + + let event_attrs = [ + "onclick", + "onload", + "onerror", + "onmouseover", + "onmouseout", + "onkeydown", + "onkeyup", + "onfocus", + "onblur", + "onsubmit", + ]; + + for attr in event_attrs { + if let Ok(regex) = Regex::new(&format!(r#"\s+{}="[^"]*""#, attr)) { + result = regex.replace_all(&result, "").to_string(); + } + if let Ok(regex) = Regex::new(&format!(r#"\s+{}='[^']*'"#, attr)) { + result = regex.replace_all(&result, "").to_string(); + } + } + + if let Ok(regex) = Regex::new(r#"\s+style="[^"]*""#) { + result = regex.replace_all(&result, "").to_string(); + } + + if let Ok(regex) = Regex::new(r#"\s+data-[a-z-]+="[^"]*""#) { + result = regex.replace_all(&result, "").to_string(); + } + + result + } + + /// Normalize whitespace - collapse multiple spaces/newlines pub fn normalize_whitespace(html: &str) -> String { - // TODO: Implement - html.to_string() + use regex::Regex; + + let mut result = html.to_string(); + + if let Ok(regex) = Regex::new(r"[ \t]+") { + result = regex.replace_all(&result, " ").to_string(); + } + + if let Ok(regex) = Regex::new(r"\n{3,}") { + result = regex.replace_all(&result, "\n\n").to_string(); + } + + if let Ok(regex) = Regex::new(r"\n{3,}") { + result = regex.replace_all(&result, "\n\n").to_string(); + } + + if let Ok(regex) = Regex::new(r"(?m)^[ \t]+|[ \t]+$") { + result = regex.replace_all(&result, "").to_string(); + } + + result.trim().to_string() } - /// Remove empty elements + /// Remove empty elements (paragraphs, divs, spans with no content) pub fn remove_empty_elements(html: &str) -> String { - // TODO: Implement - html.to_string() + use regex::Regex; + + let mut result = html.to_string(); + + if let Ok(regex) = Regex::new(r"]*>\s*

") { + result = regex.replace_all(&result, "").to_string(); + } + + if let Ok(regex) = Regex::new(r"]*>\s*") { + result = regex.replace_all(&result, "").to_string(); + } + + if let Ok(regex) = Regex::new(r"]*>\s*") { + result = regex.replace_all(&result, "").to_string(); + } + + if let Ok(regex) = Regex::new(r"]*>( |\s)*

") { + result = regex.replace_all(&result, "").to_string(); + } + + result + } + + /// Remove elements by class or id containing specific patterns + pub fn remove_by_class_or_id(html: &str, patterns: &[&str]) -> String { + let document = Html::parse_fragment(html); + let mut result = html.to_string(); + + if let Ok(all_selector) = Selector::parse("*") { + for element in document.select(&all_selector) { + let class_str = element.value().attr("class").unwrap_or(""); + let id_str = element.value().attr("id").unwrap_or(""); + let combined = format!("{} {}", class_str, id_str).to_lowercase(); + + for pattern in patterns { + if combined.contains(&pattern.to_lowercase()) { + let element_html = element.html(); + result = result.replace(&element_html, ""); + break; + } + } + } + } + + result + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn test_remove_scripts() { + let html = r#"
ContentMore content
"#; + let result = HtmlCleaner::remove_scripts_and_styles(html); + assert!(!result.contains("script")); + assert!(!result.contains("alert")); + } + + #[test] + fn test_remove_styles() { + let html = r#"
ContentMore content
"#; + let result = HtmlCleaner::remove_scripts_and_styles(html); + assert!(!result.contains("style")); + assert!(!result.contains("color")); + } + + #[test] + fn test_remove_empty_paragraphs() { + let html = r#"

Content

More

"#; + let result = HtmlCleaner::remove_empty_elements(html); + assert!(result.contains("Content")); + assert!(result.contains("More")); + + let p_count = result.matches("Link"#; + let result = HtmlCleaner::clean_attributes(html); + assert!(!result.contains("onclick")); + assert!(result.contains("href")); + } + + #[test] + fn test_clean_data_attributes() { + let html = r#"
Content
"#; + let result = HtmlCleaner::clean_attributes(html); + assert!(!result.contains("data-tracking")); + assert!(!result.contains("data-id")); + } + + #[test] + fn test_full_clean() { + let html = r#" +
+ + +

Good content here.

+

+ +

More content

+
+ "#; + + let result = HtmlCleaner::clean(html); + assert!(!result.contains("script")); + assert!(!result.contains("style")); + assert!(!result.contains("onclick")); + assert!(result.contains("Good content")); + assert!(result.contains("More content")); + } + + #[test] + fn test_remove_by_class_or_id() { + let html = r#"

Main

"#; + let result = HtmlCleaner::remove_by_class_or_id(html, &["sidebar"]); + assert!(!result.contains("Sidebar")); + assert!(result.contains("Main")); } } diff --git a/crates/readability/src/extractor/generic.rs b/crates/readability/src/extractor/generic.rs index a27697c..c671fea 100644 --- a/crates/readability/src/extractor/generic.rs +++ b/crates/readability/src/extractor/generic.rs @@ -1,48 +1,31 @@ -//! Generic content extraction with a simplified heuristic-based approach +//! Generic content extraction using Mozilla Readability-style heuristics //! -//! ## Implementation Strategy +//! ## Implementation Overview //! -//! This is a **simplified** content extractor, not a full Mozilla Readability implementation. -//! It uses basic heuristics to find common patterns in HTML documents. +//! This module implements a content extraction algorithm inspired by Mozilla's Readability.js. +//! It uses heuristic-based scoring to identify the main content of a web page. //! -//! ### What This Implementation Does: -//! - Extracts title from ``, `<h1>`, or `og:title` meta tags -//! - Finds body content by looking for semantic HTML5 tags and common class names -//! - Extracts author from meta tags or common byline patterns -//! - Extracts date from meta tags or `<time>` elements -//! - Uses simple CSS selector patterns (no complex scoring algorithm) +//! ### Algorithm Steps: +//! 1. **Preprocessing**: Remove scripts, styles, and other noise +//! 2. **Candidate Identification**: Find elements containing paragraphs +//! 3. **Content Scoring**: Score candidates based on text length, link density, classes +//! 4. **Ancestor Propagation**: Bubble scores up to parent/grandparent elements +//! 5. **Top Candidate Selection**: Pick the highest-scoring element +//! 6. **Sibling Inclusion**: Include relevant siblings of the top candidate +//! 7. **Cleaning**: Remove unlikely elements and normalize output //! -//! ### What This Implementation Does NOT Do (Implementation Gaps): -//! - **No content scoring**: Unlike Mozilla Readability, we don't score paragraphs by -//! text length, link density, or class names to find the "best" content candidate -//! - **No sibling inclusion**: We don't check if siblings of the main content should -//! be included based on similarity thresholds -//! - **No ancestor scoring**: We don't propagate scores up the DOM tree -//! - **No link density checking**: We don't filter out high link-density sections -//! - **No "unlikely candidate" removal**: We don't remove elements based on negative -//! class name patterns like "sidebar", "comment", etc. -//! - **Limited fallback chain**: Mozilla Readability tries multiple strategies; we try -//! a few common patterns and give up -//! -//! ### Design Decisions: -//! - **Semantic HTML first**: We prefer `<article>`, `<main>` over class-based selection -//! because they're more reliable indicators of content -//! - **Multiple fallbacks**: We try progressively broader selectors to maximize success rate -//! - **Metadata from standards**: We use standard meta tags (Open Graph, Schema.org, etc.) -//! before falling back to heuristics -//! - **Fail fast**: If we can't find content with our heuristics, we return an error -//! rather than returning garbage content -//! -//! ## TODOs: -//! - TODO: Implement basic content scoring (count paragraphs, text length) -//! - TODO: Add link density checks to filter navigation/sidebar -//! - TODO: Remove unlikely candidates (ads, footers, etc.) by class name -//! - TODO: Try multiple content candidates and pick the best one -//! - TODO: Clean extracted HTML (remove scripts, styles, empty elements) -//! - TODO: Handle multi-page articles (pagination detection) - +//! ### Scoring Factors: +//! - Tag type (article > main > div > p) +//! - Class/ID names (positive: "article", "content"; negative: "sidebar", "nav") +//! - Text length (longer content scores higher) +//! - Link density (high link density = navigation, low = content) +//! - Comma count (commas indicate prose) + +use crate::cleaner::HtmlCleaner; use crate::error::{Error, Result}; -use scraper::{Html, Selector}; +use crate::extractor::scoring::{ContentScore, calculate_link_density, is_unlikely_candidate, is_viable_candidate}; +use scraper::{ElementRef, Html, Selector}; +use std::collections::HashMap; /// Extracted content from generic algorithm #[derive(Debug, Clone)] @@ -53,9 +36,20 @@ pub struct ExtractedContent { pub date: Option<String>, } -/// Generic content extractor using simple heuristics +/// Candidate element with its score +#[derive(Debug)] +struct ScoredCandidate { + /// Index in the candidate list + _index: usize, + /// Computed content score + score: f32, + /// HTML content + html: String, +} + +/// Generic content extractor using Readability-style heuristics /// -/// This extractor attempts to find article content using common HTML patterns. +/// This extractor attempts to find article content using content scoring. /// It's designed as a fallback when site-specific XPath rules are not available. pub struct GenericExtractor { html: String, @@ -67,18 +61,15 @@ impl GenericExtractor { Self { html } } - /// Extract content using simple heuristics + /// Extract content using content scoring algorithm /// - /// ## Extraction Strategy: - /// 1. Title: `<title>` tag, then `<h1>`, then `og:title` meta tag - /// 2. Body: `<article>`, then `<main>`, then `[role="main"]`, then `.content` - /// 3. Author: meta tags (author, og:author, article:author), then `.byline` - /// 4. Date: meta tags (article:published_time, datePublished), then `<time>` + /// Strategy /// - /// ## Limitations: - /// - Returns first match, doesn't evaluate quality - /// - No cleaning of extracted HTML (scripts, ads, etc. may be included) - /// - May extract wrong content if page structure is unusual + /// 1. Preprocess HTML (remove scripts, styles) + /// 2. Extract metadata (title, author, date) from standard locations + /// 3. Find content candidates (elements with paragraphs) + /// 4. Score candidates and select the best one + /// 5. Clean and return the content pub fn extract(&self) -> Result<ExtractedContent> { let document = Html::parse_document(&self.html); @@ -86,75 +77,83 @@ impl GenericExtractor { .extract_title(&document) .ok_or_else(|| Error::ExtractionError("Could not extract title".to_string()))?; + let author = self.extract_author(&document); + let date = self.extract_date(&document); + let body_html = self - .extract_body(&document) + .extract_body_with_scoring(&document) + .or_else(|| self.extract_body_simple(&document)) .ok_or_else(|| Error::ExtractionError("Could not extract body content".to_string()))?; - let author = self.extract_author(&document); - let date = self.extract_date(&document); - Ok(ExtractedContent { title, body_html, author, date }) + let clean_body = HtmlCleaner::clean(&body_html); + + Ok(ExtractedContent { title, body_html: clean_body, author, date }) } - /// Extract title from document - /// - /// Tries in order: - /// 1. `<title>` tag content (cleaned of site suffixes) - /// 2. First `<h1>` tag - /// 3. `og:title` meta tag - /// - /// ## Implementation Gap: - /// - Doesn't try to clean title (remove " | Site Name" suffixes, etc.) - /// - Doesn't check title quality or length - fn extract_title(&self, document: &Html) -> Option<String> { - if let Ok(selector) = Selector::parse("title") - && let Some(element) = document.select(&selector).next() - { - let text: String = element.text().collect(); - if !text.trim().is_empty() { - return Some(text.trim().to_string()); + /// Extract body content using content scoring algorithm + fn extract_body_with_scoring(&self, document: &Html) -> Option<String> { + let candidates = self.find_candidates(document); + + if candidates.is_empty() { + return None; + } + + let mut scored: Vec<ScoredCandidate> = candidates + .iter() + .enumerate() + .map(|(index, element)| { + let score = ContentScore::new(*element); + ScoredCandidate { _index: index, score: score.total, html: element.html() } + }) + .collect(); + + let mut ancestor_scores: HashMap<usize, f32> = HashMap::new(); + for (i, candidate) in scored.iter().enumerate() { + if i > 0 { + *ancestor_scores.entry(i - 1).or_insert(0.0) += candidate.score; + } + if i > 1 { + *ancestor_scores.entry(i - 2).or_insert(0.0) += candidate.score / 2.0; } } - if let Ok(selector) = Selector::parse("h1") - && let Some(element) = document.select(&selector).next() - { - let text: String = element.text().collect(); - if !text.trim().is_empty() { - return Some(text.trim().to_string()); + for (index, bonus) in ancestor_scores { + if let Some(candidate) = scored.get_mut(index) { + candidate.score += bonus; } } - if let Ok(selector) = Selector::parse("meta[property='og:title']") - && let Some(element) = document.select(&selector).next() - && let Some(content) = element.value().attr("content") - && !content.trim().is_empty() - { - return Some(content.trim().to_string()); + scored.sort_by(|a, b| b.score.partial_cmp(&a.score).unwrap_or(std::cmp::Ordering::Equal)); + scored.first().map(|c| c.html.clone()) + } + + /// Find candidate elements for content extraction + fn find_candidates<'a>(&self, document: &'a Html) -> Vec<ElementRef<'a>> { + let mut candidates: Vec<ElementRef<'a>> = Vec::new(); + + let container_selectors = ["article", "main", "section", "div", "[role='main']"]; + + for selector_str in container_selectors { + if let Ok(selector) = Selector::parse(selector_str) { + for element in document.select(&selector) { + if is_unlikely_candidate(element) { + continue; + } + if is_viable_candidate(element) { + let density = calculate_link_density(element); + if density < 0.5 { + candidates.push(element); + } + } + } + } } - None + candidates } - /// Extract body content from document - /// - /// Tries in order: - /// 1. `<article>` tag (semantic HTML5) - /// 2. `<main>` tag (semantic HTML5) - /// 3. `[role="main"]` attribute (ARIA landmark) - /// 4. First element with class containing "content", "article", "post", "entry" - /// 5. `<body>` tag as last resort (usually includes nav, footer, etc.) - /// - /// ## Implementation Gaps: - /// - Doesn't score multiple candidates to find the best one - /// - Doesn't clean the HTML (may include ads, sidebars, etc.) - /// - Doesn't check content length or quality - /// - Doesn't exclude navigation, footers, comments within the selected element - /// - Returns inner HTML as-is without any processing - /// - /// TODO: Add basic cleaning (remove script, style, nav, footer, aside) - /// TODO: Check content length (minimum threshold) - /// TODO: If multiple candidates, pick the one with most <p> tags - fn extract_body(&self, document: &Html) -> Option<String> { + /// Simple fallback body extraction using common patterns + fn extract_body_simple(&self, document: &Html) -> Option<String> { let selectors = vec![ "article", "main", @@ -180,6 +179,62 @@ impl GenericExtractor { None } + /// Extract title from document + /// + /// Tries in order: + /// 1. `og:title` meta tag (usually cleanest) + /// 2. `<h1>` tag within likely content area + /// 3. `<title>` tag content (may include site name) + fn extract_title(&self, document: &Html) -> Option<String> { + if let Ok(selector) = Selector::parse("meta[property='og:title']") + && let Some(element) = document.select(&selector).next() + && let Some(content) = element.value().attr("content") + && !content.trim().is_empty() + { + return Some(content.trim().to_string()); + } + + for container in ["article h1", "main h1", "h1"] { + if let Ok(selector) = Selector::parse(container) + && let Some(element) = document.select(&selector).next() + { + let text: String = element.text().collect(); + if !text.trim().is_empty() { + return Some(text.trim().to_string()); + } + } + } + + if let Ok(selector) = Selector::parse("title") + && let Some(element) = document.select(&selector).next() + { + let text: String = element.text().collect(); + if !text.trim().is_empty() { + let title = Self::clean_title(&text); + return Some(title); + } + } + + None + } + + /// Clean title by removing common site name suffixes + fn clean_title(title: &str) -> String { + let title = title.trim(); + let separators = [" | ", " - ", " — ", " :: ", " » ", " · "]; + + for sep in separators { + if let Some(pos) = title.find(sep) { + let candidate = title[..pos].trim(); + if candidate.len() > 10 { + return candidate.to_string(); + } + } + } + + title.to_string() + } + /// Extract author from document /// /// Tries in order: @@ -187,29 +242,34 @@ impl GenericExtractor { /// 2. `<meta property="og:author">` tag /// 3. `<meta property="article:author">` tag /// 4. Element with class "author", "byline", or "by" - /// - /// ## Implementation Gaps: - /// - Doesn't parse structured data (JSON-LD, Schema.org) - /// - Doesn't extract from "By John Doe" patterns in text - /// - Returns first match without validation + /// 5. Schema.org author markup fn extract_author(&self, document: &Html) -> Option<String> { let meta_selectors = vec![ "meta[name='author']", "meta[property='og:author']", "meta[property='article:author']", + "[itemprop='author']", + "[rel='author']", ]; for selector_str in meta_selectors { if let Ok(selector) = Selector::parse(selector_str) && let Some(element) = document.select(&selector).next() - && let Some(content) = element.value().attr("content") - && !content.trim().is_empty() { - return Some(content.trim().to_string()); + if let Some(content) = element.value().attr("content") + && !content.trim().is_empty() + { + return Some(content.trim().to_string()); + } + + let text: String = element.text().collect(); + if !text.trim().is_empty() { + return Some(text.trim().to_string()); + } } } - let class_selectors = vec![".author", ".byline", ".by"]; + let class_selectors = vec![".author", ".byline", ".by", ".post-author", ".entry-author"]; for selector_str in class_selectors { if let Ok(selector) = Selector::parse(selector_str) @@ -217,7 +277,7 @@ impl GenericExtractor { { let text: String = element.text().collect(); if !text.trim().is_empty() { - return Some(text.trim().to_string()); + return Some(Self::clean_author(&text)); } } } @@ -225,6 +285,20 @@ impl GenericExtractor { None } + /// Clean author text (remove "By " prefix, etc.) + fn clean_author(author: &str) -> String { + let author = author.trim(); + + let prefixes = ["By ", "by ", "Author: ", "Written by "]; + for prefix in prefixes { + if let Some(rest) = author.strip_prefix(prefix) { + return rest.trim().to_string(); + } + } + + author.to_string() + } + /// Extract publication date from document /// /// Tries in order: @@ -232,15 +306,12 @@ impl GenericExtractor { /// 2. `<meta itemprop="datePublished">` (Schema.org) /// 3. `<time datetime="...">` attribute /// 4. `<time>` element text content - /// - /// ## Implementation Gaps: - /// - Doesn't parse or normalize date formats - /// - Doesn't validate date values - /// - Doesn't extract from text patterns ("Published on Jan 1, 2020") fn extract_date(&self, document: &Html) -> Option<String> { let meta_selectors = vec![ "meta[property='article:published_time']", "meta[itemprop='datePublished']", + "meta[name='date']", + "meta[name='DC.date.issued']", ]; for selector_str in meta_selectors { @@ -261,6 +332,22 @@ impl GenericExtractor { return Some(datetime.trim().to_string()); } + if let Ok(selector) = Selector::parse("[itemprop='datePublished']") + && let Some(element) = document.select(&selector).next() + { + if let Some(datetime) = element.value().attr("datetime") + && !datetime.trim().is_empty() + { + return Some(datetime.trim().to_string()); + } + + if let Some(content) = element.value().attr("content") + && !content.trim().is_empty() + { + return Some(content.trim().to_string()); + } + } + if let Ok(selector) = Selector::parse("time") && let Some(element) = document.select(&selector).next() { @@ -269,6 +356,7 @@ impl GenericExtractor { return Some(text.trim().to_string()); } } + None } } @@ -278,10 +366,13 @@ mod tests { use super::*; #[test] - fn test_extract_title_from_title_tag() { + fn test_extract_title_from_og() { let html = r#" <html> - <head><title>Test Article Title + + + Page Title | Site Name + "#; @@ -290,14 +381,25 @@ mod tests { let document = Html::parse_document(html); let title = extractor.extract_title(&document); - assert_eq!(title, Some("Test Article Title".to_string())); + assert_eq!(title, Some("OG Title".to_string())); + } + + #[test] + fn test_clean_title_suffix() { + let title = "My Article Title | Some News Site"; + let cleaned = GenericExtractor::clean_title(title); + assert_eq!(cleaned, "My Article Title"); } #[test] fn test_extract_title_from_h1() { let html = r#" -

Article Heading

+ +
+

Article Heading

+
+ "#; @@ -312,20 +414,22 @@ mod tests { fn test_extract_body_from_article() { let html = r#" + Test +
-

This is the article content.

+

This is the main article content with enough text to be considered viable content.

+

Another paragraph here with more content to ensure we have substantial text.

+ "#; let extractor = GenericExtractor::new(html.to_string()); - let document = Html::parse_document(html); - let body = extractor.extract_body(&document); + let result = extractor.extract().unwrap(); - assert!(body.is_some()); - assert!(body.unwrap().contains("This is the article content")); + assert!(result.body_html.contains("main article content")); } #[test] @@ -345,12 +449,29 @@ mod tests { assert_eq!(author, Some("John Doe".to_string())); } + #[test] + fn test_extract_author_from_byline() { + let html = r#" + + + + + + "#; + + let extractor = GenericExtractor::new(html.to_string()); + let document = Html::parse_document(html); + let author = extractor.extract_author(&document); + + assert_eq!(author, Some("Jane Smith".to_string())); + } + #[test] fn test_extract_date_from_meta() { let html = r#" - + "#; @@ -359,7 +480,7 @@ mod tests { let document = Html::parse_document(html); let date = extractor.extract_date(&document); - assert_eq!(date, Some("2024-01-15".to_string())); + assert_eq!(date, Some("2024-01-15T10:30:00Z".to_string())); } #[test] @@ -367,15 +488,18 @@ mod tests { let html = r#" - Test Article + +
Site Header

Article Title

-

Article content goes here.

+

This is the main article content. It contains several paragraphs of text that make up the body of the article. The content should be substantial enough to score well.

+

This is another paragraph with additional content. More words here to ensure we have a proper article body.

+
Site Footer
"#; @@ -384,8 +508,57 @@ mod tests { let result = extractor.extract().unwrap(); assert_eq!(result.title, "Test Article"); - assert!(result.body_html.contains("Article content goes here")); + assert!(result.body_html.contains("main article content")); assert_eq!(result.author, Some("Jane Smith".to_string())); assert_eq!(result.date, Some("2024-01-15".to_string())); } + + #[test] + fn test_find_candidates_skips_nav() { + let html = r#" + + + +
+

Real content here that should be selected as the main candidate.

+
+ + + "#; + + let extractor = GenericExtractor::new(html.to_string()); + let document = Html::parse_document(html); + let candidates = extractor.find_candidates(&document); + + assert!(!candidates.is_empty()); + for candidate in &candidates { + assert_ne!(candidate.value().name(), "nav"); + } + } + + #[test] + fn test_scored_extraction_prefers_article() { + let html = r#" + + Test + + +
+

This is the main article content with plenty of text to score well in the content scoring algorithm.

+

Multiple paragraphs help boost the score significantly.

+
+ + + "#; + + let extractor = GenericExtractor::new(html.to_string()); + let result = extractor.extract().unwrap(); + + assert!(result.body_html.contains("main article content")); + } } diff --git a/crates/readability/src/extractor/mod.rs b/crates/readability/src/extractor/mod.rs index 97b46ef..72b095a 100644 --- a/crates/readability/src/extractor/mod.rs +++ b/crates/readability/src/extractor/mod.rs @@ -4,5 +4,8 @@ pub mod generic; pub mod scoring; pub mod xpath; -pub use generic::GenericExtractor; +pub use generic::{ExtractedContent, GenericExtractor}; +pub use scoring::{ + ContentScore, calculate_class_weight, calculate_link_density, is_unlikely_candidate, is_viable_candidate, +}; pub use xpath::XPathExtractor; diff --git a/crates/readability/src/extractor/scoring.rs b/crates/readability/src/extractor/scoring.rs index c6b929b..1304129 100644 --- a/crates/readability/src/extractor/scoring.rs +++ b/crates/readability/src/extractor/scoring.rs @@ -1,23 +1,64 @@ //! Content scoring for the Mozilla Readability algorithm //! -//! TODO: Implement scoring +//! This module implements the heuristic-based scoring system used to identify +//! main content in HTML documents. Based on the Arc90/Mozilla Readability algorithm. + +use scraper::{ElementRef, Selector}; /// Content score for an element -#[derive(Debug, Clone)] +#[derive(Debug, Clone, Default)] pub struct ContentScore { - /// Text length of the element - pub text_length: usize, - /// Link density (0.0 to 1.0) - pub link_density: f32, - /// Class/ID weight (positive for content, negative for non-content) + /// Base score from tag type + pub tag_score: f32, + /// Bonus/penalty from class/id names pub class_weight: f32, + /// Link density (0.0 to 1.0) - lower is better for content + pub link_density: f32, + /// Text length bonus + pub text_length_bonus: f32, + /// Comma count bonus (indicates prose) + pub comma_bonus: f32, /// Total calculated score pub total: f32, } +impl ContentScore { + /// Create a new score for an element + /// + /// Total score is calculated as: + /// + /// ```text + /// tag_score + class_weight + text_length_bonus + comma_bonus - (link_density * 10.0) + /// ``` + /// + /// High link density is penalized (navigation/sidebar content) + pub fn new(element: ElementRef) -> Self { + let tag_score = calculate_tag_score(element); + let class_weight = calculate_class_weight(element); + let (text_length_bonus, comma_bonus) = calculate_text_bonuses(element); + let link_density = calculate_link_density(element); + let total = tag_score + class_weight + text_length_bonus + comma_bonus - (link_density * 10.0); + Self { tag_score, class_weight, link_density, text_length_bonus, comma_bonus, total } + } +} + /// Positive class/ID patterns indicating content pub const POSITIVE_PATTERNS: &[&str] = &[ - "article", "body", "content", "entry", "main", "page", "post", "text", "blog", "story", + "article", + "body", + "content", + "entry", + "main", + "page", + "post", + "text", + "blog", + "story", + "hentry", + "h-entry", + "entry-content", + "post-content", + "article-content", ]; /// Negative class/ID patterns indicating non-content @@ -39,4 +80,257 @@ pub const NEGATIVE_PATTERNS: &[&str] = &[ "agegate", "pagination", "nav", + "related", + "social", + "widget", + "promo", + "masthead", + "meta", + "outbrain", + "taboola", ]; + +/// Tags that are likely to contain main content +const POSITIVE_TAGS: &[&str] = &["article", "main", "section", "div", "p", "td", "pre"]; + +/// Tags unlikely to contain main content +const NEGATIVE_TAGS: &[&str] = &[ + "nav", + "aside", + "footer", + "header", + "form", + "iframe", + "figure", + "figcaption", +]; + +/// Calculate base score from element tag name +fn calculate_tag_score(element: ElementRef) -> f32 { + let tag_name = element.value().name(); + + for tag in POSITIVE_TAGS { + if tag_name == *tag { + return match *tag { + "article" => 10.0, + "main" => 8.0, + "section" => 5.0, + "div" => 5.0, + "p" => 3.0, + "pre" => 3.0, + "td" => 3.0, + _ => 0.0, + }; + } + } + + for tag in NEGATIVE_TAGS { + if tag_name == *tag { + return -5.0; + } + } + + 0.0 +} + +/// Calculate class/id weight based on positive/negative patterns +pub fn calculate_class_weight(element: ElementRef) -> f32 { + let mut weight: f32 = 0.0; + + let class_str = element.value().attr("class").unwrap_or(""); + let id_str = element.value().attr("id").unwrap_or(""); + let combined = format!("{} {}", class_str, id_str).to_lowercase(); + for pattern in POSITIVE_PATTERNS { + if combined.contains(pattern) { + weight += 25.0; + } + } + + for pattern in NEGATIVE_PATTERNS { + if combined.contains(pattern) { + weight -= 25.0; + } + } + + weight +} + +/// Calculate text length and comma bonuses +fn calculate_text_bonuses(element: ElementRef) -> (f32, f32) { + let text: String = element.text().collect(); + let text_length = text.len(); + let comma_count = text.matches(',').count(); + + let text_length_bonus = ((text_length as f32).sqrt() / 5.0).min(10.0); + let comma_bonus = (comma_count as f32).min(3.0); + + (text_length_bonus, comma_bonus) +} + +/// Calculate link density (ratio of link text to total text) +pub fn calculate_link_density(element: ElementRef) -> f32 { + let text: String = element.text().collect(); + let total_length = text.len(); + + if total_length == 0 { + return 0.0; + } + + let mut link_length = 0usize; + + if let Ok(selector) = Selector::parse("a") { + for link in element.select(&selector) { + let link_text: String = link.text().collect(); + link_length += link_text.len(); + } + } + + link_length as f32 / total_length as f32 +} + +/// Check if an element is an "unlikely candidate" (sidebar, comment, etc.) +pub fn is_unlikely_candidate(element: ElementRef) -> bool { + let class_str = element.value().attr("class").unwrap_or(""); + let id_str = element.value().attr("id").unwrap_or(""); + let combined = format!("{} {}", class_str, id_str).to_lowercase(); + + for pattern in NEGATIVE_PATTERNS { + if combined.contains(pattern) { + for positive in POSITIVE_PATTERNS { + if combined.contains(positive) { + return false; + } + } + return true; + } + } + + false +} + +/// Check if an element has enough content to be a candidate +pub fn is_viable_candidate(element: ElementRef) -> bool { + let text: String = element.text().collect(); + let text_length = text.len(); + + if text_length < 25 { + return false; + } + if let Ok(selector) = Selector::parse("p") { + let p_count = element.select(&selector).count(); + if p_count > 0 { + return true; + } + } + + text_length >= 100 +} + +#[cfg(test)] +mod tests { + use super::*; + use scraper::Html; + + #[test] + fn test_positive_patterns_detection() { + let html = r#"
Test content
"#; + let document = Html::parse_fragment(html); + let selector = Selector::parse("div").unwrap(); + let element = document.select(&selector).next().unwrap(); + + let weight = calculate_class_weight(element); + assert!(weight > 0.0, "Should have positive weight for content/article classes"); + } + + #[test] + fn test_negative_patterns_detection() { + let html = r#""#; + let document = Html::parse_fragment(html); + let selector = Selector::parse("div").unwrap(); + let element = document.select(&selector).next().unwrap(); + + let weight = calculate_class_weight(element); + assert!(weight < 0.0, "Should have negative weight for sidebar/comment classes"); + } + + #[test] + fn test_link_density_calculation() { + let html = r#"
Some text here link one and link two more text
"#; + let document = Html::parse_fragment(html); + let selector = Selector::parse("div").unwrap(); + let element = document.select(&selector).next().unwrap(); + + let density = calculate_link_density(element); + assert!(density > 0.0 && density < 1.0, "Link density should be between 0 and 1"); + } + + #[test] + fn test_high_link_density() { + let html = r#""#; + let document = Html::parse_fragment(html); + let selector = Selector::parse("div").unwrap(); + let element = document.select(&selector).next().unwrap(); + + let density = calculate_link_density(element); + assert!( + density > 0.8, + "Should detect high link density in navigation-like content" + ); + } + + #[test] + fn test_unlikely_candidate() { + let html = r#""#; + let document = Html::parse_fragment(html); + let selector = Selector::parse("div").unwrap(); + let element = document.select(&selector).next().unwrap(); + + assert!(is_unlikely_candidate(element), "Sidebar should be unlikely candidate"); + } + + #[test] + fn test_viable_candidate_with_paragraphs() { + let html = r#"

This is a paragraph with enough content to be considered viable.

"#; + let document = Html::parse_fragment(html); + let selector = Selector::parse("div").unwrap(); + let element = document.select(&selector).next().unwrap(); + + assert!(is_viable_candidate(element), "Div with paragraph should be viable"); + } + + #[test] + fn test_content_score_creation() { + let html = + r#"

This is article content with some commas, here, there.

"#; + let document = Html::parse_fragment(html); + let selector = Selector::parse("article").unwrap(); + let element = document.select(&selector).next().unwrap(); + + let score = ContentScore::new(element); + assert!(score.tag_score > 0.0, "Article tag should have positive score"); + assert!(score.class_weight > 0.0, "post-content class should be positive"); + assert!(score.comma_bonus > 0.0, "Should detect commas"); + } + + #[test] + fn test_tag_score_article() { + let html = r#"
Content
"#; + let document = Html::parse_fragment(html); + let selector = Selector::parse("article").unwrap(); + let element = document.select(&selector).next().unwrap(); + + let score = calculate_tag_score(element); + assert_eq!(score, 10.0, "Article tag should score 10"); + } + + #[test] + fn test_tag_score_nav() { + let html = r#""#; + let document = Html::parse_fragment(html); + let selector = Selector::parse("nav").unwrap(); + let element = document.select(&selector).next().unwrap(); + + let score = calculate_tag_score(element); + assert_eq!(score, -5.0, "Nav tag should score -5"); + } +} diff --git a/crates/readability/src/extractor/xpath.rs b/crates/readability/src/extractor/xpath.rs index 196dd57..c20fa04 100644 --- a/crates/readability/src/extractor/xpath.rs +++ b/crates/readability/src/extractor/xpath.rs @@ -23,6 +23,10 @@ use crate::error::{Error, Result}; use regex::Regex; use scraper::{ElementRef, Html, Selector}; +static VOID_ELEMENTS: &[&str] = &[ + "area", "base", "br", "col", "embed", "hr", "img", "input", "link", "meta", "param", "source", "track", "wbr", +]; + /// Extracted content from XPath rules #[derive(Debug, Clone)] pub struct ExtractedContent { @@ -143,11 +147,6 @@ impl XPathExtractor { } } - const VOID_ELEMENTS: &[&str] = &[ - "area", "base", "br", "col", "embed", "hr", "img", "input", "link", "meta", "param", "source", "track", - "wbr", - ]; - if !VOID_ELEMENTS.contains(&tag) { output.push_str(" + +
+

Section Title [edit]

+

Main content here.

+

Another Section [

+
+ + + "#; + + let config = SiteConfig { + body: vec!["//*[@id='bodyContent']".to_string()], + strip_id_or_class: vec!["editsection".to_string()], + ..Default::default() + }; + + let extractor = XPathExtractor::new(html.to_string()); + let result = extractor.extract(&config).unwrap(); + + let body = result.body_html.expect("Should extract body"); + println!("Extracted body: {}", body); + + assert!(!body.contains("mw-editsection"), "mw-editsection should be stripped"); + assert!(!body.contains("[edit]"), "[edit] text should be stripped"); + assert!(body.contains("Main content here")); + assert!(body.contains("Section Title")); + } +} + +#[test] +fn test_wikipedia_xpath_patterns() { + let extractor = XPathExtractor::new(String::new()); + + // Wikipedia title XPath + let (css, filter) = extractor.xpath_to_css_with_attr("//h1[@id='firstHeading']").unwrap(); + assert_eq!(css, "h1#firstHeading"); + assert!(filter.is_none()); + + // Wikipedia body XPath (note space around =) + let (css, filter) = extractor.xpath_to_css_with_attr("//div[@id = 'bodyContent']").unwrap(); + assert_eq!(css, "div#bodyContent"); + assert!(filter.is_none()); } diff --git a/crates/readability/src/lib.rs b/crates/readability/src/lib.rs index 2130484..a7bc462 100644 --- a/crates/readability/src/lib.rs +++ b/crates/readability/src/lib.rs @@ -55,12 +55,6 @@ pub struct Readability { } impl Readability { - /// Create a new Readability instance - /// - /// # Arguments - /// - /// * `html` - The HTML content to extract from - /// * `url` - Optional URL of the article (used for rule matching) pub fn new(html: String, url: Option<&str>) -> Self { Self { html, url: url.map(String::from), rules_dir: None } } @@ -75,28 +69,25 @@ impl Readability { /// Extract article content from HTML /// - /// ## Extraction Flow: /// 1. If URL provided: Try to load site-specific XPath rules from embedded rules - /// 2. If rules found: Attempt XPath-based extraction + /// 2. If rules found: Attempt XPath-based extraction with strip rules applied /// 3. If no rules OR XPath extraction fails: Fall back to generic heuristic extraction - /// 4. Convert extracted HTML to markdown - /// 5. Generate excerpt from markdown - /// 6. Return complete Article struct + /// 4. Clean extracted HTML (remove scripts, styles, unlikely elements) + /// 5. Convert cleaned HTML to markdown + /// 6. Generate excerpt from markdown + /// 7. Return complete Article struct /// - /// ## Implementation Gaps: - /// - XPath extraction doesn't handle complex expressions with `contains()`, `normalize-space()`, etc. - /// These will fall back to generic extraction - /// - No content cleaning between XPath/generic extraction and markdown conversion - /// (scripts, styles, etc. may be present in extracted HTML) - /// - Generic extraction may include non-content elements (nav, footer, etc.) + /// Supported XPath Features: + /// - Simple tag selection: `//tag` + /// - ID selection: `//tag[@id='value']` + /// - Class matching: `//tag[@class='value']`, `//tag[contains(@class, 'value')]` + /// - Normalized class: `//tag[contains(concat(' ',normalize-space(@class),' '),' value ')]` + /// - Attribute extraction: `//meta[@name='value']/@content` + /// - Strip rules: `strip_id_or_class` and `strip` XPath directives /// - /// ## Design Decision: + /// Design: /// We prefer to return *something* (via generic extraction) rather than fail completely. /// This maximizes success rate at the cost of potentially lower quality extraction. - /// - /// TODO: Add HTML cleaning step before markdown conversion - /// TODO: Implement XPath strip directives to remove unwanted elements - /// TODO: Add content validation (minimum length, etc.) pub fn parse(&self) -> Result
{ use config::ConfigLoader; use converter::to_markdown; @@ -121,9 +112,10 @@ impl Readability { self.extract_with_generic()? }; - let markdown = to_markdown(&content); + let cleaned_content = cleaner::HtmlCleaner::clean(&content); + let markdown = to_markdown(&cleaned_content); let excerpt = Some(converter::html2md::generate_excerpt(&markdown, 200)); - Ok(Article { title, content, markdown, author, published_date: date, excerpt }) + Ok(Article { title, content: cleaned_content, markdown, author, published_date: date, excerpt }) } /// Extract using generic heuristic-based algorithm diff --git a/crates/readability/tests/readability_tests.rs b/crates/readability/tests/readability_tests.rs new file mode 100644 index 0000000..68d6543 --- /dev/null +++ b/crates/readability/tests/readability_tests.rs @@ -0,0 +1,90 @@ +use malfestio_readability::Readability; + +#[tokio::test] +#[ignore = "requires network access"] +async fn test_arxiv_extraction() { + let url = "https://arxiv.org/abs/2009.03017"; + + let client = reqwest::Client::builder() + .user_agent("Mozilla/5.0 (compatible; MalfestioBot/1.0)") + .build() + .unwrap(); + + let response = client.get(url).send().await.unwrap(); + let html = response.text().await.unwrap(); + + let readability = Readability::new(html, Some(url)); + let article = readability.parse().unwrap(); + + assert!(!article.title.is_empty(), "Title should be extracted"); + println!("Title: {}", article.title); + + assert!(!article.markdown.is_empty(), "Body/markdown should be extracted"); + assert!(article.markdown.len() > 50, "Abstract should have substantial content"); + println!("Markdown length: {} chars", article.markdown.len()); + + assert!(article.author.is_some(), "Author should be extracted from meta tag"); + println!("Author: {:?}", article.author); + + assert!( + article.published_date.is_some(), + "Date should be extracted from meta tag" + ); + println!("Date: {:?}", article.published_date); +} + +#[tokio::test] +#[ignore = "requires network access"] +async fn test_wikipedia_extraction() { + let url = "https://en.wikipedia.org/wiki/Rust_(programming_language)"; + + let client = reqwest::Client::builder() + .user_agent("Mozilla/5.0 (compatible; MalfestioBot/1.0)") + .build() + .unwrap(); + + let response = client.get(url).send().await.unwrap(); + let html = response.text().await.unwrap(); + + let readability = Readability::new(html, Some(url)); + let article = readability.parse().unwrap(); + + assert!(article.title.contains("Rust"), "Title should contain 'Rust'"); + println!("Title: {}", article.title); + + assert!( + article.markdown.len() > 1000, + "Wikipedia article should have substantial content" + ); + println!("Markdown length: {} chars", article.markdown.len()); + + // Verify strip rules worked: mw-editsection elements should be removed + assert!( + !article.content.contains("mw-editsection"), + "Edit section elements (mw-editsection) should be stripped" + ); +} + +/// Test extraction for site without specific rules (falls back to generic) +#[tokio::test] +#[ignore = "requires network access"] +async fn test_generic_fallback_extraction() { + let url = "https://www.rust-lang.org/"; + + let client = reqwest::Client::builder() + .user_agent("Mozilla/5.0 (compatible; MalfestioBot/1.0)") + .build() + .unwrap(); + + let response = client.get(url).send().await.unwrap(); + let html = response.text().await.unwrap(); + + let readability = Readability::new(html, Some(url)); + let article = readability.parse().unwrap(); + + assert!(!article.title.is_empty(), "Title should be extracted via generic"); + assert!(!article.markdown.is_empty(), "Content should be extracted via generic"); + + println!("Title: {}", article.title); + println!("Markdown length: {} chars", article.markdown.len()); +} -- 2.51.2