diff --git a/Cargo.lock b/Cargo.lock index 34013b7..08b91c2 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -518,14 +518,14 @@ dependencies = [ [[package]] name = "cssparser" -version = "0.34.0" +version = "0.36.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b7c66d1cd8ed61bf80b38432613a7a2f09401ab8d0501110655f8b341484a3e3" +checksum = "dae61cf9c0abb83bd659dab65b7e4e38d8236824c85f0f804f173567bda257d2" dependencies = [ "cssparser-macros", "dtoa-short", "itoa", - "phf 0.11.3", + "phf 0.13.1", "smallvec", ] @@ -649,12 +649,22 @@ dependencies = [ [[package]] name = "derive_more" -version = "0.99.20" +version = "2.1.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6edb4b64a43d977b8e99788fe3a04d483834fba1215a7e02caa415b626497f7f" +checksum = "d751e9e49156b02b44f9c1815bcb94b984cdcc4396ecc32521c739452808b134" +dependencies = [ + "derive_more-impl", +] + +[[package]] +name = "derive_more-impl" +version = "2.1.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "799a97264921d8623a957f6c3b9011f3b5492f557bbb7a5a19b7fa6d06ba8dcb" dependencies = [ "proc-macro2", "quote", + "rustc_version", "syn 2.0.111", ] @@ -681,39 +691,6 @@ dependencies = [ "syn 2.0.111", ] -[[package]] -name = "dom_query" -version = "0.12.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "688b93023aba6768721b48ec5588308e45ac42d788c6dd974d1c2b9a1d04ea29" -dependencies = [ - "cssparser", - "foldhash", - "html5ever 0.29.1", - "precomputed-hash", - "selectors", - "tendril", -] - -[[package]] -name = "dom_smoothie" -version = "0.4.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d23bf500fc0a79f9bf12c38816574820929ecf4f6b39ec07743f7ed485439c31" -dependencies = [ - "dom_query", - "flagset", - "gjson", - "html-escape", - "once_cell", - "phf 0.11.3", - "regex", - "tendril", - "thiserror 2.0.17", - "unicode-segmentation", - "url", -] - [[package]] name = "dotenvy" version = "0.15.7" @@ -774,6 +751,12 @@ dependencies = [ "zeroize", ] +[[package]] +name = "ego-tree" +version = "0.10.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b2972feb8dffe7bc8c5463b1dacda1b0dfbed3710e50f977d965429692d74cd8" + [[package]] name = "elliptic-curve" version = "0.13.8" @@ -869,12 +852,6 @@ version = "0.1.6" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "645cbb3a84e60b7531617d5ae4e57f7e27308f6445f5abf653209ea76dec8dff" -[[package]] -name = "flagset" -version = "0.4.7" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b7ac824320a75a52197e8f2d787f6a38b6718bb6897a35142d749af3c0e8f4fe" - [[package]] name = "fnv" version = "1.0.7" @@ -1010,15 +987,6 @@ dependencies = [ "slab", ] -[[package]] -name = "fxhash" -version = "0.2.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c31b6d751ae2c7f11320402d34e41349dd1016f8d5d45e48c4312bc8625af50c" -dependencies = [ - "byteorder", -] - [[package]] name = "generic-array" version = "0.14.7" @@ -1030,6 +998,15 @@ dependencies = [ "zeroize", ] +[[package]] +name = "getopts" +version = "0.2.24" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cfe4fbac503b8d1f88e6676011885f34b7174f46e59956bba534ba83abded4df" +dependencies = [ + "unicode-width", +] + [[package]] name = "getrandom" version = "0.2.16" @@ -1057,12 +1034,6 @@ dependencies = [ "wasm-bindgen", ] -[[package]] -name = "gjson" -version = "0.8.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "43503cc176394dd30a6525f5f36e838339b8b5619be33ed9a7783841580a97b6" - [[package]] name = "group" version = "0.13.0" @@ -1270,14 +1241,12 @@ dependencies = [ [[package]] name = "html5ever" -version = "0.29.1" +version = "0.36.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3b7410cae13cbc75623c98ac4cbfd1f0bedddf3227afc24f370cf0f50a44a11c" +checksum = "6452c4751a24e1b99c3260d505eaeee76a050573e61f30ac2c924ddc7236f01e" dependencies = [ "log", - "mac", - "markup5ever 0.14.1", - "match_token", + "markup5ever 0.36.1", ] [[package]] @@ -1740,10 +1709,9 @@ version = "0.1.0" dependencies = [ "chrono", "clap", - "dom_smoothie", "dotenvy", - "html2md", "malfestio-core", + "malfestio-readability", "malfestio-server", "reqwest", "tokio", @@ -1760,6 +1728,22 @@ dependencies = [ "thiserror 2.0.17", ] +[[package]] +name = "malfestio-readability" +version = "0.1.0" +dependencies = [ + "html-escape", + "html2md", + "html5ever 0.36.1", + "regex", + "scraper", + "sxd-document", + "sxd-xpath", + "thiserror 2.0.17", + "tokio", + "url", +] + [[package]] name = "malfestio-server" version = "0.1.0" @@ -1771,13 +1755,12 @@ dependencies = [ "base64", "chrono", "deadpool-postgres", - "dom_smoothie", "dotenvy", "ed25519-dalek", "getrandom 0.3.4", "hickory-resolver 0.24.4", - "html2md", "malfestio-core", + "malfestio-readability", "regex", "reqwest", "serde", @@ -1804,24 +1787,21 @@ checksum = "16ce3abbeba692c8b8441d036ef91aea6df8da2c6b6e21c7e14d3c18e526be45" dependencies = [ "log", "phf 0.11.3", - "phf_codegen", - "string_cache", - "string_cache_codegen", + "phf_codegen 0.11.3", + "string_cache 0.8.9", + "string_cache_codegen 0.5.4", "tendril", ] [[package]] name = "markup5ever" -version = "0.14.1" +version = "0.36.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c7a7213d12e1864c0f002f52c2923d4556935a43dec5e71355c2760e0f6e7a18" +checksum = "6c3294c4d74d0742910f8c7b466f44dda9eb2d5742c1e430138df290a1e8451c" dependencies = [ "log", - "phf 0.11.3", - "phf_codegen", - "string_cache", - "string_cache_codegen", "tendril", + "web_atoms", ] [[package]] @@ -1847,17 +1827,6 @@ dependencies = [ "syn 1.0.109", ] -[[package]] -name = "match_token" -version = "0.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "88a9689d8d44bf9964484516275f5cd4c9b59457a6940c1d5d0ecbb94510a36b" -dependencies = [ - "proc-macro2", - "quote", - "syn 2.0.111", -] - [[package]] name = "matchers" version = "0.2.0" @@ -2125,13 +2094,18 @@ version = "2.3.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "9b4f627cb1b25917193a259e49bdad08f671f8d9708acfd5fe0a8c1455d87220" +[[package]] +name = "peresil" +version = "0.3.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f658886ed52e196e850cfbbfddab9eaa7f6d90dd0929e264c31e5cec07e09e57" + [[package]] name = "phf" version = "0.11.3" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "1fd6780a80ae0c52cc120a26a1a42c1ae51b247a253e4e06113d23d2c2edd078" dependencies = [ - "phf_macros", "phf_shared 0.11.3", ] @@ -2141,6 +2115,7 @@ version = "0.13.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "c1562dc717473dbaa4c1f85a36410e03c047b2e7df7f45ee938fbef64ae7fadf" dependencies = [ + "phf_macros", "phf_shared 0.13.1", "serde", ] @@ -2151,10 +2126,20 @@ version = "0.11.3" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "aef8048c789fa5e851558d709946d6d79a8ff88c0440c587967f8e94bfb1216a" dependencies = [ - "phf_generator", + "phf_generator 0.11.3", "phf_shared 0.11.3", ] +[[package]] +name = "phf_codegen" +version = "0.13.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "49aa7f9d80421bca176ca8dbfebe668cc7a2684708594ec9f3c0db0805d5d6e1" +dependencies = [ + "phf_generator 0.13.1", + "phf_shared 0.13.1", +] + [[package]] name = "phf_generator" version = "0.11.3" @@ -2165,14 +2150,24 @@ dependencies = [ "rand 0.8.5", ] +[[package]] +name = "phf_generator" +version = "0.13.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "135ace3a761e564ec88c03a77317a7c6b80bb7f7135ef2544dbe054243b89737" +dependencies = [ + "fastrand", + "phf_shared 0.13.1", +] + [[package]] name = "phf_macros" -version = "0.11.3" +version = "0.13.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f84ac04429c13a7ff43785d75ad27569f2951ce0ffd30a3321230db2fc727216" +checksum = "812f032b54b1e759ccd5f8b6677695d5268c588701effba24601f6932f8269ef" dependencies = [ - "phf_generator", - "phf_shared 0.11.3", + "phf_generator 0.13.1", + "phf_shared 0.13.1", "proc-macro2", "quote", "syn 2.0.111", @@ -2311,6 +2306,12 @@ dependencies = [ "unicode-ident", ] +[[package]] +name = "quick-error" +version = "1.2.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a1d01941d82fa2ab50be1e79e6714289dd7cde78eba4c074bc5a4374f650dfe0" + [[package]] name = "quinn" version = "0.11.9" @@ -2672,6 +2673,21 @@ version = "1.2.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "94143f37725109f92c262ed2cf5e59bce7498c01bcc1502d7b9afe439a4e9f49" +[[package]] +name = "scraper" +version = "0.25.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "93cecd86d6259499c844440546d02f55f3e17bd286e529e48d1f9f67e92315cb" +dependencies = [ + "cssparser", + "ego-tree", + "getopts", + "html5ever 0.36.1", + "precomputed-hash", + "selectors", + "tendril", +] + [[package]] name = "sec1" version = "0.7.3" @@ -2725,19 +2741,19 @@ dependencies = [ [[package]] name = "selectors" -version = "0.26.0" +version = "0.33.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "fd568a4c9bb598e291a08244a5c1f5a8a6650bee243b5b0f8dbb3d9cc1d87fe8" +checksum = "feef350c36147532e1b79ea5c1f3791373e61cbd9a6a2615413b3807bb164fb7" dependencies = [ "bitflags", "cssparser", "derive_more", - "fxhash", "log", "new_debug_unreachable", - "phf 0.11.3", - "phf_codegen", + "phf 0.13.1", + "phf_codegen 0.13.1", "precomputed-hash", + "rustc-hash", "servo_arc", "smallvec", ] @@ -2985,18 +3001,43 @@ dependencies = [ "serde", ] +[[package]] +name = "string_cache" +version = "0.9.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a18596f8c785a729f2819c0f6a7eae6ebeebdfffbfe4214ae6b087f690e31901" +dependencies = [ + "new_debug_unreachable", + "parking_lot", + "phf_shared 0.13.1", + "precomputed-hash", + "serde", +] + [[package]] name = "string_cache_codegen" version = "0.5.4" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "c711928715f1fe0fe509c53b43e993a9a557babc2d0a3567d0a3006f1ac931a0" dependencies = [ - "phf_generator", + "phf_generator 0.11.3", "phf_shared 0.11.3", "proc-macro2", "quote", ] +[[package]] +name = "string_cache_codegen" +version = "0.6.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "585635e46db231059f76c5849798146164652513eb9e8ab2685939dd90f29b69" +dependencies = [ + "phf_generator 0.13.1", + "phf_shared 0.13.1", + "proc-macro2", + "quote", +] + [[package]] name = "stringprep" version = "0.1.5" @@ -3020,6 +3061,27 @@ version = "2.6.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "13c2bddecc57b384dee18652358fb23172facb8a2c51ccc10d74c157bdea3292" +[[package]] +name = "sxd-document" +version = "0.3.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "94d82f37be9faf1b10a82c4bd492b74f698e40082f0f40de38ab275f31d42078" +dependencies = [ + "peresil", + "typed-arena", +] + +[[package]] +name = "sxd-xpath" +version = "0.4.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "36e39da5d30887b5690e29de4c5ebb8ddff64ebd9933f98a01daaa4fd11b36ea" +dependencies = [ + "peresil", + "quick-error", + "sxd-document", +] + [[package]] name = "syn" version = "1.0.109" @@ -3459,6 +3521,12 @@ version = "0.2.5" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "e421abadd41a4225275504ea4d6566923418b7f05506fbc9c0fe86ba7396114b" +[[package]] +name = "typed-arena" +version = "1.7.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a9b2228007eba4120145f785df0f6c92ea538f5a3635a612ecf4e334c8c1446d" + [[package]] name = "typenum" version = "1.19.0" @@ -3493,10 +3561,10 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "7df058c713841ad818f1dc5d3fd88063241cc61f49f5fbea4b951e8cf5a8d71d" [[package]] -name = "unicode-segmentation" -version = "1.12.0" +name = "unicode-width" +version = "0.2.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f6ccf251212114b54433ec949fd6a7841275f9ada20dddd2f29e9ceea4501493" +checksum = "b4ac048d71ede7ee76d585517add45da530660ef4390e49b098733c6e897f254" [[package]] name = "unsigned-varint" @@ -3700,6 +3768,18 @@ dependencies = [ "wasm-bindgen", ] +[[package]] +name = "web_atoms" +version = "0.2.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "acd0c322f146d0f8aad130ce6c187953889359584497dac6561204c8e17bb43d" +dependencies = [ + "phf 0.13.1", + "phf_codegen 0.13.1", + "string_cache 0.9.0", + "string_cache_codegen 0.6.1", +] + [[package]] name = "webpki-roots" version = "1.0.4" diff --git a/Cargo.toml b/Cargo.toml index 73e33e2..c7f90a1 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -1,6 +1,6 @@ [workspace] resolver = "2" -members = ["crates/cli", "crates/core", "crates/server"] +members = ["crates/cli", "crates/core", "crates/readability", "crates/server"] [workspace.lints.clippy] bool_comparison = "deny" diff --git a/crates/cli/Cargo.toml b/crates/cli/Cargo.toml index 6c78435..cbd1f8d 100644 --- a/crates/cli/Cargo.toml +++ b/crates/cli/Cargo.toml @@ -6,10 +6,9 @@ edition = "2024" [dependencies] chrono = "0.4" clap = { version = "4.5.53", features = ["derive"] } -dom_smoothie = "0.4" dotenvy = "0.15.7" -html2md = "0.2.15" malfestio-core = { version = "0.1.0", path = "../core" } +malfestio-readability = { version = "0.1.0", path = "../readability" } malfestio-server = { version = "0.1.0", path = "../server" } reqwest = { version = "0.12", features = ["json"] } tokio = { version = "1.48.0", features = ["full"] } diff --git a/crates/cli/src/main.rs b/crates/cli/src/main.rs index 7f1f717..3548a86 100644 --- a/crates/cli/src/main.rs +++ b/crates/cli/src/main.rs @@ -317,7 +317,7 @@ fn format_time_ago(timestamp: chrono::DateTime) -> String { #[cfg(debug_assertions)] async fn debug_article(url: &str, output_file: Option<&str>) -> malfestio_core::Result<()> { - use dom_smoothie::Readability; + use malfestio_readability::Readability; println!("Fetching article from: {}", url); @@ -327,7 +327,8 @@ async fn debug_article(url: &str, output_file: Option<&str>) -> malfestio_core:: .build() .map_err(|e| malfestio_core::Error::Other(format!("Failed to build client: {}", e)))?; - let response = client.get(url) + let response = client + .get(url) .send() .await .map_err(|e| malfestio_core::Error::Other(format!("Failed to fetch URL: {}", e)))?; @@ -339,58 +340,44 @@ async fn debug_article(url: &str, output_file: Option<&str>) -> malfestio_core:: println!("Fetched {} bytes of HTML", html_content.len()); - // Extract article using dom_smoothie + // Extract article using malfestio-readability println!("Extracting article content..."); let url_clone = url.to_string(); - let result = tokio::task::spawn_blocking( - move || -> Result<(String, String, Option, Option), String> { - let mut readability = Readability::new(html_content, Some(&url_clone), None) - .map_err(|e| format!("Readability error: {}", e))?; - let article = readability.parse().map_err(|e| format!("Parse error: {}", e))?; - Ok(( - article.title, - article.content.to_string(), - article.byline, - article.published_time, - )) - }, - ) + let result = tokio::task::spawn_blocking(move || -> Result { + let readability = Readability::new(html_content, Some(&url_clone)); + readability.parse().map_err(|e| format!("Parse error: {}", e)) + }) .await .map_err(|e| malfestio_core::Error::Other(format!("Task join error: {}", e)))? .map_err(malfestio_core::Error::Other)?; - let (title, content, author, publish_date) = result; + let article = result; println!("✓ Extracted article:"); - println!(" Title: {}", title); - if let Some(author) = &author { + println!(" Title: {}", article.title); + if let Some(ref author) = article.author { println!(" Author: {}", author); } - if let Some(date) = &publish_date { + if let Some(ref date) = article.published_date { println!(" Published: {}", date); } - println!(" Content length: {} bytes", content.len()); - - // Convert HTML to markdown - println!("\nConverting to markdown..."); - let markdown = html2md::parse_html(&content); - println!("✓ Converted to {} bytes of markdown", markdown.len()); + println!(" Content length: {} bytes", article.content.len()); + println!(" Markdown length: {} bytes", article.markdown.len()); - // Output if let Some(file_path) = output_file { println!("\nSaving to file: {}", file_path); let mut output = String::new(); - output.push_str(&format!("# {}\n\n", title)); - if let Some(author) = author { + output.push_str(&format!("# {}\n\n", article.title)); + if let Some(ref author) = article.author { output.push_str(&format!("**Author:** {}\n", author)); } - if let Some(date) = publish_date { + if let Some(ref date) = article.published_date { output.push_str(&format!("**Published:** {}\n", date)); } output.push_str(&format!("**Source:** {}\n\n", url)); output.push_str("---\n\n"); - output.push_str(&markdown); + output.push_str(&article.markdown); fs::write(file_path, output) .map_err(|e| malfestio_core::Error::Other(format!("Failed to write file: {}", e)))?; @@ -398,16 +385,16 @@ async fn debug_article(url: &str, output_file: Option<&str>) -> malfestio_core:: println!("✓ Saved to {}", file_path); } else { println!("\n{}", "=".repeat(80)); - println!("# {}", title); - if let Some(author) = author { + println!("# {}", article.title); + if let Some(ref author) = article.author { println!("\n**Author:** {}", author); } - if let Some(date) = publish_date { + if let Some(ref date) = article.published_date { println!("**Published:** {}", date); } println!("**Source:** {}", url); println!("{}", "=".repeat(80)); - println!("\n{}", markdown); + println!("\n{}", article.markdown); } Ok(()) diff --git a/crates/readability/Cargo.toml b/crates/readability/Cargo.toml new file mode 100644 index 0000000..c28fc0d --- /dev/null +++ b/crates/readability/Cargo.toml @@ -0,0 +1,18 @@ +[package] +name = "malfestio-readability" +version = "0.1.0" +edition = "2024" + +[dependencies] +scraper = "0.25" +html5ever = "0.36" +sxd-document = "0.3" +sxd-xpath = "0.4" +html2md = "0.2.15" +html-escape = "0.2" +url = "2.5" +thiserror = "2.0" +regex = "1.12" + +[dev-dependencies] +tokio = { version = "1.48", features = ["test-util"] } diff --git a/crates/readability/rules/.wikipedia.org.txt b/crates/readability/rules/.wikipedia.org.txt new file mode 100644 index 0000000..7ccd6cf --- /dev/null +++ b/crates/readability/rules/.wikipedia.org.txt @@ -0,0 +1,25 @@ +title: //h1[@id='firstHeading'] +body: //div[@id = 'bodyContent'] +strip_id_or_class: editsection +#strip_id_or_class: toc +strip_id_or_class: vertical-navbox +strip: //*[@id='toc'] +strip: //div[@id='catlinks'] +strip: //div[@id='jump-to-nav'] +strip: //div[@class='thumbcaption']//div[@class='magnify'] +strip: //table[@class='navbox'] +#strip: //table[contains(@class, 'infobox')] +strip: //div[@class='dablink'] +strip: //div[@id='contentSub'] +strip: //table[contains(@class, 'metadata')] +strip: //*[contains(@class, 'noprint')] +strip: //span[@class='noexcerpt'] +strip: //math + +http_header(user-agent): Mozilla/5.2 + +prune: no +tidy: no +test_url: http://en.wikipedia.org/wiki/Christopher_Lloyd +test_url: https://en.wikipedia.org/wiki/Ronnie_James_Dio +test_url: https://en.wikipedia.org/wiki/Metallica diff --git a/crates/readability/rules/arxiv.org.txt b/crates/readability/rules/arxiv.org.txt new file mode 100644 index 0000000..f6d70c3 --- /dev/null +++ b/crates/readability/rules/arxiv.org.txt @@ -0,0 +1,9 @@ +title: //h1[contains(concat(' ',normalize-space(@class),' '),' title ')] + +body: //blockquote[contains(concat(' ',normalize-space(@class),' '),' abstract ')] + +date: //meta[@name='citation_date']/@content +author: //meta[@name='citation_author']/@content + +test_url: https://arxiv.org/abs/2009.03017 +test_url: https://arxiv.org/abs/2012.03780 diff --git a/crates/readability/src/cleaner/mod.rs b/crates/readability/src/cleaner/mod.rs new file mode 100644 index 0000000..8006504 --- /dev/null +++ b/crates/readability/src/cleaner/mod.rs @@ -0,0 +1,5 @@ +//! HTML cleaning and sanitization + +pub mod sanitizer; + +pub use sanitizer::HtmlCleaner; diff --git a/crates/readability/src/cleaner/sanitizer.rs b/crates/readability/src/cleaner/sanitizer.rs new file mode 100644 index 0000000..fff2a45 --- /dev/null +++ b/crates/readability/src/cleaner/sanitizer.rs @@ -0,0 +1,30 @@ +//! HTML sanitization and cleaning + +/// HTML cleaner and sanitizer +pub struct HtmlCleaner; + +impl HtmlCleaner { + /// Clean HTML content + pub fn clean(html: &str) -> String { + // TODO: Implement cleaning + html.to_string() + } + + /// Remove scripts and styles + pub fn remove_scripts_and_styles(html: &str) -> String { + // TODO: Implement + html.to_string() + } + + /// Normalize whitespace + pub fn normalize_whitespace(html: &str) -> String { + // TODO: Implement + html.to_string() + } + + /// Remove empty elements + pub fn remove_empty_elements(html: &str) -> String { + // TODO: Implement + html.to_string() + } +} diff --git a/crates/readability/src/config/embedded_rules.rs b/crates/readability/src/config/embedded_rules.rs new file mode 100644 index 0000000..bb473ac --- /dev/null +++ b/crates/readability/src/config/embedded_rules.rs @@ -0,0 +1,69 @@ +//! Embedded site-specific extraction rules +//! +//! Rules are compiled into the binary at build time for fast access without filesystem dependencies. + +use std::collections::HashMap; + +/// Embedded rule files indexed by domain +/// +/// Supported domains: +/// - arxiv.org +/// - .wikipedia.org (subdomain wildcard) +pub fn get_embedded_rules() -> HashMap<&'static str, &'static str> { + let mut rules = HashMap::new(); + rules.insert("arxiv.org", include_str!("../../rules/arxiv.org.txt")); + rules.insert(".wikipedia.org", include_str!("../../rules/.wikipedia.org.txt")); + rules +} + +/// Get embedded rule content for a domain +pub fn get_rule_for_domain(domain: &str) -> Option<&'static str> { + let rules = get_embedded_rules(); + + if let Some(rule) = rules.get(domain) { + return Some(rule); + } + + let parts: Vec<&str> = domain.split('.').collect(); + if parts.len() > 2 { + let parent_domain = parts[1..].join("."); + let wildcard_key = format!(".{}", parent_domain); + if let Some(rule) = rules.get(wildcard_key.as_str()) { + return Some(rule); + } + } + + None +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn test_embedded_rules_loaded() { + let rules = get_embedded_rules(); + assert!(rules.contains_key("arxiv.org")); + assert!(rules.contains_key(".wikipedia.org")); + } + + #[test] + fn test_get_arxiv_rule() { + let rule = get_rule_for_domain("arxiv.org"); + assert!(rule.is_some()); + assert!(rule.unwrap().contains("title:")); + } + + #[test] + fn test_get_wikipedia_rule_subdomain() { + let rule = get_rule_for_domain("en.wikipedia.org"); + assert!(rule.is_some()); + assert!(rule.unwrap().contains("firstHeading")); + } + + #[test] + fn test_unknown_domain() { + let rule = get_rule_for_domain("unknown.com"); + assert!(rule.is_none()); + } +} diff --git a/crates/readability/src/config/loader.rs b/crates/readability/src/config/loader.rs new file mode 100644 index 0000000..315e75c --- /dev/null +++ b/crates/readability/src/config/loader.rs @@ -0,0 +1,132 @@ +//! Load site-specific configuration files based on URL + +use crate::config::embedded_rules; +use crate::config::parser::{SiteConfig, parse_config}; +use crate::error::Result; +use std::path::{Path, PathBuf}; +use url::Url; + +/// Loads site-specific configuration files +/// +/// First checks embedded rules, then falls back to external rules_dir if provided. +#[derive(Default)] +pub struct ConfigLoader { + rules_dir: Option, +} + +impl ConfigLoader { + /// Create a new config loader with embedded rules only + pub fn new() -> Self { + Self::default() + } + + /// Create a config loader with an external rules directory + /// + /// External rules take precedence over embedded rules. + pub fn with_rules_dir(rules_dir: PathBuf) -> Self { + Self { rules_dir: Some(rules_dir) } + } + + /// Load configuration for a given URL + /// + /// Priority: + /// 1. External rules (if rules_dir provided) + /// 2. Embedded rules + /// 3. None (if no match found) + pub fn load_for_url(&self, url: &str) -> Result> { + let Some(domain) = Self::extract_domain(url) else { + return Ok(None); + }; + + if let Some(ref rules_dir) = self.rules_dir + && let Some(config) = self.try_load_from_dir(rules_dir, &domain)? + { + return Ok(Some(config)); + } + + if let Some(rule_content) = embedded_rules::get_rule_for_domain(&domain) { + return Ok(Some(parse_config(rule_content)?)); + } + + Ok(None) + } + + /// Try to load config from external directory + fn try_load_from_dir(&self, rules_dir: &Path, domain: &str) -> Result> { + let exact_path = rules_dir.join(format!("{}.txt", domain)); + if exact_path.exists() { + let content = std::fs::read_to_string(&exact_path)?; + return Ok(Some(parse_config(&content)?)); + } + + let wildcard_path = rules_dir.join(format!(".{}.txt", domain)); + if wildcard_path.exists() { + let content = std::fs::read_to_string(&wildcard_path)?; + return Ok(Some(parse_config(&content)?)); + } + + if let Some(parent_domain) = Self::extract_parent_domain(domain) { + let parent_wildcard = rules_dir.join(format!(".{}.txt", parent_domain)); + if parent_wildcard.exists() { + let content = std::fs::read_to_string(&parent_wildcard)?; + return Ok(Some(parse_config(&content)?)); + } + } + + Ok(None) + } + + /// Extract domain from URL + fn extract_domain(url: &str) -> Option { + Url::parse(url).ok().and_then(|u| u.host_str().map(String::from)) + } + + /// Extract parent domain (e.g., "en.wikipedia.org" -> "wikipedia.org") + fn extract_parent_domain(domain: &str) -> Option { + let parts: Vec<&str> = domain.split('.').collect(); + if parts.len() > 2 { Some(parts[1..].join(".")) } else { None } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn test_extract_domain() { + assert_eq!( + ConfigLoader::extract_domain("https://arxiv.org/abs/123"), + Some("arxiv.org".to_string()) + ); + assert_eq!( + ConfigLoader::extract_domain("https://en.wikipedia.org/wiki/Article"), + Some("en.wikipedia.org".to_string()) + ); + assert_eq!(ConfigLoader::extract_domain("invalid"), None); + } + + #[test] + fn test_load_embedded_arxiv() { + let loader = ConfigLoader::new(); + let config = loader + .load_for_url("https://arxiv.org/abs/2009.03017") + .unwrap() + .expect("Should find embedded arxiv config"); + + assert_eq!(config.title.len(), 1); + assert_eq!(config.body.len(), 1); + } + + #[test] + fn test_load_embedded_wikipedia() { + let loader = ConfigLoader::new(); + let config = loader + .load_for_url("https://en.wikipedia.org/wiki/Article") + .unwrap() + .expect("Should find embedded wikipedia config"); + + assert_eq!(config.title.len(), 1); + assert_eq!(config.body.len(), 1); + assert!(!config.prune); + } +} diff --git a/crates/readability/src/config/mod.rs b/crates/readability/src/config/mod.rs new file mode 100644 index 0000000..b876c6e --- /dev/null +++ b/crates/readability/src/config/mod.rs @@ -0,0 +1,8 @@ +//! Configuration file parsing and loading for site-specific extraction rules + +pub mod embedded_rules; +pub mod loader; +pub mod parser; + +pub use loader::ConfigLoader; +pub use parser::{SiteConfig, parse_config}; diff --git a/crates/readability/src/config/parser.rs b/crates/readability/src/config/parser.rs new file mode 100644 index 0000000..adf5b76 --- /dev/null +++ b/crates/readability/src/config/parser.rs @@ -0,0 +1,156 @@ +//! Parser for ftr-site-config format extraction rules + +use crate::error::{Error, Result}; + +/// Site-specific extraction configuration +#[derive(Debug, Clone, Default)] +pub struct SiteConfig { + /// XPath expressions for title extraction (evaluated in order) + pub title: Vec, + /// XPath expressions for body extraction + pub body: Vec, + /// XPath expressions for author extraction + pub author: Vec, + /// XPath expressions for date extraction + pub date: Vec, + /// XPath expressions for elements to strip + pub strip: Vec, + /// Substrings to match in @id or @class for stripping + pub strip_id_or_class: Vec, + /// Whether to prune non-content elements (default: true) + pub prune: bool, + /// Whether to run HTML Tidy preprocessor (default: true) + pub tidy: bool, + /// Whether to fall back to generic extraction on failure (default: true) + pub autodetect_on_failure: bool, + /// Test URLs for validation + pub test_urls: Vec, +} + +/// Parse a site configuration file in ftr-site-config format +/// +/// Format: +/// ```text +/// # Comments start with hash +/// directive: value +/// directive: another value +/// +/// # Boolean directives +/// prune: yes +/// tidy: no +/// ``` +pub fn parse_config(content: &str) -> Result { + let mut config = SiteConfig { prune: true, tidy: true, autodetect_on_failure: true, ..Default::default() }; + + for line in content.lines() { + let line = line.trim(); + + if line.is_empty() || line.starts_with('#') { + continue; + } + + if let Some((directive, value)) = line.split_once(':') { + let directive = directive.trim(); + let value = value.trim(); + + match directive { + "title" => config.title.push(value.to_string()), + "body" => config.body.push(value.to_string()), + "author" => config.author.push(value.to_string()), + "date" => config.date.push(value.to_string()), + "strip" => config.strip.push(value.to_string()), + "strip_id_or_class" => config.strip_id_or_class.push(value.to_string()), + "test_url" => config.test_urls.push(value.to_string()), + "prune" => config.prune = parse_bool(value)?, + "tidy" => config.tidy = parse_bool(value)?, + "autodetect_on_failure" => config.autodetect_on_failure = parse_bool(value)?, + // TODO: Implement other directives (like http_header) + _ => {} + } + } + } + + Ok(config) +} + +/// Parse a boolean value (yes/no, true/false, 1/0) +fn parse_bool(value: &str) -> Result { + match value.to_lowercase().as_str() { + "yes" | "true" | "1" => Ok(true), + "no" | "false" | "0" => Ok(false), + _ => Err(Error::ConfigError(format!("Invalid boolean value: {}", value))), + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn test_parse_empty_config() { + let config = parse_config("").unwrap(); + assert!(config.title.is_empty()); + assert!(config.body.is_empty()); + } + + #[test] + fn test_parse_arxiv_config() { + let content = r#" +title: //h1[contains(concat(' ',normalize-space(@class),' '),' title ')] +body: //blockquote[contains(concat(' ',normalize-space(@class),' '),' abstract ')] +date: //meta[@name='citation_date']/@content +author: //meta[@name='citation_author']/@content +test_url: https://arxiv.org/abs/2009.03017 +test_url: https://arxiv.org/abs/2012.03780 + "#; + + let config = parse_config(content).unwrap(); + assert_eq!(config.title.len(), 1); + assert_eq!(config.body.len(), 1); + assert_eq!(config.author.len(), 1); + assert_eq!(config.date.len(), 1); + assert_eq!(config.test_urls.len(), 2); + } + + #[test] + fn test_parse_with_comments() { + let content = r#" +# This is a comment +title: //h1 +# Another comment +body: //article + "#; + + let config = parse_config(content).unwrap(); + assert_eq!(config.title.len(), 1); + assert_eq!(config.body.len(), 1); + } + + #[test] + fn test_parse_boolean_directives() { + let content = r#" +prune: no +tidy: yes +autodetect_on_failure: no + "#; + + let config = parse_config(content).unwrap(); + assert!(!config.prune); + assert!(config.tidy); + assert!(!config.autodetect_on_failure); + } + + #[test] + fn test_parse_strip_directives() { + let content = r#" +strip: //div[@class='sidebar'] +strip: //div[@id='footer'] +strip_id_or_class: advertisement +strip_id_or_class: nav + "#; + + let config = parse_config(content).unwrap(); + assert_eq!(config.strip.len(), 2); + assert_eq!(config.strip_id_or_class.len(), 2); + } +} diff --git a/crates/readability/src/converter/html2md.rs b/crates/readability/src/converter/html2md.rs new file mode 100644 index 0000000..afd442a --- /dev/null +++ b/crates/readability/src/converter/html2md.rs @@ -0,0 +1,39 @@ +//! Markdown conversion using html2md crate + +/// Convert HTML to Markdown +pub fn to_markdown(html: &str) -> String { + html2md::parse_html(html) +} + +/// Generate an excerpt from markdown (first ~200 chars) +pub fn generate_excerpt(markdown: &str, max_length: usize) -> String { + let cleaned: String = markdown.chars().filter(|c| !c.is_control() || *c == '\n').collect(); + + if cleaned.len() <= max_length { + cleaned + } else { + let truncated = &cleaned[..max_length]; + format!("{}...", truncated.trim_end()) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn test_generate_excerpt() { + let markdown = + "This is a long piece of markdown text that should be truncated to approximately 200 characters or so."; + let excerpt = generate_excerpt(markdown, 50); + assert!(excerpt.len() <= 53); + assert!(excerpt.ends_with("...")); + } + + #[test] + fn test_generate_excerpt_short() { + let markdown = "Short text"; + let excerpt = generate_excerpt(markdown, 50); + assert_eq!(excerpt, "Short text"); + } +} diff --git a/crates/readability/src/converter/mod.rs b/crates/readability/src/converter/mod.rs new file mode 100644 index 0000000..d44de63 --- /dev/null +++ b/crates/readability/src/converter/mod.rs @@ -0,0 +1,5 @@ +//! HTML to Markdown conversion + +pub mod html2md; + +pub use self::html2md::to_markdown; diff --git a/crates/readability/src/error.rs b/crates/readability/src/error.rs new file mode 100644 index 0000000..1a1d891 --- /dev/null +++ b/crates/readability/src/error.rs @@ -0,0 +1,23 @@ +use thiserror::Error; + +/// Errors that can occur during article extraction +#[derive(Error, Debug)] +pub enum Error { + #[error("HTML parsing failed: {0}")] + ParseError(String), + + #[error("XPath evaluation failed: {0}")] + XPathError(String), + + #[error("Config parse error: {0}")] + ConfigError(String), + + #[error("Extraction failed: {0}")] + ExtractionError(String), + + #[error("IO error: {0}")] + Io(#[from] std::io::Error), +} + +/// Result type for readability operations +pub type Result = std::result::Result; diff --git a/crates/readability/src/extractor/generic.rs b/crates/readability/src/extractor/generic.rs new file mode 100644 index 0000000..a27697c --- /dev/null +++ b/crates/readability/src/extractor/generic.rs @@ -0,0 +1,391 @@ +//! Generic content extraction with a simplified heuristic-based approach +//! +//! ## Implementation Strategy +//! +//! This is a **simplified** content extractor, not a full Mozilla Readability implementation. +//! It uses basic heuristics to find common patterns in HTML documents. +//! +//! ### What This Implementation Does: +//! - Extracts title from ``, `<h1>`, or `og:title` meta tags +//! - Finds body content by looking for semantic HTML5 tags and common class names +//! - Extracts author from meta tags or common byline patterns +//! - Extracts date from meta tags or `<time>` elements +//! - Uses simple CSS selector patterns (no complex scoring algorithm) +//! +//! ### What This Implementation Does NOT Do (Implementation Gaps): +//! - **No content scoring**: Unlike Mozilla Readability, we don't score paragraphs by +//! text length, link density, or class names to find the "best" content candidate +//! - **No sibling inclusion**: We don't check if siblings of the main content should +//! be included based on similarity thresholds +//! - **No ancestor scoring**: We don't propagate scores up the DOM tree +//! - **No link density checking**: We don't filter out high link-density sections +//! - **No "unlikely candidate" removal**: We don't remove elements based on negative +//! class name patterns like "sidebar", "comment", etc. +//! - **Limited fallback chain**: Mozilla Readability tries multiple strategies; we try +//! a few common patterns and give up +//! +//! ### Design Decisions: +//! - **Semantic HTML first**: We prefer `<article>`, `<main>` over class-based selection +//! because they're more reliable indicators of content +//! - **Multiple fallbacks**: We try progressively broader selectors to maximize success rate +//! - **Metadata from standards**: We use standard meta tags (Open Graph, Schema.org, etc.) +//! before falling back to heuristics +//! - **Fail fast**: If we can't find content with our heuristics, we return an error +//! rather than returning garbage content +//! +//! ## TODOs: +//! - TODO: Implement basic content scoring (count paragraphs, text length) +//! - TODO: Add link density checks to filter navigation/sidebar +//! - TODO: Remove unlikely candidates (ads, footers, etc.) by class name +//! - TODO: Try multiple content candidates and pick the best one +//! - TODO: Clean extracted HTML (remove scripts, styles, empty elements) +//! - TODO: Handle multi-page articles (pagination detection) + +use crate::error::{Error, Result}; +use scraper::{Html, Selector}; + +/// Extracted content from generic algorithm +#[derive(Debug, Clone)] +pub struct ExtractedContent { + pub title: String, + pub body_html: String, + pub author: Option<String>, + pub date: Option<String>, +} + +/// Generic content extractor using simple heuristics +/// +/// This extractor attempts to find article content using common HTML patterns. +/// It's designed as a fallback when site-specific XPath rules are not available. +pub struct GenericExtractor { + html: String, +} + +impl GenericExtractor { + /// Create a new generic extractor + pub fn new(html: String) -> Self { + Self { html } + } + + /// Extract content using simple heuristics + /// + /// ## Extraction Strategy: + /// 1. Title: `<title>` tag, then `<h1>`, then `og:title` meta tag + /// 2. Body: `<article>`, then `<main>`, then `[role="main"]`, then `.content` + /// 3. Author: meta tags (author, og:author, article:author), then `.byline` + /// 4. Date: meta tags (article:published_time, datePublished), then `<time>` + /// + /// ## Limitations: + /// - Returns first match, doesn't evaluate quality + /// - No cleaning of extracted HTML (scripts, ads, etc. may be included) + /// - May extract wrong content if page structure is unusual + pub fn extract(&self) -> Result<ExtractedContent> { + let document = Html::parse_document(&self.html); + + let title = self + .extract_title(&document) + .ok_or_else(|| Error::ExtractionError("Could not extract title".to_string()))?; + + let body_html = self + .extract_body(&document) + .ok_or_else(|| Error::ExtractionError("Could not extract body content".to_string()))?; + + let author = self.extract_author(&document); + let date = self.extract_date(&document); + Ok(ExtractedContent { title, body_html, author, date }) + } + + /// Extract title from document + /// + /// Tries in order: + /// 1. `<title>` tag content (cleaned of site suffixes) + /// 2. First `<h1>` tag + /// 3. `og:title` meta tag + /// + /// ## Implementation Gap: + /// - Doesn't try to clean title (remove " | Site Name" suffixes, etc.) + /// - Doesn't check title quality or length + fn extract_title(&self, document: &Html) -> Option<String> { + if let Ok(selector) = Selector::parse("title") + && let Some(element) = document.select(&selector).next() + { + let text: String = element.text().collect(); + if !text.trim().is_empty() { + return Some(text.trim().to_string()); + } + } + + if let Ok(selector) = Selector::parse("h1") + && let Some(element) = document.select(&selector).next() + { + let text: String = element.text().collect(); + if !text.trim().is_empty() { + return Some(text.trim().to_string()); + } + } + + if let Ok(selector) = Selector::parse("meta[property='og:title']") + && let Some(element) = document.select(&selector).next() + && let Some(content) = element.value().attr("content") + && !content.trim().is_empty() + { + return Some(content.trim().to_string()); + } + + None + } + + /// Extract body content from document + /// + /// Tries in order: + /// 1. `<article>` tag (semantic HTML5) + /// 2. `<main>` tag (semantic HTML5) + /// 3. `[role="main"]` attribute (ARIA landmark) + /// 4. First element with class containing "content", "article", "post", "entry" + /// 5. `<body>` tag as last resort (usually includes nav, footer, etc.) + /// + /// ## Implementation Gaps: + /// - Doesn't score multiple candidates to find the best one + /// - Doesn't clean the HTML (may include ads, sidebars, etc.) + /// - Doesn't check content length or quality + /// - Doesn't exclude navigation, footers, comments within the selected element + /// - Returns inner HTML as-is without any processing + /// + /// TODO: Add basic cleaning (remove script, style, nav, footer, aside) + /// TODO: Check content length (minimum threshold) + /// TODO: If multiple candidates, pick the one with most <p> tags + fn extract_body(&self, document: &Html) -> Option<String> { + let selectors = vec![ + "article", + "main", + "[role='main']", + "[class*='content']", + "[class*='article']", + "[class*='post']", + "[class*='entry']", + "body", + ]; + + for selector_str in selectors { + if let Ok(selector) = Selector::parse(selector_str) + && let Some(element) = document.select(&selector).next() + { + let html = element.html(); + if !html.trim().is_empty() { + return Some(html); + } + } + } + + None + } + + /// Extract author from document + /// + /// Tries in order: + /// 1. `<meta name="author">` tag + /// 2. `<meta property="og:author">` tag + /// 3. `<meta property="article:author">` tag + /// 4. Element with class "author", "byline", or "by" + /// + /// ## Implementation Gaps: + /// - Doesn't parse structured data (JSON-LD, Schema.org) + /// - Doesn't extract from "By John Doe" patterns in text + /// - Returns first match without validation + fn extract_author(&self, document: &Html) -> Option<String> { + let meta_selectors = vec![ + "meta[name='author']", + "meta[property='og:author']", + "meta[property='article:author']", + ]; + + for selector_str in meta_selectors { + if let Ok(selector) = Selector::parse(selector_str) + && let Some(element) = document.select(&selector).next() + && let Some(content) = element.value().attr("content") + && !content.trim().is_empty() + { + return Some(content.trim().to_string()); + } + } + + let class_selectors = vec![".author", ".byline", ".by"]; + + for selector_str in class_selectors { + if let Ok(selector) = Selector::parse(selector_str) + && let Some(element) = document.select(&selector).next() + { + let text: String = element.text().collect(); + if !text.trim().is_empty() { + return Some(text.trim().to_string()); + } + } + } + + None + } + + /// Extract publication date from document + /// + /// Tries in order: + /// 1. `<meta property="article:published_time">` (Open Graph) + /// 2. `<meta itemprop="datePublished">` (Schema.org) + /// 3. `<time datetime="...">` attribute + /// 4. `<time>` element text content + /// + /// ## Implementation Gaps: + /// - Doesn't parse or normalize date formats + /// - Doesn't validate date values + /// - Doesn't extract from text patterns ("Published on Jan 1, 2020") + fn extract_date(&self, document: &Html) -> Option<String> { + let meta_selectors = vec![ + "meta[property='article:published_time']", + "meta[itemprop='datePublished']", + ]; + + for selector_str in meta_selectors { + if let Ok(selector) = Selector::parse(selector_str) + && let Some(element) = document.select(&selector).next() + && let Some(content) = element.value().attr("content") + && !content.trim().is_empty() + { + return Some(content.trim().to_string()); + } + } + + if let Ok(selector) = Selector::parse("time[datetime]") + && let Some(element) = document.select(&selector).next() + && let Some(datetime) = element.value().attr("datetime") + && !datetime.trim().is_empty() + { + return Some(datetime.trim().to_string()); + } + + if let Ok(selector) = Selector::parse("time") + && let Some(element) = document.select(&selector).next() + { + let text: String = element.text().collect(); + if !text.trim().is_empty() { + return Some(text.trim().to_string()); + } + } + None + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn test_extract_title_from_title_tag() { + let html = r#" + <html> + <head><title>Test Article Title + + + "#; + + let extractor = GenericExtractor::new(html.to_string()); + let document = Html::parse_document(html); + let title = extractor.extract_title(&document); + + assert_eq!(title, Some("Test Article Title".to_string())); + } + + #[test] + fn test_extract_title_from_h1() { + let html = r#" + +

Article Heading

+ + "#; + + let extractor = GenericExtractor::new(html.to_string()); + let document = Html::parse_document(html); + let title = extractor.extract_title(&document); + + assert_eq!(title, Some("Article Heading".to_string())); + } + + #[test] + fn test_extract_body_from_article() { + let html = r#" + + +
+

This is the article content.

+
+ + + "#; + + let extractor = GenericExtractor::new(html.to_string()); + let document = Html::parse_document(html); + let body = extractor.extract_body(&document); + + assert!(body.is_some()); + assert!(body.unwrap().contains("This is the article content")); + } + + #[test] + fn test_extract_author_from_meta() { + let html = r#" + + + + + + "#; + + let extractor = GenericExtractor::new(html.to_string()); + let document = Html::parse_document(html); + let author = extractor.extract_author(&document); + + assert_eq!(author, Some("John Doe".to_string())); + } + + #[test] + fn test_extract_date_from_meta() { + let html = r#" + + + + + + "#; + + let extractor = GenericExtractor::new(html.to_string()); + let document = Html::parse_document(html); + let date = extractor.extract_date(&document); + + assert_eq!(date, Some("2024-01-15".to_string())); + } + + #[test] + fn test_full_extraction() { + let html = r#" + + + Test Article + + + + +
+

Article Title

+

Article content goes here.

+
+ + + "#; + + let extractor = GenericExtractor::new(html.to_string()); + let result = extractor.extract().unwrap(); + + assert_eq!(result.title, "Test Article"); + assert!(result.body_html.contains("Article content goes here")); + assert_eq!(result.author, Some("Jane Smith".to_string())); + assert_eq!(result.date, Some("2024-01-15".to_string())); + } +} diff --git a/crates/readability/src/extractor/mod.rs b/crates/readability/src/extractor/mod.rs new file mode 100644 index 0000000..97b46ef --- /dev/null +++ b/crates/readability/src/extractor/mod.rs @@ -0,0 +1,8 @@ +//! Content extraction using XPath rules and generic algorithms + +pub mod generic; +pub mod scoring; +pub mod xpath; + +pub use generic::GenericExtractor; +pub use xpath::XPathExtractor; diff --git a/crates/readability/src/extractor/scoring.rs b/crates/readability/src/extractor/scoring.rs new file mode 100644 index 0000000..c6b929b --- /dev/null +++ b/crates/readability/src/extractor/scoring.rs @@ -0,0 +1,42 @@ +//! Content scoring for the Mozilla Readability algorithm +//! +//! TODO: Implement scoring + +/// Content score for an element +#[derive(Debug, Clone)] +pub struct ContentScore { + /// Text length of the element + pub text_length: usize, + /// Link density (0.0 to 1.0) + pub link_density: f32, + /// Class/ID weight (positive for content, negative for non-content) + pub class_weight: f32, + /// Total calculated score + pub total: f32, +} + +/// Positive class/ID patterns indicating content +pub const POSITIVE_PATTERNS: &[&str] = &[ + "article", "body", "content", "entry", "main", "page", "post", "text", "blog", "story", +]; + +/// Negative class/ID patterns indicating non-content +pub const NEGATIVE_PATTERNS: &[&str] = &[ + "combx", + "comment", + "community", + "disqus", + "extra", + "footer", + "header", + "menu", + "remark", + "rss", + "share", + "sidebar", + "sponsor", + "ad-", + "agegate", + "pagination", + "nav", +]; diff --git a/crates/readability/src/extractor/xpath.rs b/crates/readability/src/extractor/xpath.rs new file mode 100644 index 0000000..196dd57 --- /dev/null +++ b/crates/readability/src/extractor/xpath.rs @@ -0,0 +1,494 @@ +//! XPath-based content extraction using site-specific rules +//! +//! This module provides content extraction from HTML documents using XPath-like expressions. +//! +//! ## Strategy +//! +//! Since Rust doesn't have a robust HTML-compatible XPath library, we use a hybrid approach: +//! 1. Convert simple XPath expressions to CSS selectors (scraper handles these well) +//! 2. Handle complex patterns (contains(), normalize-space()) with custom matchers +//! 3. Use regex parsing for XPath syntax to extract selector components +//! +//! ## Supported XPath Patterns +//! +//! - `//tag` - Simple tag selection +//! - `//tag[@id='value']` - ID selection +//! - `//tag[@class='value']` - Exact class match +//! - `//tag[contains(@class, 'value')]` - Class contains match +//! - `//tag[contains(concat(' ',normalize-space(@class),' '),' value ')]` - Normalized class match +//! - `//meta[@name='value']/@content` - Attribute extraction from meta tags + +use crate::config::SiteConfig; +use crate::error::{Error, Result}; +use regex::Regex; +use scraper::{ElementRef, Html, Selector}; + +/// Extracted content from XPath rules +#[derive(Debug, Clone)] +pub struct ExtractedContent { + pub title: Option, + pub body_html: Option, + pub author: Option, + pub date: Option, +} + +/// XPath-based extractor +pub struct XPathExtractor { + html: String, +} + +impl XPathExtractor { + /// Create a new XPath extractor + pub fn new(html: String) -> Self { + Self { html } + } + + /// Extract content using site-specific rules + pub fn extract(&self, config: &SiteConfig) -> Result { + let cleaned_html = self.apply_strip_rules(&self.html, config)?; + let document = Html::parse_document(&cleaned_html); + + let title = self.extract_field(&document, &config.title, false)?; + let body_html = self.extract_field(&document, &config.body, true)?; + let author = self.extract_field(&document, &config.author, false)?; + let date = self.extract_field(&document, &config.date, false)?; + + Ok(ExtractedContent { title, body_html, author, date }) + } + + /// Apply strip rules to remove unwanted elements + /// + /// Processes both `strip` (XPath) and `strip_id_or_class` (substring match) directives. + fn apply_strip_rules(&self, html: &str, config: &SiteConfig) -> Result { + let document = Html::parse_document(html); + let mut elements_to_remove: Vec = Vec::new(); + + for substring in &config.strip_id_or_class { + let substring_lower = substring.to_lowercase(); + for element in document.tree.nodes() { + if let Some(el) = ElementRef::wrap(element) { + let should_remove = el + .value() + .id() + .is_some_and(|id| id.to_lowercase().contains(&substring_lower)) + || el + .value() + .classes() + .any(|class| class.to_lowercase().contains(&substring_lower)); + + if should_remove { + elements_to_remove.push(self.element_signature(&el)); + } + } + } + } + + for xpath in &config.strip { + if let Some((css, _)) = self.xpath_to_css_with_attr(xpath) + && let Ok(selector) = Selector::parse(&css) + { + for el in document.select(&selector) { + elements_to_remove.push(self.element_signature(&el)); + } + } + } + + self.rebuild_html_without_elements(&document, &elements_to_remove) + } + + /// Generate a signature for an element to identify it during rebuild + fn element_signature(&self, el: &ElementRef) -> String { + let tag = el.value().name(); + let id = el.value().id().unwrap_or(""); + let classes: Vec<&str> = el.value().classes().collect(); + format!("{}#{}#{}", tag, id, classes.join(",")) + } + + /// Rebuild HTML without specified elements + fn rebuild_html_without_elements(&self, document: &Html, to_remove: &[String]) -> Result { + if to_remove.is_empty() { + return Ok(self.html.clone()); + } + + let mut result = String::new(); + self.rebuild_node(&document.root_element(), to_remove, &mut result); + Ok(result) + } + + /// Recursively rebuild a node and its children, skipping removed elements + fn rebuild_node(&self, element: &ElementRef, to_remove: &[String], output: &mut String) { + let sig = self.element_signature(element); + if to_remove.contains(&sig) { + return; + } + + let tag = element.value().name(); + output.push('<'); + output.push_str(tag); + + for (name, value) in element.value().attrs() { + output.push(' '); + output.push_str(name); + output.push_str("=\""); + output.push_str(&html_escape::encode_double_quoted_attribute(value)); + output.push('"'); + } + output.push('>'); + + for child in element.children() { + if let Some(el) = ElementRef::wrap(child) { + self.rebuild_node(&el, to_remove, output); + } else if let Some(text) = child.value().as_text() { + output.push_str(&html_escape::encode_text(&text.to_string())); + } + } + + const VOID_ELEMENTS: &[&str] = &[ + "area", "base", "br", "col", "embed", "hr", "img", "input", "link", "meta", "param", "source", "track", + "wbr", + ]; + + if !VOID_ELEMENTS.contains(&tag) { + output.push_str("'); + } + } + + /// Extract a field using XPath expressions (tries each in order) + fn extract_field(&self, document: &Html, xpaths: &[String], extract_html: bool) -> Result> { + for xpath_expr in xpaths { + if let Some(result) = self.evaluate_xpath(document, xpath_expr, extract_html)? { + return Ok(Some(result)); + } + } + Ok(None) + } + + /// Evaluate an XPath expression against the document + fn evaluate_xpath(&self, document: &Html, xpath: &str, extract_html: bool) -> Result> { + let (xpath_part, attr_to_extract) = if let Some(pos) = xpath.rfind("/@") { + (&xpath[..pos], Some(&xpath[pos + 2..])) + } else { + (xpath, None) + }; + + let (css, class_filter) = match self.xpath_to_css_with_attr(xpath_part) { + Some(result) => result, + None => return Ok(None), + }; + + let selector = + Selector::parse(&css).map_err(|e| Error::XPathError(format!("Invalid CSS selector '{}': {:?}", css, e)))?; + + for element in document.select(&selector) { + if let Some(ref filter) = class_filter + && !self.element_has_class_containing(&element, filter) + { + continue; + } + + if let Some(attr) = attr_to_extract { + if let Some(value) = element.value().attr(attr) { + return Ok(Some(value.to_string())); + } + continue; + } + + let content = + if extract_html { element.inner_html() } else { element.text().collect::>().join(" ") }; + + let content = content.trim().to_string(); + if !content.is_empty() { + return Ok(Some(content)); + } + } + + Ok(None) + } + + /// Convert XPath to CSS selector with optional class filter + fn xpath_to_css_with_attr(&self, xpath: &str) -> Option<(String, Option)> { + let xpath = xpath.trim(); + + if xpath.starts_with("//") && !xpath.contains('[') && !xpath.contains('@') { + let tag = xpath.trim_start_matches("//"); + return Some((tag.to_string(), None)); + } + + if let Some(css) = self.parse_id_selector(xpath) { + return Some((css, None)); + } + + if let Some((css, class_filter)) = self.parse_contains_class_normalized(xpath) { + return Some((css, Some(class_filter))); + } + + if let Some((css, class_filter)) = self.parse_contains_class_simple(xpath) { + return Some((css, Some(class_filter))); + } + + if let Some(css) = self.parse_exact_class(xpath) { + return Some((css, None)); + } + + if let Some(css) = self.parse_exact_class(xpath) { + return Some((css, None)); + } + + if let Some(css) = self.parse_any_tag_with_id(xpath) { + return Some((css, None)); + } + + if let Some(css) = self.parse_meta_selector(xpath) { + return Some((css, None)); + } + + if let Some(css) = self.parse_meta_selector(xpath) { + return Some((css, None)); + } + + None + } + + /// Parse //tag[@id='value'] pattern + fn parse_id_selector(&self, xpath: &str) -> Option { + let re = Regex::new(r#"//(\w+)\[@id\s*=\s*['"]([^'"]+)['"]\]"#).ok()?; + let caps = re.captures(xpath)?; + let tag = caps.get(1)?.as_str(); + let id = caps.get(2)?.as_str(); + Some(format!("{}#{}", tag, id)) + } + + /// Parse //*[@id='value'] pattern + fn parse_any_tag_with_id(&self, xpath: &str) -> Option { + let re = Regex::new(r#"//\*\[@id\s*=\s*['"]([^'"]+)['"]\]"#).ok()?; + let caps = re.captures(xpath)?; + let id = caps.get(1)?.as_str(); + Some(format!("#{}", id)) + } + + /// Parse //tag[@class='value'] pattern (exact class match) + fn parse_exact_class(&self, xpath: &str) -> Option { + if xpath.contains("contains") { + return None; + } + let re = Regex::new(r#"//(\w+)\[@class\s*=\s*['"]([^'"]+)['"]\]"#).ok()?; + let caps = re.captures(xpath)?; + let tag = caps.get(1)?.as_str(); + let class = caps.get(2)?.as_str(); + Some(format!("{}[class=\"{}\"]", tag, class)) + } + + /// Parse //tag[contains(@class, 'value')] pattern + fn parse_contains_class_simple(&self, xpath: &str) -> Option<(String, String)> { + let re = Regex::new(r#"//(\w+)\[contains\s*\(\s*@class\s*,\s*['"]([^'"]+)['"]\s*\)\]"#).ok()?; + let caps = re.captures(xpath)?; + let tag = caps.get(1)?.as_str(); + let class_substr = caps.get(2)?.as_str(); + Some((tag.to_string(), class_substr.to_string())) + } + + /// Parse //tag[contains(concat(' ',normalize-space(@class),' '),' value ')] pattern + fn parse_contains_class_normalized(&self, xpath: &str) -> Option<(String, String)> { + let re = Regex::new(r#"//(\w+)\[contains\s*\(\s*concat\s*\(.+\)\s*,\s*['"]([^'"]+)['"]\s*\)\]"#).ok()?; + let caps = re.captures(xpath)?; + let tag = caps.get(1)?.as_str(); + let class_name = caps.get(2)?.as_str().trim(); + Some((tag.to_string(), class_name.to_string())) + } + + /// Parse //meta[@name='value'] pattern + fn parse_meta_selector(&self, xpath: &str) -> Option { + let re = Regex::new(r#"//meta\[@(\w+)\s*=\s*['"]([^'"]+)['"]\]"#).ok()?; + let caps = re.captures(xpath)?; + let attr_name = caps.get(1)?.as_str(); + let attr_value = caps.get(2)?.as_str(); + Some(format!("meta[{}=\"{}\"]", attr_name, attr_value)) + } + + /// Check if element has a class containing the given substring + fn element_has_class_containing(&self, element: &ElementRef, class_filter: &str) -> bool { + element.value().classes().any(|class| class.contains(class_filter)) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::config::parser::SiteConfig; + + #[test] + fn test_xpath_to_css_simple_tag() { + let extractor = XPathExtractor::new(String::new()); + let (css, filter) = extractor.xpath_to_css_with_attr("//h1").unwrap(); + assert_eq!(css, "h1"); + assert!(filter.is_none()); + } + + #[test] + fn test_xpath_to_css_id_selector() { + let extractor = XPathExtractor::new(String::new()); + let (css, filter) = extractor.xpath_to_css_with_attr("//h1[@id='firstHeading']").unwrap(); + assert_eq!(css, "h1#firstHeading"); + assert!(filter.is_none()); + } + + #[test] + fn test_xpath_to_css_any_tag_with_id() { + let extractor = XPathExtractor::new(String::new()); + let (css, filter) = extractor.xpath_to_css_with_attr("//*[@id='bodyContent']").unwrap(); + assert_eq!(css, "#bodyContent"); + assert!(filter.is_none()); + } + + #[test] + fn test_xpath_contains_class_simple() { + let extractor = XPathExtractor::new(String::new()); + let (css, filter) = extractor + .xpath_to_css_with_attr("//div[contains(@class, 'content')]") + .unwrap(); + assert_eq!(css, "div"); + assert_eq!(filter, Some("content".to_string())); + } + + #[test] + fn test_xpath_contains_class_normalized() { + let extractor = XPathExtractor::new(String::new()); + let xpath = "//h1[contains(concat(' ',normalize-space(@class),' '),' title ')]"; + let (css, filter) = extractor.xpath_to_css_with_attr(xpath).unwrap(); + assert_eq!(css, "h1"); + assert_eq!(filter, Some("title".to_string())); + } + + #[test] + fn test_extract_meta_attribute() { + let html = r#" + + + + + + + "#; + + let extractor = XPathExtractor::new(html.to_string()); + let document = Html::parse_document(html); + + let date = extractor + .evaluate_xpath(&document, "//meta[@name='citation_date']/@content", false) + .unwrap(); + assert_eq!(date, Some("2020-09-07".to_string())); + + let author = extractor + .evaluate_xpath(&document, "//meta[@name='citation_author']/@content", false) + .unwrap(); + assert_eq!(author, Some("John Doe".to_string())); + } + + #[test] + fn test_extract_with_contains_class() { + let html = r#" + + +

Article Title

+
Content here
+ + + "#; + + let extractor = XPathExtractor::new(html.to_string()); + let document = Html::parse_document(html); + + let title = extractor + .evaluate_xpath(&document, "//h1[contains(@class, 'title')]", false) + .unwrap(); + assert_eq!(title, Some("Article Title".to_string())); + } + + #[test] + fn test_strip_id_or_class() { + let html = r#" + + +
Main content
+ + + + + "#; + + let config = SiteConfig { + strip_id_or_class: vec!["sidebar".to_string(), "advertisement".to_string()], + ..Default::default() + }; + + let extractor = XPathExtractor::new(html.to_string()); + let cleaned = extractor.apply_strip_rules(html, &config).unwrap(); + + assert!(cleaned.contains("Main content")); + assert!(!cleaned.contains("Sidebar")); + assert!(!cleaned.contains("Ad")); + } + + #[test] + fn test_strip_xpath() { + let html = r#" + + +
Main content
+
Table of contents
+ + + + "#; + + let config = SiteConfig { + strip: vec!["//*[@id='toc']".to_string(), "//div[@id='footer']".to_string()], + ..Default::default() + }; + + let extractor = XPathExtractor::new(html.to_string()); + let cleaned = extractor.apply_strip_rules(html, &config).unwrap(); + + assert!(cleaned.contains("Main content")); + assert!(!cleaned.contains("Table of contents")); + assert!(!cleaned.contains("Footer")); + } + + #[test] + fn test_full_extraction() { + let html = r#" + + + + + + +

Test Title

+
+

Article content here.

+
+ + + + "#; + + let config = SiteConfig { + title: vec!["//h1[@id='title']".to_string()], + body: vec!["//article".to_string()], + author: vec!["//meta[@name='author']/@content".to_string()], + date: vec!["//meta[@name='date']/@content".to_string()], + strip_id_or_class: vec!["sidebar".to_string()], + ..Default::default() + }; + + let extractor = XPathExtractor::new(html.to_string()); + let result = extractor.extract(&config).unwrap(); + + assert_eq!(result.title, Some("Test Title".to_string())); + assert!(result.body_html.unwrap().contains("Article content here")); + assert_eq!(result.author, Some("Test Author".to_string())); + assert_eq!(result.date, Some("2024-01-15".to_string())); + } +} diff --git a/crates/readability/src/lib.rs b/crates/readability/src/lib.rs new file mode 100644 index 0000000..2130484 --- /dev/null +++ b/crates/readability/src/lib.rs @@ -0,0 +1,135 @@ +//! Article extraction library with support for site-specific XPath rules and generic content extraction. +//! +//! This crate provides functionality to extract clean article content from HTML pages using: +//! - Site-specific XPath rules (ftr-site-config format) +//! - Generic content extraction (Mozilla Readability algorithm) +//! - Automatic markdown conversion +//! +//! # Example +//! +//! ```no_run +//! use malfestio_readability::Readability; +//! use std::path::PathBuf; +//! +//! let html = r#"Article..."#; +//! let readability = Readability::new(html.to_string(), Some("https://example.com/article")) +//! .with_rules_dir(PathBuf::from("rules")); +//! +//! let article = readability.parse().unwrap(); +//! println!("Title: {}", article.title); +//! println!("Markdown: {}", article.markdown); +//! ``` + +pub mod cleaner; +pub mod config; +pub mod converter; +pub mod error; +pub mod extractor; + +use std::path::PathBuf; + +pub use error::{Error, Result}; + +/// Extracted article content +#[derive(Debug, Clone)] +pub struct Article { + /// Article title + pub title: String, + /// Clean HTML content + pub content: String, + /// Markdown formatted content + pub markdown: String, + /// Article author (if found) + pub author: Option, + /// Publication date (if found) + pub published_date: Option, + /// Excerpt (first ~200 chars of content) + pub excerpt: Option, +} + +/// Main entry point for article extraction +pub struct Readability { + html: String, + url: Option, + rules_dir: Option, +} + +impl Readability { + /// Create a new Readability instance + /// + /// # Arguments + /// + /// * `html` - The HTML content to extract from + /// * `url` - Optional URL of the article (used for rule matching) + pub fn new(html: String, url: Option<&str>) -> Self { + Self { html, url: url.map(String::from), rules_dir: None } + } + + /// Set the directory containing extraction rules + /// + /// Rules files should be named `domain.com.txt` or `.domain.com.txt` for subdomain matching. + pub fn with_rules_dir(mut self, path: PathBuf) -> Self { + self.rules_dir = Some(path); + self + } + + /// Extract article content from HTML + /// + /// ## Extraction Flow: + /// 1. If URL provided: Try to load site-specific XPath rules from embedded rules + /// 2. If rules found: Attempt XPath-based extraction + /// 3. If no rules OR XPath extraction fails: Fall back to generic heuristic extraction + /// 4. Convert extracted HTML to markdown + /// 5. Generate excerpt from markdown + /// 6. Return complete Article struct + /// + /// ## Implementation Gaps: + /// - XPath extraction doesn't handle complex expressions with `contains()`, `normalize-space()`, etc. + /// These will fall back to generic extraction + /// - No content cleaning between XPath/generic extraction and markdown conversion + /// (scripts, styles, etc. may be present in extracted HTML) + /// - Generic extraction may include non-content elements (nav, footer, etc.) + /// + /// ## Design Decision: + /// We prefer to return *something* (via generic extraction) rather than fail completely. + /// This maximizes success rate at the cost of potentially lower quality extraction. + /// + /// TODO: Add HTML cleaning step before markdown conversion + /// TODO: Implement XPath strip directives to remove unwanted elements + /// TODO: Add content validation (minimum length, etc.) + pub fn parse(&self) -> Result
{ + use config::ConfigLoader; + use converter::to_markdown; + use extractor::XPathExtractor; + + let (title, content, author, date) = if let Some(ref url) = self.url { + let loader = ConfigLoader::new(); + + if let Some(config) = loader.load_for_url(url)? { + let xpath_extractor = XPathExtractor::new(self.html.clone()); + let xpath_result = xpath_extractor.extract(&config)?; + + if let (Some(title), Some(body)) = (xpath_result.title, xpath_result.body_html) { + (title, body, xpath_result.author, xpath_result.date) + } else { + self.extract_with_generic()? + } + } else { + self.extract_with_generic()? + } + } else { + self.extract_with_generic()? + }; + + let markdown = to_markdown(&content); + let excerpt = Some(converter::html2md::generate_excerpt(&markdown, 200)); + Ok(Article { title, content, markdown, author, published_date: date, excerpt }) + } + + /// Extract using generic heuristic-based algorithm + fn extract_with_generic(&self) -> Result<(String, String, Option, Option)> { + let generic_extractor = extractor::GenericExtractor::new(self.html.clone()); + let result = generic_extractor.extract()?; + Ok((result.title, result.body_html, result.author, result.date)) + } +} diff --git a/crates/server/Cargo.toml b/crates/server/Cargo.toml index 67d0614..251d96d 100644 --- a/crates/server/Cargo.toml +++ b/crates/server/Cargo.toml @@ -13,10 +13,9 @@ chrono = { version = "0.4.42", features = ["serde"] } deadpool-postgres = "0.14.0" dotenvy = "0.15.7" ed25519-dalek = { version = "2.2.0", features = ["serde"] } -dom_smoothie = "0.4" getrandom = { version = "0.3", features = ["std"] } -html2md = "0.2.15" malfestio-core = { version = "0.1.0", path = "../core" } +malfestio-readability = { version = "0.1.0", path = "../readability" } regex = "1.12.2" reqwest = { version = "0.12", features = ["json"] } serde = "1.0.228" diff --git a/crates/server/src/api/importer.rs b/crates/server/src/api/importer.rs index 96388d7..99b7f1a 100644 --- a/crates/server/src/api/importer.rs +++ b/crates/server/src/api/importer.rs @@ -1,8 +1,8 @@ use crate::middleware::auth::UserContext; use crate::state::SharedState; use axum::{Json, extract::Extension, http::StatusCode, response::IntoResponse}; -use dom_smoothie::Readability; use malfestio_core::model::Visibility; +use malfestio_readability::Readability; use serde::{Deserialize, Serialize}; use serde_json::json; @@ -33,8 +33,6 @@ pub async fn import_article(Json(payload): Json) -> impl IntoResp } let url = payload.url.clone(); - - // Fetch HTML content let html_result = reqwest::get(&url).await; let html_content = match html_result { Ok(response) => match response.text().await { @@ -56,32 +54,25 @@ pub async fn import_article(Json(payload): Json) -> impl IntoResp } }; - // Extract article using dom_smoothie let url_for_task = url.clone(); - let result = tokio::task::spawn_blocking( - move || -> Result<(String, String, Option, Option), String> { - let mut readability = Readability::new(html_content, Some(&url_for_task), None) - .map_err(|e| format!("Readability error: {}", e))?; - let article = readability.parse().map_err(|e| format!("Parse error: {}", e))?; - Ok(( - article.title, - article.content.to_string(), - article.byline, - article.published_time, - )) - }, - ) + let result = tokio::task::spawn_blocking(move || -> Result { + let readability = Readability::new(html_content, Some(&url_for_task)); + readability.parse().map_err(|e| format!("Parse error: {}", e)) + }) .await; match result { - Ok(Ok((title, content, author, publish_date))) => { - // Convert HTML content to markdown - let markdown = html2md::parse_html(&content); + Ok(Ok(article)) => { + let markdown = article.markdown; let response = ImportArticleResponse { - title, + title: article.title, markdown, - metadata: ArticleMetadata { author, publish_date, source_url: payload.url }, + metadata: ArticleMetadata { + author: article.author, + publish_date: article.published_date, + source_url: payload.url, + }, }; Json(response).into_response() @@ -121,8 +112,6 @@ pub async fn import_article_save( } let url = payload.url.clone(); - - // Fetch HTML content let html_result = reqwest::get(&url).await; let html_content = match html_result { Ok(response) => match response.text().await { @@ -144,22 +133,17 @@ pub async fn import_article_save( } }; - // Extract article using dom_smoothie let url_for_task = url.clone(); - let result = tokio::task::spawn_blocking(move || -> Result<(String, String), String> { - let mut readability = Readability::new(html_content, Some(&url_for_task), None) - .map_err(|e| format!("Readability error: {}", e))?; - let article = readability.parse().map_err(|e| format!("Parse error: {}", e))?; - Ok((article.title, article.content.to_string())) + let result = tokio::task::spawn_blocking(move || -> Result { + let readability = Readability::new(html_content, Some(&url_for_task)); + readability.parse().map_err(|e| format!("Parse error: {}", e)) }) .await; match result { - Ok(Ok((title, content))) => { - // Convert HTML content to markdown - let markdown = html2md::parse_html(&content); - - // Merge auto-tags with user-provided tags + Ok(Ok(article)) => { + let title = article.title; + let markdown = article.markdown; let mut tags = payload.tags.clone(); if !tags.contains(&"imported".to_string()) { tags.push("imported".to_string()); @@ -168,10 +152,8 @@ pub async fn import_article_save( tags.push("article".to_string()); } - // Store source URL as first link let links = vec![payload.url.clone()]; - // Create note match state .note_repo .create(&user_ctx.did, &title, &markdown, tags, payload.visibility, links) @@ -222,13 +204,11 @@ mod tests { let body_json: serde_json::Value = serde_json::from_slice(&body_bytes).unwrap(); let title = body_json["title"].as_str().unwrap(); assert!(title.contains("Rust")); - // Verify markdown field exists and is non-empty + let markdown = body_json["markdown"].as_str().unwrap(); assert!(markdown.len() > 100); - // Verify no HTML tags leak through assert!(!markdown.contains("")); - // Verify metadata structure exists assert!(body_json["metadata"].is_object()); assert_eq!( body_json["metadata"]["source_url"].as_str().unwrap(),