use std::collections::HashMap; use std::io::{Cursor, Read}; use chrono::{DateTime, NaiveDate, Utc}; use error_stack::{Report, ResultExt as _}; use quick_xml::events::Event; use quick_xml::reader::Reader; use crate::errors::{Error, Result}; /// Bibliographic metadata extracted from an EPUB's OPF package document. #[derive(Debug, Clone, Default)] pub struct EpubMeta { pub title: Option, pub authors: Vec, pub publisher: Option, pub language: Option, pub description: Option, pub series_name: Option, pub series_index: Option, pub pubdate: Option>, pub isbn: Option, /// Stable per-book identifier from a `urn:uuid:…` value /// (Calibre writes one). Survives re-export even though the file bytes /// change, so it is the reliable re-upload dedup key. pub book_uuid: Option, } /// A parsed EPUB: its metadata plus the raw cover image bytes, if one was found. #[derive(Debug, Clone)] pub struct ParsedEpub { pub meta: EpubMeta, pub cover: Option>, } /// A manifest `` entry: `id -> (href, properties)`. struct ManifestItem { href: String, properties: String, media_type: String, } /// Parse an EPUB from its raw bytes: read the OPF package document for metadata /// and extract the cover image. Synchronous (zip + XML) — call from /// `spawn_blocking`. pub fn parse_epub(bytes: &[u8]) -> Result { let mut archive = zip::ZipArchive::new(Cursor::new(bytes)) .change_context(Error::Upload) .attach("Not a valid EPUB (zip) archive")?; let opf_path = find_opf_path(&mut archive)?; let opf_xml = read_entry_string(&mut archive, &opf_path)? .ok_or_else(|| Report::new(Error::Upload).attach("OPF package document missing"))?; let (meta, cover_href) = parse_opf(&opf_xml, &opf_path); let cover = cover_href .and_then(|href| read_entry_bytes(&mut archive, &href).ok().flatten()) .filter(|bytes| !bytes.is_empty()); Ok(ParsedEpub { meta, cover }) } /// Read `META-INF/container.xml` and return the OPF package document path. fn find_opf_path(archive: &mut zip::ZipArchive) -> Result { let container = read_entry_string(archive, "META-INF/container.xml")? .ok_or_else(|| Report::new(Error::Upload).attach("EPUB missing META-INF/container.xml"))?; let mut reader = Reader::from_str(&container); reader.config_mut().trim_text(true); let mut buf = Vec::new(); loop { match reader.read_event_into(&mut buf) { Ok(Event::Empty(e)) | Ok(Event::Start(e)) if local_name(e.name().as_ref()) == "rootfile" => { if let Some(path) = attr(&e, "full-path") { return Ok(path); } } Ok(Event::Eof) => break, Err(_) => break, _ => {} } buf.clear(); } Err(Report::new(Error::Upload).attach("No rootfile in container.xml")) } /// Parse the OPF package document into metadata + the resolved cover href. fn parse_opf(opf_xml: &str, opf_path: &str) -> (EpubMeta, Option) { let mut meta = EpubMeta::default(); let mut items: HashMap = HashMap::new(); let mut cover_id: Option = None; let mut reader = Reader::from_str(opf_xml); // Do not trim per event: whitespace adjacent to an entity is real content // (`Wizards & Coast` arrives as `"Wizards "`, ref, `" Coast"`). We trim // only the fully accumulated buffer on `End`. Inter-element whitespace text // events fire while `current` is `None` and are ignored. let mut buf = Vec::new(); // Accumulate an element's text across events until its `End`. quick-xml 0.41 // splits a text node on each XML entity, emitting the surrounding text as // separate `Text` events with the entity as a `GeneralRef` in between (e.g. // `D&D` -> `Text("D")`, `GeneralRef("amp")`, `Text("D")`). Applying per // `Text` event would keep only the first fragment ("D"), so we buffer the // whole element and apply once on `End`. let mut current: Option = None; let mut text_buf = String::new(); loop { match reader.read_event_into(&mut buf) { Ok(Event::Start(e)) => { let name = local_name(e.name().as_ref()); handle_element(&e, &name, &mut items, &mut cover_id, &mut meta); current = dc_field(&name); text_buf.clear(); } Ok(Event::Empty(e)) => { let name = local_name(e.name().as_ref()); handle_element(&e, &name, &mut items, &mut cover_id, &mut meta); } Ok(Event::Text(t)) => { if current.is_some() { // Decode bytes, then resolve any entities that were left // inline (belt-and-suspenders; split ones arrive as GeneralRef). if let Ok(s) = t.decode() { let resolved = quick_xml::escape::unescape(&s) .map(|c| c.into_owned()) .unwrap_or_else(|_| s.into_owned()); text_buf.push_str(&resolved); } } } Ok(Event::GeneralRef(r)) => { if current.is_some() { if let Some(s) = resolve_entity(&r) { text_buf.push_str(&s); } } } Ok(Event::End(_)) => { if let Some(field) = current.take() { let text = text_buf.trim().to_string(); if !text.is_empty() { apply_dc_text(&mut meta, &field, text); } } text_buf.clear(); } Ok(Event::Eof) => break, Err(_) => break, _ => {} } buf.clear(); } let cover_href = resolve_cover(&items, cover_id.as_deref()).map(|href| join_href(opf_path, &href)); (meta, cover_href) } /// Handle a `` or manifest `` element (attributes only). fn handle_element( e: &quick_xml::events::BytesStart, name: &str, items: &mut HashMap, cover_id: &mut Option, meta: &mut EpubMeta, ) { match name { "meta" => { // Calibre-style: match (attr(e, "name").as_deref(), attr(e, "content")) { (Some("calibre:series"), Some(v)) => meta.series_name = Some(v), (Some("calibre:series_index"), Some(v)) => meta.series_index = v.parse().ok(), (Some("cover"), Some(v)) => *cover_id = Some(v), _ => {} } } "item" => { if let Some(id) = attr(e, "id") { items.insert( id, ManifestItem { href: attr(e, "href").unwrap_or_default(), properties: attr(e, "properties").unwrap_or_default(), media_type: attr(e, "media-type").unwrap_or_default(), }, ); } } _ => {} } } /// Map an OPF element local name to the metadata field it feeds, if any. fn dc_field(name: &str) -> Option { matches!( name, "title" | "creator" | "publisher" | "language" | "description" | "date" | "identifier" ) .then(|| name.to_string()) } /// Apply collected text for a Dublin Core element to the metadata. fn apply_dc_text(meta: &mut EpubMeta, field: &str, text: String) { match field { "title" if meta.title.is_none() => meta.title = Some(text), "creator" => meta.authors.push(text), "publisher" if meta.publisher.is_none() => meta.publisher = Some(text), "language" if meta.language.is_none() => meta.language = Some(text), "description" if meta.description.is_none() => meta.description = Some(text), "date" if meta.pubdate.is_none() => meta.pubdate = parse_date(&text), "identifier" => { if meta.book_uuid.is_none() { if let Some(uuid) = parse_book_uuid(&text) { meta.book_uuid = Some(uuid); } } if meta.isbn.is_none() { if let Some(isbn) = extract_isbn(&text) { meta.isbn = Some(isbn); } } } _ => {} } } /// Resolve the cover image href from a `meta name="cover"` id or an EPUB3 /// `properties="cover-image"` manifest item, falling back to a heuristic. fn resolve_cover(items: &HashMap, cover_id: Option<&str>) -> Option { if let Some(item) = cover_id.and_then(|id| items.get(id)) { if !item.href.is_empty() { return Some(item.href.clone()); } } if let Some(item) = items .values() .find(|i| i.properties.split_whitespace().any(|p| p == "cover-image")) { return Some(item.href.clone()); } items .values() .find(|i| i.media_type.starts_with("image/") && i.href.to_lowercase().contains("cover")) .map(|i| i.href.clone()) } /// Resolve a `GeneralRef` entity (`amp`, `#38`, `#x26`, ...) to its text by /// reconstructing `&name;` and running it through the XML unescaper, which /// handles both the predefined named entities and numeric character references. /// Unknown entities resolve to `None` and are dropped. fn resolve_entity(r: &quick_xml::events::BytesRef) -> Option { let name = r.decode().ok()?; quick_xml::escape::unescape(&format!("&{name};")) .ok() .map(|c| c.into_owned()) } /// Local name of a possibly-namespaced XML element (`dc:title` -> `title`). fn local_name(qname: &[u8]) -> String { let s = String::from_utf8_lossy(qname); s.rsplit(':').next().unwrap_or(&s).to_ascii_lowercase() } /// Read an attribute value by local name from an element. fn attr(e: &quick_xml::events::BytesStart, key: &str) -> Option { e.attributes().flatten().find_map(|a| { (local_name(a.key.as_ref()) == key) .then(|| { // quick-xml 0.41 deprecated unescape_value in favour of // normalized_value (resolves predefined entities per XML 1.0). a.normalized_value(quick_xml::XmlVersion::Implicit1_0) .map(|v| v.into_owned()) .ok() }) .flatten() }) } /// Resolve a manifest href against the OPF's directory. fn join_href(opf_path: &str, href: &str) -> String { let base = opf_path.rsplit_once('/').map(|(dir, _)| dir).unwrap_or(""); let combined = if base.is_empty() { href.to_string() } else { format!("{base}/{href}") }; // Normalize `a/b/../c` -> `a/c`. let mut parts: Vec<&str> = Vec::new(); for part in combined.split('/') { match part { "" | "." => {} ".." => { parts.pop(); } other => parts.push(other), } } parts.join("/") } /// Parse a Dublin Core `` value (RFC3339, `YYYY-MM-DD`, or `YYYY`). fn parse_date(s: &str) -> Option> { let s = s.trim(); if let Ok(dt) = DateTime::parse_from_rfc3339(s) { return Some(dt.to_utc()); } if let Ok(d) = NaiveDate::parse_from_str(s, "%Y-%m-%d") { return d.and_hms_opt(0, 0, 0).map(|ndt| ndt.and_utc()); } if let Ok(year) = s.get(0..4).unwrap_or(s).parse::() { return NaiveDate::from_ymd_opt(year, 1, 1) .and_then(|d| d.and_hms_opt(0, 0, 0)) .map(|ndt| ndt.and_utc()); } None } /// Extract a book UUID from a `` value. Calibre writes its stable /// per-book UUID either bare (`2148815d-…`) or `urn:uuid:`-prefixed; both are /// accepted. Matching by canonical 8-4-4-4-12 hex shape (not by `opf:scheme`, /// which the OPF parser does not surface here) keeps ISBNs, the small integer /// `calibre` id, and vendor ids from being mistaken for a UUID. fn parse_book_uuid(raw: &str) -> Option { let trimmed = raw.trim(); let candidate = match trimmed.get(..9) { Some(prefix) if prefix.eq_ignore_ascii_case("urn:uuid:") => trimmed[9..].trim(), _ => trimmed, }; is_uuid(candidate).then(|| candidate.to_ascii_lowercase()) } /// Whether `s` is a canonical 8-4-4-4-12 hex UUID (case-insensitive). fn is_uuid(s: &str) -> bool { let mut groups = [8usize, 4, 4, 4, 12].into_iter(); let mut parts = s.split('-'); let matched = parts .by_ref() .zip(groups.by_ref()) .all(|(part, len)| part.len() == len && part.bytes().all(|b| b.is_ascii_hexdigit())); matched && parts.next().is_none() && groups.next().is_none() } /// Pull an ISBN out of a `` value if it looks like one. fn extract_isbn(s: &str) -> Option { let digits: String = s .chars() .filter(|c| c.is_ascii_digit() || *c == 'X') .collect(); let lower = s.to_lowercase(); let looks_isbn = (lower.contains("isbn") && (digits.len() == 10 || digits.len() == 13)) || (digits.len() == 13 && digits.starts_with("978")); looks_isbn.then_some(digits) } /// Read a zip entry as bytes, trying an exact match then a case-insensitive scan. fn read_entry_bytes( archive: &mut zip::ZipArchive, name: &str, ) -> Result>> { let resolved = resolve_entry_name(archive, name); let Some(resolved) = resolved else { return Ok(None); }; let mut file = archive .by_name(&resolved) .change_context(Error::Upload) .attach("Failed to open zip entry")?; let mut out = Vec::new(); file.read_to_end(&mut out) .change_context(Error::Upload) .attach("Failed to read zip entry")?; Ok(Some(out)) } /// Read a zip entry as a UTF-8 string. fn read_entry_string( archive: &mut zip::ZipArchive, name: &str, ) -> Result> { Ok(read_entry_bytes(archive, name)?.map(|b| String::from_utf8_lossy(&b).into_owned())) } /// Find the actual archive entry name matching `name` (exact, else case-insensitive). fn resolve_entry_name( archive: &mut zip::ZipArchive, name: &str, ) -> Option { if archive.by_name(name).is_ok() { return Some(name.to_string()); } let lower = name.to_lowercase(); (0..archive.len()).find_map(|i| { let entry = archive.by_index(i).ok()?; let entry_name = entry.name().to_string(); (entry_name.to_lowercase() == lower).then_some(entry_name) }) } #[cfg(test)] mod tests { use std::io::Write as _; use super::*; /// Build a minimal in-memory EPUB with the given OPF body under ``. fn epub_with_metadata(metadata: &str) -> Vec { let container = r#" "#; let opf = format!( r#" {metadata} "# ); let mut zip = zip::ZipWriter::new(Cursor::new(Vec::new())); let opts: zip::write::SimpleFileOptions = Default::default(); for (name, body) in [("META-INF/container.xml", container), ("content.opf", &opf)] { zip.start_file(name, opts).expect("start_file"); zip.write_all(body.as_bytes()).expect("write"); } zip.finish().expect("finish").into_inner() } #[test] fn ampersand_entity_in_title_is_not_truncated() { // Regression: quick-xml splits `D&D` into Text/GeneralRef/Text; the // old per-Text apply kept only "D". The whole title must survive. let bytes = epub_with_metadata( r#"D&D 5e Players Handbook Wizards & Coast"#, ); let parsed = parse_epub(&bytes).expect("parse"); assert_eq!( parsed.meta.title.as_deref(), Some("D&D 5e Players Handbook") ); assert_eq!(parsed.meta.authors, vec!["Wizards & Coast".to_string()]); } #[test] fn numeric_char_ref_in_title_is_resolved() { let bytes = epub_with_metadata(r#"A & B & C"#); let parsed = parse_epub(&bytes).expect("parse"); assert_eq!(parsed.meta.title.as_deref(), Some("A & B & C")); } #[test] fn plain_title_still_parses() { let bytes = epub_with_metadata(r#"A Study in Scarlet"#); let parsed = parse_epub(&bytes).expect("parse"); assert_eq!(parsed.meta.title.as_deref(), Some("A Study in Scarlet")); } #[test] fn calibre_urn_uuid_identifier_becomes_dedup_key() { // Calibre writes both a scheme identifier and a urn:uuid one; the latter // is the stable re-upload dedup key. Case-insensitive prefix, lowercased. let bytes = epub_with_metadata( r#"D&D 5e Players Handbook 42 URN:UUID:5F8E1A2B-0000-4C3D-9E7F-ABCDEF012345 9780786965601"#, ); let parsed = parse_epub(&bytes).expect("parse"); assert_eq!( parsed.meta.book_uuid.as_deref(), Some("5f8e1a2b-0000-4c3d-9e7f-abcdef012345") ); // ISBN parsing still works alongside the uuid capture. assert_eq!(parsed.meta.isbn.as_deref(), Some("9780786965601")); } #[test] fn bare_calibre_uuid_identifier_is_captured() { // Real Calibre EPUBs store the book UUID *without* a urn:uuid: prefix. let bytes = epub_with_metadata( r#"Game of Thrones 7 2148815D-3C21-4625-B85E-852CE2E3F46E"#, ); let parsed = parse_epub(&bytes).expect("parse"); assert_eq!( parsed.meta.book_uuid.as_deref(), Some("2148815d-3c21-4625-b85e-852ce2e3f46e") ); } #[test] fn non_uuid_identifiers_leave_dedup_key_empty() { // Neither the small `calibre` integer id nor an ISBN is a UUID. let bytes = epub_with_metadata( r#"Plain 7 9780786965601"#, ); let parsed = parse_epub(&bytes).expect("parse"); assert_eq!(parsed.meta.book_uuid, None); assert_eq!(parsed.meta.isbn.as_deref(), Some("9780786965601")); } }