#!/usr/bin/env -S uv run --script # /// script # requires-python = ">=3.11" # dependencies = ["beautifulsoup4", "lxml"] # /// """Inline EPUB footnote links for TTS. Usage: uv run inline_epub_footnotes.py "Book.epub" uv run inline_epub_footnotes.py "Book.epub" --dry-run uv run inline_epub_footnotes.py "Book.epub" --include 'split_003.html:split_006.html' """ from __future__ import annotations import argparse import posixpath import re import shutil import tempfile import zipfile from dataclasses import dataclass from pathlib import Path from urllib.parse import unquote, urldefrag from xml.etree import ElementTree as ET from bs4 import BeautifulSoup, NavigableString, Tag HTML_EXTS = (".html", ".xhtml", ".htm") MARKER_RE = re.compile(r"^[\s\*†‡§¶#0-9\[\]\(\)\.]+$") BACK_FRONT_RE = re.compile( r"\b(table of contents|contents|toc|title page|cover|copyright|publisher|praise|acclaim|" r"about the author|about the publisher|author[’']?s note|also by|books by|other books|" r"dedication|epigraph|newsletter|preview|excerpt)\b", re.I, ) @dataclass class Replacement: source: str target: str marker: str text: str def read_text(z: zipfile.ZipFile, name: str) -> str: return z.read(name).decode("utf-8", errors="replace") def parse_opf_path(z: zipfile.ZipFile) -> str: container = ET.fromstring(read_text(z, "META-INF/container.xml")) ns = {"c": "urn:oasis:names:tc:opendocument:xmlns:container"} rootfile = container.find(".//c:rootfile", ns) if rootfile is None: raise SystemExit("Could not find OPF path in META-INF/container.xml") return rootfile.attrib["full-path"] def spine_htmls(z: zipfile.ZipFile, opf_path: str) -> list[str]: opf = ET.fromstring(read_text(z, opf_path)) ns = {"opf": "http://www.idpf.org/2007/opf"} base = posixpath.dirname(opf_path) manifest = {} for item in opf.findall(".//opf:manifest/opf:item", ns): if item.attrib.get("media-type") == "application/xhtml+xml": href = item.attrib.get("href", "") manifest[item.attrib["id"]] = posixpath.normpath(posixpath.join(base, href)) if base else href out = [] for itemref in opf.findall(".//opf:spine/opf:itemref", ns): href = manifest.get(itemref.attrib.get("idref")) if href and href.endswith(HTML_EXTS): out.append(href) return out def visible_text(html: str) -> str: soup = BeautifulSoup(html, "lxml-xml") body = soup.find("body") or soup return " ".join(body.get_text(" ", strip=True).split()) def detect_story_files(z: zipfile.ZipFile, spine: list[str]) -> list[str]: candidates = [] for name in spine: txt = visible_text(read_text(z, name)) lower = txt[:1000].lower() # Must have enough prose and not look like boilerplate/front/back matter. if len(txt) < 1200: continue if BACK_FRONT_RE.search(lower): continue candidates.append(name) # Keep the first contiguous run of substantial prose files. This matches these # Calibre-split Pratchett EPUBs and avoids later promotional matter. if not candidates: return [] idx = {n: i for i, n in enumerate(spine)} runs: list[list[str]] = [] cur = [candidates[0]] for prev, name in zip(candidates, candidates[1:]): if idx[name] == idx[prev] + 1: cur.append(name) else: runs.append(cur) cur = [name] runs.append(cur) return max(runs, key=lambda r: sum(len(visible_text(read_text(z, n))) for n in r)) def parse_include(spec: str, names: list[str]) -> list[str]: if ":" in spec: a, b = spec.split(":", 1) def match(s): m = [n for n in names if s in n] if not m: raise SystemExit(f"No spine file matches include endpoint: {s}") return m[0] start, end = match(a), match(b) ia, ib = names.index(start), names.index(end) if ia > ib: ia, ib = ib, ia return names[ia : ib + 1] selected = [] for part in spec.split(","): selected += [n for n in names if part.strip() in n] return list(dict.fromkeys(selected)) def footnote_text(z: zipfile.ZipFile, source: str, href: str) -> str | None: target_part, frag = urldefrag(href) if not target_part or not frag: return None target = posixpath.normpath(posixpath.join(posixpath.dirname(source), unquote(target_part))) if target not in z.namelist(): return None soup = BeautifulSoup(read_text(z, target), "lxml-xml") anchor = soup.find(id=frag) root: Tag | BeautifulSoup = soup.find("body") or soup if anchor: # These EPUBs put the anchor immediately before the footnote container. parent = anchor.parent if isinstance(anchor.parent, Tag) else root text = parent.get_text(" ", strip=True) if len(text) < 5: nxt = anchor.find_next(lambda t: isinstance(t, Tag) and t.get_text(strip=True)) if nxt: text = nxt.get_text(" ", strip=True) else: text = root.get_text(" ", strip=True) text = " ".join(text.split()) text = re.sub(r"^[\*†‡§¶#0-9\s\[\]\(\)\.]+", "", text).strip() return text or None def inline_file(z: zipfile.ZipFile, name: str) -> tuple[str, list[Replacement]]: soup = BeautifulSoup(read_text(z, name), "lxml-xml") reps: list[Replacement] = [] for a in list(soup.find_all("a", href=True)): marker = a.get_text("", strip=True) if not marker or not MARKER_RE.match(marker): continue text = footnote_text(z, name, a["href"]) if not text: continue target, _ = urldefrag(a["href"]) reps.append(Replacement(name, target, marker, text)) a.replace_with(NavigableString(f" Footnote: {text}")) return str(soup), reps def write_epub(src: Path, dst: Path, modified: dict[str, str]) -> None: with zipfile.ZipFile(src, "r") as zin, zipfile.ZipFile(dst, "w") as zout: names = zin.namelist() if "mimetype" in names: zout.writestr("mimetype", zin.read("mimetype"), compress_type=zipfile.ZIP_STORED) for name in names: if name == "mimetype": continue data = modified[name].encode("utf-8") if name in modified else zin.read(name) info = zin.getinfo(name) zi = zipfile.ZipInfo(name, date_time=info.date_time) zi.external_attr = info.external_attr zi.comment = info.comment zout.writestr(zi, data, compress_type=zipfile.ZIP_DEFLATED) def main() -> None: ap = argparse.ArgumentParser(description="Inline main-story EPUB footnotes for TTS.") ap.add_argument("epub", type=Path) ap.add_argument("--dry-run", action="store_true") ap.add_argument("--include", help="Comma list or range by substring, e.g. split_003.html:split_006.html") ap.add_argument("--output", type=Path) args = ap.parse_args() if not args.epub.exists(): raise SystemExit(f"Missing EPUB: {args.epub}") with zipfile.ZipFile(args.epub, "r") as z: opf = parse_opf_path(z) spine = spine_htmls(z, opf) story = parse_include(args.include, spine) if args.include else detect_story_files(z, spine) if not story: raise SystemExit("Could not detect story files; use --include.") modified: dict[str, str] = {} all_reps: list[Replacement] = [] for name in story: new_html, reps = inline_file(z, name) if reps: modified[name] = new_html all_reps.extend(reps) print(f"Input: {args.epub}") print("Story files:") for n in story: print(f" - {n}") print(f"Footnotes to inline: {len(all_reps)}") for r in all_reps[:20]: print(f" {r.source} -> {r.target}: Footnote: {r.text[:90]}{'…' if len(r.text) > 90 else ''}") if len(all_reps) > 20: print(f" ... {len(all_reps) - 20} more") if args.dry_run: print("Dry run: no EPUB written.") return out = args.output or args.epub.with_name(args.epub.stem + ".inline-footnotes.epub") tmp = Path(tempfile.mkstemp(suffix=".epub")[1]) try: write_epub(args.epub, tmp, modified) shutil.move(str(tmp), out) finally: if tmp.exists(): tmp.unlink() print(f"Output: {out}") if __name__ == "__main__": main()