Something went wrong. Try again.
[READ-ONLY] Mirror of https://github.com/agbocsardi/tools.
Something went wrong. Try again.
123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241#!/usr/bin/env -S uv run --script# /// script# requires-python = ">=3.11"# dependencies = ["beautifulsoup4", "lxml"]# ///"""Inline EPUB footnote links for TTS.
Usage: uv run inline_epub_footnotes.py "Book.epub" uv run inline_epub_footnotes.py "Book.epub" --dry-run uv run inline_epub_footnotes.py "Book.epub" --include 'split_003.html:split_006.html'"""
from __future__ import annotations
import argparseimport posixpathimport reimport shutilimport tempfileimport zipfilefrom dataclasses import dataclassfrom pathlib import Pathfrom urllib.parse import unquote, urldefragfrom xml.etree import ElementTree as ET
from bs4 import BeautifulSoup, NavigableString, Tag
HTML_EXTS = (".html", ".xhtml", ".htm")MARKER_RE = re.compile(r"^[\s\*†‡§¶#0-9\[\]\(\)\.]+$")BACK_FRONT_RE = re.compile( r"\b(table of contents|contents|toc|title page|cover|copyright|publisher|praise|acclaim|" r"about the author|about the publisher|author[’']?s note|also by|books by|other books|" r"dedication|epigraph|newsletter|preview|excerpt)\b", re.I,)
@dataclassclass Replacement: source: str target: str marker: str text: str
def read_text(z: zipfile.ZipFile, name: str) -> str: return z.read(name).decode("utf-8", errors="replace")
def parse_opf_path(z: zipfile.ZipFile) -> str: container = ET.fromstring(read_text(z, "META-INF/container.xml")) ns = {"c": "urn:oasis:names:tc:opendocument:xmlns:container"} rootfile = container.find(".//c:rootfile", ns) if rootfile is None: raise SystemExit("Could not find OPF path in META-INF/container.xml") return rootfile.attrib["full-path"]
def spine_htmls(z: zipfile.ZipFile, opf_path: str) -> list[str]: opf = ET.fromstring(read_text(z, opf_path)) ns = {"opf": "http://www.idpf.org/2007/opf"} base = posixpath.dirname(opf_path) manifest = {} for item in opf.findall(".//opf:manifest/opf:item", ns): if item.attrib.get("media-type") == "application/xhtml+xml": href = item.attrib.get("href", "") manifest[item.attrib["id"]] = posixpath.normpath(posixpath.join(base, href)) if base else href out = [] for itemref in opf.findall(".//opf:spine/opf:itemref", ns): href = manifest.get(itemref.attrib.get("idref")) if href and href.endswith(HTML_EXTS): out.append(href) return out
def visible_text(html: str) -> str: soup = BeautifulSoup(html, "lxml-xml") body = soup.find("body") or soup return " ".join(body.get_text(" ", strip=True).split())
def detect_story_files(z: zipfile.ZipFile, spine: list[str]) -> list[str]: candidates = [] for name in spine: txt = visible_text(read_text(z, name)) lower = txt[:1000].lower() # Must have enough prose and not look like boilerplate/front/back matter. if len(txt) < 1200: continue if BACK_FRONT_RE.search(lower): continue candidates.append(name)
# Keep the first contiguous run of substantial prose files. This matches these # Calibre-split Pratchett EPUBs and avoids later promotional matter. if not candidates: return [] idx = {n: i for i, n in enumerate(spine)} runs: list[list[str]] = [] cur = [candidates[0]] for prev, name in zip(candidates, candidates[1:]): if idx[name] == idx[prev] + 1: cur.append(name) else: runs.append(cur) cur = [name] runs.append(cur) return max(runs, key=lambda r: sum(len(visible_text(read_text(z, n))) for n in r))
def parse_include(spec: str, names: list[str]) -> list[str]: if ":" in spec: a, b = spec.split(":", 1) def match(s): m = [n for n in names if s in n] if not m: raise SystemExit(f"No spine file matches include endpoint: {s}") return m[0] start, end = match(a), match(b) ia, ib = names.index(start), names.index(end) if ia > ib: ia, ib = ib, ia return names[ia : ib + 1] selected = [] for part in spec.split(","): selected += [n for n in names if part.strip() in n] return list(dict.fromkeys(selected))
def footnote_text(z: zipfile.ZipFile, source: str, href: str) -> str | None: target_part, frag = urldefrag(href) if not target_part or not frag: return None target = posixpath.normpath(posixpath.join(posixpath.dirname(source), unquote(target_part))) if target not in z.namelist(): return None soup = BeautifulSoup(read_text(z, target), "lxml-xml") anchor = soup.find(id=frag) root: Tag | BeautifulSoup = soup.find("body") or soup if anchor: # These EPUBs put the anchor immediately before the footnote container. parent = anchor.parent if isinstance(anchor.parent, Tag) else root text = parent.get_text(" ", strip=True) if len(text) < 5: nxt = anchor.find_next(lambda t: isinstance(t, Tag) and t.get_text(strip=True)) if nxt: text = nxt.get_text(" ", strip=True) else: text = root.get_text(" ", strip=True) text = " ".join(text.split()) text = re.sub(r"^[\*†‡§¶#0-9\s\[\]\(\)\.]+", "", text).strip() return text or None
def inline_file(z: zipfile.ZipFile, name: str) -> tuple[str, list[Replacement]]: soup = BeautifulSoup(read_text(z, name), "lxml-xml") reps: list[Replacement] = [] for a in list(soup.find_all("a", href=True)): marker = a.get_text("", strip=True) if not marker or not MARKER_RE.match(marker): continue text = footnote_text(z, name, a["href"]) if not text: continue target, _ = urldefrag(a["href"]) reps.append(Replacement(name, target, marker, text)) a.replace_with(NavigableString(f" Footnote: {text}")) return str(soup), reps
def write_epub(src: Path, dst: Path, modified: dict[str, str]) -> None: with zipfile.ZipFile(src, "r") as zin, zipfile.ZipFile(dst, "w") as zout: names = zin.namelist() if "mimetype" in names: zout.writestr("mimetype", zin.read("mimetype"), compress_type=zipfile.ZIP_STORED) for name in names: if name == "mimetype": continue data = modified[name].encode("utf-8") if name in modified else zin.read(name) info = zin.getinfo(name) zi = zipfile.ZipInfo(name, date_time=info.date_time) zi.external_attr = info.external_attr zi.comment = info.comment zout.writestr(zi, data, compress_type=zipfile.ZIP_DEFLATED)
def main() -> None: ap = argparse.ArgumentParser(description="Inline main-story EPUB footnotes for TTS.") ap.add_argument("epub", type=Path) ap.add_argument("--dry-run", action="store_true") ap.add_argument("--include", help="Comma list or range by substring, e.g. split_003.html:split_006.html") ap.add_argument("--output", type=Path) args = ap.parse_args()
if not args.epub.exists(): raise SystemExit(f"Missing EPUB: {args.epub}")
with zipfile.ZipFile(args.epub, "r") as z: opf = parse_opf_path(z) spine = spine_htmls(z, opf) story = parse_include(args.include, spine) if args.include else detect_story_files(z, spine) if not story: raise SystemExit("Could not detect story files; use --include.")
modified: dict[str, str] = {} all_reps: list[Replacement] = [] for name in story: new_html, reps = inline_file(z, name) if reps: modified[name] = new_html all_reps.extend(reps)
print(f"Input: {args.epub}") print("Story files:") for n in story: print(f" - {n}") print(f"Footnotes to inline: {len(all_reps)}") for r in all_reps[:20]: print(f" {r.source} -> {r.target}: Footnote: {r.text[:90]}{'…' if len(r.text) > 90 else ''}") if len(all_reps) > 20: print(f" ... {len(all_reps) - 20} more")
if args.dry_run: print("Dry run: no EPUB written.") return
out = args.output or args.epub.with_name(args.epub.stem + ".inline-footnotes.epub") tmp = Path(tempfile.mkstemp(suffix=".epub")[1]) try: write_epub(args.epub, tmp, modified) shutil.move(str(tmp), out) finally: if tmp.exists(): tmp.unlink() print(f"Output: {out}")
if __name__ == "__main__": main()