Something went wrong. Try again.
Customized Bible version for reading and notetaking compiled with Typst (main-branch-only mirror of https://forge.ejuarezg.com/ejuarezg/tailcat-wormhole)
Something went wrong. Try again.
11 kB · 266 lines
Python
at main
123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267#!/usr/bin/env python3"""Acquire a configured Bible source from an archive or bible.com import."""
from __future__ import annotations
import argparseimport hashlibimport jsonfrom pathlib import Pathimport shutilimport tempfileimport urllib.requestimport zipfile
from bible_com_source import import_translation
ROOT = Path(__file__).resolve().parents[1]CATALOG = ROOT / "translations.json"ARCHIVE_PATCHES = ROOT / "patches" / "archive"
def nonnegative_float(value: str) -> float: try: result = float(value) except ValueError as error: raise argparse.ArgumentTypeError("must be a number") from error if result < 0 or result == float("inf") or result != result: raise argparse.ArgumentTypeError("must be a finite, nonnegative number") return result
def load_translation(translation_id: str) -> dict: catalog = json.loads(CATALOG.read_text(encoding="utf-8")) try: return catalog[translation_id] except KeyError as error: choices = ", ".join(sorted(catalog)) raise SystemExit(f"Unknown translation {translation_id!r}. Choose one of: {choices}") from error
def read_stamp_hash(stamp: Path) -> str | None: """Return the SHA-256 recorded in a .source-url stamp, if present.""" return read_stamp_value(stamp, "sha256")
def read_stamp_value(stamp: Path, key: str) -> str | None: """Return a value recorded as ``key:value`` in a source stamp.""" prefix = f"{key}:" for line in stamp.read_text(encoding="utf-8").splitlines(): if line.startswith(prefix): return line.split(":", 1)[1].strip() return None
def archive_patch_manifest(translation_id: str, archive_hash: str) -> Path | None: """Return the patch manifest that is pinned to this archive revision.""" path = ARCHIVE_PATCHES / translation_id / f"{archive_hash}.json" return path if path.is_file() else None
def archive_patch_digest(manifest: Path | None) -> str: """Return a reproducible stamp value for an archive's applied patch manifest.""" return hashlib.sha256(manifest.read_bytes()).hexdigest() if manifest else "none"
def load_archive_patches( translation_id: str, archive_hash: str, manifest: Path | None) -> list[dict]: """Load patches only when their manifest explicitly matches the archive hash.""" if manifest is None: return [] try: data = json.loads(manifest.read_text(encoding="utf-8")) except json.JSONDecodeError as error: raise RuntimeError(f"Invalid archive patch manifest {manifest}: {error}") from error if not isinstance(data, dict) or data.get("archive_sha256") != archive_hash: raise RuntimeError( f"Archive patch manifest {manifest} does not match the configured " f"SHA-256 for {translation_id}" ) patches = data.get("patches") if not isinstance(patches, list) or not patches: raise RuntimeError(f"Archive patch manifest {manifest} has no patches") return patches
def apply_archive_patches(output: Path, manifest: Path | None, patches: list[dict]) -> None: """Apply checked, line-anchored source repairs after archive extraction.""" if manifest is None: return output_root = output.resolve() for index, patch in enumerate(patches, start=1): if not isinstance(patch, dict): raise RuntimeError(f"Archive patch {index} in {manifest} is not an object") filename = patch.get("file") line_number = patch.get("line") replacements = patch.get("replacements") if ( not isinstance(filename, str) or Path(filename).name != filename or not isinstance(line_number, int) or line_number < 1 or not isinstance(replacements, list) or not replacements ): raise RuntimeError(f"Archive patch {index} in {manifest} has invalid fields") target = (output / filename).resolve() if target.parent != output_root or not target.is_file(): raise RuntimeError(f"Archive patch {index} in {manifest} targets missing file {filename}") lines = target.read_text(encoding="utf-8").splitlines(keepends=True) if line_number > len(lines): raise RuntimeError( f"Archive patch {index} in {manifest} targets missing line " f"{line_number} in {filename}" ) source_line = lines[line_number - 1] for replacement in replacements: if not isinstance(replacement, dict): raise RuntimeError(f"Archive patch {index} in {manifest} has an invalid replacement") old = replacement.get("old") new = replacement.get("new") if not isinstance(old, str) or not old or not isinstance(new, str): raise RuntimeError(f"Archive patch {index} in {manifest} has an invalid replacement") if source_line.count(old) != 1: raise RuntimeError( f"Archive patch {index} in {manifest} no longer applies to " f"{filename}:{line_number}; review this patch for an upstream update" ) source_line = source_line.replace(old, new, 1) lines[line_number - 1] = source_line target.write_text("".join(lines), encoding="utf-8") print(f"Applied {len(patches)} archive source patch(es) from {manifest}")
def count_usfm_books(output: Path) -> int: """Count regular .usfm files in output (case-insensitive, like extraction).""" return sum( 1 for p in output.iterdir() if p.is_file() and p.suffix.lower() == ".usfm" )
def fetch_archive(translation_id: str, metadata: dict, output: Path, refresh: bool = False) -> None: expected_hash = metadata["archive_sha256"] expected_books = metadata["expected_books"] manifest = archive_patch_manifest(translation_id, expected_hash) patches = load_archive_patches(translation_id, expected_hash, manifest) expected_patch_digest = archive_patch_digest(manifest) stamp = output / ".source-url"
# Short-circuit when the local source already matches translations.json. # The stamp is intentionally not re-touched here: leaving its mtime alone # keeps Make re-invoking this recipe (cheap: a glob plus a stamp read) so a # real archive_sha256 change in translations.json still triggers a re-download. if not refresh and stamp.exists(): recorded_hash = read_stamp_hash(stamp) if recorded_hash is None: print( f"Local source stamp for {translation_id} is malformed " "(missing sha256). Re-downloading." ) elif recorded_hash == expected_hash: book_count = count_usfm_books(output) if book_count == expected_books: recorded_patch_digest = read_stamp_value(stamp, "patches-sha256") if recorded_patch_digest == expected_patch_digest: print( f"Source for {translation_id} is up to date " f"({book_count} books, sha256 {expected_hash[:12]}…)." ) return print( f"Local source for {translation_id} has outdated archive patches. " "Re-downloading." ) else: print( f"Local source for {translation_id} has {book_count} books; " f"expected {expected_books}. Re-downloading." ) else: print( f"Local source for {translation_id} is stale (sha256 changed). " "Re-downloading." )
output.mkdir(parents=True, exist_ok=True) url = metadata["archive_url"]
with tempfile.NamedTemporaryFile(suffix=".zip") as archive: print(f"Downloading {url}") request = urllib.request.Request(url, headers={"User-Agent": "typst-bible-builder/1.0"}) with urllib.request.urlopen(request, timeout=60) as response: shutil.copyfileobj(response, archive) archive.flush()
digest = hashlib.sha256() archive.seek(0) for chunk in iter(lambda: archive.read(1024 * 1024), b""): digest.update(chunk) actual_hash = digest.hexdigest() if actual_hash != expected_hash: raise RuntimeError( f"SHA-256 mismatch for {translation_id}: expected {expected_hash}, " f"downloaded {actual_hash}. Review the upstream update before changing " "translations.json." )
with zipfile.ZipFile(archive.name) as source: names = [name for name in source.namelist() if name.lower().endswith(".usfm")] if len(names) != expected_books: raise RuntimeError( f"Expected {expected_books} USFM books for {translation_id}; archive contains {len(names)}" ) for old_file in output.glob("*.usfm"): old_file.unlink() for name in names: destination = output / Path(name).name with source.open(name) as incoming, destination.open("wb") as outgoing: shutil.copyfileobj(incoming, outgoing)
apply_archive_patches(output, manifest, patches) (output / ".source-url").write_text( f"{url}\nsha256:{expected_hash}\npatches-sha256:{expected_patch_digest}\n", encoding="utf-8", ) print(f"Extracted {len(names)} books to {output}")
def main() -> None: parser = argparse.ArgumentParser() parser.add_argument("--translation", default="spav1602p") parser.add_argument("--output", type=Path) parser.add_argument( "--bible-scraper", type=Path, default=Path("../bible-scraper"), help="path to a tobiashellerslien/bible-scraper checkout", ) parser.add_argument("--rate-limit", type=nonnegative_float, default=0.5, metavar="SECONDS") parser.add_argument( "--refresh", action="store_true", help="re-download even if the local source is present and matches the expected hash", ) args = parser.parse_args()
metadata = load_translation(args.translation) output = args.output or ROOT / "source" / args.translation / "usfm" source_type = metadata.get("source_type", "archive") if source_type == "archive": fetch_archive(args.translation, metadata, output, refresh=args.refresh) elif source_type == "bible.com": import_translation( args.translation, metadata, args.bible_scraper, output, args.rate_limit, refresh=args.refresh, ) else: raise SystemExit(f"Unsupported source_type {source_type!r} for {args.translation}")
if __name__ == "__main__": main()