From bd588dc6e8cd94de506e6329db2549e927aaaccf Mon Sep 17 00:00:00 2001 From: Jer Miller Date: Tue, 14 Jul 2026 16:46:05 -0600 Subject: [PATCH] Rework document importer for per-page provenance-carrying extraction Rebuild the PDF document importer on the sol-pdf/1 worker: two extract invocations per document (authoritative text pass, then an explicit render pass), a per-page state machine, and honest per-slot failure markers. All model-generated content is blockquote-delimited so a later reader can always separate deterministic extraction from model output; verbatim text-layer pages carry zero model involvement. Rasters for every model-touched page are persisted under pages/ as model-free ground truth, and the derived transcript is written last so a crash never leaves a transcript citing a missing original or raster. Adds ImportResult.hard_failures and document-only --force plumbing: the CLI exits non-zero after a completed batch when the document importer reports hard per-file failures, and re-importing already-imported content is a safe no-op unless --force is given. Deletes the pypdf/ pdf2image extraction path and the document importer's entity seeding. Co-Authored-By: Claude Opus 4.8 (1M context) --- solstone/apps/import/tests/test_routes.py | 145 ++++ solstone/think/importers/cli.py | 10 +- solstone/think/importers/documents.py | 918 +++++++++++++++----- solstone/think/importers/file_importer.py | 2 + tests/pdf_worker_fixtures.py | 66 ++ tests/test_importer.py | 46 +- tests/test_importer_documents.py | 981 ++++++++++++++++------ tests/test_pdf_lazy_imports.py | 5 +- 8 files changed, 1666 insertions(+), 507 deletions(-) diff --git a/solstone/apps/import/tests/test_routes.py b/solstone/apps/import/tests/test_routes.py index 8dc0114ac..891f44175 100644 --- a/solstone/apps/import/tests/test_routes.py +++ b/solstone/apps/import/tests/test_routes.py @@ -4,10 +4,59 @@ from __future__ import annotations import json +import sys +from io import BytesIO +from pathlib import Path +from unittest.mock import MagicMock import pytest +def _pdf_literal(value: str) -> bytes: + out = bytearray() + for char in value: + if char == "\\": + out += b"\\\\" + elif char == "(": + out += b"\\(" + elif char == ")": + out += b"\\)" + else: + out += char.encode("ascii") + return b"(" + bytes(out) + b")" + + +def _write_text_pdf(path: Path, text: str) -> Path: + content = b"BT /F1 12 Tf 72 720 Td " + _pdf_literal(text) + b" Tj ET\n" + objects = [ + b"<< /Type /Catalog /Pages 2 0 R >>", + b"<< /Type /Pages /Kids [ 4 0 R ] /Count 1 >>", + b"<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>", + ( + b"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] " + b"/Resources << /Font << /F1 3 0 R >> >> /Contents 5 0 R >>" + ), + b"<< /Length %d >>\nstream\n" % len(content) + content + b"endstream", + ] + out = bytearray(b"%PDF-1.4\n%\xe2\xe3\xcf\xd3\n") + offsets = [0] + for index, obj in enumerate(objects, start=1): + offsets.append(len(out)) + out += f"{index} 0 obj\n".encode("ascii") + obj + b"\nendobj\n" + xref_offset = len(out) + out += f"xref\n0 {len(objects) + 1}\n".encode("ascii") + out += b"0000000000 65535 f \n" + for offset in offsets[1:]: + out += f"{offset:010d} 00000 n \n".encode("ascii") + out += ( + b"trailer\n<< /Size 6 /Root 1 0 R >>\nstartxref\n" + + str(xref_offset).encode("ascii") + + b"\n%%EOF\n" + ) + path.write_bytes(bytes(out)) + return path + + @pytest.fixture def client(tmp_path, monkeypatch): journal = tmp_path / "journal" @@ -52,3 +101,99 @@ def test_import_detail_api_path_resolves(client): endpoint, _args = adapter.match("/app/import/api/missing-import", method="GET") assert endpoint == "app:import.import_detail_api" + + +def test_document_upload_stages_emits_command_and_imports_new_shape_segment( + client, tmp_path, monkeypatch +): + import importlib + + import_routes = importlib.import_module("solstone.apps.import.routes") + cli_mod = importlib.import_module("solstone.think.importers.cli") + doc_mod = importlib.import_module("solstone.think.importers.documents") + journal = Path(import_routes.state.journal_root) + source_pdf = _write_text_pdf( + tmp_path / "upload.pdf", + ( + "Route upload document has enough extractable text for the worker-backed " + "document importer to create a new transcript." + ), + ) + emitted: list[dict] = [] + + monkeypatch.setattr( + import_routes, + "emit", + lambda tract, event, **kwargs: emitted.append( + {"tract": tract, "event": event, **kwargs} + ), + ) + monkeypatch.setattr(cli_mod, "CallosumConnection", lambda **kwargs: MagicMock()) + monkeypatch.setattr(cli_mod, "_status_emitter", lambda: None) + monkeypatch.setattr(cli_mod, "index_file", lambda *args, **kwargs: None) + monkeypatch.setattr( + doc_mod, + "generate", + lambda *, contents, context, **kwargs: pytest.fail("unexpected model call"), + ) + + save_response = client.post( + "/app/import/api/save", + data={ + "file": (BytesIO(source_pdf.read_bytes()), "contract.pdf"), + "client_item_id": "document-upload", + "facet": "work", + "setting": "review", + "source_hint": "document", + }, + content_type="multipart/form-data", + ) + + assert save_response.status_code == 200 + saved = save_response.get_json() + timestamp = saved["timestamp"] + staged_pdf = journal / "imports" / timestamp / "contract.pdf" + import_json = journal / "imports" / timestamp / "import.json" + assert staged_pdf.read_bytes() == source_pdf.read_bytes() + assert json.loads(import_json.read_text(encoding="utf-8"))["source_hint"] == ( + "document" + ) + + start_response = client.post( + "/app/import/api/start", + json={"path": str(staged_pdf), "timestamp": timestamp, "force": True}, + ) + + assert start_response.status_code == 200 + assert emitted + cmd = emitted[-1]["cmd"] + assert cmd == [ + "journal", + "importer", + str(staged_pdf), + timestamp, + "--facet", + "work", + "--setting", + "review", + "--source", + "document", + "--force", + ] + + cli_argv = [cmd[0], cmd[2], "--timestamp", cmd[3], *cmd[4:]] + monkeypatch.setattr(sys, "argv", cli_argv) + monkeypatch.setenv("SOL_SKIP_SUPERVISOR_CHECK", "1") + cli_mod.main() + + segments = list((journal / "chronicle").glob("*/import.document/*")) + assert len(segments) == 1 + segment_dir = segments[0] + transcript = segment_dir / "document_transcript.md" + assert (segment_dir / "original.pdf").read_bytes() == source_pdf.read_bytes() + assert transcript.is_file() + text = transcript.read_text(encoding="utf-8") + assert "**Type:** Document" in text + assert "**Extraction:" in text + assert "Route upload document has enough extractable text" in text + assert doc_mod.MARKER_MODEL_EXTRACTED.split("{NNNN}", 1)[0] not in text diff --git a/solstone/think/importers/cli.py b/solstone/think/importers/cli.py index 044856739..2e572e4a9 100644 --- a/solstone/think/importers/cli.py +++ b/solstone/think/importers/cli.py @@ -505,6 +505,8 @@ def _process_file_importer( "with_day_summaries": getattr(args, "with_day_summaries", False), } ) + if importer.name == "document": + kwargs["force"] = getattr(args, "force", False) return importer.process(path, journal_root, **kwargs) @@ -992,11 +994,12 @@ def _import_one_from_args(args: argparse.Namespace) -> dict[str, Any] | None: ) processing_results["entries_written"] = result.entries_written processing_results["entities_seeded"] = result.entities_seeded + processing_results["summary_errors"] = list(result.errors) + processing_results["hard_failures"] = list(result.hard_failures) if result.merge_summary is not None: processing_results["merge_summary"] = result.merge_summary processing_results["merge_log_path"] = result.merge_log_path processing_results["merge_staging_path"] = result.merge_staging_path - processing_results["summary_errors"] = list(result.errors) if result.principal_collision is not None: processing_results["principal_collision"] = result.principal_collision @@ -1135,6 +1138,7 @@ def _import_one_from_args(args: argparse.Namespace) -> dict[str, Any] | None: "entities_seeded": result.entities_seeded, "files_created": result.files_created, "errors": result.errors, + "hard_failures": list(result.hard_failures), "summary": result.summary, "merge_summary": result.merge_summary, "principal_collision": result.principal_collision, @@ -1678,7 +1682,7 @@ def main() -> None: parser.error("the following arguments are required: media") try: - import_one( + result = import_one( args.media, timestamp=args.timestamp, facet=args.facet, @@ -1695,6 +1699,8 @@ def main() -> None: date_to=args.date_to, with_day_summaries=args.with_day_summaries, ) + if result and result.get("hard_failures"): + raise SystemExit(1) except Exception as exc: raise SystemExit(str(exc)) from exc diff --git a/solstone/think/importers/documents.py b/solstone/think/importers/documents.py index 50964d7a5..ee2eb1732 100644 --- a/solstone/think/importers/documents.py +++ b/solstone/think/importers/documents.py @@ -1,32 +1,139 @@ # SPDX-License-Identifier: AGPL-3.0-only # Copyright (c) 2026 sol pbc -"""PDF document importer.""" +"""PDF document importer backed by the isolated PDFium worker.""" from __future__ import annotations import datetime as dt +import functools import logging -import re +import tempfile +from dataclasses import dataclass from pathlib import Path -from typing import TYPE_CHECKING, Callable - -if TYPE_CHECKING: - from pypdf import PdfReader - -from solstone.think.entities.seeding import seed_entities +from typing import Any, Callable + +from solstone.observe import pdf_worker +from solstone.observe.pdf_worker import ( + PdfWorkerCorruptError, + PdfWorkerEncryptedError, + PdfWorkerEngineError, + PdfWorkerError, + PdfWorkerRenderIOError, + PdfWorkerTimeoutError, + run_pdf_worker, +) from solstone.think.features import require_extra from solstone.think.importers.file_importer import ImportPreview, ImportResult -from solstone.think.importers.shared import install_source_file, write_content_manifest +from solstone.think.importers.shared import ( + PRIVATE_IMPORT_FILE_MODE, + hash_source, + install_source_file, + write_content_manifest, +) from solstone.think.journal_io import write_text -from solstone.think.models import generate -from solstone.think.utils import day_path +from solstone.think.models import NoBrainConfiguredError, generate +from solstone.think.prompts import load_prompt logger = logging.getLogger(__name__) +PAGE_TEXT_MIN_CHARS = 50 +PAGE_IMAGE_DESCRIBE_MIN = 0.10 +MODEL_CALLS_MAX_PER_DOCUMENT = 50 + +REASON_MODEL_CALL_LIMIT = "model-call limit reached" +REASON_EMPTY_MODEL_RESPONSE = "empty model response" +REASON_NO_BRAIN_CONFIGURED = "no brain configured" + +MARKER_MODEL_EXTRACTED = ( + "> [model-extracted from page image — may contain errors; " + "original: pages/page-{NNNN}.png]" +) +MARKER_IMAGE_DESCRIPTION = ( + "> [image description — model-generated; original: pages/page-{NNNN}.png]" +) +MARKER_PAGE_TEXT_UNAVAILABLE_WITH_RASTER = ( + "> [page text unavailable — {reason}; " + "page image preserved at pages/page-{NNNN}.png]" +) +MARKER_IMAGE_DESCRIPTION_UNAVAILABLE_WITH_RASTER = ( + "> [image description unavailable — {reason}; " + "page image preserved at pages/page-{NNNN}.png]" +) +MARKER_PAGE_TEXT_UNAVAILABLE_NO_RASTER = ( + "> [page text unavailable — {reason}; no page image could be produced]" +) +MARKER_IMAGE_DESCRIPTION_UNAVAILABLE_NO_RASTER = ( + "> [image description unavailable — {reason}; no page image could be produced]" +) + +_DOCUMENT_STREAM = "import.document" +_TRANSCRIPT_FILENAME = "document_transcript.md" +_ORIGINAL_FILENAME = "original.pdf" +_DESCRIBE_PROMPT = ( + "Describe this image in detail. Include any visible text, people, objects, " + "setting, and notable context. Return a concise natural-language description." +) + + +@dataclass(frozen=True) +class _TimestampChoice: + timestamp: float + source: str + + +@dataclass(frozen=True) +class _PreparedDocument: + payload: dict[str, Any] + transcript: str + rasters: dict[int, Path] + warnings: tuple[str, ...] + timestamp_source: str + text_layer_pages: int + model_extracted_pages: int + unavailable_pages: int + image_described_pages: int + model_calls: int + + +@dataclass(frozen=True) +class _WorkerOutputs: + payload: dict[str, Any] + rasters: dict[int, Path] + render_errors: dict[int, str] + warnings: tuple[str, ...] + + +@dataclass(frozen=True) +class _SegmentClaim: + day: str + seg_key: str + timestamp: float + already_imported: bool = False + + +@dataclass +class _RenderStats: + text_layer_pages: int = 0 + model_extracted_pages: int = 0 + unavailable_pages: int = 0 + image_described_pages: int = 0 + model_calls: int = 0 + + +@dataclass(frozen=True) +class _ModelOutcome: + text: str | None + reason: str | None + + @property + def ok(self) -> bool: + return self.text is not None + def _find_pdfs(path: Path) -> list[Path]: """Return matching PDF files for a file or directory path.""" + if path.is_file() and path.suffix.lower() == ".pdf": return [path] if path.is_dir(): @@ -38,156 +145,520 @@ def _find_pdfs(path: Path) -> list[Path]: return [] -def _get_pdf_timestamp(reader: PdfReader, pdf_path: Path) -> float: - """Return a best-effort timestamp for a PDF.""" - metadata = reader.metadata or {} - for key in ("/ModDate", "/CreationDate"): - value = metadata.get(key) - if not value: +def _now_local() -> dt.datetime: + return dt.datetime.now().astimezone() + + +def _collapse_line(value: Any) -> str: + return " ".join(str(value).split()) + + +def _page_name(index: int) -> str: + return f"page-{index:04d}.png" + + +def _marker( + template: str, *, index: int | None = None, reason: str | None = None +) -> str: + rendered = template + if index is not None: + rendered = rendered.replace("{NNNN}", f"{index:04d}") + if reason is not None: + rendered = rendered.replace("{reason}", _collapse_line(reason)) + return rendered + + +def _blockquote_lines(text: str) -> list[str]: + return [">" if line == "" else f"> {line}" for line in text.splitlines()] + + +def _model_block(marker: str, text: str) -> str: + return "\n".join([marker, *_blockquote_lines(text)]) + + +@functools.lru_cache(maxsize=1) +def _reading_prompt() -> str: + categories_dir = Path(pdf_worker.__file__).resolve().parent / "categories" + return load_prompt("reading", base_dir=categories_dir).text + + +def _image_for_model(path: Path) -> Any: + from PIL import Image + + from solstone.observe.utils import resize_for_vlm + + with Image.open(path) as img: + img.load() + return resize_for_vlm(img).copy() + + +def _generate_for_page( + *, + prompt: str, + raster_path: Path, + context: str, + stats: _RenderStats, +) -> _ModelOutcome: + if stats.model_calls >= MODEL_CALLS_MAX_PER_DOCUMENT: + return _ModelOutcome(text=None, reason=REASON_MODEL_CALL_LIMIT) + + stats.model_calls += 1 + try: + response = generate( + contents=[prompt, _image_for_model(raster_path)], + context=context, + ) + except NoBrainConfiguredError: + return _ModelOutcome(text=None, reason=REASON_NO_BRAIN_CONFIGURED) + except Exception as exc: + return _ModelOutcome( + text=None, reason=_collapse_line(str(exc) or type(exc).__name__) + ) + + text = response.strip() + if not text: + return _ModelOutcome(text=None, reason=REASON_EMPTY_MODEL_RESPONSE) + return _ModelOutcome(text=text, reason=None) + + +def _merge_warnings(*groups: tuple[str, ...] | list[str]) -> tuple[str, ...]: + seen: set[str] = set() + merged: list[str] = [] + for group in groups: + for warning in group: + collapsed = _collapse_line(warning) + if collapsed and collapsed not in seen: + seen.add(collapsed) + merged.append(collapsed) + return tuple(merged) + + +def _render_set(payload: dict[str, Any]) -> set[int]: + pages: set[int] = set() + for page in payload.get("pages", []): + if page.get("error") is not None: continue - match = re.search(r"(\d{4})(\d{2})(\d{2})(\d{2})(\d{2})(\d{2})", str(value)) - if not match: + index = int(page["index"]) + chars = int(page.get("chars") or 0) + image_area_fraction = float(page.get("image_area_fraction") or 0.0) + if ( + chars < PAGE_TEXT_MIN_CHARS + or image_area_fraction >= PAGE_IMAGE_DESCRIBE_MIN + ): + pages.add(index) + return pages + + +def _rendered_pages( + payload: dict[str, Any], + render_dir: Path | None, +) -> tuple[dict[int, Path], dict[int, str]]: + rasters: dict[int, Path] = {} + render_errors: dict[int, str] = {} + if render_dir is None: + return rasters, render_errors + + for page in payload.get("pages", []): + index = int(page["index"]) + error = page.get("error") + if error is not None: + render_errors[index] = _collapse_line(error) continue - try: - parsed = dt.datetime( - int(match.group(1)), - int(match.group(2)), - int(match.group(3)), - int(match.group(4)), - int(match.group(5)), - int(match.group(6)), - ) - return parsed.timestamp() - except ValueError: + rendered = page.get("rendered") + if not rendered: continue + raster_path = render_dir / str(rendered) + if raster_path.exists(): + rasters[index] = raster_path + else: + render_errors[index] = f"page {index}: rendered page image missing" + return rasters, render_errors + + +def _parse_pdf_metadata_date(value: Any, *, now: dt.datetime) -> float | None: + if not value: + return None + raw = str(value) + if raw.endswith("Z"): + raw = raw[:-1] + "+00:00" try: - return pdf_path.stat().st_mtime + parsed = dt.datetime.fromisoformat(raw) + except ValueError: + return None + if parsed.tzinfo is None: + return None + local = parsed.astimezone() + if _timestamp_in_window(local.timestamp(), now=now): + return local.timestamp() + return None + + +def _timestamp_in_window(timestamp: float, *, now: dt.datetime) -> bool: + lower = dt.datetime(1970, 1, 1, tzinfo=dt.timezone.utc) + upper = now.astimezone(dt.timezone.utc) + dt.timedelta(days=1) + candidate = dt.datetime.fromtimestamp(timestamp, tz=dt.timezone.utc) + return lower <= candidate <= upper + + +def _choose_timestamp(payload: dict[str, Any], pdf_path: Path) -> _TimestampChoice: + now = _now_local() + metadata = payload.get("metadata") or {} + for key in ("mod_date", "creation_date"): + timestamp = _parse_pdf_metadata_date(metadata.get(key), now=now) + if timestamp is not None: + return _TimestampChoice(timestamp=timestamp, source="pdf-metadata") + + try: + mtime = pdf_path.stat().st_mtime except OSError: - return dt.datetime.now().timestamp() - - -def _extract_text_pypdf(reader: PdfReader) -> tuple[str, int, bool]: - """Extract text and classify whether the PDF appears scanned.""" - parts: list[str] = [] - total_chars = 0 - for page in reader.pages: - text = page.extract_text() or "" - parts.append(text) - total_chars += len(text) - page_count = len(reader.pages) - is_scanned = (total_chars / max(page_count, 1)) < 50 - return "\n\n".join(parts).strip(), page_count, is_scanned - - -def _extract_text_vision(pdf_path: Path, page_count: int) -> str: - """Extract text from scanned PDFs using vision models.""" - from pdf2image import convert_from_path - - prompt = ( - "Extract all text content from this document. Preserve the document " - "structure including headings, paragraphs, lists, and tables. Return " - "the content as clean markdown." + mtime = None + if mtime is not None and _timestamp_in_window(mtime, now=now): + return _TimestampChoice(timestamp=mtime, source="file-mtime") + + return _TimestampChoice(timestamp=now.timestamp(), source="import-time") + + +def _segment_matches_sha(segment_dir: Path, sha256: str) -> bool: + original_path = segment_dir / _ORIGINAL_FILENAME + return original_path.is_file() and hash_source(original_path) == sha256 + + +def _claim_segment( + journal_root: Path, + *, + timestamp: float, + sha256: str, + used_keys: set[tuple[str, str]], + force: bool, +) -> _SegmentClaim: + ts = timestamp + while True: + local_dt = dt.datetime.fromtimestamp(ts).astimezone() + day = local_dt.strftime("%Y%m%d") + seg_key = f"{local_dt.strftime('%H%M%S')}_0" + segment_dir = journal_root / "chronicle" / day / _DOCUMENT_STREAM / seg_key + candidate = (day, seg_key) + if candidate in used_keys: + ts += 1 + continue + if not segment_dir.exists(): + used_keys.add(candidate) + return _SegmentClaim(day=day, seg_key=seg_key, timestamp=ts) + if _segment_matches_sha(segment_dir, sha256): + used_keys.add(candidate) + return _SegmentClaim( + day=day, + seg_key=seg_key, + timestamp=ts, + already_imported=not force, + ) + ts += 1 + + +def _page_failure_reason( + *, + page: dict[str, Any], + render_errors: dict[int, str], +) -> str: + index = int(page["index"]) + return _collapse_line( + page.get("error") + or render_errors.get(index) + or f"page {index}: page image missing" ) - images = convert_from_path(str(pdf_path), dpi=200) - if page_count <= 10: - return generate( - contents=[prompt, *images], context="import.document.vision" - ).strip() - - pages: list[str] = [] - for image in images: - page_text = generate( - contents=[prompt, image], context="import.document.vision" - ).strip() - if page_text: - pages.append(page_text) - return "\n\n".join(pages).strip() - - -def extract_pdf_text(pdf_path: Path) -> tuple[str, dict]: - """Extract text from a PDF, using vision fallback for scanned PDFs. - - Returns (text, meta). meta keys: page_count (int), is_scanned (bool), - extraction_method ("pypdf"|"vision"), vision_error (str|None). On a scanned - PDF whose vision extraction fails, returns the sparse pypdf text with - vision_error set (does NOT raise). Hard failures (unreadable PDF, missing - deps) propagate. - """ - require_extra("pdf") - from pypdf import PdfReader - - reader = PdfReader(str(pdf_path)) - text, page_count, is_scanned = _extract_text_pypdf(reader) - meta = { - "page_count": page_count, - "is_scanned": is_scanned, - "extraction_method": "pypdf", - "vision_error": None, - } - if is_scanned: - try: - text = _extract_text_vision(pdf_path, page_count) - meta["extraction_method"] = "vision" - except Exception as vision_exc: - logger.warning("Vision extraction failed for %s: %s", pdf_path, vision_exc) - meta["vision_error"] = str(vision_exc) - return text, meta - - -def _render_document_markdown(title: str, text: str, metadata: dict) -> str: - """Render extracted document text as markdown.""" - lines = [f"# {title}", "", "**Type:** Document"] - if metadata.get("page_count") is not None: - lines.append(f"**Pages:** {metadata['page_count']}") - if metadata.get("date"): - lines.append(f"**Date:** {metadata['date']}") - lines.extend(["", "---", "", text.strip()]) - return "\n".join(lines).rstrip() + "\n" - - -def _extract_entities(text: str, title: str) -> list[dict]: - """Extract simple named people and organizations from document text.""" - names = set( - m.group(1).strip(" ,.") - for m in re.finditer( - r"(?i)\b(?:by|from|to|between|signed by)\s+([A-Z][a-z]+(?:\s+[A-Z][a-z]+){1,3})", - text, + + +def _append_text_layer( + section: list[str], + text: str, +) -> None: + section.append(text) + if not text.endswith("\n"): + section.append("\n") + + +def _render_page_section( + page: dict[str, Any], + *, + rasters: dict[int, Path], + render_errors: dict[int, str], + stats: _RenderStats, +) -> str: + index = int(page["index"]) + chars = int(page.get("chars") or 0) + image_area_fraction = float(page.get("image_area_fraction") or 0.0) + raster_path = rasters.get(index) + page_error = page.get("error") + section: list[str] = [f"## Page {index}\n\n"] + + if page_error is None and chars >= PAGE_TEXT_MIN_CHARS: + stats.text_layer_pages += 1 + _append_text_layer(section, str(page.get("text") or "")) + if image_area_fraction >= PAGE_IMAGE_DESCRIBE_MIN: + if raster_path is not None: + outcome = _generate_for_page( + prompt=_DESCRIBE_PROMPT, + raster_path=raster_path, + context="import.document.describe", + stats=stats, + ) + if outcome.ok: + stats.image_described_pages += 1 + section.append("\n") + section.append( + _model_block( + _marker(MARKER_IMAGE_DESCRIPTION, index=index), + outcome.text or "", + ) + ) + section.append("\n") + else: + section.append("\n") + section.append( + _marker( + MARKER_IMAGE_DESCRIPTION_UNAVAILABLE_WITH_RASTER, + index=index, + reason=outcome.reason or REASON_EMPTY_MODEL_RESPONSE, + ) + ) + section.append("\n") + else: + section.append("\n") + section.append( + _marker( + MARKER_IMAGE_DESCRIPTION_UNAVAILABLE_NO_RASTER, + reason=_page_failure_reason( + page=page, + render_errors=render_errors, + ), + ) + ) + section.append("\n") + return "".join(section).rstrip("\n") + + if page_error is None and chars < PAGE_TEXT_MIN_CHARS and raster_path is not None: + outcome = _generate_for_page( + prompt=_reading_prompt(), + raster_path=raster_path, + context="import.document.vision", + stats=stats, ) - ) - orgs = set( - m.group(1).strip(" ,.") - for m in re.finditer( - r"\b([A-Z][A-Za-z0-9&.,' -]{1,80}\s+(?:LLC|Inc|Corp|Corporation|Trust|Ltd|Company)(?:\s+(?:LLC|Inc|Corp|Corporation|Trust|Ltd|Company))*)\b", - text, + if outcome.ok: + stats.model_extracted_pages += 1 + section.append( + _model_block( + _marker(MARKER_MODEL_EXTRACTED, index=index), + outcome.text or "", + ) + ) + section.append("\n") + else: + stats.unavailable_pages += 1 + section.append( + _marker( + MARKER_PAGE_TEXT_UNAVAILABLE_WITH_RASTER, + index=index, + reason=outcome.reason or REASON_EMPTY_MODEL_RESPONSE, + ) + ) + section.append("\n") + return "".join(section).rstrip("\n") + + stats.unavailable_pages += 1 + section.append( + _marker( + MARKER_PAGE_TEXT_UNAVAILABLE_NO_RASTER, + reason=_page_failure_reason(page=page, render_errors=render_errors), ) ) - observation = f"Named in {title}" - entities = [ - {"name": name, "type": "Person", "observations": [observation]} - for name in sorted(names) + section.append("\n") + return "".join(section).rstrip("\n") + + +def _render_header( + *, + title: str, + payload: dict[str, Any], + date: str, + timestamp_source: str, + stats: _RenderStats, + warnings: tuple[str, ...], +) -> str: + page_count = int(payload.get("page_count") or 0) + lines = [ + f"# {title}", + "", + "**Type:** Document", + f"**Pages:** {page_count}", + f"**Date:** {date} ({timestamp_source})", + ( + f"**Extraction:** {payload.get('engine', 'unknown')} — " + f"{stats.text_layer_pages} text-layer, " + f"{stats.model_extracted_pages} model-extracted, " + f"{stats.unavailable_pages} unavailable of {page_count} pages; " + f"{stats.image_described_pages} image-described; " + f"{stats.model_calls} model calls" + ), + ] + if warnings: + lines.extend(["", "**Worker warnings:**"]) + lines.extend(f"- {warning}" for warning in warnings) + lines.extend(["", "---"]) + return "\n".join(lines) + + +def _render_transcript( + *, + title: str, + payload: dict[str, Any], + rasters: dict[int, Path], + render_errors: dict[int, str], + timestamp_choice: _TimestampChoice, + segment_timestamp: float, + warnings: tuple[str, ...], +) -> tuple[str, _RenderStats]: + stats = _RenderStats() + pages = sorted(payload.get("pages", []), key=lambda page: int(page["index"])) + sections = [ + _render_page_section( + page, + rasters=rasters, + render_errors=render_errors, + stats=stats, + ) + for page in pages ] - entities.extend( - {"name": name, "type": "Organization", "observations": [observation]} - for name in sorted(orgs) + header = _render_header( + title=title, + payload=payload, + date=dt.datetime.fromtimestamp(segment_timestamp).strftime("%Y-%m-%d"), + timestamp_source=timestamp_choice.source, + stats=stats, + warnings=warnings, + ) + body = "\n\n".join(sections) + return f"{header}\n\n{body}\n", stats + + +def _collect_worker_outputs(pdf_path: Path, *, render_dir: Path) -> _WorkerOutputs: + first = run_pdf_worker("extract", pdf_path).payload + render_pages = _render_set(first) + second: dict[str, Any] | None = None + if render_pages: + second = run_pdf_worker( + "extract", + pdf_path, + render_pages=sorted(render_pages), + render_dir=render_dir, + ).payload + + warnings = _merge_warnings( + first.get("warnings", []), + second.get("warnings", []) if second is not None else [], + ) + rasters, render_errors = _rendered_pages(second or first, render_dir) + return _WorkerOutputs( + payload=first, + rasters=rasters, + render_errors=render_errors, + warnings=warnings, ) - return entities + + +def _prepare_document( + outputs: _WorkerOutputs, + *, + pdf_path: Path, + timestamp_choice: _TimestampChoice, + segment_timestamp: float, +) -> _PreparedDocument: + transcript, stats = _render_transcript( + title=pdf_path.stem, + payload=outputs.payload, + rasters=outputs.rasters, + render_errors=outputs.render_errors, + timestamp_choice=timestamp_choice, + segment_timestamp=segment_timestamp, + warnings=outputs.warnings, + ) + return _PreparedDocument( + payload=outputs.payload, + transcript=transcript, + rasters=outputs.rasters, + warnings=outputs.warnings, + timestamp_source=timestamp_choice.source, + text_layer_pages=stats.text_layer_pages, + model_extracted_pages=stats.model_extracted_pages, + unavailable_pages=stats.unavailable_pages, + image_described_pages=stats.image_described_pages, + model_calls=stats.model_calls, + ) + + +def _install_artifacts( + *, + pdf_path: Path, + segment_dir: Path, + prepared: _PreparedDocument, +) -> Path: + original_path = segment_dir / _ORIGINAL_FILENAME + install_source_file(pdf_path, original_path) + + pages_dir = segment_dir / "pages" + for index, raster_path in sorted(prepared.rasters.items()): + install_source_file(raster_path, pages_dir / _page_name(index)) + + transcript_path = segment_dir / _TRANSCRIPT_FILENAME + write_text(transcript_path, prepared.transcript, mode=PRIVATE_IMPORT_FILE_MODE) + return transcript_path + + +def _worker_error_message(pdf_path: Path, exc: PdfWorkerError) -> str: + name = pdf_path.name + detail = "" + if exc.payload: + detail = _collapse_line(exc.payload.get("detail") or "") + if not detail: + detail = _collapse_line(str(exc)) + + if isinstance(exc, PdfWorkerEncryptedError): + return f"{name}: password-protected PDF" + if isinstance(exc, PdfWorkerCorruptError): + return f"{name}: corrupt PDF ({detail})" + if isinstance(exc, PdfWorkerRenderIOError): + return f"{name}: PDF render I/O failed ({detail})" + if isinstance(exc, PdfWorkerTimeoutError): + return f"{name}: PDF worker timed out after {exc.timeout_seconds:g}s" + if isinstance(exc, PdfWorkerEngineError): + return f"{name}: PDF worker failed ({detail})" + return f"{name}: PDF worker failed ({detail})" + + +def _manifest_meta(prepared: _PreparedDocument) -> dict[str, Any]: + return { + "page_count": prepared.payload.get("page_count"), + "engine": prepared.payload.get("engine"), + "timestamp_source": prepared.timestamp_source, + "text_layer_pages": prepared.text_layer_pages, + "model_extracted_pages": prepared.model_extracted_pages, + "unavailable_pages": prepared.unavailable_pages, + "image_described_pages": prepared.image_described_pages, + "model_calls": prepared.model_calls, + "warnings": list(prepared.warnings), + } class DocumentImporter: name = "document" display_name = "Documents" file_patterns = ["*.pdf"] - description = ( - "Import PDF documents with text extraction and vision fallback for scanned PDFs" - ) + description = "Import PDF documents with worker-backed text and raster extraction" def detect(self, path: Path) -> bool: return bool(_find_pdfs(path)) def preview(self, path: Path) -> ImportPreview: - require_extra("pdf") - from pypdf import PdfReader - + require_extra("pdf-import") pdfs = _find_pdfs(path) if not pdfs: return ImportPreview( @@ -199,19 +670,28 @@ class DocumentImporter: timestamps: list[float] = [] total_pages = 0 + failures: list[str] = [] for pdf_path in pdfs: - reader = PdfReader(str(pdf_path)) - timestamps.append(_get_pdf_timestamp(reader, pdf_path)) - total_pages += len(reader.pages) + try: + payload = run_pdf_worker("inspect", pdf_path).payload + except PdfWorkerError as exc: + failures.append(_worker_error_message(pdf_path, exc)) + continue + timestamps.append(_choose_timestamp(payload, pdf_path).timestamp) + total_pages += int(payload.get("page_count") or 0) dates = sorted( dt.datetime.fromtimestamp(ts).strftime("%Y%m%d") for ts in timestamps ) + date_range = (dates[0], dates[-1]) if dates else ("", "") + summary = f"{len(pdfs)} PDF documents, {total_pages} total pages" + if failures: + summary += f"; {len(failures)} unreadable ({'; '.join(failures)})" return ImportPreview( - date_range=(dates[0], dates[-1]), + date_range=date_range, item_count=len(pdfs), entity_count=0, - summary=f"{len(pdfs)} PDF documents, {total_pages} total pages", + summary=summary, ) def process( @@ -223,10 +703,9 @@ class DocumentImporter: import_id: str | None = None, progress_callback: Callable | None = None, dry_run: bool = False, + force: bool = False, ) -> ImportResult: - require_extra("pdf") - from pypdf import PdfReader - + require_extra("pdf-import") pdfs = _find_pdfs(path) import_id = import_id or dt.datetime.now().strftime("%Y%m%d_%H%M%S") if not pdfs: @@ -238,83 +717,82 @@ class DocumentImporter: summary="No PDF documents found to import", ) + journal_root = Path(journal_root) created_files: list[str] = [] errors: list[str] = [] + hard_failures: list[str] = [] segments: list[tuple[str, str]] = [] - manifest_entries: list[dict] = [] - entities_seeded = 0 + manifest_entries: list[dict[str, Any]] = [] timestamps: list[float] = [] - used_keys: dict[str, set[str]] = {} + used_keys: set[tuple[str, str]] = set() for index, pdf_path in enumerate(pdfs): - try: - reader = PdfReader(str(pdf_path)) - ts = _get_pdf_timestamp(reader, pdf_path) - text, meta = extract_pdf_text(pdf_path) - page_count = meta["page_count"] - extraction_method = meta["extraction_method"] - if meta["vision_error"]: - errors.append( - f"{pdf_path.name}: scanned PDF — vision failed ({meta['vision_error']}); using sparse pypdf text" + with tempfile.TemporaryDirectory() as render_root: + try: + outputs = _collect_worker_outputs( + pdf_path=pdf_path, + render_dir=Path(render_root), ) - - seg_dt = dt.datetime.fromtimestamp(ts) - day = seg_dt.strftime("%Y%m%d") - seg_key = f"{seg_dt.strftime('%H%M%S')}_0" - day_used = used_keys.setdefault(day, set()) - while seg_key in day_used: - ts += 1 - seg_dt = dt.datetime.fromtimestamp(ts) - day = seg_dt.strftime("%Y%m%d") - seg_key = f"{seg_dt.strftime('%H%M%S')}_0" - day_used = used_keys.setdefault(day, set()) - day_used.add(seg_key) - timestamps.append(ts) - - title = pdf_path.stem - date_str = seg_dt.strftime("%Y-%m-%d") - segment_dir = day_path(day) / "import.document" / seg_key - segment_dir.mkdir(parents=True, exist_ok=True) - - original_path = segment_dir / "original.pdf" - install_source_file(pdf_path, original_path) - md_path = segment_dir / "document_transcript.md" - md_text = _render_document_markdown( - title, - text, - {"page_count": page_count, "date": date_str}, - ) - write_text(md_path, md_text) - - created_files.append(str(md_path)) - segments.append((day, seg_key)) - manifest_entries.append( - { - "id": f"document-{index}", - "title": title, - "date": day, - "type": "document", - "preview": text[:200], - "meta": { - "page_count": page_count, - "extraction_method": extraction_method, - }, - "segments": [{"day": day, "key": seg_key}], - } - ) - - if facet: - try: - resolved = seed_entities( - facet, day, _extract_entities(text, title) - ) - entities_seeded += len(resolved) - except Exception as exc: + timestamp_choice = _choose_timestamp(outputs.payload, pdf_path) + claim = _claim_segment( + journal_root, + timestamp=timestamp_choice.timestamp, + sha256=str(outputs.payload.get("sha256") or ""), + used_keys=used_keys, + force=force, + ) + if claim.already_imported: errors.append( - f"Failed to seed entities for {pdf_path.name}: {exc}" + f"{pdf_path.name}: skipped (already imported; use --force to regenerate)" + ) + continue + + prepared = _prepare_document( + outputs, + pdf_path=pdf_path, + timestamp_choice=timestamp_choice, + segment_timestamp=claim.timestamp, + ) + errors.extend( + f"{pdf_path.name}: {warning}" for warning in prepared.warnings + ) + timestamps.append(claim.timestamp) + + segment_dir = ( + journal_root + / "chronicle" + / claim.day + / _DOCUMENT_STREAM + / claim.seg_key + ) + md_path = segment_dir / _TRANSCRIPT_FILENAME + if not dry_run: + md_path = _install_artifacts( + pdf_path=pdf_path, + segment_dir=segment_dir, + prepared=prepared, ) - except Exception as exc: - errors.append(f"Failed to process {pdf_path.name}: {exc}") + created_files.append(str(md_path)) + + segments.append((claim.day, claim.seg_key)) + manifest_entries.append( + { + "id": f"document-{index}", + "title": pdf_path.stem, + "date": claim.day, + "type": "document", + "preview": prepared.transcript[:200], + "meta": _manifest_meta(prepared), + "segments": [{"day": claim.day, "key": claim.seg_key}], + } + ) + except PdfWorkerError as exc: + message = _worker_error_message(pdf_path, exc) + errors.append(message) + hard_failures.append(message) + except Exception as exc: + message = f"{pdf_path.name}: document import failed ({_collapse_line(str(exc) or type(exc).__name__)})" + errors.append(message) if progress_callback: earliest = None @@ -331,10 +809,13 @@ class DocumentImporter: len(pdfs), earliest_date=earliest, latest_date=latest, - entities_found=entities_seeded, + entities_found=0, ) - write_content_manifest(import_id, manifest_entries) + if not dry_run and manifest_entries: + write_content_manifest( + import_id, manifest_entries, journal_root=journal_root + ) if timestamps: earliest = dt.datetime.fromtimestamp(min(timestamps)).strftime("%Y%m%d") @@ -345,9 +826,10 @@ class DocumentImporter: return ImportResult( entries_written=len(segments), - entities_seeded=entities_seeded, + entities_seeded=0, files_created=created_files, errors=errors, + hard_failures=tuple(hard_failures), summary=( f"Imported {len(segments)} PDF documents across " f"{len({day for day, _ in segments})} days into {len(segments)} segments" diff --git a/solstone/think/importers/file_importer.py b/solstone/think/importers/file_importer.py index 5d7151e67..1f92e14ec 100644 --- a/solstone/think/importers/file_importer.py +++ b/solstone/think/importers/file_importer.py @@ -30,6 +30,7 @@ class ImportResult: files_created: list[str] errors: list[str] summary: str + hard_failures: tuple[str, ...] = () segments: list[tuple[str, str]] | None = None date_range: tuple[str, str] | None = None merge_summary: dict[str, Any] | None = None @@ -59,6 +60,7 @@ class FileImporter(Protocol): import_id: str | None = None, progress_callback: Callable | None = None, dry_run: bool = False, + force: bool = False, ) -> ImportResult: ... diff --git a/tests/pdf_worker_fixtures.py b/tests/pdf_worker_fixtures.py index 9405e72a4..7e747358a 100644 --- a/tests/pdf_worker_fixtures.py +++ b/tests/pdf_worker_fixtures.py @@ -8,6 +8,9 @@ from pathlib import Path from pypdf import PdfReader, PdfWriter TEXT_SENTINEL = "SOLPDF_SENTINEL_PAGE_2" +TEXT_RICH_SENTINEL = "SOLSTONE_TEXT_RICH_PAGE" +MIXED_TEXT_SENTINEL = "SOLSTONE_MIXED_TEXT_LAYER" +IMAGE_TEXT_SENTINEL = "SOLSTONE_TEXT_WITH_IMAGE_LAYER" PAGE_WIDTH_PT = 612 PAGE_HEIGHT_PT = 792 @@ -135,6 +138,26 @@ def write_text_fixture(path: Path) -> Path: ) +def write_text_rich_fixture(path: Path) -> Path: + return write_pdf( + path, + [ + { + "text": ( + f"{TEXT_RICH_SENTINEL} first page has more than fifty " + "non-whitespace characters for text-layer importer coverage." + ) + }, + { + "text": ( + "Second rich text page also stays above the threshold so " + "the document importer makes no model calls." + ) + }, + ], + ) + + def write_image_only_fixture(path: Path) -> Path: return write_pdf( path, @@ -145,6 +168,49 @@ def write_image_only_fixture(path: Path) -> Path: ) +def write_text_with_image_rich_fixture(path: Path) -> Path: + return write_pdf( + path, + [ + { + "text": ( + f"{IMAGE_TEXT_SENTINEL} has a healthy text layer and a " + "large embedded image area that requires an overlay description." + ), + "image": (72, 144, 420, 420), + }, + { + "text": ( + "Pure text companion page remains above the threshold and " + "must not trigger a model description." + ) + }, + ], + ) + + +def write_importer_mixed_fixture(path: Path) -> Path: + return write_pdf( + path, + [ + { + "text": ( + f"{MIXED_TEXT_SENTINEL} page has enough extractable text " + "to be emitted verbatim without model assistance." + ) + }, + {"image": (0, 0, PAGE_WIDTH_PT, PAGE_HEIGHT_PT)}, + { + "text": ( + f"{IMAGE_TEXT_SENTINEL} mixed page has text plus a large " + "embedded image so the importer emits a description overlay." + ), + "image": (72, 144, 420, 420), + }, + ], + ) + + def write_mixed_fixture(path: Path) -> Path: return write_pdf( path, diff --git a/tests/test_importer.py b/tests/test_importer.py index 1e4175fce..c498bfbfe 100644 --- a/tests/test_importer.py +++ b/tests/test_importer.py @@ -7,6 +7,7 @@ import importlib import json import os import subprocess +import sys import time import uuid import zipfile @@ -23,6 +24,7 @@ from solstone.convey.secure_listener import ConveyIdentity from solstone.think.importers.file_importer import ImportPreview, ImportResult from solstone.think.importers.shared import install_source_file from solstone.think.utils import day_path +from tests.pdf_worker_fixtures import write_pdf PDF_SENTINEL = "SOLSTONE_PDF_SENTINEL_DOCUMENT_ROUTING" @@ -80,12 +82,16 @@ def _configure_text_import_runtime(monkeypatch, mod): def _write_text_pdf(path: Path, text: str) -> None: - pytest.importorskip("weasyprint") - pytest.importorskip("pypdf") - from weasyprint import HTML - - HTML(string=f"

Fixture

{text}

").write_pdf( - path + write_pdf( + path, + [ + { + "text": ( + f"{text} has enough extractable text for the document importer " + "to use the text layer without model calls." + ) + } + ], ) @@ -93,7 +99,14 @@ def _configure_document_import_runtime(monkeypatch, mod, doc_mod, *, timestamp: monkeypatch.setattr(mod, "CallosumConnection", lambda **kwargs: MagicMock()) monkeypatch.setattr(mod, "_status_emitter", lambda: None) monkeypatch.setattr(mod, "index_file", lambda *args, **kwargs: None) - monkeypatch.setattr(doc_mod, "_get_pdf_timestamp", lambda reader, path: timestamp) + monkeypatch.setattr( + doc_mod, + "_choose_timestamp", + lambda payload, path: doc_mod._TimestampChoice( + timestamp=timestamp, + source="file-mtime", + ), + ) def _read_action_entries(journal_root: Path) -> list[dict]: @@ -821,6 +834,25 @@ def test_corrupt_pdf_routes_to_document_importer_and_reports_error( assert not (tmp_path / "chronicle" / "20251205" / "import.text").exists() +def test_importer_main_exits_nonzero_after_completed_hard_failures(monkeypatch): + mod = importlib.import_module("solstone.think.importers.cli") + monkeypatch.setattr( + sys, + "argv", + ["journal", "importer", "/tmp/corrupt.pdf", "20251205_163000"], + ) + monkeypatch.setattr( + mod, + "import_one", + lambda *args, **kwargs: {"hard_failures": ["corrupt.pdf: corrupt PDF"]}, + ) + + with pytest.raises(SystemExit) as exc: + mod.main() + + assert exc.value.code == 1 + + def test_write_segment(tmp_path): """Test write_segment creates a segment directory and JSONL file.""" mod = importlib.import_module("solstone.think.importers.shared") diff --git a/tests/test_importer_documents.py b/tests/test_importer_documents.py index ec894ab5e..65141a2be 100644 --- a/tests/test_importer_documents.py +++ b/tests/test_importer_documents.py @@ -1,382 +1,809 @@ # SPDX-License-Identifier: AGPL-3.0-only # Copyright (c) 2026 sol pbc +from __future__ import annotations + import datetime as dt +import hashlib import importlib +import json import os - +import shutil +import time +from pathlib import Path + +from solstone.observe.pdf_worker import ( + PdfWorkerEncryptedError, + PdfWorkerRenderIOError, + PdfWorkerSuccess, +) from solstone.think.importers.file_importer import FILE_IMPORTER_REGISTRY +from solstone.think.models import NoBrainConfiguredError +from tests.pdf_worker_fixtures import ( + IMAGE_TEXT_SENTINEL, + MIXED_TEXT_SENTINEL, + TEXT_RICH_SENTINEL, + write_encrypted_fixture_pair, + write_image_only_fixture, + write_importer_mixed_fixture, + write_pdf, + write_text_rich_fixture, + write_text_with_image_rich_fixture, +) + + +def _mod(): + return importlib.import_module("solstone.think.importers.documents") + + +def _set_mtime(path: Path, when: dt.datetime) -> None: + ts = when.timestamp() + os.utime(path, (ts, ts)) + + +def _fixed_mtime(path: Path) -> None: + _set_mtime(path, dt.datetime(2026, 3, 4, 12, 0, 0).astimezone()) + + +def _install_generate(monkeypatch, mod, outcomes): + calls: list[dict] = [] + iterator = iter(outcomes) + def fake_generate(*, contents, context, **kwargs): + del kwargs + calls.append({"contents": contents, "context": context}) + outcome = next(iterator) + if isinstance(outcome, BaseException): + raise outcome + return outcome + + monkeypatch.setattr(mod, "generate", fake_generate) + return calls + + +def _import_pdf(mod, pdf: Path, journal: Path, **kwargs): + return mod.importer.process( + pdf, + journal, + import_id=kwargs.pop("import_id", "import-test"), + **kwargs, + ) + + +def _segment_dir(journal: Path, result) -> Path: + day, seg_key = result.segments[0] + return journal / "chronicle" / day / "import.document" / seg_key + + +def _transcript(journal: Path, result) -> str: + return (_segment_dir(journal, result) / "document_transcript.md").read_text( + encoding="utf-8" + ) -class MockPage: - def __init__(self, text: str): - self._text = text - def extract_text(self): - return self._text +def _assert_cited_rasters_exist(segment_dir: Path) -> None: + transcript = (segment_dir / "document_transcript.md").read_text(encoding="utf-8") + for line in transcript.splitlines(): + marker = "pages/page-" + if marker not in line: + continue + rel = line[line.index(marker) :].split("]", 1)[0] + assert (segment_dir / rel).is_file() -class MockPdfReader: - def __init__(self, path): - self.path = str(path) - self.pages = [ - MockPage( - "Page text content here with enough characters to pass threshold." - ), - MockPage( - "Second page text content here with enough characters to pass threshold." - ), - ] - self.metadata = {"/CreationDate": "D:20260115120000"} +def _assert_no_bare_single_h1(transcript: str) -> None: + for index, line in enumerate(transcript.splitlines()): + if line.startswith("# "): + assert index == 0 + + +def _snapshot_tree(path: Path) -> dict[str, bytes]: + return { + str(child.relative_to(path)): child.read_bytes() + for child in sorted(path.rglob("*")) + if child.is_file() + } + + +def _payload(pdf: Path, *, pages: list[dict], warnings=(), metadata=None) -> dict: + return { + "schema": "sol-pdf/1", + "engine": "pdfium fixture / pypdfium2 fixture", + "sha256": hashlib.sha256(pdf.read_bytes()).hexdigest(), + "page_count": len(pages), + "encrypted": False, + "warnings": list(warnings), + "render": None, + "metadata": metadata + or { + "title": None, + "author": None, + "creation_date": None, + "mod_date": None, + "producer": None, + }, + "pages": pages, + } def test_detect_pdf_file(tmp_path): - mod = importlib.import_module("solstone.think.importers.documents") + mod = _mod() pdf = tmp_path / "file.pdf" pdf.write_bytes(b"%PDF-1.4") assert mod.importer.detect(pdf) is True def test_detect_non_pdf(tmp_path): - mod = importlib.import_module("solstone.think.importers.documents") + mod = _mod() txt = tmp_path / "file.txt" txt.write_text("hello", encoding="utf-8") assert mod.importer.detect(txt) is False def test_detect_directory_with_pdfs(tmp_path): - mod = importlib.import_module("solstone.think.importers.documents") + mod = _mod() (tmp_path / "a.pdf").write_bytes(b"%PDF-1.4") (tmp_path / "b.pdf").write_bytes(b"%PDF-1.4") assert mod.importer.detect(tmp_path) is True def test_detect_empty_directory(tmp_path): - mod = importlib.import_module("solstone.think.importers.documents") + mod = _mod() assert mod.importer.detect(tmp_path) is False -def test_preview_single_pdf(tmp_path, monkeypatch): - mod = importlib.import_module("solstone.think.importers.documents") - pdf = tmp_path / "file.pdf" - pdf.write_bytes(b"%PDF-1.4") - monkeypatch.setattr("pypdf.PdfReader", MockPdfReader) +def test_pure_text_layer_pdf_imports_verbatim_with_zero_model_calls( + tmp_path, monkeypatch +): + mod = _mod() + pdf = write_text_rich_fixture(tmp_path / "contract.pdf") + _fixed_mtime(pdf) + calls = _install_generate( + monkeypatch, mod, [AssertionError("unexpected model call")] + ) - preview = mod.importer.preview(pdf) + result = _import_pdf(mod, pdf, tmp_path) - assert preview.date_range == ("20260115", "20260115") - assert preview.item_count == 1 - assert preview.entity_count == 0 - assert preview.summary == "1 PDF documents, 2 total pages" + segment_dir = _segment_dir(tmp_path, result) + transcript = _transcript(tmp_path, result) + assert result.errors == [] + assert result.hard_failures == () + assert result.entries_written == 1 + assert result.entities_seeded == 0 + assert result.files_created == [str(segment_dir / "document_transcript.md")] + assert (segment_dir / "original.pdf").read_bytes() == pdf.read_bytes() + assert calls == [] + assert "## Page 1" in transcript + assert "## Page 2" in transcript + assert TEXT_RICH_SENTINEL in transcript + assert mod.MARKER_MODEL_EXTRACTED.split("{NNNN}", 1)[0] not in transcript + assert ( + "**Extraction:" in transcript + and "2 text-layer, 0 model-extracted, 0 unavailable of 2 pages; " + "0 image-described; 0 model calls" + in transcript + ) -def test_extract_pdf_text_digital(tmp_path, monkeypatch): - mod = importlib.import_module("solstone.think.importers.documents") - pdf = tmp_path / "digital.pdf" - pdf.write_bytes(b"%PDF-1.4") - monkeypatch.setattr("pypdf.PdfReader", MockPdfReader) +def test_scanned_pdf_uses_reading_prompt_once_per_page_and_installs_rasters( + tmp_path, monkeypatch +): + mod = _mod() + pdf = write_image_only_fixture(tmp_path / "scan.pdf") + _fixed_mtime(pdf) + calls = _install_generate( + monkeypatch, + mod, + ["# Extracted page 1\n\nBody 1", "Extracted page 2"], + ) - text, meta = mod.extract_pdf_text(pdf) + result = _import_pdf(mod, pdf, tmp_path) - assert "Page text content here" in text - assert meta == { - "page_count": 2, - "is_scanned": False, - "extraction_method": "pypdf", - "vision_error": None, - } + segment_dir = _segment_dir(tmp_path, result) + transcript = _transcript(tmp_path, result) + assert [call["context"] for call in calls] == [ + "import.document.vision", + "import.document.vision", + ] + assert all("# [Document Title or Type]" in call["contents"][0] for call in calls) + assert mod._marker(mod.MARKER_MODEL_EXTRACTED, index=1) in transcript + assert mod._marker(mod.MARKER_MODEL_EXTRACTED, index=2) in transcript + assert "> # Extracted page 1" in transcript + assert ">\n> Body 1" in transcript + assert (segment_dir / "pages" / "page-0001.png").is_file() + assert (segment_dir / "pages" / "page-0002.png").is_file() + assert ( + "0 text-layer, 2 model-extracted, 0 unavailable of 2 pages; " + "0 image-described; 2 model calls" in transcript + ) + _assert_no_bare_single_h1(transcript) + _assert_cited_rasters_exist(segment_dir) -def test_extract_pdf_text_scanned_vision_success(tmp_path, monkeypatch): - mod = importlib.import_module("solstone.think.importers.documents") +def test_mixed_pdf_keeps_text_extracts_scanned_and_describes_text_image_page( + tmp_path, monkeypatch +): + mod = _mod() + pdf = write_importer_mixed_fixture(tmp_path / "mixed.pdf") + _fixed_mtime(pdf) + calls = _install_generate( + monkeypatch, + mod, + ["Scanned page text", "Description of embedded chart"], + ) - class ScannedReader(MockPdfReader): - def __init__(self, path): - self.path = str(path) - self.pages = [MockPage("x"), MockPage("y")] - self.metadata = {"/CreationDate": "D:20260115120000"} + result = _import_pdf(mod, pdf, tmp_path) + + segment_dir = _segment_dir(tmp_path, result) + transcript = _transcript(tmp_path, result) + assert [call["context"] for call in calls] == [ + "import.document.vision", + "import.document.describe", + ] + assert MIXED_TEXT_SENTINEL in transcript + assert IMAGE_TEXT_SENTINEL in transcript + assert mod._marker(mod.MARKER_MODEL_EXTRACTED, index=2) in transcript + assert mod._marker(mod.MARKER_IMAGE_DESCRIPTION, index=3) in transcript + assert "> Scanned page text" in transcript + assert "> Description of embedded chart" in transcript + assert ( + "2 text-layer, 1 model-extracted, 0 unavailable of 3 pages; " + "1 image-described; 2 model calls" in transcript + ) + _assert_cited_rasters_exist(segment_dir) - pdf = tmp_path / "scan.pdf" - pdf.write_bytes(b"%PDF-1.4") - monkeypatch.setattr("pypdf.PdfReader", ScannedReader) - calls = [] - def fake_vision(pdf_path, page_count): - calls.append((pdf_path.name, page_count)) - return "Vision extracted text" +def test_text_page_with_large_image_gets_one_description_only(tmp_path, monkeypatch): + mod = _mod() + pdf = write_text_with_image_rich_fixture(tmp_path / "image-text.pdf") + _fixed_mtime(pdf) + calls = _install_generate(monkeypatch, mod, ["Image description"]) - monkeypatch.setattr(mod, "_extract_text_vision", fake_vision) + result = _import_pdf(mod, pdf, tmp_path) - text, meta = mod.extract_pdf_text(pdf) + transcript = _transcript(tmp_path, result) + assert [call["context"] for call in calls] == ["import.document.describe"] + assert mod._marker(mod.MARKER_IMAGE_DESCRIPTION, index=1) in transcript + assert "page-0002.png" not in transcript + assert "2 text-layer, 0 model-extracted, 0 unavailable of 2 pages" in transcript + assert "1 image-described; 1 model calls" in transcript - assert calls == [("scan.pdf", 2)] - assert text == "Vision extracted text" - assert meta["page_count"] == 2 - assert meta["is_scanned"] is True - assert meta["extraction_method"] == "vision" - assert meta["vision_error"] is None +def test_marker_bytes_are_exact(): + mod = _mod() -def test_extract_pdf_text_scanned_vision_failure_returns_sparse_text( + assert mod.MARKER_MODEL_EXTRACTED == ( + "> [model-extracted from page image — may contain errors; " + "original: pages/page-{NNNN}.png]" + ) + assert mod.MARKER_IMAGE_DESCRIPTION == ( + "> [image description — model-generated; original: pages/page-{NNNN}.png]" + ) + assert mod.MARKER_PAGE_TEXT_UNAVAILABLE_WITH_RASTER == ( + "> [page text unavailable — {reason}; " + "page image preserved at pages/page-{NNNN}.png]" + ) + assert mod.MARKER_IMAGE_DESCRIPTION_UNAVAILABLE_WITH_RASTER == ( + "> [image description unavailable — {reason}; " + "page image preserved at pages/page-{NNNN}.png]" + ) + assert mod.MARKER_PAGE_TEXT_UNAVAILABLE_NO_RASTER == ( + "> [page text unavailable — {reason}; no page image could be produced]" + ) + assert mod.MARKER_IMAGE_DESCRIPTION_UNAVAILABLE_NO_RASTER == ( + "> [image description unavailable — {reason}; no page image could be produced]" + ) + assert mod._marker(mod.MARKER_MODEL_EXTRACTED, index=2) == ( + "> [model-extracted from page image — may contain errors; " + "original: pages/page-0002.png]" + ) + assert mod._marker(mod.MARKER_IMAGE_DESCRIPTION, index=12) == ( + "> [image description — model-generated; original: pages/page-0012.png]" + ) + assert ( + mod._marker( + mod.MARKER_PAGE_TEXT_UNAVAILABLE_WITH_RASTER, + index=3, + reason="boom", + ) + == "> [page text unavailable — boom; page image preserved at pages/page-0003.png]" + ) + assert mod._marker( + mod.MARKER_IMAGE_DESCRIPTION_UNAVAILABLE_WITH_RASTER, + index=4, + reason="boom", + ) == ( + "> [image description unavailable — boom; " + "page image preserved at pages/page-0004.png]" + ) + assert ( + mod._marker( + mod.MARKER_PAGE_TEXT_UNAVAILABLE_NO_RASTER, + reason="boom", + ) + == "> [page text unavailable — boom; no page image could be produced]" + ) + assert ( + mod._marker( + mod.MARKER_IMAGE_DESCRIPTION_UNAVAILABLE_NO_RASTER, + reason="boom", + ) + == "> [image description unavailable — boom; no page image could be produced]" + ) + + +def test_vision_extraction_failure_marks_one_scanned_page_unavailable( tmp_path, monkeypatch ): - mod = importlib.import_module("solstone.think.importers.documents") + mod = _mod() + pdf = write_image_only_fixture(tmp_path / "scan.pdf") + _fixed_mtime(pdf) + calls = _install_generate( + monkeypatch, + mod, + [RuntimeError("vision failed"), "Second page model text"], + ) - class ScannedReader(MockPdfReader): - def __init__(self, path): - self.path = str(path) - self.pages = [MockPage("x")] - self.metadata = {"/CreationDate": "D:20260115120000"} + result = _import_pdf(mod, pdf, tmp_path) - pdf = tmp_path / "scan.pdf" - pdf.write_bytes(b"%PDF-1.4") - monkeypatch.setattr("pypdf.PdfReader", ScannedReader) - monkeypatch.setattr( - mod, - "_extract_text_vision", - lambda pdf_path, page_count: (_ for _ in ()).throw( - RuntimeError("vision failed") - ), + segment_dir = _segment_dir(tmp_path, result) + transcript = _transcript(tmp_path, result) + assert len(calls) == 2 + assert result.hard_failures == () + assert ( + mod._marker( + mod.MARKER_PAGE_TEXT_UNAVAILABLE_WITH_RASTER, + index=1, + reason="vision failed", + ) + in transcript ) + assert mod._marker(mod.MARKER_MODEL_EXTRACTED, index=2) in transcript + assert (segment_dir / "pages" / "page-0001.png").is_file() + assert (segment_dir / "pages" / "page-0002.png").is_file() - text, meta = mod.extract_pdf_text(pdf) - assert text == "x" - assert meta["page_count"] == 1 - assert meta["is_scanned"] is True - assert meta["extraction_method"] == "pypdf" - assert meta["vision_error"] == "vision failed" +def test_description_failure_keeps_full_text_and_uses_description_marker( + tmp_path, monkeypatch +): + mod = _mod() + pdf = write_text_with_image_rich_fixture(tmp_path / "image-text.pdf") + _fixed_mtime(pdf) + _install_generate(monkeypatch, mod, [NoBrainConfiguredError()]) + result = _import_pdf(mod, pdf, tmp_path) -def test_process_text_pdf(tmp_path, monkeypatch): - mod = importlib.import_module("solstone.think.importers.documents") - pdf = tmp_path / "contract.pdf" - pdf.write_bytes(b"%PDF-1.4") - monkeypatch.setenv("SOLSTONE_JOURNAL", str(tmp_path)) - monkeypatch.setattr("pypdf.PdfReader", MockPdfReader) - monkeypatch.setattr(mod, "day_path", lambda day: tmp_path / "chronicle" / day) - monkeypatch.setattr( - mod, - "write_content_manifest", - lambda import_id, entries: tmp_path / "manifest.jsonl", + transcript = _transcript(tmp_path, result) + assert IMAGE_TEXT_SENTINEL in transcript + assert ( + mod._marker( + mod.MARKER_IMAGE_DESCRIPTION_UNAVAILABLE_WITH_RASTER, + index=1, + reason=mod.REASON_NO_BRAIN_CONFIGURED, + ) + in transcript ) - monkeypatch.setattr(mod, "seed_entities", lambda facet, day, entities: entities) + assert "page text unavailable" not in transcript + assert result.hard_failures == () - result = mod.importer.process( - pdf, tmp_path, facet="work", import_id="20260115_120000" - ) - seg_dir = tmp_path / "chronicle" / "20260115" / "import.document" / "120000_0" - md_path = seg_dir / "document_transcript.md" +def test_model_call_cap_marks_overflow_pages_unavailable(tmp_path, monkeypatch): + mod = _mod() + pdf = write_image_only_fixture(tmp_path / "scan.pdf") + _fixed_mtime(pdf) + monkeypatch.setattr(mod, "MODEL_CALLS_MAX_PER_DOCUMENT", 1) + calls = _install_generate(monkeypatch, mod, ["Only first page"]) - assert result.entries_written == 1 - assert result.entities_seeded >= 0 - assert result.segments == [("20260115", "120000_0")] - assert result.files_created == [str(md_path)] - assert md_path.exists() - content = md_path.read_text(encoding="utf-8") - assert content.startswith("# contract") - assert "**Pages:** 2" in content - assert "Page text content here" in content - - -def test_process_creates_original_pdf(tmp_path, monkeypatch): - mod = importlib.import_module("solstone.think.importers.documents") - pdf = tmp_path / "original.pdf" - pdf.write_bytes(b"%PDF-1.4 fake") - os.utime(pdf, (1_600_000_000, 1_600_000_000)) - pdf_mtime = pdf.stat().st_mtime - monkeypatch.setenv("SOLSTONE_JOURNAL", str(tmp_path)) - monkeypatch.setattr("pypdf.PdfReader", MockPdfReader) - monkeypatch.setattr(mod, "day_path", lambda day: tmp_path / "chronicle" / day) - monkeypatch.setattr( - mod, - "write_content_manifest", - lambda import_id, entries: tmp_path / "manifest.jsonl", + result = _import_pdf(mod, pdf, tmp_path) + + transcript = _transcript(tmp_path, result) + assert len(calls) == 1 + assert mod._marker(mod.MARKER_MODEL_EXTRACTED, index=1) in transcript + assert ( + mod._marker( + mod.MARKER_PAGE_TEXT_UNAVAILABLE_WITH_RASTER, + index=2, + reason=mod.REASON_MODEL_CALL_LIMIT, + ) + in transcript ) - monkeypatch.setattr(mod, "seed_entities", lambda facet, day, entities: entities) + assert "1 model calls" in transcript - result = mod.importer.process(pdf, tmp_path, import_id="20260115_120000") - copied = ( - tmp_path - / "chronicle" - / "20260115" - / "import.document" - / "120000_0" - / "original.pdf" - ) - assert copied.exists() - assert copied.read_bytes() == b"%PDF-1.4 fake" - assert copied.stat().st_mtime == pdf_mtime - assert str(copied) not in result.files_created +def test_encrypted_fixture_is_hard_failure_without_segment(tmp_path): + mod = _mod() + clear = write_text_rich_fixture(tmp_path / "clear.pdf") + user_pdf = tmp_path / "encrypted.pdf" + owner_pdf = tmp_path / "owner.pdf" + write_encrypted_fixture_pair(clear, user_pdf, owner_pdf) + result = _import_pdf(mod, user_pdf, tmp_path) -def test_process_scanned_detection(tmp_path, monkeypatch): - mod = importlib.import_module("solstone.think.importers.documents") + assert result.entries_written == 0 + assert result.hard_failures == ("encrypted.pdf: password-protected PDF",) + assert result.errors == ["encrypted.pdf: password-protected PDF"] + assert not (tmp_path / "chronicle").exists() - class ScannedReader(MockPdfReader): - def __init__(self, path): - self.path = str(path) - self.pages = [MockPage("x"), MockPage("y")] - self.metadata = {"/CreationDate": "D:20260115120000"} - pdf = tmp_path / "scan.pdf" - pdf.write_bytes(b"%PDF-1.4") - monkeypatch.setenv("SOLSTONE_JOURNAL", str(tmp_path)) - monkeypatch.setattr("pypdf.PdfReader", ScannedReader) - monkeypatch.setattr(mod, "day_path", lambda day: tmp_path / "chronicle" / day) - monkeypatch.setattr( - mod, - "write_content_manifest", - lambda import_id, entries: tmp_path / "manifest.jsonl", +def test_corrupt_pdf_batched_with_good_pdf_imports_good_and_returns_hard_failure( + tmp_path, monkeypatch +): + mod = _mod() + good = write_text_rich_fixture(tmp_path / "good.pdf") + corrupt = tmp_path / "corrupt.pdf" + corrupt.write_bytes(b"%PDF-1.4\nnot a valid xref\n%%EOF\n") + _fixed_mtime(good) + calls = _install_generate( + monkeypatch, mod, [AssertionError("unexpected model call")] ) - monkeypatch.setattr(mod, "seed_entities", lambda facet, day, entities: entities) - calls = [] - def fake_vision(pdf_path, page_count): - calls.append((pdf_path.name, page_count)) - return "Vision extracted text" + result = _import_pdf(mod, tmp_path, tmp_path) - monkeypatch.setattr(mod, "_extract_text_vision", fake_vision) + assert calls == [] + assert result.entries_written == 1 + assert result.segments + assert "good" in _transcript(tmp_path, result) + assert len(result.hard_failures) == 1 + assert "corrupt.pdf" in result.hard_failures[0] + assert "corrupt" in result.hard_failures[0] - result = mod.importer.process(pdf, tmp_path, import_id="20260115_120000") - assert calls == [("scan.pdf", 2)] - assert result.errors == [] - md_path = ( - tmp_path - / "chronicle" - / "20260115" - / "import.document" - / "120000_0" - / "document_transcript.md" +def test_render_io_failure_is_hard_failure_before_segment_creation( + tmp_path, monkeypatch +): + mod = _mod() + pdf = tmp_path / "scan.pdf" + pdf.write_bytes(b"%PDF-1.4 synthetic") + render_root = tmp_path / "render-root" + payload = _payload( + pdf, + pages=[ + { + "index": 1, + "width_pt": 612.0, + "height_pt": 792.0, + "chars": 0, + "image_area_fraction": 1.0, + "rendered": None, + "error": None, + "text": "", + } + ], + ) + + class FakeTemporaryDirectory: + def __enter__(self): + render_root.mkdir() + return str(render_root) + + def __exit__(self, exc_type, exc, tb): + del exc_type, exc, tb + shutil.rmtree(render_root, ignore_errors=True) + + def fake_worker(command, pdf_path, **kwargs): + del command, pdf_path + if kwargs.get("render_pages"): + raise PdfWorkerRenderIOError( + "synthetic render I/O failure", + returncode=5, + stdout="", + stderr="", + payload={ + "schema": "sol-pdf/1", + "error": "render-io", + "detail": "synthetic render I/O failure", + }, + ) + return PdfWorkerSuccess(payload=payload, warnings=(), stderr="") + + monkeypatch.setattr(mod.tempfile, "TemporaryDirectory", FakeTemporaryDirectory) + monkeypatch.setattr(mod, "run_pdf_worker", fake_worker) + + result = _import_pdf(mod, pdf, tmp_path) + + assert result.entries_written == 0 + assert len(result.hard_failures) == 1 + assert "scan.pdf" in result.hard_failures[0] + assert "I/O" in result.hard_failures[0] + assert not (tmp_path / "chronicle").exists() + assert not render_root.exists() + + +def test_worker_warnings_land_in_header_and_import_result(tmp_path, monkeypatch): + mod = _mod() + pdf = tmp_path / "warning.pdf" + pdf.write_bytes(b"%PDF-1.4 synthetic") + warning = ( + "page 1: page render failed: synthetic render failure after text extraction" + ) + first = _payload( + pdf, + pages=[ + { + "index": 1, + "width_pt": 612.0, + "height_pt": 792.0, + "chars": 80, + "image_area_fraction": 0.25, + "rendered": None, + "error": None, + "text": "Healthy text layer survives even when render later fails.", + } + ], + ) + second = _payload( + pdf, + pages=[ + { + **first["pages"][0], + "chars": 0, + "image_area_fraction": 0.0, + "error": warning, + "text": "", + } + ], + warnings=[warning], ) - assert "Vision extracted text" in md_path.read_text(encoding="utf-8") + calls = [] + def fake_worker(command, pdf_path, **kwargs): + calls.append((command, Path(pdf_path).name, kwargs)) + return PdfWorkerSuccess( + payload=second if kwargs.get("render_pages") else first, + warnings=tuple(second["warnings"] if kwargs.get("render_pages") else ()), + stderr="", + ) -def test_process_scanned_all_fallback(tmp_path, monkeypatch): - mod = importlib.import_module("solstone.think.importers.documents") + monkeypatch.setattr(mod, "run_pdf_worker", fake_worker) - class ScannedReader(MockPdfReader): - def __init__(self, path): - self.path = str(path) - self.pages = [MockPage("x")] - self.metadata = {"/CreationDate": "D:20260115120000"} + result = _import_pdf(mod, pdf, tmp_path) - pdf = tmp_path / "scan.pdf" - pdf.write_bytes(b"%PDF-1.4") - monkeypatch.setenv("SOLSTONE_JOURNAL", str(tmp_path)) - monkeypatch.setattr("pypdf.PdfReader", ScannedReader) - monkeypatch.setattr(mod, "day_path", lambda day: tmp_path / "chronicle" / day) - monkeypatch.setattr( - mod, - "write_content_manifest", - lambda import_id, entries: tmp_path / "manifest.jsonl", - ) - monkeypatch.setattr(mod, "seed_entities", lambda facet, day, entities: entities) - monkeypatch.setattr( - mod, - "_extract_text_vision", - lambda pdf_path, page_count: (_ for _ in ()).throw( - RuntimeError("vision failed") - ), + transcript = _transcript(tmp_path, result) + assert len(calls) == 2 + assert "**Worker warnings:**" in transcript + assert f"- {warning}" in transcript + assert result.errors == [f"warning.pdf: {warning}"] + assert ( + mod._marker( + mod.MARKER_IMAGE_DESCRIPTION_UNAVAILABLE_NO_RASTER, + reason=warning, + ) + in transcript ) - result = mod.importer.process(pdf, tmp_path, import_id="20260115_120000") - assert result.errors == [ - "scan.pdf: scanned PDF — vision failed (vision failed); using sparse pypdf text" +def test_timestamp_metadata_offset_and_year_2999_fallback(tmp_path, monkeypatch): + mod = _mod() + old_tz = os.environ.get("TZ") + monkeypatch.setenv("TZ", "America/Denver") + time.tzset() + offset_pdf = write_pdf( + tmp_path / "offset.pdf", + [ + { + "text": ( + "Offset metadata document has enough text to avoid model " + "calls while proving the local day conversion." + ) + } + ], + {"ModDate": "D:20260304003000+02'00'"}, + ) + future_pdf = write_pdf( + tmp_path / "future.pdf", + [ + { + "text": ( + "Future metadata document falls back to the ordinary file " + "mtime because year 2999 is outside the sanity window." + ) + } + ], + {"ModDate": "D:29990101000000+00'00'"}, + ) + _set_mtime(future_pdf, dt.datetime(2026, 5, 6, 12, 0, 0).astimezone()) + + try: + offset_result = _import_pdf(mod, offset_pdf, tmp_path) + future_result = _import_pdf(mod, future_pdf, tmp_path) + + assert offset_result.segments[0][0] == "20260303" + assert "**Date:** 2026-03-03 (pdf-metadata)" in _transcript( + tmp_path, offset_result + ) + assert future_result.segments[0][0] == "20260506" + assert "**Date:** 2026-05-06 (file-mtime)" in _transcript( + tmp_path, future_result + ) + finally: + if old_tz is None: + os.environ.pop("TZ", None) + else: + os.environ["TZ"] = old_tz + time.tzset() + + +def test_segment_identity_force_controls_same_content_regeneration( + tmp_path, monkeypatch +): + mod = _mod() + first = write_importer_mixed_fixture(tmp_path / "first.pdf") + second = write_pdf( + tmp_path / "second.pdf", + [ + { + "text": ( + "Second different document deliberately shares the exact " + "same metadata timestamp as the first document." + ) + } + ], + ) + _fixed_mtime(first) + _fixed_mtime(second) + + _install_generate(monkeypatch, mod, ["Initial scanned text", "Initial description"]) + first_result = _import_pdf(mod, first, tmp_path) + first_segment = _segment_dir(tmp_path, first_result) + initial_snapshot = _snapshot_tree(first_segment) + manifest_path = tmp_path / "imports" / "import-test" / "content_manifest.jsonl" + initial_manifest = manifest_path.read_bytes() + + def fail_generate(**_kwargs): + raise AssertionError("same-content skip must not call generate") + + monkeypatch.setattr(mod, "generate", fail_generate) + skipped_result = _import_pdf(mod, first, tmp_path) + + assert skipped_result.entries_written == 0 + assert skipped_result.segments == [] + assert skipped_result.hard_failures == () + assert skipped_result.errors == [ + "first.pdf: skipped (already imported; use --force to regenerate)" ] - md_path = ( - tmp_path - / "chronicle" - / "20260115" - / "import.document" - / "120000_0" - / "document_transcript.md" - ) - assert "\nx\n" in md_path.read_text(encoding="utf-8") - - -def test_process_multi_file(tmp_path, monkeypatch): - mod = importlib.import_module("solstone.think.importers.documents") - pdf_a = tmp_path / "a.pdf" - pdf_b = tmp_path / "b.pdf" - pdf_a.write_bytes(b"%PDF-1.4") - pdf_b.write_bytes(b"%PDF-1.4") - monkeypatch.setenv("SOLSTONE_JOURNAL", str(tmp_path)) - monkeypatch.setattr("pypdf.PdfReader", MockPdfReader) - monkeypatch.setattr(mod, "day_path", lambda day: tmp_path / "chronicle" / day) - monkeypatch.setattr( - mod, - "write_content_manifest", - lambda import_id, entries: tmp_path / "manifest.jsonl", + assert _snapshot_tree(first_segment) == initial_snapshot + assert manifest_path.read_bytes() == initial_manifest + + _install_generate(monkeypatch, mod, ["Forced scanned text", "Forced description"]) + same_result = _import_pdf(mod, first, tmp_path, force=True) + + assert same_result.segments == first_result.segments + assert same_result.entries_written == 1 + assert "Forced scanned text" in _transcript(tmp_path, same_result) + assert "Forced description" in _transcript(tmp_path, same_result) + assert (first_segment / "original.pdf").read_bytes() == first.read_bytes() + _assert_cited_rasters_exist(first_segment) + + forced_snapshot = _snapshot_tree(first_segment) + second_result = _import_pdf(mod, second, tmp_path, force=True) + + assert second_result.segments != first_result.segments + assert _snapshot_tree(first_segment) == forced_snapshot + assert (_segment_dir(tmp_path, second_result) / "original.pdf").read_bytes() == ( + second.read_bytes() ) - monkeypatch.setattr(mod, "seed_entities", lambda facet, day, entities: entities) - result = mod.importer.process(tmp_path, tmp_path, import_id="20260115_120000") - assert result.entries_written == 2 - assert result.segments == [("20260115", "120000_0"), ("20260115", "120001_0")] +def test_write_order_original_exists_without_transcript_when_transcript_write_fails( + tmp_path, monkeypatch +): + mod = _mod() + pdf = write_text_rich_fixture(tmp_path / "contract.pdf") + _fixed_mtime(pdf) + def fail_write_text(*_args, **_kwargs): + raise RuntimeError("stop after original") -def test_process_entity_seeding(tmp_path, monkeypatch): - mod = importlib.import_module("solstone.think.importers.documents") + monkeypatch.setattr(mod, "write_text", fail_write_text) + result = _import_pdf(mod, pdf, tmp_path) - class EntityReader(MockPdfReader): - def __init__(self, path): - self.path = str(path) - self.pages = [ - MockPage( - "Signed by Jane Doe on behalf of Example Corp Inc for the purchase agreement and related closing documents." - ) - ] - self.metadata = {"/CreationDate": "D:20260115120000"} + assert result.segments == [] + [segment_dir] = list((tmp_path / "chronicle").glob("*/import.document/*")) + assert (segment_dir / "original.pdf").is_file() + assert not (segment_dir / "document_transcript.md").exists() - pdf = tmp_path / "parties.pdf" - pdf.write_bytes(b"%PDF-1.4") - monkeypatch.setenv("SOLSTONE_JOURNAL", str(tmp_path)) - monkeypatch.setattr("pypdf.PdfReader", EntityReader) - monkeypatch.setattr(mod, "day_path", lambda day: tmp_path / "chronicle" / day) - monkeypatch.setattr( - mod, - "write_content_manifest", - lambda import_id, entries: tmp_path / "manifest.jsonl", - ) - calls = [] - def fake_seed_entities(facet, day, entities): - calls.append((facet, day, entities)) - return entities +def test_write_order_rasters_exist_without_transcript_when_final_write_fails( + tmp_path, monkeypatch +): + mod = _mod() + pdf = write_image_only_fixture(tmp_path / "scan.pdf") + _fixed_mtime(pdf) + _install_generate(monkeypatch, mod, ["Page one", "Page two"]) + + def fail_write_text(*_args, **_kwargs): + raise RuntimeError("stop after rasters") - monkeypatch.setattr(mod, "seed_entities", fake_seed_entities) + monkeypatch.setattr(mod, "write_text", fail_write_text) + result = _import_pdf(mod, pdf, tmp_path) - result = mod.importer.process( - pdf, tmp_path, facet="legal", import_id="20260115_120000" - ) + assert result.segments == [] + [segment_dir] = list((tmp_path / "chronicle").glob("*/import.document/*")) + assert (segment_dir / "original.pdf").is_file() + assert (segment_dir / "pages" / "page-0001.png").is_file() + assert (segment_dir / "pages" / "page-0002.png").is_file() + assert not (segment_dir / "document_transcript.md").exists() - assert result.entities_seeded == len(calls[0][2]) - assert calls[0][0] == "legal" - assert calls[0][1] == "20260115" - assert any(entity["type"] == "Person" for entity in calls[0][2]) - assert any(entity["type"] == "Organization" for entity in calls[0][2]) - assert all(entity["observations"] == ["Named in parties"] for entity in calls[0][2]) +def test_preview_uses_inspect_only_and_aggregates_worker_failures( + tmp_path, monkeypatch +): + mod = _mod() + readable = tmp_path / "readable.pdf" + encrypted = tmp_path / "encrypted.pdf" + readable.write_bytes(b"%PDF-1.4 readable") + encrypted.write_bytes(b"%PDF-1.4 encrypted") + payload = _payload( + readable, + pages=[ + { + "index": 1, + "width_pt": 612.0, + "height_pt": 792.0, + "chars": 0, + "image_area_fraction": 0.0, + "rendered": None, + "error": None, + } + ], + metadata={"mod_date": "2026-03-04T12:00:00+00:00"}, + ) + calls = [] -def test_timestamp_from_metadata(tmp_path): - mod = importlib.import_module("solstone.think.importers.documents") - pdf = tmp_path / "file.pdf" - pdf.write_bytes(b"%PDF-1.4") - reader = MockPdfReader(pdf) + def fake_worker(command, pdf_path, **kwargs): + calls.append((command, Path(pdf_path).name, kwargs)) + if Path(pdf_path).name == "encrypted.pdf": + raise PdfWorkerEncryptedError( + "encrypted", + returncode=3, + stdout="", + stderr="", + payload={"schema": "sol-pdf/1", "error": "encrypted"}, + ) + return PdfWorkerSuccess(payload=payload, warnings=(), stderr="") + + monkeypatch.setattr(mod, "run_pdf_worker", fake_worker) + + preview = mod.importer.preview(tmp_path) + + assert [call[0] for call in calls] == ["inspect", "inspect"] + assert all("render_pages" not in call[2] for call in calls) + assert preview.item_count == 2 + assert preview.entity_count == 0 + assert preview.summary.startswith("2 PDF documents, 1 total pages") + assert "encrypted.pdf: password-protected PDF" in preview.summary + assert not (tmp_path / "chronicle").exists() + assert not (tmp_path / "imports").exists() - timestamp = mod._get_pdf_timestamp(reader, pdf) - assert ( - dt.datetime.fromtimestamp(timestamp).strftime("%Y%m%d%H%M%S") - == "20260115120000" - ) +def test_document_importer_no_longer_seeds_entities(): + mod = _mod() + assert not hasattr(mod, "_extract_entities") + assert not hasattr(mod, "seed_entities") def test_registry_entry(): assert FILE_IMPORTER_REGISTRY["document"] == "solstone.think.importers.documents" + + +def test_manifest_meta_is_new_shape(tmp_path): + mod = _mod() + pdf = write_text_rich_fixture(tmp_path / "contract.pdf") + _fixed_mtime(pdf) + + _import_pdf(mod, pdf, tmp_path) + + manifest = tmp_path / "imports" / "import-test" / "content_manifest.jsonl" + entry = json.loads(manifest.read_text(encoding="utf-8")) + assert entry["meta"] == { + "page_count": 2, + "engine": entry["meta"]["engine"], + "timestamp_source": "file-mtime", + "text_layer_pages": 2, + "model_extracted_pages": 0, + "unavailable_pages": 0, + "image_described_pages": 0, + "model_calls": 0, + "warnings": [], + } + assert "extraction_method" not in entry["meta"] diff --git a/tests/test_pdf_lazy_imports.py b/tests/test_pdf_lazy_imports.py index b87e02275..9434c6172 100644 --- a/tests/test_pdf_lazy_imports.py +++ b/tests/test_pdf_lazy_imports.py @@ -59,7 +59,6 @@ def test_pdf_modules_are_not_loaded_by_static_imports(): assert "weasyprint" not in modules assert "pypdf" not in modules assert "pypdfium2" not in modules - assert "pdf2image" not in modules def _force_missing_feature(monkeypatch, name: str): @@ -72,7 +71,7 @@ def _force_missing_feature(monkeypatch, name: str): def _force_missing_pdf(monkeypatch): - _force_missing_feature(monkeypatch, "pdf") + _force_missing_feature(monkeypatch, "pdf-import") def test_render_reflection_pdf_missing_extra(monkeypatch, tmp_path): @@ -95,4 +94,4 @@ def test_document_importer_process_pdf_missing_extra(monkeypatch, tmp_path): with pytest.raises(MissingExtraError) as exc: DocumentImporter().process(pdf, tmp_path) - assert "pip install 'solstone[pdf]'" in str(exc.value) + assert "pip install 'solstone[pdf-import]'" in str(exc.value) -- 2.51.2