From 5cfd7c2d1df5eca2dbae89c32710a3a3dcdf0e74 Mon Sep 17 00:00:00 2001 From: Jer Miller Date: Thu, 4 Jun 2026 14:28:10 -0600 Subject: [PATCH] feat(import): byte-hash dedup + batch-safe collisions for audio/text imports Bring the generic audio/text import path to parity with the file-importer path on byte-hash duplicate detection, and make collisions catchable in a batch loop. - Re-importing the same audio/text bytes is detected as a duplicate before the timestamp-detection LLM call, printing the same "already imported" message the file importers give; --force still re-imports. - import_one() directory collisions raise a catchable FileExistsError instead of SystemExit; the human `sol import` CLI still surfaces the same message and exit code 1 via main()'s wrapper. - Successful audio/text imports now write a dedup manifest.json with correct days_affected for the YYYYMMDD/stream/segment/file layout. - write_manifest now records import_id (it already took the param) and accepts an optional days_affected to avoid path-derivation across two layouts. - detect_created() no longer writes gemini_debug_* files to /tmp. Co-Authored-By: Claude Opus 4.8 (1M context) --- solstone/think/detect_created.py | 17 ---- solstone/think/importers/cli.py | 42 ++++++++ solstone/think/importers/shared.py | 19 ++-- tests/test_detect_created_schema.py | 4 + tests/test_import_dedup.py | 2 + tests/test_importer.py | 149 ++++++++++++++++++++++++++++ 6 files changed, 208 insertions(+), 25 deletions(-) diff --git a/solstone/think/detect_created.py b/solstone/think/detect_created.py index a0c10c8d9..63f6c0386 100644 --- a/solstone/think/detect_created.py +++ b/solstone/think/detect_created.py @@ -6,9 +6,7 @@ from __future__ import annotations import json -import os import subprocess -import sys from datetime import datetime, timezone from pathlib import Path from typing import Optional @@ -39,18 +37,6 @@ def _extract_metadata(path: str) -> str: return f"Error extracting metadata: {exc}" -def _debug_write_content(content: str, path: str) -> None: - """Write content to a debug file in /tmp for diagnosis.""" - timestamp = datetime.now().strftime("%Y%m%d_%H%M%S") - filename = f"gemini_debug_{timestamp}_{os.path.basename(path)}.md" - debug_path = os.path.join("/tmp", filename) - - with open(debug_path, "w", encoding="utf-8") as f: - f.write(content) - - print(f"Debug: Content written to {debug_path}", file=sys.stderr) - - def detect_created( path: str, original_filename: Optional[str] = None, guidance: Optional[str] = None ) -> Optional[dict]: @@ -90,9 +76,6 @@ def detect_created( if guidance: markdown += f"\n\nImportant guidance from the user: {guidance}" - # Debug: write content to temp file - _debug_write_content(markdown, path) - from solstone.think.models import generate response_text = generate( diff --git a/solstone/think/importers/cli.py b/solstone/think/importers/cli.py index b814cc4ba..fd66aa671 100644 --- a/solstone/think/importers/cli.py +++ b/solstone/think/importers/cli.py @@ -402,6 +402,32 @@ def _import_one_from_args(args: argparse.Namespace) -> dict[str, Any] | None: _file_importer = detected import_source = detected.name + # Source-level dedup for audio/text (before timestamp detection and before + # _setup_import rewrites args.media). Mirrors the file-importer dedup at the + # file-importer branch; skipped for --force and --dry-run (which still preview). + if _file_importer is None and not args.dry_run: + from solstone.think.importers.shared import ( + find_manifest_by_hash, + hash_source, + ) + + _source_hash = hash_source(Path(args.media)) + if not args.force: + existing = find_manifest_by_hash(Path(get_journal()), _source_hash) + if existing: + imported_at = existing.get("imported_at", "unknown date") + entry_count = existing.get("entry_count", 0) + print( + f"This file was already imported on {imported_at} " + f"({entry_count} entries). Use --force to re-import." + ) + return { + "skipped": True, + "reason": "already_imported", + "imported_at": imported_at, + "entry_count": entry_count, + } + # --- Timestamp resolution --- if _file_importer is not None and not args.timestamp: # File importers don't need an external timestamp — auto-generate for metadata @@ -1087,6 +1113,22 @@ def _import_one_from_args(args: argparse.Namespace) -> dict[str, Any] | None: [processing_results["target_day"], processing_results["target_day"]], ) + # Write dedup manifest for audio/text imports. File importers already wrote + # theirs in the file-importer branch above (line ~887); guard prevents a + # double write since all branches fall through this common tail. + if _file_importer is None: + from solstone.think.importers.shared import write_manifest + + write_manifest( + journal_root, + import_id=args.timestamp, + source_type=import_source, + source_hash=_source_hash, + entry_count=len(all_created_files), + files_created=all_created_files, + days_affected=[day], + ) + imported_path = import_dir / "imported.json" # Write imported.json with all processing metadata try: diff --git a/solstone/think/importers/shared.py b/solstone/think/importers/shared.py index 90ee6ddf8..ce45d7d31 100644 --- a/solstone/think/importers/shared.py +++ b/solstone/think/importers/shared.py @@ -360,7 +360,7 @@ def _setup_import( logger.info(f"Removing existing import directory: {import_dir}") shutil.rmtree(import_dir) else: - raise SystemExit( + raise FileExistsError( f"Error: Import already exists for timestamp {timestamp}\n" f"To re-import, use --force to delete existing data and start over" ) @@ -551,19 +551,22 @@ def write_manifest( source_hash: str, entry_count: int, files_created: list[str], + days_affected: list[str] | None = None, ) -> Path: """Write an import manifest for deduplication tracking. Returns path to the manifest file. """ - days_affected = sorted( - { - os.path.basename(os.path.dirname(os.path.dirname(f))) - for f in files_created - if os.path.basename(os.path.dirname(os.path.dirname(f))).isdigit() - } - ) + if days_affected is None: + days_affected = sorted( + { + os.path.basename(os.path.dirname(os.path.dirname(f))) + for f in files_created + if os.path.basename(os.path.dirname(os.path.dirname(f))).isdigit() + } + ) manifest = { + "import_id": import_id, "source_type": source_type, "source_hash": source_hash, "entry_count": entry_count, diff --git a/tests/test_detect_created_schema.py b/tests/test_detect_created_schema.py index 3b8c3957a..143f77378 100644 --- a/tests/test_detect_created_schema.py +++ b/tests/test_detect_created_schema.py @@ -1,6 +1,7 @@ # SPDX-License-Identifier: AGPL-3.0-only # Copyright (c) 2026 sol pbc +import glob import importlib import json from pathlib import Path @@ -71,9 +72,12 @@ def test_detect_created_passes_schema_to_generate(monkeypatch): lambda path: "QuickTime Create Date : 2024:03:15 14:30:52", ) + before = set(glob.glob("/tmp/gemini_debug_*")) result = detect_created_mod.detect_created("/dev/null") + after = set(glob.glob("/tmp/gemini_debug_*")) assert captured["json_schema"] is detect_created_mod._SCHEMA + assert after == before assert result == { "day": "20240315", "time": "143052", diff --git a/tests/test_import_dedup.py b/tests/test_import_dedup.py index aa8e0e7d2..9ac2a6353 100644 --- a/tests/test_import_dedup.py +++ b/tests/test_import_dedup.py @@ -119,6 +119,7 @@ def test_write_and_find_manifest(): # Read it back with open(manifest_path) as f: data = json.load(f) + assert data["import_id"] == "20260115_120000" assert data["source_type"] == "ics" assert data["source_hash"] == "abc123" assert data["entry_count"] == 42 @@ -128,6 +129,7 @@ def test_write_and_find_manifest(): # Find by hash found = find_manifest_by_hash(Path(journal), "abc123") assert found is not None + assert found["import_id"] == "20260115_120000" assert found["source_type"] == "ics" # Not found for different hash diff --git a/tests/test_importer.py b/tests/test_importer.py index fdfe7341b..4996c0c39 100644 --- a/tests/test_importer.py +++ b/tests/test_importer.py @@ -1144,6 +1144,25 @@ def test_importer_existing_import_without_force_still_errors(tmp_path, monkeypat assert _read_action_entries(tmp_path) == [] +def test_import_one_collision_raises_catchable_file_exists_error(tmp_path, monkeypatch): + mod = importlib.import_module("solstone.think.importers.cli") + + timestamp = "20240101_120000" + import_dir = tmp_path / "imports" / timestamp + import_dir.mkdir(parents=True) + (import_dir / "stale.txt").write_text("old import payload") + + media = tmp_path / "note.txt" + media.write_text("new transcript") + + monkeypatch.setenv("SOLSTONE_JOURNAL", str(tmp_path)) + + with pytest.raises(FileExistsError) as ei: + mod.import_one(media, timestamp=timestamp) + + assert not isinstance(ei.value, SystemExit) + + def test_file_importer_without_timestamp(tmp_path, monkeypatch, capsys): """File importers auto-generate timestamp and skip import setup.""" mod = importlib.import_module("solstone.think.importers.cli") @@ -1390,6 +1409,136 @@ def test_import_one_skips_wait_when_disabled(tmp_path, monkeypatch): assert "failed_segments" not in result +def test_import_one_audio_reimport_is_deduped(tmp_path, monkeypatch): + mod = importlib.import_module("solstone.think.importers.cli") + + audio_file = tmp_path / "test.mp3" + audio_file.write_bytes(b"fake audio") + calls = [] + callosum = MagicMock() + + def fake_prepare_audio_segments(media_path, day_dir, base_dt, import_id, stream): + calls.append(media_path) + seg_dir = Path(day_dir) / stream / "120000_300" + seg_dir.mkdir(parents=True, exist_ok=True) + (seg_dir / "imported_audio.mp3").write_bytes(b"sliced audio") + return [("120000_300", seg_dir, ["imported_audio.mp3"])] + + monkeypatch.setenv("SOLSTONE_JOURNAL", str(tmp_path)) + monkeypatch.setattr(mod, "CallosumConnection", lambda **kwargs: callosum) + monkeypatch.setattr(mod, "get_rev", lambda: "test-rev") + monkeypatch.setattr(mod, "_status_emitter", lambda: None) + monkeypatch.setattr(mod, "prepare_audio_segments", fake_prepare_audio_segments) + monkeypatch.setattr( + mod, + "update_stream", + lambda stream, day, seg, **kwargs: { + "prev_day": None, + "prev_segment": None, + "seq": 1, + }, + ) + monkeypatch.setattr(mod, "write_segment_stream", lambda *args, **kwargs: None) + + result = mod.import_one( + audio_file, + timestamp="20260303_120000", + source="audio", + wait_for_processing=False, + ) + + assert result is not None + assert len(calls) == 1 + manifest_path = tmp_path / "imports" / "20260303_120000" / "manifest.json" + assert manifest_path.exists() + manifest = json.loads(manifest_path.read_text(encoding="utf-8")) + assert manifest["source_hash"] == hashlib.sha256(b"fake audio").hexdigest() + assert manifest["source_type"] == "audio" + assert manifest["import_id"] == "20260303_120000" + assert manifest["days_affected"] == ["20260303"] + + audio_file2 = tmp_path / "same-audio.mp3" + audio_file2.write_bytes(b"fake audio") + result = mod.import_one( + audio_file2, + timestamp="20260303_120000", + source="audio", + wait_for_processing=False, + ) + + assert result is not None + assert result["reason"] == "already_imported" + assert len(calls) == 1 + + +def test_import_one_text_reimport_is_deduped(tmp_path, monkeypatch): + mod = importlib.import_module("solstone.think.importers.cli") + text_mod = importlib.import_module("solstone.think.importers.text") + + transcript = "hello\nworld" + txt = tmp_path / "sample.txt" + txt.write_text(transcript) + calls = [] + + def fake_detect_segment(text, start_time): + calls.append((text, start_time)) + return [("12:00:00", text)] + + monkeypatch.setenv("SOLSTONE_JOURNAL", str(tmp_path)) + monkeypatch.setattr( + mod, "detect_created", lambda p, **kw: {"day": "20240101", "time": "120000"} + ) + monkeypatch.setattr(text_mod, "detect_transcript_segment", fake_detect_segment) + monkeypatch.setattr( + text_mod, + "detect_transcript_json", + lambda text, segment_start: { + "entries": [ + { + "start": segment_start, + "speaker": "Unknown", + "text": text, + } + ], + "topics": "", + "setting": "", + }, + ) + monkeypatch.setattr(mod, "CallosumConnection", lambda **kwargs: MagicMock()) + monkeypatch.setattr(mod, "get_rev", lambda: "test-rev") + monkeypatch.setattr(mod, "_status_emitter", lambda: None) + + result = mod.import_one( + txt, + timestamp="20240101_120000", + source="text", + ) + + assert result is not None + assert len(calls) == 1 + manifest_path = tmp_path / "imports" / "20240101_120000" / "manifest.json" + assert manifest_path.exists() + manifest = json.loads(manifest_path.read_text(encoding="utf-8")) + assert manifest["source_type"] == "text" + assert manifest["import_id"] == "20240101_120000" + assert manifest["days_affected"] == ["20240101"] + segment_dir = day_path("20240101") / "import.text" + assert len(list(segment_dir.iterdir())) == 1 + + txt2 = tmp_path / "same-sample.txt" + txt2.write_text(transcript) + result = mod.import_one( + txt2, + timestamp="20240101_120000", + source="text", + ) + + assert result is not None + assert result["reason"] == "already_imported" + assert len(calls) == 1 + assert len(list(segment_dir.iterdir())) == 1 + + def test_file_importer_indexes_created_files_in_process(tmp_path, monkeypatch): mod = importlib.import_module("solstone.think.importers.cli") -- 2.51.2