diff --git a/tests/test_gemini_importer.py b/tests/test_gemini_importer.py index eb487a3e4..427fc5717 100644 --- a/tests/test_gemini_importer.py +++ b/tests/test_gemini_importer.py @@ -189,20 +189,18 @@ def test_process_json(): os.environ["JOURNAL_PATH"] = journal result = importer.process(Path(f.name), Path(journal)) assert result.entries_written == 2 - assert len(result.files_created) == 1 # same day assert result.errors == [] - - # Verify JSONL content - jsonl_path = Path(result.files_created[0]) - assert jsonl_path.exists() - lines = jsonl_path.read_text().strip().split("\n") - header = json.loads(lines[0]) - assert header["import"]["source"] == "gemini" - assert header["entry_count"] == 2 - - entry = json.loads(lines[1]) - assert entry["type"] == "ai_chat" - assert entry["source"] == "gemini" + assert result.segments is not None + assert len(result.segments) >= 1 + assert any(Path(p).name == "imported.md" for p in result.files_created) + + md = "" + for file_path in result.files_created: + md_path = Path(file_path) + assert md_path.exists() + md += md_path.read_text() + assert "Python" in md + assert "sorted" in md finally: os.unlink(f.name) os.environ.pop("JOURNAL_PATH", None) @@ -221,12 +219,40 @@ def test_process_zip(): os.environ["JOURNAL_PATH"] = journal result = importer.process(Path(tmp.name), Path(journal)) assert result.entries_written == 1 - assert len(result.files_created) == 1 + assert result.segments is not None + assert len(result.segments) == 1 + assert any(Path(p).suffix == ".md" for p in result.files_created) finally: os.unlink(tmp.name) os.environ.pop("JOURNAL_PATH", None) +def test_process_multiple_windows(): + """Activities more than 5 minutes apart land in different segments.""" + with tempfile.NamedTemporaryFile(suffix=".json", mode="w", delete=False) as f: + activities = [ + _sample_activity(time="2026-01-15T10:00:00Z"), + _sample_activity( + prompt="Second question", + response="Second answer", + time="2026-01-15T10:10:00Z", + ), + ] + json.dump(activities, f) + f.flush() + try: + with tempfile.TemporaryDirectory() as journal: + os.environ["JOURNAL_PATH"] = journal + result = importer.process(Path(f.name), Path(journal)) + assert result.entries_written == 2 + assert result.segments is not None + assert len(result.segments) == 2 + assert len(result.files_created) == 2 + finally: + os.unlink(f.name) + os.environ.pop("JOURNAL_PATH", None) + + # --- Registry test --- diff --git a/tests/test_import_formatting.py b/tests/test_import_formatting.py index ec64cf5b0..942be55a8 100644 --- a/tests/test_import_formatting.py +++ b/tests/test_import_formatting.py @@ -233,8 +233,17 @@ def test_formatter_registration_ics_segment_markdown(): def test_formatter_registration_kindle(): from think.formatters import get_formatter - formatter = get_formatter("20260115/import.kindle/imported.jsonl") + formatter = get_formatter("20260115/import.kindle/103000_300/imported.md") assert formatter is not None + assert formatter.__name__ == "format_markdown" + + +def test_formatter_registration_gemini_segment(): + from think.formatters import get_formatter + + formatter = get_formatter("20260115/import.gemini/100000_300/imported.md") + assert formatter is not None + assert formatter.__name__ == "format_markdown" def test_path_metadata_extraction(): diff --git a/tests/test_importer.py b/tests/test_importer.py index fe414e84f..2fabdf59e 100644 --- a/tests/test_importer.py +++ b/tests/test_importer.py @@ -1007,8 +1007,8 @@ def test_ics_creation_timestamp_none(): assert mod._creation_timestamp(EmptyComponent()) is None -def test_ics_window_events_single_window(): - mod = importlib.import_module("think.importers.ics") +def test_window_items_single_window(): + mod = importlib.import_module("think.importers.shared") base = dt.datetime(2026, 3, 1, 12, 0, 0, tzinfo=dt.timezone.utc).timestamp() events = [ @@ -1017,13 +1017,13 @@ def test_ics_window_events_single_window(): {"title": "C", "create_ts": base + 120}, ] - windows = mod._window_events(events) + windows = mod.window_items(events, "create_ts") assert windows == [("20260301", "120000_300", events)] -def test_ics_window_events_time_gap_split(): - mod = importlib.import_module("think.importers.ics") +def test_window_items_time_gap_split(): + mod = importlib.import_module("think.importers.shared") base = dt.datetime(2026, 3, 1, 12, 0, 0, tzinfo=dt.timezone.utc).timestamp() events = [ @@ -1033,7 +1033,7 @@ def test_ics_window_events_time_gap_split(): {"title": "D", "create_ts": base + 600}, ] - windows = mod._window_events(events) + windows = mod.window_items(events, "create_ts") assert len(windows) == 2 assert windows[0][0] == "20260301" @@ -1043,8 +1043,8 @@ def test_ics_window_events_time_gap_split(): assert windows[1][2] == [events[3]] -def test_ics_window_events_day_boundary(): - mod = importlib.import_module("think.importers.ics") +def test_window_items_day_boundary(): + mod = importlib.import_module("think.importers.shared") first_day = dt.datetime(2026, 3, 1, 12, 0, 0, tzinfo=dt.timezone.utc).timestamp() second_day = dt.datetime(2026, 3, 2, 12, 0, 0, tzinfo=dt.timezone.utc).timestamp() @@ -1053,7 +1053,7 @@ def test_ics_window_events_day_boundary(): {"title": "B", "create_ts": second_day}, ] - windows = mod._window_events(events) + windows = mod.window_items(events, "create_ts") assert windows == [ ("20260301", "120000_300", [events[0]]), diff --git a/tests/test_kindle_importer.py b/tests/test_kindle_importer.py new file mode 100644 index 000000000..ad0a57fcc --- /dev/null +++ b/tests/test_kindle_importer.py @@ -0,0 +1,217 @@ +# SPDX-License-Identifier: AGPL-3.0-only +# Copyright (c) 2026 sol pbc + +"""Tests for think.importers.kindle — Kindle My Clippings.txt importer.""" + +import os +import tempfile +from pathlib import Path + +from think.importers.kindle import KindleImporter, _parse_block, _parse_date + +importer = KindleImporter() + + +def _make_clipping( + title: str = "Test Book (Author Name)", + meta: str = "- Your Highlight on page 42 | location 100-101 | Added on Saturday, March 15, 2025 10:30:00 AM", + content: str = "This is a highlighted passage.", +) -> str: + return f"{title}\n{meta}\n\n{content}\n" + + +def _make_clippings_file(clippings: list[str]) -> str: + return "==========\n".join(clippings) + "==========\n" + + +# --- Unit tests for helpers --- + + +def test_parse_date(): + result = _parse_date("Saturday, March 15, 2025 10:30:00 AM") + assert result is not None + assert result.month == 3 + assert result.day == 15 + + +def test_parse_block_basic(): + block = _make_clipping() + entry = _parse_block(block) + assert entry is not None + assert entry["type"] == "highlight" + assert entry["book_title"] == "Test Book" + assert entry["author"] == "Author Name" + assert entry["content"] == "This is a highlighted passage." + assert entry["page"] == 42 + assert entry["location"] == "100-101" + + +def test_parse_block_note(): + block = _make_clipping( + meta="- Your Note on page 10 | Added on Saturday, March 15, 2025 10:30:00 AM", + content="My personal note", + ) + entry = _parse_block(block) + assert entry is not None + assert entry["clip_type"] == "note" + + +def test_parse_block_no_author(): + block = _make_clipping(title="Title Without Author") + entry = _parse_block(block) + assert entry is not None + assert entry["book_title"] == "Title Without Author" + assert entry["author"] == "" + + +# --- Detection tests --- + + +def test_detect_valid(): + content = _make_clippings_file([_make_clipping()]) + with tempfile.NamedTemporaryFile(suffix=".txt", mode="w", delete=False) as f: + f.write(content) + f.flush() + try: + assert importer.detect(Path(f.name)) is True + finally: + os.unlink(f.name) + + +def test_detect_wrong_format(): + with tempfile.NamedTemporaryFile(suffix=".txt", mode="w", delete=False) as f: + f.write("Just some random text file.\n") + f.flush() + try: + assert importer.detect(Path(f.name)) is False + finally: + os.unlink(f.name) + + +# --- Preview tests --- + + +def test_preview(): + content = _make_clippings_file( + [ + _make_clipping(), + _make_clipping( + title="Another Book (Jane Doe)", + meta="- Your Note on page 5 | Added on Sunday, March 16, 2025 02:00:00 PM", + content="A note", + ), + ] + ) + with tempfile.NamedTemporaryFile(suffix=".txt", mode="w", delete=False) as f: + f.write(content) + f.flush() + try: + preview = importer.preview(Path(f.name)) + assert preview.item_count == 2 + assert preview.entity_count > 0 + assert "2 books" in preview.summary + finally: + os.unlink(f.name) + + +# --- Process tests --- + + +def test_process_basic(): + content = _make_clippings_file( + [ + _make_clipping(), + _make_clipping( + meta="- Your Highlight on page 43 | Added on Saturday, March 15, 2025 10:31:00 AM", + content="Another highlight from same session.", + ), + ] + ) + with tempfile.NamedTemporaryFile(suffix=".txt", mode="w", delete=False) as f: + f.write(content) + f.flush() + try: + with tempfile.TemporaryDirectory() as journal: + os.environ["JOURNAL_PATH"] = journal + result = importer.process(Path(f.name), Path(journal)) + assert result.entries_written == 2 + assert result.errors == [] + assert result.segments is not None + assert len(result.segments) >= 1 + + md_path = Path(result.files_created[0]) + assert md_path.exists() + assert md_path.name == "imported.md" + md = md_path.read_text() + assert "Test Book" in md + assert "> This is a highlighted passage." in md + assert "Page 42" in md + finally: + os.unlink(f.name) + os.environ.pop("JOURNAL_PATH", None) + + +def test_process_multiple_windows(): + """Highlights more than 5 minutes apart land in different segments.""" + content = _make_clippings_file( + [ + _make_clipping( + meta="- Your Highlight on page 1 | Added on Saturday, March 15, 2025 10:00:00 AM", + content="First highlight", + ), + _make_clipping( + meta="- Your Highlight on page 2 | Added on Saturday, March 15, 2025 10:10:00 AM", + content="Second highlight, 10 min later", + ), + ] + ) + with tempfile.NamedTemporaryFile(suffix=".txt", mode="w", delete=False) as f: + f.write(content) + f.flush() + try: + with tempfile.TemporaryDirectory() as journal: + os.environ["JOURNAL_PATH"] = journal + result = importer.process(Path(f.name), Path(journal)) + assert result.entries_written == 2 + assert result.segments is not None + assert len(result.segments) == 2 + assert len(result.files_created) == 2 + finally: + os.unlink(f.name) + os.environ.pop("JOURNAL_PATH", None) + + +def test_process_note_markdown(): + """Notes render with Note: prefix instead of blockquote.""" + content = _make_clippings_file( + [ + _make_clipping( + meta="- Your Note on page 10 | Added on Saturday, March 15, 2025 10:30:00 AM", + content="My personal note", + ), + ] + ) + with tempfile.NamedTemporaryFile(suffix=".txt", mode="w", delete=False) as f: + f.write(content) + f.flush() + try: + with tempfile.TemporaryDirectory() as journal: + os.environ["JOURNAL_PATH"] = journal + result = importer.process(Path(f.name), Path(journal)) + md = Path(result.files_created[0]).read_text() + assert "Note: My personal note" in md + finally: + os.unlink(f.name) + os.environ.pop("JOURNAL_PATH", None) + + +# --- Registry test --- + + +def test_registered_in_registry(): + from think.importers.file_importer import FILE_IMPORTER_REGISTRY, get_file_importer + + assert "kindle" in FILE_IMPORTER_REGISTRY + imp = get_file_importer("kindle") + assert imp is not None + assert imp.name == "kindle" diff --git a/think/importers/gemini.py b/think/importers/gemini.py index db36c097d..f54f4f034 100644 --- a/think/importers/gemini.py +++ b/think/importers/gemini.py @@ -23,7 +23,8 @@ from pathlib import Path from typing import Any, Callable from think.importers.file_importer import ImportPreview, ImportResult -from think.importers.shared import write_structured_import +from think.importers.shared import window_items +from think.utils import day_path logger = logging.getLogger(__name__) @@ -150,6 +151,27 @@ def _parse_activity(activity: dict[str, Any]) -> dict[str, Any] | None: return entry +def _render_activity_markdown(activity: dict) -> str: + """Render a Gemini activity as markdown.""" + title = activity.get("title", "Gemini activity") + lines = [f"## {title}"] + + content = activity.get("content", "") + if content: + # Content already has "Human: ..." and "Assistant: ..." format + # Convert to bold labels + for part in content.split("\n\n"): + part = part.strip() + if part.startswith("Human: "): + lines.append(f"**Human:** {part[7:]}") + elif part.startswith("Assistant: "): + lines.append(f"**Assistant:** {part[11:]}") + elif part: + lines.append(part) + + return "\n\n".join(lines) + + class GeminiImporter: name = "gemini" display_name = "Gemini Activity History" @@ -233,7 +255,6 @@ class GeminiImporter: progress_callback: Callable | None = None, ) -> ImportResult: activities = _load_activities(path) - import_id = dt.datetime.now().strftime("%Y%m%d_%H%M%S") entries: list[dict[str, Any]] = [] errors: list[str] = [] @@ -244,18 +265,41 @@ class GeminiImporter: if entry is None: skipped += 1 continue + + # Add epoch timestamp for windowing + entry["create_ts"] = dt.datetime.fromisoformat(entry["ts"]).timestamp() entries.append(entry) if progress_callback and (i + 1) % 100 == 0: progress_callback(i + 1, len(activities)) - # Write to journal - created_files = write_structured_import( - "gemini", - entries, - import_id=import_id, - facet=facet, - ) + if not entries: + return ImportResult( + entries_written=0, + entities_seeded=0, + files_created=[], + errors=errors, + summary="No activities found to import", + ) + + entries.sort(key=lambda e: e["create_ts"]) + + windows = window_items(entries, "create_ts") + created_files: list[str] = [] + segments: list[tuple[str, str]] = [] + + for day, seg_key, window_activities in windows: + segment_dir = day_path(day) / "import.gemini" / seg_key + segment_dir.mkdir(parents=True, exist_ok=True) + md_path = segment_dir / "imported.md" + markdown = "\n\n".join( + _render_activity_markdown(act) for act in window_activities + ) + md_path.write_text(markdown + "\n", encoding="utf-8") + created_files.append(str(md_path)) + segments.append((day, seg_key)) + + segment_days = {day for day, _ in segments} if skipped: logger.info("Skipped %d activities with no content", skipped) @@ -268,7 +312,11 @@ class GeminiImporter: entities_seeded=0, files_created=created_files, errors=errors, - summary=f"Imported {len(entries)} Gemini activities{bard_info} across {len(created_files)} days", + summary=( + f"Imported {len(entries)} Gemini activities{bard_info} across " + f"{len(segment_days)} days into {len(segments)} segments" + ), + segments=segments, ) diff --git a/think/importers/ics.py b/think/importers/ics.py index 5dc12212b..4ab473524 100644 --- a/think/importers/ics.py +++ b/think/importers/ics.py @@ -10,7 +10,7 @@ from pathlib import Path from typing import Any, Callable from think.importers.file_importer import ImportPreview, ImportResult -from think.importers.shared import seed_entities +from think.importers.shared import seed_entities, window_items from think.utils import day_path logger = logging.getLogger(__name__) @@ -126,48 +126,6 @@ def _creation_timestamp(component: Any) -> float | None: return None -def _window_events( - events: list[dict[str, Any]], - window_duration: int = 300, -) -> list[tuple[str, str, list[dict[str, Any]]]]: - """Group sorted events into fixed-duration windows per creation time.""" - if not events: - return [] - - windows: list[tuple[str, str, list[dict[str, Any]]]] = [] - window_start: float | None = None - window_day: str | None = None - window_events: list[dict[str, Any]] = [] - - for event in events: - create_ts = event["create_ts"] - event_dt = dt.datetime.fromtimestamp(create_ts, tz=dt.timezone.utc) - event_day = event_dt.strftime("%Y%m%d") - - if ( - window_start is None - or event_day != window_day - or create_ts - window_start >= window_duration - ): - if window_events and window_day and window_start is not None: - start_dt = dt.datetime.fromtimestamp(window_start, tz=dt.timezone.utc) - seg_key = f"{start_dt.strftime('%H%M%S')}_{window_duration}" - windows.append((window_day, seg_key, window_events)) - - window_start = create_ts - window_day = event_day - window_events = [] - - window_events.append(event) - - if window_events and window_day and window_start is not None: - start_dt = dt.datetime.fromtimestamp(window_start, tz=dt.timezone.utc) - seg_key = f"{start_dt.strftime('%H%M%S')}_{window_duration}" - windows.append((window_day, seg_key, window_events)) - - return windows - - def _render_event_markdown(event: dict[str, Any]) -> str: """Render a calendar event as markdown.""" title = event.get("title", "Untitled event") @@ -398,7 +356,7 @@ class ICSImporter: all_entries.sort(key=lambda entry: entry["create_ts"]) - windows = _window_events(all_entries) + windows = window_items(all_entries, "create_ts") created_files: list[str] = [] segments: list[tuple[str, str]] = [] diff --git a/think/importers/kindle.py b/think/importers/kindle.py index 3ff7b66dd..0fc87f80f 100644 --- a/think/importers/kindle.py +++ b/think/importers/kindle.py @@ -10,7 +10,8 @@ from pathlib import Path from typing import Callable from think.importers.file_importer import ImportPreview, ImportResult -from think.importers.shared import seed_entities, write_structured_import +from think.importers.shared import seed_entities, window_items +from think.utils import day_path logger = logging.getLogger(__name__) @@ -128,6 +129,47 @@ def _parse_block(block: str) -> dict | None: return entry +def _render_highlight_markdown(highlights: list[dict]) -> str: + """Render highlights grouped by book as markdown.""" + # Group by book + by_book: dict[str, list[dict]] = {} + for h in highlights: + key = h["book_title"] + by_book.setdefault(key, []).append(h) + + sections: list[str] = [] + for book_title, book_highlights in by_book.items(): + # Use first highlight's author (all from same book share author) + author = book_highlights[0].get("author", "") + if author: + heading = f"## {book_title} by {author}" + else: + heading = f"## {book_title}" + + lines = [heading] + for h in book_highlights: + content = h.get("content", "") + clip_type = h.get("clip_type", "highlight") + + if clip_type == "note": + lines.append(f"Note: {content}") + else: + lines.append(f"> {content}") + + # Page / location metadata + meta_parts: list[str] = [] + if h.get("page") is not None: + meta_parts.append(f"Page {h['page']}") + if h.get("location"): + meta_parts.append(f"Location {h['location']}") + if meta_parts: + lines.append(" | ".join(meta_parts)) + + sections.append("\n".join(lines)) + + return "\n\n".join(sections) + + class KindleImporter: name = "kindle" display_name = "Kindle Highlights" @@ -211,7 +253,6 @@ class KindleImporter: ) -> ImportResult: text = path.read_text(encoding="utf-8-sig") blocks = text.split(DELIMITER) - import_id = dt.datetime.now().strftime("%Y%m%d_%H%M%S") entries: list[dict] = [] errors: list[str] = [] @@ -223,12 +264,13 @@ class KindleImporter: continue entry = _parse_block(block) if entry is None: - # Only log as error if block had real content (not just whitespace) stripped = block.strip() if stripped and "\n" in stripped: errors.append(f"Failed to parse clipping block {i + 1}") continue + # Add epoch timestamp for windowing + entry["create_ts"] = dt.datetime.fromisoformat(entry["ts"]).timestamp() entries.append(entry) books.add(entry["book_title"]) if entry["author"]: @@ -237,13 +279,31 @@ class KindleImporter: if progress_callback and (i + 1) % 100 == 0: progress_callback(i + 1, len(blocks)) - # Write to journal - created_files = write_structured_import( - "kindle", - entries, - import_id=import_id, - facet=facet, - ) + if not entries: + return ImportResult( + entries_written=0, + entities_seeded=0, + files_created=[], + errors=errors, + summary="No clippings found to import", + ) + + entries.sort(key=lambda e: e["create_ts"]) + + windows = window_items(entries, "create_ts", tz=None) + created_files: list[str] = [] + segments: list[tuple[str, str]] = [] + + for day, seg_key, window_highlights in windows: + segment_dir = day_path(day) / "import.kindle" / seg_key + segment_dir.mkdir(parents=True, exist_ok=True) + md_path = segment_dir / "imported.md" + markdown = _render_highlight_markdown(window_highlights) + md_path.write_text(markdown + "\n", encoding="utf-8") + created_files.append(str(md_path)) + segments.append((day, seg_key)) + + segment_days = {day for day, _ in segments} # Seed entities (books and authors) entities_seeded = 0 @@ -266,7 +326,11 @@ class KindleImporter: entities_seeded=entities_seeded, files_created=created_files, errors=errors, - summary=f"Imported {len(entries)} Kindle clippings from {len(books)} books across {len(created_files)} days", + summary=( + f"Imported {len(entries)} Kindle clippings from {len(books)} books " + f"across {len(segment_days)} days into {len(segments)} segments" + ), + segments=segments, ) diff --git a/think/importers/shared.py b/think/importers/shared.py index 903344351..b7cf5e883 100644 --- a/think/importers/shared.py +++ b/think/importers/shared.py @@ -178,6 +178,69 @@ def _window_messages( return windows +def window_items( + items: list[dict[str, Any]], + ts_key: str, + *, + window_duration: int = 300, + tz: dt.timezone | None = dt.timezone.utc, +) -> list[tuple[str, str, list[dict[str, Any]]]]: + """Group sorted items into fixed-duration windows per day. + + Parameters + ---------- + items : list[dict] + Items sorted by ts_key. The ts_key field must be a float epoch. + ts_key : str + Key name for the float epoch timestamp in each item. + window_duration : int + Window size in seconds (default 300 = 5 minutes). + tz : timezone or None + Timezone for day grouping and seg_key formatting. + Use dt.timezone.utc for UTC timestamps, None for local time. + + Returns + ------- + list[tuple[str, str, list[dict]]] + (day_str, seg_key, items) tuples. + """ + if not items: + return [] + + windows: list[tuple[str, str, list[dict[str, Any]]]] = [] + window_start: float | None = None + window_day: str | None = None + window_items_acc: list[dict[str, Any]] = [] + + for item in items: + ts = item[ts_key] + item_dt = dt.datetime.fromtimestamp(ts, tz=tz) + item_day = item_dt.strftime("%Y%m%d") + + if ( + window_start is None + or item_day != window_day + or ts - window_start >= window_duration + ): + if window_items_acc and window_day and window_start is not None: + start_dt = dt.datetime.fromtimestamp(window_start, tz=tz) + seg_key = f"{start_dt.strftime('%H%M%S')}_{window_duration}" + windows.append((window_day, seg_key, window_items_acc)) + + window_start = ts + window_day = item_day + window_items_acc = [] + + window_items_acc.append(item) + + if window_items_acc and window_day and window_start is not None: + start_dt = dt.datetime.fromtimestamp(window_start, tz=tz) + seg_key = f"{start_dt.strftime('%H%M%S')}_{window_duration}" + windows.append((window_day, seg_key, window_items_acc)) + + return windows + + # MIME type mapping for import metadata _MIME_TYPES = { ".m4a": "audio/mp4",