From c57f08cd70fb01eeeac8fcbb02b785103bd1d8bd Mon Sep 17 00:00:00 2001 From: Jer Miller Date: Fri, 6 Mar 2026 18:29:09 -0700 Subject: [PATCH] Rewrite Kindle and Gemini importers: creation-moment segments with markdown output Add a shared window_items helper and switch the ICS importer to use it for\ncreation-time segmenting. Rewrite the Kindle and Gemini file importers\nto emit windowed imported.md segment files with populated ImportResult\nsegment metadata instead of structured imported.jsonl output.\n\nUpdate importer and formatter tests to cover the new segment markdown\npaths and the shared windowing helper. --- tests/test_gemini_importer.py | 54 +++++--- tests/test_import_formatting.py | 11 +- tests/test_importer.py | 18 +-- tests/test_kindle_importer.py | 217 ++++++++++++++++++++++++++++++++ think/importers/gemini.py | 68 ++++++++-- think/importers/ics.py | 46 +------ think/importers/kindle.py | 86 +++++++++++-- think/importers/shared.py | 63 ++++++++++ 8 files changed, 474 insertions(+), 89 deletions(-) create mode 100644 tests/test_kindle_importer.py diff --git a/tests/test_gemini_importer.py b/tests/test_gemini_importer.py index eb487a3e4..427fc5717 100644 --- a/tests/test_gemini_importer.py +++ b/tests/test_gemini_importer.py @@ -189,20 +189,18 @@ def test_process_json(): os.environ["JOURNAL_PATH"] = journal result = importer.process(Path(f.name), Path(journal)) assert result.entries_written == 2 - assert len(result.files_created) == 1 # same day assert result.errors == [] - - # Verify JSONL content - jsonl_path = Path(result.files_created[0]) - assert jsonl_path.exists() - lines = jsonl_path.read_text().strip().split("\n") - header = json.loads(lines[0]) - assert header["import"]["source"] == "gemini" - assert header["entry_count"] == 2 - - entry = json.loads(lines[1]) - assert entry["type"] == "ai_chat" - assert entry["source"] == "gemini" + assert result.segments is not None + assert len(result.segments) >= 1 + assert any(Path(p).name == "imported.md" for p in result.files_created) + + md = "" + for file_path in result.files_created: + md_path = Path(file_path) + assert md_path.exists() + md += md_path.read_text() + assert "Python" in md + assert "sorted" in md finally: os.unlink(f.name) os.environ.pop("JOURNAL_PATH", None) @@ -221,12 +219,40 @@ def test_process_zip(): os.environ["JOURNAL_PATH"] = journal result = importer.process(Path(tmp.name), Path(journal)) assert result.entries_written == 1 - assert len(result.files_created) == 1 + assert result.segments is not None + assert len(result.segments) == 1 + assert any(Path(p).suffix == ".md" for p in result.files_created) finally: os.unlink(tmp.name) os.environ.pop("JOURNAL_PATH", None) +def test_process_multiple_windows(): + """Activities more than 5 minutes apart land in different segments.""" + with tempfile.NamedTemporaryFile(suffix=".json", mode="w", delete=False) as f: + activities = [ + _sample_activity(time="2026-01-15T10:00:00Z"), + _sample_activity( + prompt="Second question", + response="Second answer", + time="2026-01-15T10:10:00Z", + ), + ] + json.dump(activities, f) + f.flush() + try: + with tempfile.TemporaryDirectory() as journal: + os.environ["JOURNAL_PATH"] = journal + result = importer.process(Path(f.name), Path(journal)) + assert result.entries_written == 2 + assert result.segments is not None + assert len(result.segments) == 2 + assert len(result.files_created) == 2 + finally: + os.unlink(f.name) + os.environ.pop("JOURNAL_PATH", None) + + # --- Registry test --- diff --git a/tests/test_import_formatting.py b/tests/test_import_formatting.py index ec64cf5b0..942be55a8 100644 --- a/tests/test_import_formatting.py +++ b/tests/test_import_formatting.py @@ -233,8 +233,17 @@ def test_formatter_registration_ics_segment_markdown(): def test_formatter_registration_kindle(): from think.formatters import get_formatter - formatter = get_formatter("20260115/import.kindle/imported.jsonl") + formatter = get_formatter("20260115/import.kindle/103000_300/imported.md") assert formatter is not None + assert formatter.__name__ == "format_markdown" + + +def test_formatter_registration_gemini_segment(): + from think.formatters import get_formatter + + formatter = get_formatter("20260115/import.gemini/100000_300/imported.md") + assert formatter is not None + assert formatter.__name__ == "format_markdown" def test_path_metadata_extraction(): diff --git a/tests/test_importer.py b/tests/test_importer.py index fe414e84f..2fabdf59e 100644 --- a/tests/test_importer.py +++ b/tests/test_importer.py @@ -1007,8 +1007,8 @@ def test_ics_creation_timestamp_none(): assert mod._creation_timestamp(EmptyComponent()) is None -def test_ics_window_events_single_window(): - mod = importlib.import_module("think.importers.ics") +def test_window_items_single_window(): + mod = importlib.import_module("think.importers.shared") base = dt.datetime(2026, 3, 1, 12, 0, 0, tzinfo=dt.timezone.utc).timestamp() events = [ @@ -1017,13 +1017,13 @@ def test_ics_window_events_single_window(): {"title": "C", "create_ts": base + 120}, ] - windows = mod._window_events(events) + windows = mod.window_items(events, "create_ts") assert windows == [("20260301", "120000_300", events)] -def test_ics_window_events_time_gap_split(): - mod = importlib.import_module("think.importers.ics") +def test_window_items_time_gap_split(): + mod = importlib.import_module("think.importers.shared") base = dt.datetime(2026, 3, 1, 12, 0, 0, tzinfo=dt.timezone.utc).timestamp() events = [ @@ -1033,7 +1033,7 @@ def test_ics_window_events_time_gap_split(): {"title": "D", "create_ts": base + 600}, ] - windows = mod._window_events(events) + windows = mod.window_items(events, "create_ts") assert len(windows) == 2 assert windows[0][0] == "20260301" @@ -1043,8 +1043,8 @@ def test_ics_window_events_time_gap_split(): assert windows[1][2] == [events[3]] -def test_ics_window_events_day_boundary(): - mod = importlib.import_module("think.importers.ics") +def test_window_items_day_boundary(): + mod = importlib.import_module("think.importers.shared") first_day = dt.datetime(2026, 3, 1, 12, 0, 0, tzinfo=dt.timezone.utc).timestamp() second_day = dt.datetime(2026, 3, 2, 12, 0, 0, tzinfo=dt.timezone.utc).timestamp() @@ -1053,7 +1053,7 @@ def test_ics_window_events_day_boundary(): {"title": "B", "create_ts": second_day}, ] - windows = mod._window_events(events) + windows = mod.window_items(events, "create_ts") assert windows == [ ("20260301", "120000_300", [events[0]]), diff --git a/tests/test_kindle_importer.py b/tests/test_kindle_importer.py new file mode 100644 index 000000000..ad0a57fcc --- /dev/null +++ b/tests/test_kindle_importer.py @@ -0,0 +1,217 @@ +# SPDX-License-Identifier: AGPL-3.0-only +# Copyright (c) 2026 sol pbc + +"""Tests for think.importers.kindle — Kindle My Clippings.txt importer.""" + +import os +import tempfile +from pathlib import Path + +from think.importers.kindle import KindleImporter, _parse_block, _parse_date + +importer = KindleImporter() + + +def _make_clipping( + title: str = "Test Book (Author Name)", + meta: str = "- Your Highlight on page 42 | location 100-101 | Added on Saturday, March 15, 2025 10:30:00 AM", + content: str = "This is a highlighted passage.", +) -> str: + return f"{title}\n{meta}\n\n{content}\n" + + +def _make_clippings_file(clippings: list[str]) -> str: + return "==========\n".join(clippings) + "==========\n" + + +# --- Unit tests for helpers --- + + +def test_parse_date(): + result = _parse_date("Saturday, March 15, 2025 10:30:00 AM") + assert result is not None + assert result.month == 3 + assert result.day == 15 + + +def test_parse_block_basic(): + block = _make_clipping() + entry = _parse_block(block) + assert entry is not None + assert entry["type"] == "highlight" + assert entry["book_title"] == "Test Book" + assert entry["author"] == "Author Name" + assert entry["content"] == "This is a highlighted passage." + assert entry["page"] == 42 + assert entry["location"] == "100-101" + + +def test_parse_block_note(): + block = _make_clipping( + meta="- Your Note on page 10 | Added on Saturday, March 15, 2025 10:30:00 AM", + content="My personal note", + ) + entry = _parse_block(block) + assert entry is not None + assert entry["clip_type"] == "note" + + +def test_parse_block_no_author(): + block = _make_clipping(title="Title Without Author") + entry = _parse_block(block) + assert entry is not None + assert entry["book_title"] == "Title Without Author" + assert entry["author"] == "" + + +# --- Detection tests --- + + +def test_detect_valid(): + content = _make_clippings_file([_make_clipping()]) + with tempfile.NamedTemporaryFile(suffix=".txt", mode="w", delete=False) as f: + f.write(content) + f.flush() + try: + assert importer.detect(Path(f.name)) is True + finally: + os.unlink(f.name) + + +def test_detect_wrong_format(): + with tempfile.NamedTemporaryFile(suffix=".txt", mode="w", delete=False) as f: + f.write("Just some random text file.\n") + f.flush() + try: + assert importer.detect(Path(f.name)) is False + finally: + os.unlink(f.name) + + +# --- Preview tests --- + + +def test_preview(): + content = _make_clippings_file( + [ + _make_clipping(), + _make_clipping( + title="Another Book (Jane Doe)", + meta="- Your Note on page 5 | Added on Sunday, March 16, 2025 02:00:00 PM", + content="A note", + ), + ] + ) + with tempfile.NamedTemporaryFile(suffix=".txt", mode="w", delete=False) as f: + f.write(content) + f.flush() + try: + preview = importer.preview(Path(f.name)) + assert preview.item_count == 2 + assert preview.entity_count > 0 + assert "2 books" in preview.summary + finally: + os.unlink(f.name) + + +# --- Process tests --- + + +def test_process_basic(): + content = _make_clippings_file( + [ + _make_clipping(), + _make_clipping( + meta="- Your Highlight on page 43 | Added on Saturday, March 15, 2025 10:31:00 AM", + content="Another highlight from same session.", + ), + ] + ) + with tempfile.NamedTemporaryFile(suffix=".txt", mode="w", delete=False) as f: + f.write(content) + f.flush() + try: + with tempfile.TemporaryDirectory() as journal: + os.environ["JOURNAL_PATH"] = journal + result = importer.process(Path(f.name), Path(journal)) + assert result.entries_written == 2 + assert result.errors == [] + assert result.segments is not None + assert len(result.segments) >= 1 + + md_path = Path(result.files_created[0]) + assert md_path.exists() + assert md_path.name == "imported.md" + md = md_path.read_text() + assert "Test Book" in md + assert "> This is a highlighted passage." in md + assert "Page 42" in md + finally: + os.unlink(f.name) + os.environ.pop("JOURNAL_PATH", None) + + +def test_process_multiple_windows(): + """Highlights more than 5 minutes apart land in different segments.""" + content = _make_clippings_file( + [ + _make_clipping( + meta="- Your Highlight on page 1 | Added on Saturday, March 15, 2025 10:00:00 AM", + content="First highlight", + ), + _make_clipping( + meta="- Your Highlight on page 2 | Added on Saturday, March 15, 2025 10:10:00 AM", + content="Second highlight, 10 min later", + ), + ] + ) + with tempfile.NamedTemporaryFile(suffix=".txt", mode="w", delete=False) as f: + f.write(content) + f.flush() + try: + with tempfile.TemporaryDirectory() as journal: + os.environ["JOURNAL_PATH"] = journal + result = importer.process(Path(f.name), Path(journal)) + assert result.entries_written == 2 + assert result.segments is not None + assert len(result.segments) == 2 + assert len(result.files_created) == 2 + finally: + os.unlink(f.name) + os.environ.pop("JOURNAL_PATH", None) + + +def test_process_note_markdown(): + """Notes render with Note: prefix instead of blockquote.""" + content = _make_clippings_file( + [ + _make_clipping( + meta="- Your Note on page 10 | Added on Saturday, March 15, 2025 10:30:00 AM", + content="My personal note", + ), + ] + ) + with tempfile.NamedTemporaryFile(suffix=".txt", mode="w", delete=False) as f: + f.write(content) + f.flush() + try: + with tempfile.TemporaryDirectory() as journal: + os.environ["JOURNAL_PATH"] = journal + result = importer.process(Path(f.name), Path(journal)) + md = Path(result.files_created[0]).read_text() + assert "Note: My personal note" in md + finally: + os.unlink(f.name) + os.environ.pop("JOURNAL_PATH", None) + + +# --- Registry test --- + + +def test_registered_in_registry(): + from think.importers.file_importer import FILE_IMPORTER_REGISTRY, get_file_importer + + assert "kindle" in FILE_IMPORTER_REGISTRY + imp = get_file_importer("kindle") + assert imp is not None + assert imp.name == "kindle" diff --git a/think/importers/gemini.py b/think/importers/gemini.py index db36c097d..f54f4f034 100644 --- a/think/importers/gemini.py +++ b/think/importers/gemini.py @@ -23,7 +23,8 @@ from pathlib import Path from typing import Any, Callable from think.importers.file_importer import ImportPreview, ImportResult -from think.importers.shared import write_structured_import +from think.importers.shared import window_items +from think.utils import day_path logger = logging.getLogger(__name__) @@ -150,6 +151,27 @@ def _parse_activity(activity: dict[str, Any]) -> dict[str, Any] | None: return entry +def _render_activity_markdown(activity: dict) -> str: + """Render a Gemini activity as markdown.""" + title = activity.get("title", "Gemini activity") + lines = [f"## {title}"] + + content = activity.get("content", "") + if content: + # Content already has "Human: ..." and "Assistant: ..." format + # Convert to bold labels + for part in content.split("\n\n"): + part = part.strip() + if part.startswith("Human: "): + lines.append(f"**Human:** {part[7:]}") + elif part.startswith("Assistant: "): + lines.append(f"**Assistant:** {part[11:]}") + elif part: + lines.append(part) + + return "\n\n".join(lines) + + class GeminiImporter: name = "gemini" display_name = "Gemini Activity History" @@ -233,7 +255,6 @@ class GeminiImporter: progress_callback: Callable | None = None, ) -> ImportResult: activities = _load_activities(path) - import_id = dt.datetime.now().strftime("%Y%m%d_%H%M%S") entries: list[dict[str, Any]] = [] errors: list[str] = [] @@ -244,18 +265,41 @@ class GeminiImporter: if entry is None: skipped += 1 continue + + # Add epoch timestamp for windowing + entry["create_ts"] = dt.datetime.fromisoformat(entry["ts"]).timestamp() entries.append(entry) if progress_callback and (i + 1) % 100 == 0: progress_callback(i + 1, len(activities)) - # Write to journal - created_files = write_structured_import( - "gemini", - entries, - import_id=import_id, - facet=facet, - ) + if not entries: + return ImportResult( + entries_written=0, + entities_seeded=0, + files_created=[], + errors=errors, + summary="No activities found to import", + ) + + entries.sort(key=lambda e: e["create_ts"]) + + windows = window_items(entries, "create_ts") + created_files: list[str] = [] + segments: list[tuple[str, str]] = [] + + for day, seg_key, window_activities in windows: + segment_dir = day_path(day) / "import.gemini" / seg_key + segment_dir.mkdir(parents=True, exist_ok=True) + md_path = segment_dir / "imported.md" + markdown = "\n\n".join( + _render_activity_markdown(act) for act in window_activities + ) + md_path.write_text(markdown + "\n", encoding="utf-8") + created_files.append(str(md_path)) + segments.append((day, seg_key)) + + segment_days = {day for day, _ in segments} if skipped: logger.info("Skipped %d activities with no content", skipped) @@ -268,7 +312,11 @@ class GeminiImporter: entities_seeded=0, files_created=created_files, errors=errors, - summary=f"Imported {len(entries)} Gemini activities{bard_info} across {len(created_files)} days", + summary=( + f"Imported {len(entries)} Gemini activities{bard_info} across " + f"{len(segment_days)} days into {len(segments)} segments" + ), + segments=segments, ) diff --git a/think/importers/ics.py b/think/importers/ics.py index 5dc12212b..4ab473524 100644 --- a/think/importers/ics.py +++ b/think/importers/ics.py @@ -10,7 +10,7 @@ from pathlib import Path from typing import Any, Callable from think.importers.file_importer import ImportPreview, ImportResult -from think.importers.shared import seed_entities +from think.importers.shared import seed_entities, window_items from think.utils import day_path logger = logging.getLogger(__name__) @@ -126,48 +126,6 @@ def _creation_timestamp(component: Any) -> float | None: return None -def _window_events( - events: list[dict[str, Any]], - window_duration: int = 300, -) -> list[tuple[str, str, list[dict[str, Any]]]]: - """Group sorted events into fixed-duration windows per creation time.""" - if not events: - return [] - - windows: list[tuple[str, str, list[dict[str, Any]]]] = [] - window_start: float | None = None - window_day: str | None = None - window_events: list[dict[str, Any]] = [] - - for event in events: - create_ts = event["create_ts"] - event_dt = dt.datetime.fromtimestamp(create_ts, tz=dt.timezone.utc) - event_day = event_dt.strftime("%Y%m%d") - - if ( - window_start is None - or event_day != window_day - or create_ts - window_start >= window_duration - ): - if window_events and window_day and window_start is not None: - start_dt = dt.datetime.fromtimestamp(window_start, tz=dt.timezone.utc) - seg_key = f"{start_dt.strftime('%H%M%S')}_{window_duration}" - windows.append((window_day, seg_key, window_events)) - - window_start = create_ts - window_day = event_day - window_events = [] - - window_events.append(event) - - if window_events and window_day and window_start is not None: - start_dt = dt.datetime.fromtimestamp(window_start, tz=dt.timezone.utc) - seg_key = f"{start_dt.strftime('%H%M%S')}_{window_duration}" - windows.append((window_day, seg_key, window_events)) - - return windows - - def _render_event_markdown(event: dict[str, Any]) -> str: """Render a calendar event as markdown.""" title = event.get("title", "Untitled event") @@ -398,7 +356,7 @@ class ICSImporter: all_entries.sort(key=lambda entry: entry["create_ts"]) - windows = _window_events(all_entries) + windows = window_items(all_entries, "create_ts") created_files: list[str] = [] segments: list[tuple[str, str]] = [] diff --git a/think/importers/kindle.py b/think/importers/kindle.py index 3ff7b66dd..0fc87f80f 100644 --- a/think/importers/kindle.py +++ b/think/importers/kindle.py @@ -10,7 +10,8 @@ from pathlib import Path from typing import Callable from think.importers.file_importer import ImportPreview, ImportResult -from think.importers.shared import seed_entities, write_structured_import +from think.importers.shared import seed_entities, window_items +from think.utils import day_path logger = logging.getLogger(__name__) @@ -128,6 +129,47 @@ def _parse_block(block: str) -> dict | None: return entry +def _render_highlight_markdown(highlights: list[dict]) -> str: + """Render highlights grouped by book as markdown.""" + # Group by book + by_book: dict[str, list[dict]] = {} + for h in highlights: + key = h["book_title"] + by_book.setdefault(key, []).append(h) + + sections: list[str] = [] + for book_title, book_highlights in by_book.items(): + # Use first highlight's author (all from same book share author) + author = book_highlights[0].get("author", "") + if author: + heading = f"## {book_title} by {author}" + else: + heading = f"## {book_title}" + + lines = [heading] + for h in book_highlights: + content = h.get("content", "") + clip_type = h.get("clip_type", "highlight") + + if clip_type == "note": + lines.append(f"Note: {content}") + else: + lines.append(f"> {content}") + + # Page / location metadata + meta_parts: list[str] = [] + if h.get("page") is not None: + meta_parts.append(f"Page {h['page']}") + if h.get("location"): + meta_parts.append(f"Location {h['location']}") + if meta_parts: + lines.append(" | ".join(meta_parts)) + + sections.append("\n".join(lines)) + + return "\n\n".join(sections) + + class KindleImporter: name = "kindle" display_name = "Kindle Highlights" @@ -211,7 +253,6 @@ class KindleImporter: ) -> ImportResult: text = path.read_text(encoding="utf-8-sig") blocks = text.split(DELIMITER) - import_id = dt.datetime.now().strftime("%Y%m%d_%H%M%S") entries: list[dict] = [] errors: list[str] = [] @@ -223,12 +264,13 @@ class KindleImporter: continue entry = _parse_block(block) if entry is None: - # Only log as error if block had real content (not just whitespace) stripped = block.strip() if stripped and "\n" in stripped: errors.append(f"Failed to parse clipping block {i + 1}") continue + # Add epoch timestamp for windowing + entry["create_ts"] = dt.datetime.fromisoformat(entry["ts"]).timestamp() entries.append(entry) books.add(entry["book_title"]) if entry["author"]: @@ -237,13 +279,31 @@ class KindleImporter: if progress_callback and (i + 1) % 100 == 0: progress_callback(i + 1, len(blocks)) - # Write to journal - created_files = write_structured_import( - "kindle", - entries, - import_id=import_id, - facet=facet, - ) + if not entries: + return ImportResult( + entries_written=0, + entities_seeded=0, + files_created=[], + errors=errors, + summary="No clippings found to import", + ) + + entries.sort(key=lambda e: e["create_ts"]) + + windows = window_items(entries, "create_ts", tz=None) + created_files: list[str] = [] + segments: list[tuple[str, str]] = [] + + for day, seg_key, window_highlights in windows: + segment_dir = day_path(day) / "import.kindle" / seg_key + segment_dir.mkdir(parents=True, exist_ok=True) + md_path = segment_dir / "imported.md" + markdown = _render_highlight_markdown(window_highlights) + md_path.write_text(markdown + "\n", encoding="utf-8") + created_files.append(str(md_path)) + segments.append((day, seg_key)) + + segment_days = {day for day, _ in segments} # Seed entities (books and authors) entities_seeded = 0 @@ -266,7 +326,11 @@ class KindleImporter: entities_seeded=entities_seeded, files_created=created_files, errors=errors, - summary=f"Imported {len(entries)} Kindle clippings from {len(books)} books across {len(created_files)} days", + summary=( + f"Imported {len(entries)} Kindle clippings from {len(books)} books " + f"across {len(segment_days)} days into {len(segments)} segments" + ), + segments=segments, ) diff --git a/think/importers/shared.py b/think/importers/shared.py index 903344351..b7cf5e883 100644 --- a/think/importers/shared.py +++ b/think/importers/shared.py @@ -178,6 +178,69 @@ def _window_messages( return windows +def window_items( + items: list[dict[str, Any]], + ts_key: str, + *, + window_duration: int = 300, + tz: dt.timezone | None = dt.timezone.utc, +) -> list[tuple[str, str, list[dict[str, Any]]]]: + """Group sorted items into fixed-duration windows per day. + + Parameters + ---------- + items : list[dict] + Items sorted by ts_key. The ts_key field must be a float epoch. + ts_key : str + Key name for the float epoch timestamp in each item. + window_duration : int + Window size in seconds (default 300 = 5 minutes). + tz : timezone or None + Timezone for day grouping and seg_key formatting. + Use dt.timezone.utc for UTC timestamps, None for local time. + + Returns + ------- + list[tuple[str, str, list[dict]]] + (day_str, seg_key, items) tuples. + """ + if not items: + return [] + + windows: list[tuple[str, str, list[dict[str, Any]]]] = [] + window_start: float | None = None + window_day: str | None = None + window_items_acc: list[dict[str, Any]] = [] + + for item in items: + ts = item[ts_key] + item_dt = dt.datetime.fromtimestamp(ts, tz=tz) + item_day = item_dt.strftime("%Y%m%d") + + if ( + window_start is None + or item_day != window_day + or ts - window_start >= window_duration + ): + if window_items_acc and window_day and window_start is not None: + start_dt = dt.datetime.fromtimestamp(window_start, tz=tz) + seg_key = f"{start_dt.strftime('%H%M%S')}_{window_duration}" + windows.append((window_day, seg_key, window_items_acc)) + + window_start = ts + window_day = item_day + window_items_acc = [] + + window_items_acc.append(item) + + if window_items_acc and window_day and window_start is not None: + start_dt = dt.datetime.fromtimestamp(window_start, tz=tz) + seg_key = f"{start_dt.strftime('%H%M%S')}_{window_duration}" + windows.append((window_day, seg_key, window_items_acc)) + + return windows + + # MIME type mapping for import metadata _MIME_TYPES = { ".m4a": "audio/mp4", -- 2.51.2