diff --git a/AGENTS.md b/AGENTS.md index 8c81adcae..7d8a6a413 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -182,6 +182,7 @@ Each domain has exactly **one** write-owning module (or one tightly-scoped famil |--------|------------------------| | Entities (`entities/*/entity.json`, `entities/*/*.npz`) | `solstone/think/entities/journal.py` + `solstone/think/entities/consolidation.py` + `solstone/think/entities/saving.py` + `solstone/think/entities/merge.py` + `solstone/apps/entities/call.py` | | Entity merge candidates (`entities/review-candidates.jsonl`) | `solstone/think/entities/review_candidates.py` + `solstone/apps/entities/call.py` | +| Facet review candidates (`facets/review-candidates.jsonl`) | `solstone/think/facet_review_candidates.py` | | Facets (`facets/*/facet.json`, `facets/*/relationships/`) | `solstone/think/facets.py` + `solstone/apps/facets/*` (if/when created) | | Observations (`observations.jsonl`) | `solstone/think/entities/observations.py` | | Activities (`facets/*/activities/*.jsonl`) | `solstone/think/activities.py` | diff --git a/solstone/think/facet_candidates_cli.py b/solstone/think/facet_candidates_cli.py new file mode 100644 index 000000000..5c8a0f077 --- /dev/null +++ b/solstone/think/facet_candidates_cli.py @@ -0,0 +1,46 @@ +# SPDX-License-Identifier: AGPL-3.0-only +# Copyright (c) 2026 sol pbc + +"""CLI entrypoint for recurring facet review-candidate recording.""" + +from __future__ import annotations + +import argparse +from datetime import datetime + +from solstone.think.facet_review_candidates import record_facet_candidate +from solstone.think.facets import aggregate_speculative_facets +from solstone.think.utils import require_solstone, setup_cli + + +def run() -> int: + """Record recurring facet candidates and return the number refreshed.""" + day = datetime.now().strftime("%Y%m%d") + candidates = aggregate_speculative_facets() + for candidate in candidates: + record_facet_candidate( + name=candidate["name"], + name_key=candidate["name_key"], + count=candidate["count"], + window_days=candidate["window_days"], + samples=candidate["samples"], + day=day, + ) + + count = len(candidates) + print(f"Recorded/updated {count} facet candidate(s).") + return count + + +def main() -> None: + """Entry point for ``journal facet-candidates``.""" + parser = argparse.ArgumentParser( + description="Record recurring facet review candidates." + ) + setup_cli(parser) + require_solstone() + run() + + +if __name__ == "__main__": + main() diff --git a/solstone/think/facet_review_candidates.py b/solstone/think/facet_review_candidates.py new file mode 100644 index 000000000..b18c2cc24 --- /dev/null +++ b/solstone/think/facet_review_candidates.py @@ -0,0 +1,179 @@ +# SPDX-License-Identifier: AGPL-3.0-only +# Copyright (c) 2026 sol pbc +"""Facet review-candidate storage helpers. + +Sole write-owner of: + journal/facets/review-candidates.jsonl +""" + +from __future__ import annotations + +import fcntl +import json +import logging +from datetime import datetime, timezone +from pathlib import Path +from typing import Any, Callable + +from solstone.think.entities.core import atomic_write +from solstone.think.utils import get_journal + +logger = logging.getLogger(__name__) + + +def facet_review_candidates_dir() -> Path: + """Return the facet review-candidates directory, creating it if needed.""" + path = Path(get_journal()) / "facets" + path.mkdir(parents=True, exist_ok=True) + return path + + +def facet_review_candidates_path() -> Path: + """Return the facet review-candidates JSONL path.""" + return facet_review_candidates_dir() / "review-candidates.jsonl" + + +def facet_review_candidates_lock_path() -> Path: + """Return the sibling lock path for review-candidates.jsonl.""" + return facet_review_candidates_dir() / ".review-candidates.lock" + + +def _load_jsonl_rows(path: Path) -> list[dict[str, Any]]: + """Load JSONL rows from *path*, skipping blanks and malformed lines.""" + if not path.exists(): + return [] + + rows: list[dict[str, Any]] = [] + with open(path, encoding="utf-8") as handle: + for lineno, line in enumerate(handle, start=1): + raw = line.strip() + if not raw: + continue + try: + data = json.loads(raw) + except json.JSONDecodeError: + logger.warning( + "facet review candidates: malformed JSONL line %s in %s", + lineno, + path, + ) + continue + if not isinstance(data, dict): + logger.warning( + "facet review candidates: non-object JSONL line %s in %s (got %s)", + lineno, + path, + type(data).__name__, + ) + continue + rows.append(data) + return rows + + +def load_candidates() -> list[dict[str, Any]]: + """Load facet review candidates from JSONL.""" + return _load_jsonl_rows(facet_review_candidates_path()) + + +def _save_jsonl_rows(path: Path, rows: list[dict[str, Any]]) -> None: + """Write *rows* to *path* as JSONL using an atomic replace.""" + content = "" + if rows: + content = "\n".join(json.dumps(row, ensure_ascii=False) for row in rows) + "\n" + atomic_write(path, content) + + +def save_candidates(rows: list[dict[str, Any]]) -> None: + """Persist facet review candidates atomically.""" + _save_jsonl_rows(facet_review_candidates_path(), rows) + + +def candidate_key(name_key: str) -> str: + """Return the deterministic key for one facet review candidate.""" + return name_key + + +def find_candidate(rows: list[dict[str, Any]], name_key: str) -> dict[str, Any] | None: + """Return one facet review candidate by key, or None when not found.""" + target_key = candidate_key(name_key) + for row in rows: + row_key = candidate_key(str(row.get("name_key") or "")) + if row_key == target_key: + return row + return None + + +def locked_modify_candidates( + fn: Callable[[list[dict[str, Any]]], list[dict[str, Any]]], +) -> list[dict[str, Any]]: + """Apply a locked read-modify-write cycle to review-candidates.jsonl.""" + facet_review_candidates_dir() + lock_path = facet_review_candidates_lock_path() + # Lock file contents are irrelevant; opening with "w" matches the existing pattern. + with open(lock_path, "w", encoding="utf-8") as lock_file: + fcntl.flock(lock_file, fcntl.LOCK_EX) + try: + rows = load_candidates() + new_rows = fn(rows) + save_candidates(new_rows) + return new_rows + finally: + fcntl.flock(lock_file, fcntl.LOCK_UN) + + +def utc_now_iso() -> str: + """Return the current UTC time as an ISO-8601 string ending in Z.""" + return ( + datetime.now(timezone.utc).isoformat(timespec="seconds").replace("+00:00", "Z") + ) + + +def touch_updated(row: dict[str, Any]) -> None: + """Update a candidate row's updated_at timestamp in place.""" + row["updated_at"] = utc_now_iso() + + +def record_facet_candidate( + name: str, + name_key: str, + count: int, + window_days: int, + samples: list[dict[str, Any]], + day: str, +) -> dict[str, Any]: + """Record or refresh one proposed facet candidate.""" + row: dict[str, Any] | None = None + + def mutate(rows: list[dict[str, Any]]) -> list[dict[str, Any]]: + nonlocal row + existing = find_candidate(rows, name_key) + now = utc_now_iso() + if existing is None: + row = { + "name": name, + "name_key": name_key, + "status": "open", + "count": count, + "window_days": window_days, + "evidence": {"samples": samples}, + "first_surfaced": day, + "last_surfaced": day, + "created_at": now, + "updated_at": now, + } + return list(rows) + [row] + + ev = existing.setdefault("evidence", {}) + ev["samples"] = samples + existing["count"] = count + existing["window_days"] = window_days + existing["last_surfaced"] = day + existing["updated_at"] = now + row = existing + return rows + + locked_modify_candidates(mutate) + + if row is None: # pragma: no cover - defensive assertion + raise RuntimeError("record_facet_candidate produced no row") + return row diff --git a/solstone/think/facets.py b/solstone/think/facets.py index 39206ae8a..df8ded463 100644 --- a/solstone/think/facets.py +++ b/solstone/think/facets.py @@ -8,12 +8,12 @@ import logging import os import re import shutil -from datetime import datetime, timezone +from datetime import datetime, timedelta, timezone from pathlib import Path from typing import Any, Optional from solstone.think.entities import get_identity_names -from solstone.think.utils import DATE_RE, day_path, get_journal, iter_segments +from solstone.think.utils import day_dirs, day_path, get_journal, iter_segments def _get_principal_display_name() -> str | None: @@ -636,80 +636,87 @@ def get_active_facets(day: str) -> set[str]: return active -def aggregate_speculative_facets(days: list[str] | None = None) -> list[dict]: - """Aggregate speculative facet outputs from segment classifiers across days. +# speculative_facet is sparse, so use a recent rolling window before surfacing. +FACET_CANDIDATE_WINDOW_DAYS = 14 +# Prefer one strong recurring candidate over several weak early ones. +FACET_CANDIDATE_MIN_SEGMENTS = 3 - Scans per-segment agents/facets.json files produced by the facets classifier - and counts facet name frequency. Useful during onboarding to suggest journal - organization to the user. - Args: - days: Optional list of days in YYYYMMDD format. If None, scans all days. +def _normalize_speculative_name(name: str) -> tuple[str, str]: + """Return display and normalized key for one speculative facet name.""" + display = " ".join(name.split()) + return display, display.casefold() - Returns: - List of dicts with keys: - - "facet": facet name (str) - - "count": number of segments where this facet appeared (int) - - "sample_activities": up to 3 activity descriptions for this facet (list[str]) - Sorted by count descending, capped at 8 entries. - """ - journal_path = Path(get_journal()) - if days is not None: - scan_days = days +def aggregate_speculative_facets( + days: list[str] | None = None, + min_count: int = FACET_CANDIDATE_MIN_SEGMENTS, +) -> list[dict[str, Any]]: + """Aggregate recurring speculative facet proposals from segment sense output. + + Scans per-segment ``talents/sense.json`` files over a rolling recent-day + window and returns proposed-name candidates at or above ``min_count``. + Side-effect-free. + """ + if days is None: + cutoff = ( + datetime.now() - timedelta(days=FACET_CANDIDATE_WINDOW_DAYS) + ).strftime("%Y%m%d") + scan_days = [day for day in day_dirs() if day >= cutoff] else: - scan_days = [] - if journal_path.exists(): - for entry in sorted(journal_path.iterdir()): - if entry.is_dir() and DATE_RE.fullmatch(entry.name): - scan_days.append(entry.name) + scan_days = days + scan_days = sorted(scan_days) - facet_counts: dict[str, int] = {} - facet_activities: dict[str, list[str]] = {} + groups: dict[str, dict[str, Any]] = {} for day in scan_days: - for _stream, _seg_key, seg_path in iter_segments(day): - facets_file = seg_path / "talents" / "facets.json" - if not facets_file.exists(): + for stream, seg_key, seg_path in iter_segments(day): + sense_file = seg_path / "talents" / "sense.json" + if not sense_file.exists(): continue try: - content = facets_file.read_text().strip() - if not content: - continue - - data = json.loads(content) - if not isinstance(data, list): - continue - - for item in data: - if not isinstance(item, dict): - continue - facet_name = item.get("facet") - if not facet_name: - continue - facet_counts[facet_name] = facet_counts.get(facet_name, 0) + 1 - - activity = item.get("activity", "") - if activity: - samples = facet_activities.setdefault(facet_name, []) - if len(samples) < 3: - samples.append(activity) - + data = json.loads(sense_file.read_text(encoding="utf-8")) except (json.JSONDecodeError, OSError): continue + if not isinstance(data, dict): + continue + + raw_name = data.get("speculative_facet") + if not isinstance(raw_name, str): + continue + + display, name_key = _normalize_speculative_name(raw_name) + if not display: + continue + + group = groups.setdefault( + name_key, + { + "name": display, + "name_key": name_key, + "count": 0, + "samples": [], + }, + ) + group["count"] += 1 + samples = group["samples"] + if len(samples) < 3: + samples.append({"day": day, "stream": stream, "segment": seg_key}) result = [ { - "facet": facet_name, - "count": count, - "sample_activities": facet_activities.get(facet_name, []), + "name": row["name"], + "name_key": row["name_key"], + "count": row["count"], + "window_days": FACET_CANDIDATE_WINDOW_DAYS, + "samples": row["samples"], } - for facet_name, count in sorted( - facet_counts.items(), key=lambda x: x[1], reverse=True - ) + for row in groups.values() + if row["count"] >= min_count ] - return result[:8] + result.sort(key=lambda row: (-row["count"], row["name_key"])) + return result def set_facet_muted(facet: str, muted: bool) -> None: diff --git a/solstone/think/scheduler.py b/solstone/think/scheduler.py index f6c00d942..21c844072 100644 --- a/solstone/think/scheduler.py +++ b/solstone/think/scheduler.py @@ -345,8 +345,14 @@ def register_defaults() -> None: need_heartbeat = "heartbeat" not in _entries need_weekly = "weekly-agents" not in _entries need_providers = "providers" not in _entries - - if not need_heartbeat and not need_weekly and not need_providers: + need_facet_candidates = "facet-candidates" not in _entries + + if ( + not need_heartbeat + and not need_weekly + and not need_providers + and not need_facet_candidates + ): return # Read raw config (preserving daily_time and other entries) @@ -394,6 +400,15 @@ def register_defaults() -> None: } changed = True + if need_facet_candidates and "facet-candidates" not in raw: + raw["facet-candidates"] = { + "cmd": ["journal", "facet-candidates"], + "every": "weekly", + "enabled": True, + "max_runtime": "10m", + } + changed = True + if not changed: return diff --git a/solstone/think/sol_cli.py b/solstone/think/sol_cli.py index 9cf189852..673a67f4c 100644 --- a/solstone/think/sol_cli.py +++ b/solstone/think/sol_cli.py @@ -106,6 +106,7 @@ COMMANDS: dict[str, Command] = { "observer": Command("solstone.observe.observer_cli", "service"), # AI providers and talent execution "providers": Command("solstone.think.providers_cli", "service"), + "facet-candidates": Command("solstone.think.facet_candidates_cli", "service"), "cortex": Command("solstone.think.cortex", "service"), "talent": Command("solstone.think.talent_cli", "service"), "link": Command("solstone.think.link", "access"), diff --git a/tests/test_facet_candidates_cli.py b/tests/test_facet_candidates_cli.py new file mode 100644 index 000000000..f3a21cb3e --- /dev/null +++ b/tests/test_facet_candidates_cli.py @@ -0,0 +1,59 @@ +# SPDX-License-Identifier: AGPL-3.0-only +# Copyright (c) 2026 sol pbc + +from __future__ import annotations + +import json +from datetime import datetime +from pathlib import Path + +from solstone.think.facet_candidates_cli import run +from solstone.think.facet_review_candidates import ( + facet_review_candidates_path, + load_candidates, +) + + +def _write_segment_sense( + journal: Path, + day: str, + segment: str, + speculative_facet: str, + *, + stream: str = "archon", +) -> None: + talents_dir = journal / "chronicle" / day / stream / segment / "talents" + talents_dir.mkdir(parents=True, exist_ok=True) + (talents_dir / "sense.json").write_text( + json.dumps({"speculative_facet": speculative_facet}), + encoding="utf-8", + ) + + +def test_run_records_and_upserts_facet_candidates(monkeypatch, tmp_path): + monkeypatch.setenv("SOLSTONE_JOURNAL", str(tmp_path)) + day = datetime.now().strftime("%Y%m%d") + for segment in ("090000_300", "093000_300", "100000_300"): + _write_segment_sense(tmp_path, day, segment, "Home Reno") + + first_count = run() + rows = load_candidates() + + assert first_count == 1 + assert facet_review_candidates_path().exists() + assert len(rows) == 1 + row = rows[0] + assert row["name"] == "Home Reno" + assert row["name_key"] == "home reno" + assert row["status"] == "open" + assert row["count"] == 3 + first_surfaced = row["first_surfaced"] + created_at = row["created_at"] + + second_count = run() + rows = load_candidates() + + assert second_count == 1 + assert len(rows) == 1 + assert rows[0]["first_surfaced"] == first_surfaced + assert rows[0]["created_at"] == created_at diff --git a/tests/test_facet_review_candidates.py b/tests/test_facet_review_candidates.py new file mode 100644 index 000000000..b2e3cb003 --- /dev/null +++ b/tests/test_facet_review_candidates.py @@ -0,0 +1,285 @@ +# SPDX-License-Identifier: AGPL-3.0-only +# Copyright (c) 2026 sol pbc + +from __future__ import annotations + +import logging +import threading +from pathlib import Path + +import pytest + +import solstone.think.facet_review_candidates as mod +from solstone.think.facet_review_candidates import ( + candidate_key, + facet_review_candidates_dir, + facet_review_candidates_lock_path, + facet_review_candidates_path, + find_candidate, + load_candidates, + locked_modify_candidates, + record_facet_candidate, + save_candidates, + touch_updated, + utc_now_iso, +) + + +@pytest.fixture +def candidate_journal(monkeypatch, tmp_path): + monkeypatch.setenv("SOLSTONE_JOURNAL", str(tmp_path)) + return Path(tmp_path) + + +def test_facet_review_candidates_dir_creates_dir(candidate_journal): + path = facet_review_candidates_dir() + + assert path == candidate_journal / "facets" + assert path.exists() + assert path.is_dir() + + +def test_path_helpers_return_expected_names(candidate_journal): + assert ( + facet_review_candidates_path() + == candidate_journal / "facets" / "review-candidates.jsonl" + ) + assert ( + facet_review_candidates_lock_path() + == candidate_journal / "facets" / ".review-candidates.lock" + ) + + +def test_load_candidates_missing_file_returns_empty(candidate_journal): + assert load_candidates() == [] + + +def test_save_and_load_candidates_roundtrip(candidate_journal): + rows = [ + {"name": "Home Reno", "name_key": "home reno"}, + {"name": "Field Notes", "name_key": "field notes"}, + ] + + save_candidates(rows) + + assert load_candidates() == rows + + +def test_save_candidates_empty_list_writes_empty_file(candidate_journal): + save_candidates([]) + + assert facet_review_candidates_path().read_text(encoding="utf-8") == "" + + +def test_load_candidates_skips_malformed_line(candidate_journal, caplog): + facet_review_candidates_path().write_text( + '{"name_key": "home reno"}\nnot-json\n', + encoding="utf-8", + ) + + rows = load_candidates() + + assert rows == [{"name_key": "home reno"}] + assert "malformed JSONL line 2" in caplog.text + + +def test_load_candidates_warns_on_non_dict_line(candidate_journal, caplog): + facet_review_candidates_path().write_text( + '{"name_key": "home reno"}\n[1, 2]\n', + encoding="utf-8", + ) + + with caplog.at_level(logging.WARNING): + rows = load_candidates() + + assert rows == [{"name_key": "home reno"}] + assert "non-object JSONL line 2" in caplog.text + assert "list" in caplog.text + + +def test_candidate_key_is_deterministic_and_distinct(candidate_journal): + assert candidate_key("home reno") == candidate_key("home reno") + assert candidate_key("home reno") != candidate_key("home-reno") + assert candidate_key("home reno") != candidate_key("field notes") + + +def test_find_candidate_returns_row_or_none(candidate_journal): + rows = [ + {"name": "Home Reno", "name_key": "home reno"}, + {"name": "Field Notes", "name_key": "field notes"}, + ] + + assert find_candidate(rows, "field notes") == rows[1] + assert find_candidate(rows, "missing") is None + + +def test_utc_now_iso_ends_with_z(candidate_journal): + assert utc_now_iso().endswith("Z") + + +def test_touch_updated_sets_updated_at(candidate_journal): + row = {} + + touch_updated(row) + + assert row["updated_at"].endswith("Z") + + +def test_locked_modify_candidates_applies_fn_and_persists(candidate_journal): + def mutate(rows): + return list(rows) + [{"name": "Home Reno", "name_key": "home reno"}] + + updated = locked_modify_candidates(mutate) + + assert updated == [{"name": "Home Reno", "name_key": "home reno"}] + assert load_candidates() == [{"name": "Home Reno", "name_key": "home reno"}] + + +def test_locked_modify_candidates_serializes_threads(candidate_journal): + barrier = threading.Barrier(4) + exceptions: list[BaseException] = [] + + def worker(i: int) -> None: + try: + barrier.wait() + + def mutate(rows): + next_rows = list(rows) + next_rows.append( + { + "name": f"Candidate {i}", + "name_key": f"candidate {i}", + } + ) + return next_rows + + locked_modify_candidates(mutate) + except BaseException as exc: # pragma: no cover - assertion surface + exceptions.append(exc) + + threads = [threading.Thread(target=worker, args=(i,)) for i in range(4)] + for thread in threads: + thread.start() + for thread in threads: + thread.join() + + assert exceptions == [] + rows = load_candidates() + assert sorted(row["name_key"] for row in rows) == [ + "candidate 0", + "candidate 1", + "candidate 2", + "candidate 3", + ] + + +def test_record_facet_candidate_creates_one_row(candidate_journal, monkeypatch): + monkeypatch.setattr(mod, "utc_now_iso", lambda: "2026-06-02T17:30:00Z") + samples = [{"day": "20260602", "stream": "archon", "segment": "090000_300"}] + + row = record_facet_candidate( + "Home Reno", + "home reno", + 3, + 14, + samples, + "20260602", + ) + + rows = load_candidates() + assert rows == [row] + assert row["name"] == "Home Reno" + assert row["name_key"] == "home reno" + assert row["status"] == "open" + assert row["count"] == 3 + assert row["window_days"] == 14 + assert row["evidence"] == {"samples": samples} + assert row["first_surfaced"] == "20260602" + assert row["last_surfaced"] == "20260602" + assert row["created_at"] == "2026-06-02T17:30:00Z" + assert row["updated_at"] == "2026-06-02T17:30:00Z" + + +def test_record_facet_candidate_upserts_idempotently(candidate_journal): + first_samples = [{"day": "20260602", "stream": "archon", "segment": "090000_300"}] + second_samples = [{"day": "20260603", "stream": "archon", "segment": "100000_300"}] + + record_facet_candidate("Home Reno", "home reno", 3, 14, first_samples, "20260602") + record_facet_candidate("home reno", "home reno", 4, 14, second_samples, "20260603") + + rows = load_candidates() + assert len(rows) == 1 + assert rows[0]["name"] == "Home Reno" + assert rows[0]["count"] == 4 + assert rows[0]["evidence"]["samples"] == second_samples + + +def test_record_facet_candidate_update_refreshes_expected_fields( + candidate_journal, monkeypatch +): + times = iter(["2026-06-02T17:30:00Z", "2026-06-03T17:30:00Z"]) + monkeypatch.setattr(mod, "utc_now_iso", lambda: next(times)) + first_samples = [{"day": "20260602", "stream": "archon", "segment": "090000_300"}] + second_samples = [ + {"day": "20260603", "stream": "archon", "segment": "100000_300"}, + {"day": "20260603", "stream": "archon", "segment": "103000_300"}, + ] + + record_facet_candidate("Home Reno", "home reno", 3, 14, first_samples, "20260602") + row = record_facet_candidate( + "home reno", + "home reno", + 5, + 21, + second_samples, + "20260603", + ) + + assert row["name"] == "Home Reno" + assert row["name_key"] == "home reno" + assert row["count"] == 5 + assert row["window_days"] == 21 + assert row["evidence"]["samples"] == second_samples + assert row["first_surfaced"] == "20260602" + assert row["last_surfaced"] == "20260603" + assert row["created_at"] == "2026-06-02T17:30:00Z" + assert row["updated_at"] == "2026-06-03T17:30:00Z" + + +def test_record_facet_candidate_preserves_status_and_unknown_keys( + candidate_journal, monkeypatch +): + monkeypatch.setattr(mod, "utc_now_iso", lambda: "2026-06-03T17:30:00Z") + old_samples = [{"day": "20260602", "stream": "archon", "segment": "090000_300"}] + new_samples = [{"day": "20260603", "stream": "archon", "segment": "100000_300"}] + save_candidates( + [ + { + "name": "Home Reno", + "name_key": "home reno", + "status": "dismissed", + "count": 3, + "window_days": 14, + "evidence": {"samples": old_samples, "review_note": "preserve"}, + "first_surfaced": "20260602", + "last_surfaced": "20260602", + "created_at": "2026-06-02T17:30:00Z", + "updated_at": "2026-06-02T17:30:00Z", + "note": "keep me", + } + ] + ) + + row = record_facet_candidate( + "home reno", + "home reno", + 4, + 14, + new_samples, + "20260603", + ) + + assert row["status"] == "dismissed" + assert row["note"] == "keep me" + assert row["evidence"]["review_note"] == "preserve" + assert row["evidence"]["samples"] == new_samples diff --git a/tests/test_facets.py b/tests/test_facets.py index 37fae1dbf..af3de85ed 100644 --- a/tests/test_facets.py +++ b/tests/test_facets.py @@ -4,16 +4,18 @@ """Tests for think.facets module.""" import json -from datetime import datetime +from datetime import datetime, timedelta from pathlib import Path import pytest from slugify import slugify from solstone.think.facets import ( + FACET_CANDIDATE_WINDOW_DAYS, _format_principal_role, _get_principal_display_name, _rank_entities_by_signal, + aggregate_speculative_facets, ensure_facet, facet_summaries, facet_summary, @@ -583,6 +585,108 @@ def test_get_active_facets_malformed_json(monkeypatch, tmp_path): assert active == {"work"} +_MISSING = object() + + +def _write_segment_sense( + journal: Path, + day: str, + segment: str, + speculative_facet: object = _MISSING, + *, + stream: str = "archon", +) -> None: + talents_dir = journal / "chronicle" / day / stream / segment / "talents" + talents_dir.mkdir(parents=True, exist_ok=True) + payload = {} + if speculative_facet is not _MISSING: + payload["speculative_facet"] = speculative_facet + (talents_dir / "sense.json").write_text(json.dumps(payload), encoding="utf-8") + + +def test_aggregate_speculative_facets_surfaces_above_threshold(monkeypatch, tmp_path): + monkeypatch.setenv("SOLSTONE_JOURNAL", str(tmp_path)) + day = "20260602" + for segment in ("090000_300", "093000_300", "100000_300"): + _write_segment_sense(tmp_path, day, segment, "Home Reno") + + result = aggregate_speculative_facets(days=[day]) + + assert result == [ + { + "name": "Home Reno", + "name_key": "home reno", + "count": 3, + "window_days": FACET_CANDIDATE_WINDOW_DAYS, + "samples": [ + {"day": day, "stream": "archon", "segment": "090000_300"}, + {"day": day, "stream": "archon", "segment": "093000_300"}, + {"day": day, "stream": "archon", "segment": "100000_300"}, + ], + } + ] + + +def test_aggregate_speculative_facets_skips_one_off(monkeypatch, tmp_path): + monkeypatch.setenv("SOLSTONE_JOURNAL", str(tmp_path)) + day = "20260602" + _write_segment_sense(tmp_path, day, "090000_300", "Home Reno") + _write_segment_sense(tmp_path, day, "093000_300", None) + _write_segment_sense(tmp_path, day, "100000_300", None) + + assert aggregate_speculative_facets(days=[day]) == [] + + +def test_aggregate_speculative_facets_skips_invalid_values(monkeypatch, tmp_path): + monkeypatch.setenv("SOLSTONE_JOURNAL", str(tmp_path)) + day = "20260602" + _write_segment_sense(tmp_path, day, "090000_300", None) + _write_segment_sense(tmp_path, day, "093000_300") + _write_segment_sense(tmp_path, day, "100000_300", 123) + _write_segment_sense(tmp_path, day, "103000_300", "") + _write_segment_sense(tmp_path, day, "110000_300", "Home Reno") + + result = aggregate_speculative_facets(days=[day], min_count=1) + + assert len(result) == 1 + assert result[0]["name_key"] == "home reno" + assert result[0]["count"] == 1 + + +def test_aggregate_speculative_facets_uses_rolling_window(monkeypatch, tmp_path): + monkeypatch.setenv("SOLSTONE_JOURNAL", str(tmp_path)) + now = datetime.now() + today = now.strftime("%Y%m%d") + two_days_ago = (now - timedelta(days=2)).strftime("%Y%m%d") + old_day = (now - timedelta(days=FACET_CANDIDATE_WINDOW_DAYS + 5)).strftime("%Y%m%d") + _write_segment_sense(tmp_path, today, "090000_300", "Home Reno") + _write_segment_sense(tmp_path, today, "093000_300", "Home Reno") + _write_segment_sense(tmp_path, two_days_ago, "100000_300", "Home Reno") + _write_segment_sense(tmp_path, old_day, "090000_300", "Home Reno") + _write_segment_sense(tmp_path, old_day, "093000_300", "Home Reno") + + result = aggregate_speculative_facets() + + assert len(result) == 1 + assert result[0]["name_key"] == "home reno" + assert result[0]["count"] == 3 + + +def test_aggregate_speculative_facets_groups_case_and_whitespace(monkeypatch, tmp_path): + monkeypatch.setenv("SOLSTONE_JOURNAL", str(tmp_path)) + day = "20260602" + _write_segment_sense(tmp_path, day, "090000_300", "Home Reno") + _write_segment_sense(tmp_path, day, "093000_300", " home reno ") + _write_segment_sense(tmp_path, day, "100000_300", "HOME RENO") + + result = aggregate_speculative_facets(days=[day]) + + assert len(result) == 1 + assert result[0]["name"] == "Home Reno" + assert result[0]["name_key"] == "home reno" + assert result[0]["count"] == 3 + + # ============================================================================ # Principal role in facet summaries tests # ============================================================================ diff --git a/tests/test_scheduler.py b/tests/test_scheduler.py index 963b26639..c84d6043f 100644 --- a/tests/test_scheduler.py +++ b/tests/test_scheduler.py @@ -1265,6 +1265,26 @@ class TestHeartbeatSchedule: } assert mod._entries["providers"]["max_runtime"] == 300 + def test_register_defaults_creates_facet_candidates(self, journal_path): + """register_defaults() creates a facet candidates entry.""" + import solstone.think.scheduler as mod + + mock_cal = Mock() + mod.init(mock_cal) + mod.register_defaults() + + config_path = journal_path / "config" / "schedules.json" + with open(config_path) as f: + raw = json.load(f) + + assert raw["facet-candidates"] == { + "cmd": ["journal", "facet-candidates"], + "every": "weekly", + "enabled": True, + "max_runtime": "10m", + } + assert mod._entries["facet-candidates"]["max_runtime"] == 600 + def test_register_defaults_idempotent(self, journal_path): """register_defaults() does not overwrite existing heartbeat config.""" import solstone.think.scheduler as mod