From 7501c5a29b17a2296cbeabfd92e57db08b9038da Mon Sep 17 00:00:00 2001 From: Jer Miller Date: Sat, 18 Apr 2026 00:29:43 -0600 Subject: [PATCH] feat(talents): add participation talent with structured role/source per span MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Introduce a per-span participation talent that produces structured participation records (role, source, confidence, context, entity_id) for each activity, consolidating per-segment Sense entity drafts. Three coordinated changes: 1. Extend talent/sense.md entities schema with role (attendee|mentioned) and source (voice|speaker_label|transcript|screen|other), plus a prompt-level contamination guard for screen-only names and tool/app UI. 2. Add talent/participation.md (quality-judge, activity-scheduled, tier 3) and talent/participation.py post-hook. The hook parses the LLM JSON, resolves entities read-only via think.entities.matching.find_matching_entity, and atomically merges participation + participation_confidence onto the activity record. 3. Generalize think/activities.update_record_description into update_record_fields, preserving the atomic tempfile+rename write. update_record_description becomes a thin wrapper. Sense drafts reach participation via think/sense_splitter.py, which now also writes a sense.md summary alongside sense.json. This was the only viable load path because think/cluster.py only discovers talents/**/*.md — sense.json is invisible to the existing load mechanism. The alternative (widening cluster) was rejected as out of scope. active_entities on activity records is preserved unchanged for backward compatibility. The state-machine flatten in think/activity_state_machine.py is untouched. The entity resolver is read-only: it calls find_matching_entity against already-loaded entities and injects entity_id (or null) per participation entry. It never writes under facets/{facet}/entities/. Tests cover: Sense schema, contamination guard prose, participation frontmatter/prompt, resolver file-count invariance, contamination regression at the plumbing layer, and atomic activity-record merge including malformed-JSON early return. Co-authored-by: Codex --- talent/participation.md | 59 ++++++++ talent/participation.py | 70 ++++++++++ talent/sense.md | 15 +- tests/baselines/api/settings/providers.json | 8 ++ tests/baselines/api/sol/talents-day.json | 11 ++ tests/baselines/api/stats/stats.json | 26 ++++ tests/test_activity_record_merge.py | 130 ++++++++++++++++++ ..._participation_contamination_regression.py | 52 +++++++ tests/test_participation_resolver.py | 95 +++++++++++++ tests/test_participation_talent.py | 42 ++++++ tests/test_sense_contamination_guard.py | 36 +++++ tests/test_sense_schema.py | 54 ++++++++ tests/test_sense_splitter.py | 50 ++++++- think/activities.py | 17 ++- think/sense_splitter.py | 15 ++ 15 files changed, 673 insertions(+), 7 deletions(-) create mode 100644 talent/participation.md create mode 100644 talent/participation.py create mode 100644 tests/test_activity_record_merge.py create mode 100644 tests/test_participation_contamination_regression.py create mode 100644 tests/test_participation_resolver.py create mode 100644 tests/test_participation_talent.py create mode 100644 tests/test_sense_contamination_guard.py create mode 100644 tests/test_sense_schema.py diff --git a/talent/participation.md b/talent/participation.md new file mode 100644 index 000000000..1667ceddc --- /dev/null +++ b/talent/participation.md @@ -0,0 +1,59 @@ +{ + "type": "generate", + "title": "Participation", + "description": "Consolidates per-segment Sense entity drafts into a structured per-activity participation list.", + "hook": {"post": "participation"}, + "schedule": "activity", + "activities": ["*"], + "priority": 10, + "tier": 3, + "output": "json", + "load": { + "transcripts": true, + "percepts": true, + "talents": { + "sense": true + } + } +} + +$facets + +$activity_context + +$activity_preamble + +# Participation Consolidation + +You are a quality-judge consolidating per-segment Sense drafts of entities into a single per-activity `participation` list. The loaded `sense.md` snippets provide entity candidates gathered across the activity span. Deduplicate name variants, preserve the strongest role/source signal, and keep only grounded entities that genuinely participated in or were mentioned during this activity. + +## Output Schema + +```json +{ + "participation": [ + { + "name": "Full Name", + "role": "attendee|mentioned", + "source": "voice|speaker_label|transcript|screen|other", + "confidence": 0.0, + "context": "Short explanation of why this entity belongs in the activity", + "entity_id": null + } + ], + "participation_confidence": 0.0 +} +``` + +`entity_id` must always be `null`; the post-hook resolves it after generation. + +## Rules + +1. Exclude the journal owner. +2. Never mark someone `role: attendee` in a non-meeting activity. +3. No fabrication — if you didn't see them, don't list them. +4. Empty `participation: []` when no entities were involved. +5. Confidence is subjective but should reflect signal strength (`voice` > `speaker_label` > `transcript` > `screen`). +6. Dedupe variants (e.g., "JB" and "John B." → one entry with the richer name). + +Return only the JSON object with `participation` and optional `participation_confidence`. diff --git a/talent/participation.py b/talent/participation.py new file mode 100644 index 000000000..9da8f5b46 --- /dev/null +++ b/talent/participation.py @@ -0,0 +1,70 @@ +# SPDX-License-Identifier: AGPL-3.0-only +# Copyright (c) 2026 sol pbc + +"""Hook for merging participation data onto activity records.""" + +import json +import logging + +from think.activities import update_record_fields +from think.entities.loading import load_entities +from think.entities.matching import find_matching_entity + +logger = logging.getLogger(__name__) + + +def post_process(result: str, context: dict) -> str | None: + """Resolve participation entries and merge them onto an activity record.""" + try: + data = json.loads(result.strip()) + except (json.JSONDecodeError, ValueError) as exc: + logger.warning("participation hook: failed to parse JSON: %s", exc) + return None + + if not isinstance(data, dict): + logger.warning("participation hook: expected top-level object") + return None + + activity = context.get("activity") + if not isinstance(activity, dict): + logger.warning("participation hook: missing activity context") + return None + + record_id = activity.get("id") + if not record_id: + logger.warning("participation hook: missing activity record id") + return None + + facet = context.get("facet") + day = context.get("day") + if not facet or not day: + logger.warning("participation hook: missing facet/day context") + return None + + participation = data.get("participation") + if not isinstance(participation, list): + logger.warning("participation hook: missing participation list") + return None + + entities_list = load_entities(facet=facet, day=day) + + resolved_entries = [] + for entry in participation: + if not isinstance(entry, dict): + logger.warning("participation hook: skipping non-object entry") + continue + + resolved_entry = dict(entry) + match = find_matching_entity(resolved_entry.get("name", ""), entities_list) + resolved_entry["entity_id"] = match.get("id") if match else None + resolved_entries.append(resolved_entry) + + payload = {"participation": resolved_entries} + participation_confidence = data.get("participation_confidence") + if participation_confidence is not None: + payload["participation_confidence"] = participation_confidence + + if not update_record_fields(facet, day, record_id, payload): + logger.warning("participation hook: activity record not found: %s", record_id) + + return None diff --git a/talent/sense.md b/talent/sense.md index e69c2b244..eb199647a 100644 --- a/talent/sense.md +++ b/talent/sense.md @@ -33,7 +33,7 @@ Read the transcript and screen data. Produce a JSON object with ALL of the follo "content_type": "meeting|coding|browsing|email|messaging|reading|idle|mixed", "activity_summary": "1-3 sentence description of what happened", "entities": [ - {"type": "Person|Company|Project|Tool", "name": "Full Name", "context": "Why this entity matters in this segment"} + {"type": "Person|Company|Project|Tool", "name": "Full Name", "role": "attendee|mentioned", "source": "voice|speaker_label|transcript|screen|other", "context": "Why this entity matters in this segment"} ], "facets": [ {"facet": "facet_id", "activity": "1-sentence description for this facet", "level": "high|medium|low"} @@ -82,6 +82,19 @@ Extract ALL named entities mentioned in the content. Be thorough — extract eve Skip URLs, domains, filenames, paths. Each entity needs type, name, and context (brief description of the entity's role in this segment). +#### role +- **attendee**: The entity was directly participating in the live interaction during this segment. Use only for people who were actively present in the meeting or call. +- **mentioned**: The entity was referenced, quoted, shown on screen, or otherwise relevant, but was not directly participating. + +Contamination guard: tool or product names visible on screen must be `source: screen` and `role: mentioned`, never `attendee`. Video-conference app names such as Google Meet or Zoom are platform/tool entities, not attendees. People quoted or referenced in transcripts are `role: mentioned` unless they were actively speaking as participants in the live meeting. + +#### source +- **voice**: Use when the entity is identified from spoken audio content. +- **speaker_label**: Use when the entity comes from an explicit speaker/participant label in meeting UI or transcript metadata. +- **transcript**: Use when the entity appears in transcript text but not as an actively speaking participant signal. +- **screen**: Use when the entity is visible in screen content such as UI, documents, headlines, or app chrome. +- **other**: Use only when the entity is grounded in another clear signal that does not fit the categories above. + ### facets Classify into the owner's configured facets. Only include facets with clear, direct evidence of activity. Be precise — assign exactly ONE primary facet in most cases. Only add a second facet if there is genuinely distinct secondary activity. For each: - `facet`: The facet ID slug — MUST be one of the configured facets listed in the input diff --git a/tests/baselines/api/settings/providers.json b/tests/baselines/api/settings/providers.json index 9ea7a764c..7f6f23700 100644 --- a/tests/baselines/api/settings/providers.json +++ b/tests/baselines/api/settings/providers.json @@ -335,6 +335,14 @@ "tier": 2, "type": null }, + "talent.system.participation": { + "disabled": false, + "group": "Think", + "label": "Participation", + "schedule": "activity", + "tier": 3, + "type": "generate" + }, "talent.system.partner": { "disabled": false, "group": "Think", diff --git a/tests/baselines/api/sol/talents-day.json b/tests/baselines/api/sol/talents-day.json index d2b64df7b..8b1ca006e 100644 --- a/tests/baselines/api/sol/talents-day.json +++ b/tests/baselines/api/sol/talents-day.json @@ -319,6 +319,17 @@ "title": "Occurrence Extraction", "type": null }, + "participation": { + "app": null, + "color": "#6c757d", + "description": "Consolidates per-segment Sense entity drafts into a structured per-activity participation list.", + "multi_facet": false, + "output_format": "json", + "schedule": "activity", + "source": "system", + "title": "Participation", + "type": "generate" + }, "partner": { "app": null, "color": "#6c757d", diff --git a/tests/baselines/api/stats/stats.json b/tests/baselines/api/stats/stats.json index fb04925a5..2ed5280ad 100644 --- a/tests/baselines/api/stats/stats.json +++ b/tests/baselines/api/stats/stats.json @@ -250,6 +250,32 @@ "title": "Messaging Summary", "type": "generate" }, + "participation": { + "activities": [ + "*" + ], + "color": "#6c757d", + "description": "Consolidates per-segment Sense entity drafts into a structured per-activity participation list.", + "hook": { + "post": "participation" + }, + "load": { + "percepts": true, + "talents": { + "sense": true + }, + "transcripts": true + }, + "mtime": 0, + "output": "json", + "path": "/talent/participation.md", + "priority": 10, + "schedule": "activity", + "source": "system", + "tier": 3, + "title": "Participation", + "type": "generate" + }, "schedule": { "color": "#5e35b1", "description": "Identifies all future calendar events and scheduled activities noted in transcripts. Extracts dates, times, participants, and event details for anything scheduled beyond today.", diff --git a/tests/test_activity_record_merge.py b/tests/test_activity_record_merge.py new file mode 100644 index 000000000..5b6bd62de --- /dev/null +++ b/tests/test_activity_record_merge.py @@ -0,0 +1,130 @@ +# SPDX-License-Identifier: AGPL-3.0-only +# Copyright (c) 2026 sol pbc + +import json + + +def _write_detected_entities(tmp_path, facet: str, day: str, rows: list[dict]) -> None: + entities_path = tmp_path / "facets" / facet / "entities" / f"{day}.jsonl" + entities_path.parent.mkdir(parents=True, exist_ok=True) + entities_path.write_text( + "".join(json.dumps(row, ensure_ascii=False) + "\n" for row in rows), + encoding="utf-8", + ) + + +def _activity_record(): + return { + "id": "meeting_090000_300", + "activity": "meeting", + "segments": ["090000_300"], + "level_avg": 1.0, + "description": "Team sync", + "active_entities": ["JB", "Alex"], + "created_at": 1, + } + + +def test_participation_post_hook_merges_fields_and_preserves_active_entities( + tmp_path, monkeypatch +): + from talent.participation import post_process + from think.activities import append_activity_record, load_activity_records + + facet = "work" + day = "20260418" + monkeypatch.setenv("_SOLSTONE_JOURNAL_OVERRIDE", str(tmp_path)) + + _write_detected_entities( + tmp_path, + facet, + day, + [ + { + "id": "john_borthwick", + "type": "Person", + "name": "John Borthwick", + "aka": ["JB"], + } + ], + ) + append_activity_record(facet, day, _activity_record()) + + post_process( + json.dumps( + { + "participation": [ + { + "name": "JB", + "role": "attendee", + "source": "voice", + "confidence": 0.91, + "context": "Spoke during the meeting", + "entity_id": None, + }, + { + "name": "Alex", + "role": "mentioned", + "source": "transcript", + "confidence": 0.42, + "context": "Mentioned as a collaborator", + "entity_id": None, + }, + ], + "participation_confidence": 0.77, + } + ), + {"facet": facet, "day": day, "activity": {"id": "meeting_090000_300"}}, + ) + + record = load_activity_records(facet, day)[0] + assert record["active_entities"] == ["JB", "Alex"] + assert record["participation_confidence"] == 0.77 + assert record["participation"][0]["entity_id"] == "john_borthwick" + assert record["participation"][1]["entity_id"] is None + + +def test_participation_post_hook_leaves_file_unchanged_on_malformed_json( + tmp_path, monkeypatch, caplog +): + from talent.participation import post_process + from think.activities import append_activity_record + + facet = "work" + day = "20260418" + monkeypatch.setenv("_SOLSTONE_JOURNAL_OVERRIDE", str(tmp_path)) + + append_activity_record(facet, day, _activity_record()) + record_path = tmp_path / "facets" / facet / "activities" / f"{day}.jsonl" + before = record_path.read_bytes() + + post_process( + "{not valid json", + {"facet": facet, "day": day, "activity": {"id": "meeting_090000_300"}}, + ) + + assert record_path.read_bytes() == before + assert "failed to parse JSON" in caplog.text + + +def test_participation_post_hook_requires_activity_context( + tmp_path, monkeypatch, caplog +): + from talent.participation import post_process + from think.activities import append_activity_record + + facet = "work" + day = "20260418" + monkeypatch.setenv("_SOLSTONE_JOURNAL_OVERRIDE", str(tmp_path)) + + append_activity_record(facet, day, _activity_record()) + record_path = tmp_path / "facets" / facet / "activities" / f"{day}.jsonl" + before = record_path.read_bytes() + + post_process( + json.dumps({"participation": []}), + {"facet": facet, "day": day}, + ) + + assert record_path.read_bytes() == before + assert "missing activity context" in caplog.text diff --git a/tests/test_participation_contamination_regression.py b/tests/test_participation_contamination_regression.py new file mode 100644 index 000000000..78e95b3dd --- /dev/null +++ b/tests/test_participation_contamination_regression.py @@ -0,0 +1,52 @@ +# SPDX-License-Identifier: AGPL-3.0-only +# Copyright (c) 2026 sol pbc + +from pathlib import Path + +import pytest + + +@pytest.mark.parametrize( + ("entity_name", "entity_type", "source"), + [ + ("Claude Code", "Tool", "screen"), + ("Mozilla Ventures", "Company", "transcript"), + ("Google Meet", "Tool", "screen"), + ], +) +def test_sense_entity_role_source_survive_splitter_without_changing_active_entity_flatten( + tmp_path, entity_name: str, entity_type: str, source: str +): + from think.activity_state_machine import ActivityStateMachine + from think.sense_splitter import write_sense_outputs + + # This is a schema/plumbing regression test, not an LLM-behavioral test. + day = "20260418" + segment = "090000_300" + sense_json = { + "density": "active", + "content_type": "coding", + "activity_summary": "Worked through captured activity.", + "entities": [ + { + "type": entity_type, + "name": entity_name, + "role": "mentioned", + "source": source, + "context": "Detected by fixture-driven test input", + } + ], + "facets": [{"facet": "work", "activity": "coding", "level": "high"}], + "meeting_detected": False, + "speakers": [], + "recommend": {}, + } + + seg_dir = Path(tmp_path) / day / "default" / segment + write_sense_outputs(sense_json, seg_dir) + + sense_md = (seg_dir / "talents" / "sense.md").read_text(encoding="utf-8") + assert f"{entity_name} (role=mentioned, source={source})" in sense_md + + changes = ActivityStateMachine().update(sense_json, segment, day) + assert changes[0]["active_entities"] == [entity_name] diff --git a/tests/test_participation_resolver.py b/tests/test_participation_resolver.py new file mode 100644 index 000000000..eb0be1764 --- /dev/null +++ b/tests/test_participation_resolver.py @@ -0,0 +1,95 @@ +# SPDX-License-Identifier: AGPL-3.0-only +# Copyright (c) 2026 sol pbc + +import json + + +def _write_detected_entities(tmp_path, facet: str, day: str, rows: list[dict]) -> None: + entities_path = tmp_path / "facets" / facet / "entities" / f"{day}.jsonl" + entities_path.parent.mkdir(parents=True, exist_ok=True) + entities_path.write_text( + "".join(json.dumps(row, ensure_ascii=False) + "\n" for row in rows), + encoding="utf-8", + ) + + +def test_participation_post_hook_resolves_entity_ids_without_mutating_entities( + tmp_path, monkeypatch +): + from talent.participation import post_process + from think.activities import append_activity_record, load_activity_records + + facet = "work" + day = "20260418" + monkeypatch.setenv("_SOLSTONE_JOURNAL_OVERRIDE", str(tmp_path)) + + _write_detected_entities( + tmp_path, + facet, + day, + [ + { + "id": "john_borthwick", + "type": "Person", + "name": "John Borthwick", + "aka": ["JB"], + }, + { + "id": "other_person", + "type": "Person", + "name": "Other Person", + }, + ], + ) + + entities_dir = tmp_path / "facets" / facet / "entities" + snapshot_before = {p.name: p.stat().st_size for p in entities_dir.iterdir()} + + append_activity_record( + facet, + day, + { + "id": "meeting_090000_300", + "activity": "meeting", + "segments": ["090000_300"], + "level_avg": 1.0, + "description": "Team sync", + "active_entities": ["JB", "Alex"], + "created_at": 1, + }, + ) + + result = json.dumps( + { + "participation": [ + { + "name": "JB", + "role": "attendee", + "source": "voice", + "confidence": 0.98, + "context": "Spoke during the meeting", + "entity_id": "fake_id", + }, + { + "name": "Alex", + "role": "mentioned", + "source": "transcript", + "confidence": 0.55, + "context": "Mentioned as a follow-up owner", + "entity_id": "fake_id", + }, + ] + } + ) + + post_process( + result, + {"facet": facet, "day": day, "activity": {"id": "meeting_090000_300"}}, + ) + + snapshot_after = {p.name: p.stat().st_size for p in entities_dir.iterdir()} + assert snapshot_after == snapshot_before + + record = load_activity_records(facet, day)[0] + assert record["participation"][0]["entity_id"] == "john_borthwick" + assert record["participation"][1]["entity_id"] is None diff --git a/tests/test_participation_talent.py b/tests/test_participation_talent.py new file mode 100644 index 000000000..728e41767 --- /dev/null +++ b/tests/test_participation_talent.py @@ -0,0 +1,42 @@ +# SPDX-License-Identifier: AGPL-3.0-only +# Copyright (c) 2026 sol pbc + +from pathlib import Path + +import frontmatter + +PARTICIPATION_PATH = Path(__file__).resolve().parents[1] / "talent" / "participation.md" + + +def test_participation_talent_frontmatter_and_placeholders(): + post = frontmatter.load(PARTICIPATION_PATH) + + assert post.metadata["schedule"] == "activity" + assert post.metadata["activities"] == ["*"] + assert post.metadata["tier"] == 3 + assert post.metadata["output"] == "json" + assert post.metadata["priority"] == 10 + assert post.metadata["load"]["talents"]["sense"] is True + + body = post.content + assert "$facets" in body + assert "$activity_context" in body + assert "$activity_preamble" in body + + +def test_participation_talent_rules_block_is_present(): + body = frontmatter.load(PARTICIPATION_PATH).content + + expected_rules = [ + "1. Exclude the journal owner.", + "2. Never mark someone `role: attendee` in a non-meeting activity.", + "3. No fabrication — if you didn't see them, don't list them.", + "4. Empty `participation: []` when no entities were involved.", + "5. Confidence is subjective but should reflect signal strength (`voice` > `speaker_label` > `transcript` > `screen`).", + '6. Dedupe variants (e.g., "JB" and "John B." → one entry with the richer name).', + ] + + for rule in expected_rules: + assert rule in body + + assert "`entity_id` must always be `null`" in body diff --git a/tests/test_sense_contamination_guard.py b/tests/test_sense_contamination_guard.py new file mode 100644 index 000000000..78a3b5a0b --- /dev/null +++ b/tests/test_sense_contamination_guard.py @@ -0,0 +1,36 @@ +# SPDX-License-Identifier: AGPL-3.0-only +# Copyright (c) 2026 sol pbc + +from pathlib import Path + +import frontmatter + +SENSE_PATH = Path(__file__).resolve().parents[1] / "talent" / "sense.md" + + +def _role_section() -> str: + content = frontmatter.load(SENSE_PATH).content + start = content.index("#### role") + end = content.index("#### source", start) + return content[start:end] + + +def test_sense_role_section_contains_contamination_guard(): + role_section = _role_section() + + assert "tool or product names visible on screen" in role_section + assert "`source: screen`" in role_section + assert "`role: mentioned`" in role_section + assert "Google Meet" in role_section + assert "Zoom" in role_section + assert "quoted or referenced in transcripts" in role_section + assert "actively speaking as participants" in role_section + + +def test_sense_role_section_has_screen_and_mentioned_guidance_for_tools_and_apps(): + role_section = _role_section() + + assert "screen" in role_section + assert "mentioned" in role_section + assert "tool" in role_section + assert "Video-conference app names" in role_section diff --git a/tests/test_sense_schema.py b/tests/test_sense_schema.py new file mode 100644 index 000000000..ad2ec2fe2 --- /dev/null +++ b/tests/test_sense_schema.py @@ -0,0 +1,54 @@ +# SPDX-License-Identifier: AGPL-3.0-only +# Copyright (c) 2026 sol pbc + +from pathlib import Path + +import frontmatter + +SENSE_PATH = Path(__file__).resolve().parents[1] / "talent" / "sense.md" + + +def _section(text: str, start: str, end: str | None = None) -> str: + section_start = text.index(start) + if end is None: + return text[section_start:] + section_end = text.index(end, section_start) + return text[section_start:section_end] + + +def test_sense_prompt_parses_and_documents_role_and_source(): + post = frontmatter.load(SENSE_PATH) + + assert post.metadata["tier"] == 3 + + schema = _section( + post.content, "## Output Schema", "## Field-by-Field Instructions" + ) + entities = _section(post.content, "### entities", "### facets") + + assert '"role": "attendee|mentioned"' in schema + assert '"source": "voice|speaker_label|transcript|screen|other"' in schema + assert "#### role" in entities + assert "#### source" in entities + + +def test_role_and_source_do_not_leak_into_other_sense_sections(): + content = frontmatter.load(SENSE_PATH).content + + sections = [ + _section(content, "### density", "### content_type"), + _section(content, "### content_type", "### activity_summary"), + _section(content, "### activity_summary", "### entities"), + _section(content, "### facets", "### meeting_detected"), + _section(content, "### meeting_detected", "### speakers"), + _section(content, "### speakers", "### recommend"), + _section(content, "### recommend", "### emotional_register"), + _section(content, "### emotional_register", "## Rules"), + _section(content, "## Rules"), + ] + + for section in sections: + assert "attendee|mentioned" not in section + assert "voice|speaker_label|transcript|screen|other" not in section + assert "#### role" not in section + assert "#### source" not in section diff --git a/tests/test_sense_splitter.py b/tests/test_sense_splitter.py index d21e9166b..be59a1965 100644 --- a/tests/test_sense_splitter.py +++ b/tests/test_sense_splitter.py @@ -14,7 +14,13 @@ def _make_sense_output(**overrides): "content_type": "coding", "activity_summary": "Writing unit tests for the API module.", "entities": [ - {"type": "Project", "name": "SolAPI", "context": "main project"}, + { + "type": "Project", + "name": "SolAPI", + "role": "mentioned", + "source": "screen", + "context": "main project", + }, ], "facets": [ {"facet": "work", "activity": "coding", "level": "high"}, @@ -81,6 +87,48 @@ class TestWriteSenseOutputs: assert stored["foo"] == "bar" assert stored == sense_json + def test_writes_sense_markdown_when_entities_exist(self, tmp_path): + from think.sense_splitter import write_sense_outputs + + seg_dir = Path(tmp_path) / "20260304" / "default" / "090000_300" + sense_json = _make_sense_output( + entities=[ + { + "type": "Project", + "name": "SolAPI", + "role": "mentioned", + "source": "screen", + "context": "main project", + }, + { + "type": "Person", + "name": "John Borthwick", + "role": "attendee", + "source": "voice", + "context": "active meeting participant", + }, + ] + ) + + write_sense_outputs(sense_json, seg_dir) + + sense_md = (seg_dir / "talents" / "sense.md").read_text(encoding="utf-8") + assert sense_md == ( + "# Sense Entities\n\n" + "- Project — SolAPI (role=mentioned, source=screen) — main project\n" + "- Person — John Borthwick (role=attendee, source=voice) " + "— active meeting participant" + ) + + def test_skips_sense_markdown_when_entities_empty(self, tmp_path): + from think.sense_splitter import write_sense_outputs + + seg_dir = Path(tmp_path) / "20260304" / "default" / "090000_300" + + write_sense_outputs(_make_sense_output(entities=[]), seg_dir) + + assert not (seg_dir / "talents" / "sense.md").exists() + class TestMeetingDetection: def test_writes_speakers_when_meeting_detected(self, tmp_path): diff --git a/think/activities.py b/think/activities.py index 1c34fc44c..090b840e8 100644 --- a/think/activities.py +++ b/think/activities.py @@ -776,13 +776,13 @@ def append_activity_record( return True -def update_record_description( - facet: str, day: str, record_id: str, description: str +def update_record_fields( + facet: str, day: str, record_id: str, fields: dict[str, Any] ) -> bool: - """Update the description of an existing activity record. + """Update fields on an existing activity record. Rewrites the JSONL file atomically (write temp + rename) with the updated - description for the matching record. + fields for the matching record. Returns True if record was found and updated, False otherwise. """ @@ -806,7 +806,7 @@ def update_record_description( continue if record.get("id") == record_id: - record["description"] = description + record.update(fields) updated = True new_lines.append(json.dumps(record, ensure_ascii=False)) @@ -825,6 +825,13 @@ def update_record_description( return updated +def update_record_description( + facet: str, day: str, record_id: str, description: str +) -> bool: + """Update the description of an existing activity record.""" + return update_record_fields(facet, day, record_id, {"description": description}) + + def estimate_duration_minutes(segments: list[str]) -> int: """Estimate total duration in minutes from a list of segment keys. diff --git a/think/sense_splitter.py b/think/sense_splitter.py index 83ec3e781..cd05638cc 100644 --- a/think/sense_splitter.py +++ b/think/sense_splitter.py @@ -32,6 +32,7 @@ def write_sense_outputs( density = sense_json.get("density") or "active" activity_summary = sense_json.get("activity_summary") or "" + entities = sense_json.get("entities") or [] facets = sense_json.get("facets") or [] meeting_detected = bool(sense_json.get("meeting_detected")) speakers = sense_json.get("speakers") or [] @@ -49,6 +50,20 @@ def write_sense_outputs( ) _write_json_atomic(agents_dir / "sense.json", sense_json) + if entities: + lines = ["# Sense Entities", ""] + for entity in entities: + if not isinstance(entity, dict): + continue + lines.append( + "- " + f"{entity.get('type', '')} — {entity.get('name', '')} " + f"(role={entity.get('role', '')}, source={entity.get('source', '')}) " + f"— {entity.get('context', '')}" + ) + if len(lines) > 2: + _write_text_atomic(agents_dir / "sense.md", "\n".join(lines)) + if meeting_detected: _write_json_atomic(agents_dir / "speakers.json", speakers) -- 2.51.2