diff --git a/docs/design/story-talent-refactor.md b/docs/design/story-talent-refactor.md new file mode 100644 index 000000000..110ad81a6 --- /dev/null +++ b/docs/design/story-talent-refactor.md @@ -0,0 +1,265 @@ +# Story-Talent Refactor +This refactor replaces the old storyteller span-row write path with activity-record +story merges. It is a clean break: +- storytellers stop writing `facets/*/spans/*.jsonl` +- story content lives on the activity record itself +- `think/activities.py` remains the only writer for activity records +- priority ordering, not extra locking, serializes participation before story +## 1. New writer: `merge_story_fields` +Location: +- add `merge_story_fields(...)` in `think/activities.py` +- place it next to `update_record_fields()` and `update_activity_record()` +- this is the only activity-record write added by the refactor, so L2 stays + satisfied inside the domain owner +Signature and docstring: +```python +def merge_story_fields( + facet: str, + day: str, + record_id: str, + *, + story: dict, + commitments: list[dict], + closures: list[dict], + decisions: list[dict], + actor: str, + note: str | None = None, +) -> bool: + """Replace story-derived fields on an activity record and append one edit.""" +``` +Behavior and return semantics: +- use one `locked_modify(...)` call only, following the same pattern as + `update_activity_record()` and `_set_activity_hidden_state()` + (`think/activities.py:762-790`, `1046-1089`, `1092-1133`) +- inside the callback, find the record with the same `record.get("id") == record_id` + match used by `update_activity_record()` +- if found: + - normalize the current record + - replace `story`, `commitments`, `closures`, and `decisions` wholesale + - call `append_edit(...)` exactly once with + `fields=["story", "commitments", "closures", "decisions"]` + - pass through `actor` + - pass through `note` + - return `True` +- if the day file is missing or the record is absent: + - log a warning + - return `False` + - do not raise +Why this shape: +- `update_activity_record()` is intentionally narrow and only allows + `title`, `description`, and `details` (`think/activities.py:1046-1062`) +- the CLI mirrors that exact scope + (`apps/activities/call.py:503-532`, + `apps/activities/talent/activities/SKILL.md:116-130`) +- `_set_activity_hidden_state()` already establishes the specialized-writer + pattern in this module (`think/activities.py:1092-1133`) +- `update_record_fields()` stays the generic no-edit helper used by participation + (`think/activities.py:979-1011`, `talent/participation.py:99-105`) +## 2. `story.py` `post_process` flow +The new hook lives at `talent/story.py` and always returns `""` so the JSON +generator artifact is suppressed. +Dispatcher context shape, confirmed: +- `run_activity_prompts()` sends `facet`, `day`, `span`, `activity`, and + `output_path` in the activity request (`think/thinking.py:2064-2083`) +- `prepare_config()` merges those request keys into the full talent config and + always carries `name` (`think/talents.py:438-520`) +- `_run_post_hooks()` passes the full prepared config dict directly to the hook + (`think/talents.py:712-734`) +The hook can rely on: +- `context["name"]` +- `context["facet"]` +- `context["day"]` +- `context["activity"]` +- `context["span"]` +- `context["output_path"]` +Execution order: +1. Parse `result` with `json.loads(result.strip())`. + On failure: log and return `""`. +2. Require a top-level `dict`. + Otherwise: log and return `""`. +3. Validate required top-level fields. + - `body`: `str`, non-empty after strip + - `topics`: `list[str]`, may be empty + - `confidence`: numeric in `0.0..1.0` + - `commitments`: `list` + - `closures`: `list` + - `decisions`: `list` + Any missing field, wrong type, or out-of-range `confidence` logs and returns + `""`. +4. Validate required context. + - `context["activity"]` must be a `dict` + - `context["activity"]["id"]` must exist + - `context["facet"]` and `context["day"]` must exist + Missing context logs and returns `""`. +5. Load entities once with: + `load_entities(facet=context["facet"], day=context["day"])`. +6. Validate `commitments` entry by entry. + - each entry must be a `dict` + - required keys: `owner`, `action`, `counterparty`, `when`, `context` + - each required value must be a `str` + - invalid entries are skipped with a per-entry log +7. Validate `closures` entry by entry. + - each entry must be a `dict` + - required keys: `owner`, `action`, `counterparty`, `resolution`, `context` + - each required value must be a `str` + - `resolution` must be one of: + `sent`, `done`, `signed`, `dropped`, `deferred` + - invalid entries are skipped with a per-entry log +8. Validate `decisions` entry by entry. + - each entry must be a `dict` + - required keys: `owner`, `action`, `context` + - each required value must be a `str` + - invalid entries are skipped with a per-entry log +9. Resolve entity ids for every valid entry with + `find_matching_entity(name, entities, fuzzy_threshold=90)`. + - commitments: add `owner_entity_id` and `counterparty_entity_id` + - closures: add `owner_entity_id` and `counterparty_entity_id` + - decisions: add `owner_entity_id` + - unmatched values become `None` + - preserve the original `owner` and `counterparty` strings +10. Build: + `story = {"body": body, "topics": topics, "confidence": confidence}`. +11. Extract: + - `record_id = context["activity"]["id"]` + - `facet = context["facet"]` + - `day = context["day"]` +12. Call: + `merge_story_fields(facet, day, record_id, story=..., commitments=..., closures=..., decisions=..., actor="story", note=None)`. + If it returns `False`: log a warning and return `""`. +13. Return `""`. + This is required because `_execute_with_tools()` only writes the output file + when `result` is truthy (`think/talents.py:837-846`); returning `None` would + fall back to the original JSON result (`think/talents.py:726-734`). +Intentional differences from `talent/spans.py`: +- no spans-file write +- no topic dedupe/normalization +- no confidence clamping +- no fence-stripping carryover unless explicitly added during implementation +## 3. Activity-record formatter extension +Target: +- extend `think/activities.py::format_activities()` +- current order is: + title, activity, facet, day, time, level, description, details, participation, + hidden (`think/activities.py:1271-1307`) +Chosen insertion point: +- add the story block after participation and before hidden +Behavior: +- if `record.get("story")` is not a `dict`, do nothing +- if `story["body"]` is a non-empty string, render it as prose rather than + `- Story: ...` +- if `story["topics"]` is a non-empty list of strings, render one line as + `Topics: a, b, c` +- if `body` is missing/empty, skip the prose block +- if `topics` is missing, non-list, or empty, skip the topics line +- keep all other formatter output unchanged +- keep the existing activity formatter registration; no new registry entry is + needed because activities are already mapped to `format_activities()` + (`think/formatters.py:143-144`) +Why this insertion point is best: +- description/details remain raw activity metadata +- participation remains the structured who-was-involved summary +- story reads naturally after those structured fields +- hidden stays last because it is record state, not content +## 4. Storyteller prompt changes +Common frontmatter changes for all three storyteller talents: +- `priority: 10` -> `priority: 20` +- `hook: {"post": "spans"}` -> `hook: {"post": "story"}` +- keep `schedule: "activity"` +- keep `output: "json"` +- keep existing activity filters per talent +Common schema changes for all three: +- require exactly: + `body`, `topics`, `confidence`, `commitments`, `closures`, `decisions` +- all six fields are required on every response +- `topics` may be `[]` +- `commitments`, `closures`, and `decisions` may be `[]` +- add the explicit instruction: + `Return [] if you do not observe a clear commitment / closure / decision. Better to omit than invent.` +- state the controlled closure `resolution` vocabulary exactly: + `sent`, `done`, `signed`, `dropped`, `deferred` +`talent/conversation.md` +- keep the meeting/call/messaging/email narrative focus +- expand the schema block to the six-field JSON shape +- inline examples: + - commitment: send a follow-up, draft, or deck by a date + - closure: an open item was `sent` or `done` + - decision: the group chose a direction, owner, or timing +- keep the current guidance that brief quotes are allowed when they sharpen a + decision, commitment, or disagreement +`talent/work.md` +- keep the coding/browsing/reading progress focus +- expand the schema block to the six-field JSON shape +- inline examples: + - commitment: ship a patch, benchmark, or send results + - closure: a task was `done` or a review was `sent` + - decision: a code-path, retry strategy, or API choice was made +- keep the instruction to emphasize actual work performed over UI description +`talent/event.md` +- keep the appointment/event/travel/errand outcome focus +- expand the schema block to the six-field JSON shape +- inline examples: + - commitment: a travel or logistics follow-up + - closure: a form was `signed`, a reservation was `done`, or a task was + `deferred` + - decision: a route, plan, or next-step choice was made +- keep the guidance to prefer what actually happened over generic event labels +## 5. Test matrix +| test name | file | pins | +| --- | --- | --- | +| `test_story_hook_parses_and_writes` | `tests/test_story_hook.py` | Valid JSON writes `story`, `commitments`, `closures`, `decisions` onto the activity record and appends one edit with actor `story`. | +| `test_story_hook_empty_arrays` | `tests/test_story_hook.py` | Empty `commitments`/`closures`/`decisions` still persist alongside the story payload. | +| `test_story_hook_bad_resolution_skipped` | `tests/test_story_hook.py` | Invalid closure `resolution` is dropped while valid sibling closures survive. | +| `test_story_hook_missing_required_field_skipped` | `tests/test_story_hook.py` | Missing required per-entry fields skip only the bad item. | +| `test_story_hook_resolves_entities` | `tests/test_story_hook.py` | `owner`/`counterparty` resolve to `*_entity_id` with `fuzzy_threshold=90`; misses become `None`. | +| `test_story_hook_idempotent_rerun` | `tests/test_story_hook.py` | Second run replaces story/list fields wholesale and appends one more edit entry. | +| `test_story_hook_missing_record_logs_and_returns` | `tests/test_story_hook.py` | `merge_story_fields()` returns `False`, hook logs warning, nothing raises. | +| `test_story_hook_no_json_file_written` | `tests/test_story_hook.py` | Returning `""` suppresses the storyteller JSON artifact. | +| `test_format_activities_renders_story` | `tests/test_activities.py` | Story prose and `Topics:` line appear when present and disappear cleanly when absent. | +| `test_no_spans_formatter_registered` | `tests/test_formatters.py` | `"facets/*/spans/*.jsonl"` is removed from `FORMATTERS`; spans paths no longer resolve to a formatter. | +| `test_no_spans_writes` | `tests/test_formatters.py` | Search-style assertion that no `format_spans` or `spans/` write targets remain in `think/`, `talent/`, or `apps/`. | +Existing test templates to reuse: +- `tests/test_activity_record_merge.py` for temp-journal setup, activity seeding, + hook execution, and record reload assertions +- `tests/test_schedule_hook.py` for per-entry skip behavior and entity-resolution + patterns +## 6. Files touched / deleted +Create: +- `docs/design/story-talent-refactor.md` +- `talent/story.py` +- `tests/test_story_hook.py` +Modify: +- `think/activities.py` +- `think/formatters.py` +- `talent/conversation.md` +- `talent/work.md` +- `talent/event.md` +- `tests/test_activities.py` +- `tests/test_formatters.py` +- `tests/baselines/api/stats/stats.json` +- `talent/journal/references/captures.md` +Delete: +- `talent/spans.py` +- `think/spans.py` +- `tests/test_spans_hook.py` +- `tests/test_spans_formatter.py` +Intentionally untouched: +- `think/thinking.py` because priority-group serialization already does the job +- `apps/activities/call.py` because no CLI surface change is needed +## 7. Risks / gotchas +- Preserve the hook-return behavior exactly: `""`, not `None`. + `None` would fall back to the original JSON result and write a generator file + (`think/talents.py:726-734`, `837-846`). +- Preserve the missing-record behavior of `update_record_fields()`: + no raise, but the story path must log the failure like participation does + (`think/activities.py:1007-1011`, `talent/participation.py:104-105`). +- `format_activities()` is already registered for + `"facets/*/activities/*.jsonl"` (`think/formatters.py:144`). + Do not add a new formatter entry for story data. +- Layer hygiene L2 stays strict: + only `think/activities.py` writes the activity record. + `talent/story.py` imports and calls the new writer; it does not perform raw + file I/O. +- Story merges serialize after participation via priority ordering, not new locks. + Keep participation at `10`, storytellers at `20`, and rely on the existing + group ordering/drain in `run_activity_prompts()` + (`think/thinking.py:1925-1928`, `2150-2183`). diff --git a/talent/conversation.md b/talent/conversation.md index d44bbdaeb..5530023f8 100644 --- a/talent/conversation.md +++ b/talent/conversation.md @@ -1,13 +1,13 @@ { "type": "generate", "title": "Conversation Story", - "description": "Writes a structured narrative span row for meeting, call, messaging, and email activities.", + "description": "Generates a conversation story, topics, and structured commitments, closures, and decisions to merge onto the activity record.", "color": "#00796b", "schedule": "activity", "activities": ["meeting", "call", "messaging", "email"], - "priority": 10, + "priority": 20, "output": "json", - "hook": {"post": "spans"}, + "hook": {"post": "story"}, "load": { "transcripts": true, "percepts": true, @@ -29,10 +29,18 @@ Summarize this conversation as one coherent narrative for the full activity. Participation and entity extraction already happened upstream. Reuse that context; do not re-extract people or entities into new structures. -Return exactly these three fields: +Return exactly this six-field JSON object: - `body`: string narrative prose covering what was discussed, what moved, and any commitments. -- `topics`: array of 3-8 short string tags. +- `topics`: array of short string tags; use `[]` when there are no durable topics worth preserving. - `confidence`: float from 0.0 to 1.0. +- `commitments`: array of objects with required string fields `owner`, `action`, `counterparty`, `when`, `context`. + Example: `{"owner":"Mina","action":"send the revised deck","counterparty":"Ravi","when":"Friday morning","context":"Mina committed to send the deck before the investor follow-up."}` +- `closures`: array of objects with required string fields `owner`, `action`, `counterparty`, `resolution`, `context`. `resolution` must be one of `sent`, `done`, `signed`, `dropped`, `deferred`. + Example: `{"owner":"Ravi","action":"intro email","counterparty":"Mina","resolution":"sent","context":"Ravi confirmed the intro email already went out during the call."}` +- `decisions`: array of objects with required string fields `owner`, `action`, `context`. + Example: `{"owner":"Team","action":"schedule the launch review for next Tuesday","context":"The group agreed to move the review to Tuesday after checking calendars."}` + +Return `[]` if you do not observe a clear commitment / closure / decision. Better to omit than invent. Body requirements: - Write one tight paragraph in chronological order. @@ -42,4 +50,4 @@ Body requirements: - If the activity mixes channels, unify them into one narrative rather than listing separate threads. -Output a single JSON object with only `body`, `topics`, and `confidence`. +Output a single JSON object with all six required fields: `body`, `topics`, `confidence`, `commitments`, `closures`, and `decisions`. diff --git a/talent/event.md b/talent/event.md index 362b163f7..f76d51af0 100644 --- a/talent/event.md +++ b/talent/event.md @@ -1,13 +1,13 @@ { "type": "generate", "title": "Event Story", - "description": "Writes a structured narrative span row for appointment, event, travel, errand, celebration, deadline, and reminder activities.", + "description": "Generates an event story, topics, and structured commitments, closures, and decisions to merge onto the activity record.", "color": "#ff7043", "schedule": "activity", "activities": ["appointment", "event", "travel", "errand", "celebration", "deadline", "reminder"], - "priority": 10, + "priority": 20, "output": "json", - "hook": {"post": "spans"}, + "hook": {"post": "story"}, "load": { "transcripts": true, "percepts": true, @@ -30,10 +30,18 @@ deadline-related activity. Participation and entity extraction already happened upstream. Use that context; do not re-extract people or entities into new structures. -Return exactly these three fields: +Return exactly this six-field JSON object: - `body`: string narrative prose describing what happened and any outcome. -- `topics`: array of 3-8 short string tags. +- `topics`: array of short string tags; use `[]` when there are no durable topics worth preserving. - `confidence`: float from 0.0 to 1.0. +- `commitments`: array of objects with required string fields `owner`, `action`, `counterparty`, `when`, `context`. + Example: `{"owner":"Jordan","action":"send the updated itinerary","counterparty":"Taylor","when":"tonight","context":"Jordan said the revised travel plan would be sent after the delay was confirmed."}` +- `closures`: array of objects with required string fields `owner`, `action`, `counterparty`, `resolution`, `context`. `resolution` must be one of `sent`, `done`, `signed`, `dropped`, `deferred`. + Example: `{"owner":"Jordan","action":"hotel confirmation","counterparty":"Taylor","resolution":"signed","context":"Jordan completed and signed the hotel check-in form during the event."}` +- `decisions`: array of objects with required string fields `owner`, `action`, `context`. + Example: `{"owner":"Travel group","action":"take the shuttle instead of renting a car","context":"After the delay, the group agreed the shuttle was the fastest remaining option."}` + +Return `[]` if you do not observe a clear commitment / closure / decision. Better to omit than invent. Body requirements: - Write one tight paragraph in chronological order. @@ -41,4 +49,4 @@ Body requirements: - Prefer what actually occurred over generic labels from the activity type. - If evidence is thin, keep the narrative modest and confidence honest. -Output a single JSON object with only `body`, `topics`, and `confidence`. +Output a single JSON object with all six required fields: `body`, `topics`, `confidence`, `commitments`, `closures`, and `decisions`. diff --git a/talent/journal/references/captures.md b/talent/journal/references/captures.md index 241bec128..0eb368a41 100644 --- a/talent/journal/references/captures.md +++ b/talent/journal/references/captures.md @@ -271,6 +271,6 @@ Each template is a `.md` file with JSON frontmatter containing metadata (title, - System outputs: `talents/{agent}.md` (e.g., `talents/briefing.md`, `talents/default.md`) - App outputs: `talents/_{app}_{agent}.md` (e.g., `talents/_entities_observer.md`) - JSON output: `talents/{agent}.json` when metadata specifies `"output": "json"` -- Story span rows: `facets/{facet}/spans/{day}.jsonl` +- Story fields (`story`, `commitments`, `closures`, `decisions`) live on the activity record in `facets/{facet}/activities/{day}.jsonl` Each generator type has a corresponding template file (`{name}.md`) that defines how the AI synthesizes extracts into narrative form. diff --git a/talent/spans.py b/talent/spans.py deleted file mode 100644 index bb39e60e2..000000000 --- a/talent/spans.py +++ /dev/null @@ -1,170 +0,0 @@ -# SPDX-License-Identifier: AGPL-3.0-only -# Copyright (c) 2026 sol pbc - -"""Post-hook for structured storytelling span rows.""" - -from __future__ import annotations - -import json -import logging -import math -import re -from pathlib import Path -from typing import Any - -from think.activities import locked_modify -from think.utils import get_journal, segment_parse - -logger = logging.getLogger(__name__) - - -def _strip_code_fences(result: str) -> str: - stripped = result.strip() - stripped = re.sub(r"^```(?:json)?\s*", "", stripped) - return re.sub(r"\s*```$", "", stripped) - - -def _normalize_topics(value: Any) -> list[str] | None: - if not isinstance(value, list): - logger.warning("spans hook: missing topics list") - return None - - topics: list[str] = [] - seen: set[str] = set() - for item in value: - if not isinstance(item, str): - logger.warning("spans hook: invalid topics list") - return None - topic = item.strip().lower() - if not topic or topic in seen: - continue - seen.add(topic) - topics.append(topic) - if len(topics) >= 10: - break - - if not topics: - logger.warning("spans hook: empty topics after normalization") - return None - - return topics - - -def _normalize_confidence(value: Any) -> float | None: - if isinstance(value, bool) or not isinstance(value, (int, float)): - logger.warning("spans hook: invalid confidence value") - return None - - confidence = float(value) - if math.isnan(confidence): - logger.warning("spans hook: invalid confidence value") - return None - - clamped = min(1.0, max(0.0, confidence)) - if clamped != confidence: - logger.warning( - "spans hook: clamped confidence %.3f to %.3f", confidence, clamped - ) - return clamped - - -def _activity_time_bounds(segments: Any) -> tuple[str, str] | None: - if not isinstance(segments, list) or not segments: - logger.warning("spans hook: missing activity segments") - return None - - start_time, _ = segment_parse(str(segments[0])) - _, end_time = segment_parse(str(segments[-1])) - if start_time is None or end_time is None: - logger.warning("spans hook: invalid activity segments") - return None - - return start_time.strftime("%H:%M:%S"), end_time.strftime("%H:%M:%S") - - -def _spans_path(facet: str, day: str) -> Path: - return Path(get_journal()) / "facets" / facet / "spans" / f"{day}.jsonl" - - -def post_process(result: str, context: dict) -> str: - """Parse model JSON and persist a single storytelling span row.""" - try: - try: - data = json.loads(_strip_code_fences(result)) - except (json.JSONDecodeError, ValueError) as exc: - logger.warning("spans hook: failed to parse JSON: %s", exc) - return "" - - if not isinstance(data, dict): - logger.warning("spans hook: expected top-level object") - return "" - - body = data.get("body") - if not isinstance(body, str) or not body.strip(): - logger.warning("spans hook: missing body") - return "" - normalized_body = body.strip() - - topics = _normalize_topics(data.get("topics")) - if topics is None: - return "" - - confidence = _normalize_confidence(data.get("confidence")) - if confidence is None: - return "" - - activity = context.get("activity") - if not isinstance(activity, dict): - logger.warning("spans hook: missing activity context") - return "" - - facet = str(context.get("facet") or "").strip() - day = str(context.get("day") or "").strip() - if not facet or not day: - logger.warning("spans hook: missing facet/day context") - return "" - - span_id = str(activity.get("id") or "").strip() - activity_type = str(activity.get("activity") or "").strip() - talent = str(context.get("name") or "").strip() - if not span_id or not activity_type or not talent: - logger.warning("spans hook: missing span metadata") - return "" - - bounds = _activity_time_bounds(activity.get("segments")) - if bounds is None: - return "" - start, end = bounds - - row = { - "span_id": span_id, - "talent": talent, - "facet": facet, - "day": day, - "activity_type": activity_type, - "start": start, - "end": end, - "body": normalized_body, - "topics": topics, - "confidence": confidence, - } - - def modify_fn(records: list[dict[str, Any]]) -> list[dict[str, Any]]: - updated: list[dict[str, Any]] = [] - replaced = False - for record in records: - if record.get("span_id") == span_id and record.get("talent") == talent: - if not replaced: - updated.append(dict(row)) - replaced = True - continue - updated.append(record) - if not replaced: - updated.append(dict(row)) - return updated - - locked_modify(_spans_path(facet, day), modify_fn, create_if_missing=True) - except Exception as exc: - logger.warning("spans hook: failed to persist row: %s", exc) - - return "" diff --git a/talent/story.py b/talent/story.py new file mode 100644 index 000000000..e2ef1a3ca --- /dev/null +++ b/talent/story.py @@ -0,0 +1,191 @@ +# SPDX-License-Identifier: AGPL-3.0-only +# Copyright (c) 2026 sol pbc + +"""Hook for merging storyteller outputs onto activity records.""" + +from __future__ import annotations + +import json +import logging +from typing import Any + +from think.activities import merge_story_fields +from think.entities.loading import load_entities +from think.entities.matching import find_matching_entity + +logger = logging.getLogger(__name__) + +ALLOWED_RESOLUTIONS = frozenset({"sent", "done", "signed", "dropped", "deferred"}) + + +def _resolve_entity_id(name: str, entities: list[dict[str, Any]]) -> str | None: + match = find_matching_entity(name, entities, fuzzy_threshold=90) + return match.get("id") if match else None + + +def _validate_fields( + entry: dict[str, Any], required_fields: tuple[str, ...] +) -> dict[str, str] | None: + normalized: dict[str, str] = {} + for field in required_fields: + value = entry.get(field) + if not isinstance(value, str): + return None + normalized[field] = value + return normalized + + +def post_process(result: str, context: dict) -> str: + """Validate storyteller JSON and merge it onto an activity record.""" + try: + data = json.loads(result.strip()) + except (json.JSONDecodeError, ValueError) as exc: + logger.error("story hook: failed to parse JSON: %s", exc) + return "" + + if not isinstance(data, dict): + logger.warning("story hook: expected top-level object") + return "" + + body = data.get("body") + topics = data.get("topics") + confidence = data.get("confidence") + commitments = data.get("commitments") + closures = data.get("closures") + decisions = data.get("decisions") + + if not isinstance(body, str) or not body.strip(): + logger.warning("story hook: missing body") + return "" + if not isinstance(topics, list) or any( + not isinstance(topic, str) for topic in topics + ): + logger.warning("story hook: invalid topics") + return "" + if ( + isinstance(confidence, bool) + or not isinstance(confidence, (int, float)) + or not 0.0 <= float(confidence) <= 1.0 + ): + logger.warning("story hook: invalid confidence") + return "" + if not isinstance(commitments, list): + logger.warning("story hook: missing commitments list") + return "" + if not isinstance(closures, list): + logger.warning("story hook: missing closures list") + return "" + if not isinstance(decisions, list): + logger.warning("story hook: missing decisions list") + return "" + + activity = context.get("activity") + if not isinstance(activity, dict): + logger.warning("story hook: missing activity context") + return "" + + record_id = activity.get("id") + if not isinstance(record_id, str) or not record_id: + logger.warning("story hook: missing activity record id") + return "" + + facet = context.get("facet") + day = context.get("day") + if not isinstance(facet, str) or not facet or not isinstance(day, str) or not day: + logger.warning("story hook: missing facet/day context") + return "" + + entities = load_entities(facet=facet, day=day) + + resolved_commitments: list[dict[str, Any]] = [] + for index, entry in enumerate(commitments): + if not isinstance(entry, dict): + logger.warning( + "story hook: skipping commitment[%d]: expected object", index + ) + continue + normalized = _validate_fields( + entry, ("owner", "action", "counterparty", "when", "context") + ) + if normalized is None: + logger.warning( + "story hook: skipping commitment[%d]: missing required string field", + index, + ) + continue + resolved_commitment = dict(normalized) + resolved_commitment["owner_entity_id"] = _resolve_entity_id( + normalized["owner"], entities + ) + resolved_commitment["counterparty_entity_id"] = _resolve_entity_id( + normalized["counterparty"], entities + ) + resolved_commitments.append(resolved_commitment) + + resolved_closures: list[dict[str, Any]] = [] + for index, entry in enumerate(closures): + if not isinstance(entry, dict): + logger.warning("story hook: skipping closure[%d]: expected object", index) + continue + normalized = _validate_fields( + entry, ("owner", "action", "counterparty", "resolution", "context") + ) + if normalized is None: + logger.warning( + "story hook: skipping closure[%d]: missing required string field", + index, + ) + continue + if normalized["resolution"] not in ALLOWED_RESOLUTIONS: + logger.warning( + "story hook: skipping closure[%d]: invalid resolution '%s'", + index, + normalized["resolution"], + ) + continue + resolved_closure = dict(normalized) + resolved_closure["owner_entity_id"] = _resolve_entity_id( + normalized["owner"], entities + ) + resolved_closure["counterparty_entity_id"] = _resolve_entity_id( + normalized["counterparty"], entities + ) + resolved_closures.append(resolved_closure) + + resolved_decisions: list[dict[str, Any]] = [] + for index, entry in enumerate(decisions): + if not isinstance(entry, dict): + logger.warning("story hook: skipping decision[%d]: expected object", index) + continue + normalized = _validate_fields(entry, ("owner", "action", "context")) + if normalized is None: + logger.warning( + "story hook: skipping decision[%d]: missing required string field", + index, + ) + continue + resolved_decision = dict(normalized) + resolved_decision["owner_entity_id"] = _resolve_entity_id( + normalized["owner"], entities + ) + resolved_decisions.append(resolved_decision) + + story = { + "body": body.strip(), + "topics": list(topics), + "confidence": float(confidence), + } + + merge_story_fields( + facet, + day, + record_id, + story=story, + commitments=resolved_commitments, + closures=resolved_closures, + decisions=resolved_decisions, + actor="story", + note=None, + ) + + return "" diff --git a/talent/work.md b/talent/work.md index feaa1e0d6..79e7f911a 100644 --- a/talent/work.md +++ b/talent/work.md @@ -1,13 +1,13 @@ { "type": "generate", "title": "Work Story", - "description": "Writes a structured narrative span row for coding, browsing, and reading activities.", + "description": "Generates a work story, topics, and structured commitments, closures, and decisions to merge onto the activity record.", "color": "#6d4c41", "schedule": "activity", "activities": ["coding", "browsing", "reading"], - "priority": 10, + "priority": 20, "output": "json", - "hook": {"post": "spans"}, + "hook": {"post": "story"}, "load": { "transcripts": true, "percepts": true, @@ -29,10 +29,18 @@ Summarize what this person accomplished, investigated, or worked through during the activity. Participation and entity extraction already happened upstream. Use that context; do not re-extract people or entities into new structures. -Return exactly these three fields: +Return exactly this six-field JSON object: - `body`: string narrative prose about the work performed and what changed. -- `topics`: array of 3-8 short string tags. +- `topics`: array of short string tags; use `[]` when there are no durable topics worth preserving. - `confidence`: float from 0.0 to 1.0. +- `commitments`: array of objects with required string fields `owner`, `action`, `counterparty`, `when`, `context`. + Example: `{"owner":"Avery","action":"post the benchmark results","counterparty":"Priya","when":"after lunch","context":"Avery said the new retry benchmark would be shared once the run completed."}` +- `closures`: array of objects with required string fields `owner`, `action`, `counterparty`, `resolution`, `context`. `resolution` must be one of `sent`, `done`, `signed`, `dropped`, `deferred`. + Example: `{"owner":"Avery","action":"follow-up PR","counterparty":"Priya","resolution":"done","context":"Avery noted the cleanup PR was merged during this work block."}` +- `decisions`: array of objects with required string fields `owner`, `action`, `context`. + Example: `{"owner":"Avery","action":"switch the retry path to queue-backed backoff","context":"The work session concluded that queue-backed backoff was simpler than the timer-based branch."}` + +Return `[]` if you do not observe a clear commitment / closure / decision. Better to omit than invent. Body requirements: - Write one tight paragraph in chronological order. @@ -41,4 +49,4 @@ Body requirements: - If evidence is partial, describe the most defensible story and keep the confidence honest. -Output a single JSON object with only `body`, `topics`, and `confidence`. +Output a single JSON object with all six required fields: `body`, `topics`, `confidence`, `commitments`, `closures`, and `decisions`. diff --git a/tests/baselines/api/sol/talents-day.json b/tests/baselines/api/sol/talents-day.json index b21712bcb..03f564ec0 100644 --- a/tests/baselines/api/sol/talents-day.json +++ b/tests/baselines/api/sol/talents-day.json @@ -85,7 +85,7 @@ "conversation": { "app": null, "color": "#00796b", - "description": "Writes a structured narrative span row for meeting, call, messaging, and email activities.", + "description": "Generates a conversation story, topics, and structured commitments, closures, and decisions to merge onto the activity record.", "multi_facet": false, "output_format": "json", "schedule": "activity", @@ -184,7 +184,7 @@ "event": { "app": null, "color": "#ff7043", - "description": "Writes a structured narrative span row for appointment, event, travel, errand, celebration, deadline, and reminder activities.", + "description": "Generates an event story, topics, and structured commitments, closures, and decisions to merge onto the activity record.", "multi_facet": false, "output_format": "json", "schedule": "activity", @@ -459,7 +459,7 @@ "work": { "app": null, "color": "#6d4c41", - "description": "Writes a structured narrative span row for coding, browsing, and reading activities.", + "description": "Generates a work story, topics, and structured commitments, closures, and decisions to merge onto the activity record.", "multi_facet": false, "output_format": "json", "schedule": "activity", diff --git a/tests/baselines/api/stats/stats.json b/tests/baselines/api/stats/stats.json index 602626b45..66190a14d 100644 --- a/tests/baselines/api/stats/stats.json +++ b/tests/baselines/api/stats/stats.json @@ -8,9 +8,9 @@ "email" ], "color": "#00796b", - "description": "Writes a structured narrative span row for meeting, call, messaging, and email activities.", + "description": "Generates a conversation story, topics, and structured commitments, closures, and decisions to merge onto the activity record.", "hook": { - "post": "spans" + "post": "story" }, "load": { "percepts": true, @@ -20,7 +20,7 @@ "mtime": 0, "output": "json", "path": "/talent/conversation.md", - "priority": 10, + "priority": 20, "schedule": "activity", "source": "system", "title": "Conversation Story", @@ -130,9 +130,9 @@ "reminder" ], "color": "#ff7043", - "description": "Writes a structured narrative span row for appointment, event, travel, errand, celebration, deadline, and reminder activities.", + "description": "Generates an event story, topics, and structured commitments, closures, and decisions to merge onto the activity record.", "hook": { - "post": "spans" + "post": "story" }, "load": { "percepts": true, @@ -142,7 +142,7 @@ "mtime": 0, "output": "json", "path": "/talent/event.md", - "priority": 10, + "priority": 20, "schedule": "activity", "source": "system", "title": "Event Story", @@ -287,9 +287,9 @@ "reading" ], "color": "#6d4c41", - "description": "Writes a structured narrative span row for coding, browsing, and reading activities.", + "description": "Generates a work story, topics, and structured commitments, closures, and decisions to merge onto the activity record.", "hook": { - "post": "spans" + "post": "story" }, "load": { "percepts": true, @@ -299,7 +299,7 @@ "mtime": 0, "output": "json", "path": "/talent/work.md", - "priority": 10, + "priority": 20, "schedule": "activity", "source": "system", "title": "Work Story", diff --git a/tests/test_activities.py b/tests/test_activities.py index d7d10f67b..55c256daa 100644 --- a/tests/test_activities.py +++ b/tests/test_activities.py @@ -781,6 +781,44 @@ class TestActivityRecordIO: note="bad field", ) + def test_format_activities_renders_story(self): + from think.activities import format_activities + + chunks, _meta = format_activities( + [ + { + "id": "meeting_090000_300", + "activity": "meeting", + "description": "Team sync", + "segments": ["090000_300"], + "created_at": 1, + "participation": [{"name": "Mina"}], + "story": { + "body": "Aligned on the launch plan and assigned owners.", + "topics": ["launch", "owners"], + "confidence": 0.9, + }, + }, + { + "id": "coding_100000_300", + "activity": "coding", + "description": "Implementation block", + "segments": ["100000_300"], + "created_at": 2, + }, + ] + ) + + assert ( + "Aligned on the launch plan and assigned owners." in chunks[0]["markdown"] + ) + assert "Topics: launch, owners" in chunks[0]["markdown"] + assert "Topics:" not in chunks[1]["markdown"] + assert ( + "Aligned on the launch plan and assigned owners." + not in chunks[1]["markdown"] + ) + def test_hidden_records_filtered_by_default(self, monkeypatch): from think.activities import ( append_activity_record, diff --git a/tests/test_formatters.py b/tests/test_formatters.py index d03d82584..946a08d05 100644 --- a/tests/test_formatters.py +++ b/tests/test_formatters.py @@ -79,6 +79,41 @@ class TestRegistry: formatter = get_formatter("random/path/unknown.jsonl") assert formatter is None + def test_no_spans_formatter_registered(self): + """Spans JSONL is no longer registered after the story refactor.""" + from think.formatters import FORMATTERS, get_formatter + + assert "facets/*/spans/*.jsonl" not in FORMATTERS + assert get_formatter("facets/work/spans/20260418.jsonl") is None + + def test_no_spans_writes(self): + """No spans formatter or spans JSONL write targets remain in source dirs.""" + repo_root = Path(__file__).resolve().parent.parent + patterns = [ + "format_spans", + "talent.spans", + "think.spans", + ' / "spans" / ', + "facets/*/spans", + "spans/{day}.jsonl", + ] + + hits: list[str] = [] + for directory in ("think", "talent", "apps"): + for path in (repo_root / directory).rglob("*"): + if not path.is_file() or path.suffix not in {".py", ".md"}: + continue + for lineno, line in enumerate( + path.read_text(encoding="utf-8").splitlines(), start=1 + ): + for pattern in patterns: + if pattern in line: + hits.append( + f"{path.relative_to(repo_root)}:{lineno}:{pattern}" + ) + + assert hits == [] + class TestLoadJsonl: """Tests for JSONL loading utility.""" diff --git a/tests/test_spans_formatter.py b/tests/test_spans_formatter.py deleted file mode 100644 index 572a59441..000000000 --- a/tests/test_spans_formatter.py +++ /dev/null @@ -1,108 +0,0 @@ -# SPDX-License-Identifier: AGPL-3.0-only -# Copyright (c) 2026 sol pbc - -"""Tests for spans JSONL formatting.""" - -from __future__ import annotations - -from datetime import datetime -from pathlib import Path - - -def test_format_spans_builds_chunks_and_metadata(): - from think.spans import format_spans - - entries = [ - { - "span_id": "meeting_090000_300", - "talent": "conversation", - "facet": "work", - "day": "20260101", - "activity_type": "meeting", - "start": "09:00:00", - "end": "09:15:00", - "body": "Aligned on the launch plan and confirmed owners.", - "topics": ["launch", "owners", "planning"], - "confidence": 0.93, - }, - { - "span_id": "coding_130000_300", - "talent": "work", - "facet": "work", - "day": "20260101", - "activity_type": "coding", - "start": "13:00:00", - "end": "13:10:00", - "body": "Implemented the migration and updated the tests.", - "topics": ["migration", "tests", "backend"], - "confidence": 0.81, - }, - ] - - file_path = Path("/tmp/journal/facets/work/spans/20260101.jsonl") - chunks, meta = format_spans(entries, {"file_path": file_path}) - - assert len(chunks) == 2 - assert meta["header"] == "# Spans for 'work' facet on 2026-01-01" - assert meta["indexer"] == {"agent": "span"} - - first = chunks[0] - expected_ts = int(datetime.strptime("20260101", "%Y%m%d").timestamp() * 1000) - expected_ts += 9 * 3600 * 1000 - assert first["timestamp"] == expected_ts - assert first["source"] == entries[0] - assert "### Meeting: meeting_090000_300" in first["markdown"] - assert "**Time:** 09:00:00-09:15:00" in first["markdown"] - assert "**Activity Type:** meeting" in first["markdown"] - assert "**Topics:** launch, owners, planning" in first["markdown"] - assert "**Confidence:** 0.93" in first["markdown"] - assert "**Talent:** conversation" in first["markdown"] - assert "Aligned on the launch plan" in first["markdown"] - - -def test_format_spans_skips_invalid_rows_and_reports_error(): - from think.spans import format_spans - - entries = [ - { - "span_id": "valid_1", - "talent": "work", - "facet": "work", - "day": "20260101", - "activity_type": "coding", - "start": "08:00:00", - "end": "08:05:00", - "body": "Valid row.", - "topics": ["alpha", "beta", "gamma"], - "confidence": 0.5, - }, - { - "span_id": "invalid_1", - "talent": "work", - "facet": "work", - "day": "20260101", - "activity_type": "coding", - "start": "08:05:00", - "end": "08:10:00", - "topics": ["alpha", "beta", "gamma"], - "confidence": 0.5, - }, - ] - - chunks, meta = format_spans( - entries, {"file_path": Path("/tmp/journal/facets/work/spans/20260101.jsonl")} - ) - - assert len(chunks) == 1 - assert "Skipped 1 entries missing required fields" in meta["error"] - assert "20260101.jsonl" in meta["error"] - assert meta["indexer"] == {"agent": "span"} - - -def test_get_formatter_returns_spans_formatter(): - from think.formatters import get_formatter - - formatter = get_formatter("facets/foo/spans/20260101.jsonl") - - assert formatter is not None - assert formatter.__name__ == "format_spans" diff --git a/tests/test_spans_hook.py b/tests/test_spans_hook.py deleted file mode 100644 index 5287768ef..000000000 --- a/tests/test_spans_hook.py +++ /dev/null @@ -1,297 +0,0 @@ -# SPDX-License-Identifier: AGPL-3.0-only -# Copyright (c) 2026 sol pbc - -"""Tests for the storytelling spans post-hook.""" - -from __future__ import annotations - -import json -from pathlib import Path - -from talent.spans import post_process - - -def _activity( - *, - activity_id: str = "coding_100000_300", - activity_type: str = "coding", - segments: list[str] | None = None, -) -> dict: - return { - "id": activity_id, - "activity": activity_type, - "segments": segments or ["100000_300", "100500_300"], - } - - -def _context( - *, - name: str = "work", - facet: str = "work", - day: str = "20260418", - activity: dict | None = None, -) -> dict: - return { - "name": name, - "facet": facet, - "day": day, - "activity": activity or _activity(), - } - - -def _rows(tmp_path: Path, *, facet: str = "work", day: str = "20260418") -> list[dict]: - path = tmp_path / "facets" / facet / "spans" / f"{day}.jsonl" - if not path.exists(): - return [] - return [json.loads(line) for line in path.read_text(encoding="utf-8").splitlines()] - - -def test_post_process_writes_all_fields_and_renders_coding_span(monkeypatch, tmp_path): - from think.spans import format_spans - - monkeypatch.setenv("_SOLSTONE_JOURNAL_OVERRIDE", str(tmp_path)) - - result = json.dumps( - { - "body": "Implemented the retry path and verified the failing case.", - "topics": ["Retry Logic", "Testing", "retry logic", " Testing "], - "confidence": 0.82, - } - ) - - returned = post_process(result, _context()) - - assert returned == "" - - rows = _rows(tmp_path) - assert len(rows) == 1 - assert rows[0] == { - "span_id": "coding_100000_300", - "talent": "work", - "facet": "work", - "day": "20260418", - "activity_type": "coding", - "start": "10:00:00", - "end": "10:10:00", - "body": "Implemented the retry path and verified the failing case.", - "topics": ["retry logic", "testing"], - "confidence": 0.82, - } - - file_path = tmp_path / "facets" / "work" / "spans" / "20260418.jsonl" - chunks, meta = format_spans(rows, {"file_path": file_path}) - assert len(chunks) == 1 - assert chunks[0]["source"] == rows[0] - assert "### Coding: coding_100000_300" in chunks[0]["markdown"] - assert meta["indexer"] == {"agent": "span"} - - -def test_post_process_writes_single_conversation_row_for_meeting(monkeypatch, tmp_path): - monkeypatch.setenv("_SOLSTONE_JOURNAL_OVERRIDE", str(tmp_path)) - - result = json.dumps( - { - "body": 'Aligned on next steps and confirmed "ship it Friday".', - "topics": ["planning", "alignment", "delivery"], - "confidence": 0.91, - } - ) - ctx = _context( - name="conversation", - activity=_activity( - activity_id="meeting_090000_300", - activity_type="meeting", - segments=["090000_300", "091500_300"], - ), - ) - - returned = post_process(result, ctx) - - assert returned == "" - rows = _rows(tmp_path) - assert len(rows) == 1 - assert rows[0]["talent"] == "conversation" - assert rows[0]["activity_type"] == "meeting" - assert rows[0]["start"] == "09:00:00" - assert rows[0]["end"] == "09:20:00" - assert not ( - tmp_path - / "facets" - / "work" - / "activities" - / "20260418" - / "meeting_090000_300" - / "conversation.json" - ).exists() - - -def test_post_process_clamps_confidence_and_logs(monkeypatch, tmp_path, caplog): - monkeypatch.setenv("_SOLSTONE_JOURNAL_OVERRIDE", str(tmp_path)) - - returned = post_process( - json.dumps( - { - "body": "Shipped the fix.", - "topics": ["release", "shipping", "qa"], - "confidence": 1.4, - } - ), - _context(), - ) - - assert returned == "" - assert _rows(tmp_path)[0]["confidence"] == 1.0 - assert "clamped confidence" in caplog.text - - -def test_post_process_rejects_bad_confidence(monkeypatch, tmp_path, caplog): - monkeypatch.setenv("_SOLSTONE_JOURNAL_OVERRIDE", str(tmp_path)) - - returned = post_process( - json.dumps( - { - "body": "Investigated the issue.", - "topics": ["debugging", "logs", "triage"], - "confidence": "high", - } - ), - _context(), - ) - - assert returned == "" - assert _rows(tmp_path) == [] - assert "invalid confidence" in caplog.text - - caplog.clear() - returned = post_process( - json.dumps( - { - "body": "Investigated the issue.", - "topics": ["debugging", "logs", "triage"], - "confidence": float("nan"), - } - ), - _context(), - ) - - assert returned == "" - assert _rows(tmp_path) == [] - assert "invalid confidence" in caplog.text - - -def test_post_process_rejects_missing_or_empty_topics(monkeypatch, tmp_path, caplog): - monkeypatch.setenv("_SOLSTONE_JOURNAL_OVERRIDE", str(tmp_path)) - - missing_topics = json.dumps({"body": "Worked through the task.", "confidence": 0.7}) - returned = post_process(missing_topics, _context()) - assert returned == "" - assert _rows(tmp_path) == [] - assert "missing topics" in caplog.text - - caplog.clear() - empty_topics = json.dumps( - {"body": "Worked through the task.", "topics": [" ", "\t"], "confidence": 0.7} - ) - returned = post_process(empty_topics, _context()) - assert returned == "" - assert _rows(tmp_path) == [] - assert "empty topics" in caplog.text - - caplog.clear() - invalid_topics = json.dumps( - { - "body": "Worked through the task.", - "topics": ["valid", 7, "other"], - "confidence": 0.7, - } - ) - returned = post_process(invalid_topics, _context()) - assert returned == "" - assert _rows(tmp_path) == [] - assert "invalid topics" in caplog.text - - -def test_post_process_replaces_existing_row_by_span_and_talent(monkeypatch, tmp_path): - monkeypatch.setenv("_SOLSTONE_JOURNAL_OVERRIDE", str(tmp_path)) - ctx = _context() - - first = json.dumps( - {"body": "First pass.", "topics": ["alpha", "beta", "gamma"], "confidence": 0.5} - ) - second = json.dumps( - { - "body": "Second pass.", - "topics": ["delta", "epsilon", "zeta"], - "confidence": 0.9, - } - ) - - assert post_process(first, ctx) == "" - assert post_process(second, ctx) == "" - - rows = _rows(tmp_path) - assert len(rows) == 1 - assert rows[0]["body"] == "Second pass." - assert rows[0]["topics"] == ["delta", "epsilon", "zeta"] - assert rows[0]["confidence"] == 0.9 - - -def test_post_process_appends_distinct_talent_rows(monkeypatch, tmp_path): - monkeypatch.setenv("_SOLSTONE_JOURNAL_OVERRIDE", str(tmp_path)) - activity = _activity(activity_id="event_130000_300", activity_type="event") - - event_ctx = _context(name="event", activity=activity) - conversation_ctx = _context(name="conversation", activity=activity) - - assert ( - post_process( - json.dumps( - { - "body": "Wrapped the event.", - "topics": ["planning", "venue", "timeline"], - "confidence": 0.66, - } - ), - event_ctx, - ) - == "" - ) - assert ( - post_process( - json.dumps( - { - "body": "Captured the side conversation.", - "topics": ["alignment", "follow-up", "owners"], - "confidence": 0.72, - } - ), - conversation_ctx, - ) - == "" - ) - - rows = _rows(tmp_path) - assert len(rows) == 2 - assert {(row["span_id"], row["talent"]) for row in rows} == { - ("event_130000_300", "event"), - ("event_130000_300", "conversation"), - } - - -def test_post_process_handles_parse_failures_and_fenced_json( - monkeypatch, tmp_path, caplog -): - monkeypatch.setenv("_SOLSTONE_JOURNAL_OVERRIDE", str(tmp_path)) - - assert post_process("{not-json", _context()) == "" - assert _rows(tmp_path) == [] - assert "failed to parse JSON" in caplog.text - - caplog.clear() - fenced = """```json -{"body":"Recovered.","topics":["alpha","beta","gamma"],"confidence":0.6} -```""" - assert post_process(fenced, _context()) == "" - rows = _rows(tmp_path) - assert len(rows) == 1 - assert rows[0]["body"] == "Recovered." diff --git a/tests/test_story_hook.py b/tests/test_story_hook.py new file mode 100644 index 000000000..712beff89 --- /dev/null +++ b/tests/test_story_hook.py @@ -0,0 +1,376 @@ +# SPDX-License-Identifier: AGPL-3.0-only +# Copyright (c) 2026 sol pbc + +import json +from pathlib import Path + + +def _write_detected_entities(tmp_path, facet: str, day: str, rows: list[dict]) -> None: + path = tmp_path / "facets" / facet / "entities" / f"{day}.jsonl" + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text( + "".join(json.dumps(row, ensure_ascii=False) + "\n" for row in rows), + encoding="utf-8", + ) + + +def _activity_record(record_id: str = "meeting_090000_300") -> dict: + return { + "id": record_id, + "activity": "meeting", + "description": "Team sync", + "segments": ["090000_300"], + "created_at": 1, + } + + +def _context( + tmp_path: Path, + *, + facet: str = "work", + day: str = "20260418", + record_id: str = "meeting_090000_300", +) -> dict: + return { + "facet": facet, + "day": day, + "activity": {"id": record_id}, + "output_path": str( + tmp_path / "facets" / facet / "activities" / day / record_id / "story.json" + ), + } + + +def _valid_result(**overrides) -> str: + payload = { + "body": "Aligned on launch work and assigned the follow-up.", + "topics": ["launch", "follow-up"], + "confidence": 0.82, + "commitments": [ + { + "owner": "Mina", + "action": "send the revised deck", + "counterparty": "Ravi", + "when": "Friday morning", + "context": "Mina committed to send the deck before the next investor call.", + } + ], + "closures": [ + { + "owner": "Ravi", + "action": "intro email", + "counterparty": "Mina", + "resolution": "sent", + "context": "Ravi confirmed the intro email already went out.", + } + ], + "decisions": [ + { + "owner": "Team", + "action": "move the launch review to Tuesday", + "context": "The group aligned on Tuesday after checking calendars.", + } + ], + } + payload.update(overrides) + return json.dumps(payload) + + +def _load_record(facet: str, day: str): + from think.activities import load_activity_records + + return load_activity_records(facet, day, include_hidden=True)[0] + + +def test_story_hook_parses_and_writes(tmp_path, monkeypatch): + from talent.story import post_process + from think.activities import append_activity_record + + monkeypatch.setenv("_SOLSTONE_JOURNAL_OVERRIDE", str(tmp_path)) + + append_activity_record("work", "20260418", _activity_record()) + + returned = post_process( + _valid_result(body=" Wrapped the launch prep and assigned follow-up. "), + _context(tmp_path), + ) + + record = _load_record("work", "20260418") + assert returned == "" + assert record["story"] == { + "body": "Wrapped the launch prep and assigned follow-up.", + "topics": ["launch", "follow-up"], + "confidence": 0.82, + } + assert record["commitments"][0]["owner"] == "Mina" + assert record["closures"][0]["resolution"] == "sent" + assert record["decisions"][0]["owner"] == "Team" + assert record["edits"][-1]["actor"] == "story" + assert record["edits"][-1]["fields"] == [ + "story", + "commitments", + "closures", + "decisions", + ] + + +def test_story_hook_empty_arrays(tmp_path, monkeypatch): + from talent.story import post_process + from think.activities import append_activity_record + + monkeypatch.setenv("_SOLSTONE_JOURNAL_OVERRIDE", str(tmp_path)) + append_activity_record("work", "20260418", _activity_record()) + + post_process( + _valid_result(commitments=[], closures=[], decisions=[]), + _context(tmp_path), + ) + + record = _load_record("work", "20260418") + assert ( + record["story"]["body"] == "Aligned on launch work and assigned the follow-up." + ) + assert record["commitments"] == [] + assert record["closures"] == [] + assert record["decisions"] == [] + + +def test_story_hook_bad_resolution_skipped(tmp_path, monkeypatch, caplog): + from talent.story import post_process + from think.activities import append_activity_record + + monkeypatch.setenv("_SOLSTONE_JOURNAL_OVERRIDE", str(tmp_path)) + append_activity_record("work", "20260418", _activity_record()) + + post_process( + _valid_result( + closures=[ + { + "owner": "Ravi", + "action": "intro email", + "counterparty": "Mina", + "resolution": "sent", + "context": "The intro email went out.", + }, + { + "owner": "Ravi", + "action": "budget request", + "counterparty": "Finance", + "resolution": "approved", + "context": "This resolution is invalid for the schema.", + }, + ] + ), + _context(tmp_path), + ) + + record = _load_record("work", "20260418") + assert [closure["action"] for closure in record["closures"]] == ["intro email"] + assert "invalid resolution 'approved'" in caplog.text + + +def test_story_hook_missing_required_field_skipped(tmp_path, monkeypatch, caplog): + from talent.story import post_process + from think.activities import append_activity_record + + monkeypatch.setenv("_SOLSTONE_JOURNAL_OVERRIDE", str(tmp_path)) + append_activity_record("work", "20260418", _activity_record()) + + post_process( + _valid_result( + commitments=[ + { + "owner": "Mina", + "action": "send the revised deck", + "counterparty": "Ravi", + "when": "Friday morning", + "context": "Valid commitment.", + }, + { + "owner": "Mina", + "action": "book the room", + "when": "tomorrow", + "context": "Missing counterparty should skip.", + }, + ], + closures=[ + { + "owner": "Ravi", + "action": "intro email", + "counterparty": "Mina", + "resolution": "sent", + "context": "Valid closure.", + }, + { + "action": "parking pass", + "counterparty": "Travel desk", + "resolution": "done", + "context": "Missing owner should skip.", + }, + ], + decisions=[ + { + "owner": "Team", + "action": "move the launch review to Tuesday", + "context": "Valid decision.", + }, + { + "owner": "Team", + "context": "Missing action should skip.", + }, + ], + ), + _context(tmp_path), + ) + + record = _load_record("work", "20260418") + assert len(record["commitments"]) == 1 + assert len(record["closures"]) == 1 + assert len(record["decisions"]) == 1 + assert "missing required string field" in caplog.text + + +def test_story_hook_resolves_entities(tmp_path, monkeypatch): + from talent.story import post_process + from think.activities import append_activity_record + + monkeypatch.setenv("_SOLSTONE_JOURNAL_OVERRIDE", str(tmp_path)) + _write_detected_entities( + tmp_path, + "work", + "20260418", + [ + {"id": "mina_lee", "type": "Person", "name": "Mina Lee", "aka": ["Mina"]}, + {"id": "ravi_shah", "type": "Person", "name": "Ravi Shah", "aka": ["Ravi"]}, + ], + ) + append_activity_record("work", "20260418", _activity_record()) + + post_process( + _valid_result( + commitments=[ + { + "owner": "Mina", + "action": "send the revised deck", + "counterparty": "Ravi", + "when": "Friday morning", + "context": "Valid commitment.", + }, + { + "owner": "Unknown Owner", + "action": "draft the note", + "counterparty": "Unknown Counterparty", + "when": "later", + "context": "Unmatched names should stay null.", + }, + ], + closures=[ + { + "owner": "Ravi", + "action": "intro email", + "counterparty": "Mina", + "resolution": "sent", + "context": "Valid closure.", + } + ], + decisions=[ + { + "owner": "Mina Lee", + "action": "move the launch review to Tuesday", + "context": "Valid decision.", + } + ], + ), + _context(tmp_path), + ) + + record = _load_record("work", "20260418") + assert record["commitments"][0]["owner_entity_id"] == "mina_lee" + assert record["commitments"][0]["counterparty_entity_id"] == "ravi_shah" + assert record["commitments"][1]["owner_entity_id"] is None + assert record["commitments"][1]["counterparty_entity_id"] is None + assert record["closures"][0]["owner_entity_id"] == "ravi_shah" + assert record["closures"][0]["counterparty_entity_id"] == "mina_lee" + assert record["decisions"][0]["owner_entity_id"] == "mina_lee" + assert record["commitments"][0]["owner"] == "Mina" + assert record["closures"][0]["counterparty"] == "Mina" + + +def test_story_hook_idempotent_rerun(tmp_path, monkeypatch): + from talent.story import post_process + from think.activities import append_activity_record + + monkeypatch.setenv("_SOLSTONE_JOURNAL_OVERRIDE", str(tmp_path)) + append_activity_record("work", "20260418", _activity_record()) + + post_process(_valid_result(), _context(tmp_path)) + first = _load_record("work", "20260418") + assert len(first["edits"]) == 1 + + post_process( + _valid_result( + body="Second pass with a clearer summary.", + topics=["handoff"], + commitments=[], + closures=[], + decisions=[ + { + "owner": "Lead", + "action": "ship the patch on Wednesday", + "context": "The second pass reached a more specific plan.", + } + ], + ), + _context(tmp_path), + ) + + second = _load_record("work", "20260418") + assert second["story"] == { + "body": "Second pass with a clearer summary.", + "topics": ["handoff"], + "confidence": 0.82, + } + assert second["commitments"] == [] + assert second["closures"] == [] + assert second["decisions"] == [ + { + "owner": "Lead", + "action": "ship the patch on Wednesday", + "context": "The second pass reached a more specific plan.", + "owner_entity_id": None, + } + ] + assert len(second["edits"]) == 2 + + +def test_story_hook_missing_record_logs_and_returns(tmp_path, monkeypatch, caplog): + from talent.story import post_process + + monkeypatch.setenv("_SOLSTONE_JOURNAL_OVERRIDE", str(tmp_path)) + + returned = post_process(_valid_result(), _context(tmp_path)) + + assert returned == "" + assert "activity record not found" in caplog.text + + +def test_story_hook_no_json_file_written(tmp_path, monkeypatch): + from talent.story import post_process + from think.activities import append_activity_record + + monkeypatch.setenv("_SOLSTONE_JOURNAL_OVERRIDE", str(tmp_path)) + append_activity_record("work", "20260418", _activity_record()) + + output_path = ( + tmp_path + / "facets" + / "work" + / "activities" + / "20260418" + / "meeting_090000_300" + / "story.json" + ) + returned = post_process(_valid_result(), _context(tmp_path)) + + assert returned == "" + assert not output_path.exists() diff --git a/think/activities.py b/think/activities.py index c8677f631..37c2b4a83 100644 --- a/think/activities.py +++ b/think/activities.py @@ -803,7 +803,7 @@ def locked_modify( def append_edit( - record: dict[str, Any], *, actor: str, fields: list[str], note: str + record: dict[str, Any], *, actor: str, fields: list[str], note: str | None ) -> dict[str, Any]: """Append an edit entry to an activity record and return the record.""" normalized = _normalize_activity_record(record) @@ -1089,6 +1089,55 @@ def update_activity_record( return updated_record +def merge_story_fields( + facet: str, + day: str, + record_id: str, + *, + story: dict, + commitments: list[dict], + closures: list[dict], + decisions: list[dict], + actor: str, + note: str | None = None, +) -> bool: + """Replace story-derived fields on an activity record and append one edit.""" + updated = False + path = _get_records_path(facet, day) + + def modify_fn(records: list[dict[str, Any]]) -> list[dict[str, Any]]: + nonlocal updated + new_records: list[dict[str, Any]] = [] + for record in records: + if record.get("id") == record_id: + merged = _normalize_activity_record(record) + merged["story"] = dict(story) + merged["commitments"] = [dict(entry) for entry in commitments] + merged["closures"] = [dict(entry) for entry in closures] + merged["decisions"] = [dict(entry) for entry in decisions] + merged = append_edit( + merged, + actor=actor, + fields=["story", "commitments", "closures", "decisions"], + note=note, + ) + new_records.append(merged) + updated = True + else: + new_records.append(record) + return new_records + + try: + locked_modify(path, modify_fn, create_if_missing=False) + except FileNotFoundError: + logger.warning("story hook: activity record not found: %s", record_id) + return False + + if not updated: + logger.warning("story hook: activity record not found: %s", record_id) + return updated + + def _set_activity_hidden_state( facet: str, day: str, @@ -1302,6 +1351,23 @@ def format_activities( if participants: lines.append(f"- Participation: {participants}") + story = record.get("story") + if isinstance(story, dict): + body = story.get("body") + if isinstance(body, str) and body.strip(): + lines.append("") + lines.append(body.strip()) + + topics = story.get("topics") + if isinstance(topics, list): + topic_values = [ + topic.strip() + for topic in topics + if isinstance(topic, str) and topic.strip() + ] + if topic_values: + lines.append(f"Topics: {', '.join(topic_values)}") + if record.get("hidden", False): lines.append("- Hidden: yes") diff --git a/think/formatters.py b/think/formatters.py index 83aa8fd58..cd8362d0b 100644 --- a/think/formatters.py +++ b/think/formatters.py @@ -140,7 +140,6 @@ FORMATTERS: dict[str, tuple[str, str, bool]] = { False, # Indexed via _index_entity_search_chunks (enriched with relationship data) ), "facets/*/events/*.jsonl": ("think.event_formatter", "format_events", True), - "facets/*/spans/*.jsonl": ("think.spans", "format_spans", True), "facets/*/activities/*.jsonl": ("think.activities", "format_activities", True), "facets/*/todos/*.jsonl": ("apps.todos.todo", "format_todos", True), "facets/*/logs/*.jsonl": ("think.facets", "format_logs", True), diff --git a/think/spans.py b/think/spans.py deleted file mode 100644 index fa04d8ce1..000000000 --- a/think/spans.py +++ /dev/null @@ -1,128 +0,0 @@ -# SPDX-License-Identifier: AGPL-3.0-only -# Copyright (c) 2026 sol pbc - -"""Formatting helpers for storytelling spans JSONL files.""" - -from __future__ import annotations - -import logging -import re -from datetime import datetime -from pathlib import Path -from typing import Any - - -def _extract_spans_path_context(file_path: str | Path | None) -> tuple[str, str | None]: - facet_name = "unknown" - day_str: str | None = None - - if not file_path: - return facet_name, day_str - - path = Path(file_path) - path_str = str(path) - facet_match = re.search(r"facets/([^/]+)/spans", path_str) - if facet_match: - facet_name = facet_match.group(1) - - if path.stem.isdigit() and len(path.stem) == 8: - day_str = path.stem - - return facet_name, day_str - - -def _start_seconds(start: str) -> int | None: - try: - parts = start.split(":") - hours = int(parts[0]) - minutes = int(parts[1]) if len(parts) > 1 else 0 - seconds = int(parts[2]) if len(parts) > 2 else 0 - except (IndexError, ValueError, TypeError): - return None - return hours * 3600 + minutes * 60 + seconds - - -def format_spans( - entries: list[dict], - context: dict | None = None, -) -> tuple[list[dict], dict]: - """Format storytelling span JSONL rows into markdown chunks.""" - ctx = context or {} - file_path = ctx.get("file_path") - meta: dict[str, Any] = {"indexer": {"agent": "span"}} - chunks: list[dict[str, Any]] = [] - skipped_count = 0 - - facet_name, day_str = _extract_spans_path_context(file_path) - - base_ts = 0 - if day_str: - try: - dt = datetime.strptime(day_str, "%Y%m%d") - base_ts = int(dt.timestamp() * 1000) - except ValueError: - pass - - if day_str: - formatted_day = f"{day_str[:4]}-{day_str[4:6]}-{day_str[6:8]}" - meta["header"] = f"# Spans for '{facet_name}' facet on {formatted_day}" - else: - meta["header"] = f"# Spans for '{facet_name}' facet" - - for entry in entries: - body = str(entry.get("body") or "").strip() - start = str(entry.get("start") or "").strip() - talent = str(entry.get("talent") or "").strip() - activity_type = str(entry.get("activity_type") or "").strip() - span_id = str(entry.get("span_id") or "").strip() - - if not all((body, start, talent, activity_type, span_id)): - skipped_count += 1 - continue - - ts = base_ts - start_seconds = _start_seconds(start) - if start_seconds is not None and base_ts: - ts = base_ts + start_seconds * 1000 - - end = str(entry.get("end") or "").strip() - time_display = start if not end else f"{start}-{end}" - topics = entry.get("topics", []) - topics_display = ( - ", ".join(str(topic).strip() for topic in topics if str(topic).strip()) - if isinstance(topics, list) - else "" - ) - - confidence = entry.get("confidence") - if isinstance(confidence, (int, float)): - confidence_display = f"{float(confidence):.2f}" - else: - confidence_display = str(confidence or "") - - lines = [f"### {activity_type.capitalize()}: {span_id}\n", ""] - lines.append(f"**Time:** {time_display}") - lines.append(f"**Activity Type:** {activity_type}") - lines.append(f"**Topics:** {topics_display}") - lines.append(f"**Confidence:** {confidence_display}") - lines.append(f"**Talent:** {talent}") - lines.append("") - lines.append(body) - lines.append("") - - chunks.append( - { - "timestamp": ts, - "markdown": "\n".join(lines), - "source": entry, - } - ) - - if skipped_count > 0: - error_msg = f"Skipped {skipped_count} entries missing required fields" - if file_path: - error_msg += f" in {file_path}" - meta["error"] = error_msg - logging.info(error_msg) - - return chunks, meta