diff --git a/solstone/talent/conversation.md b/solstone/talent/conversation.md index 218d1fa12..a619d02af 100644 --- a/solstone/talent/conversation.md +++ b/solstone/talent/conversation.md @@ -9,6 +9,7 @@ "output": "json", "schema": "story.schema.json", "hook": {"post": "story"}, + "degradation_check": true, "load": { "transcripts": true, "percepts": true, diff --git a/solstone/talent/documents.md b/solstone/talent/documents.md index fdbc765b3..601d1eeb3 100644 --- a/solstone/talent/documents.md +++ b/solstone/talent/documents.md @@ -10,6 +10,7 @@ "thinking_budget": 8192, "max_output_tokens": 8192, "output": "md", + "degradation_check": true, "load": {"transcripts": true, "percepts": false, "talents": false} } diff --git a/solstone/talent/event.md b/solstone/talent/event.md index 52bae355b..54cbbc3df 100644 --- a/solstone/talent/event.md +++ b/solstone/talent/event.md @@ -9,6 +9,7 @@ "output": "json", "schema": "story.schema.json", "hook": {"post": "story"}, + "degradation_check": true, "load": { "transcripts": true, "percepts": true, diff --git a/solstone/talent/morning_briefing.md b/solstone/talent/morning_briefing.md index d639a6b2a..e78439246 100644 --- a/solstone/talent/morning_briefing.md +++ b/solstone/talent/morning_briefing.md @@ -7,6 +7,7 @@ "schedule": "daily", "priority": 50, "output": "md", + "degradation_check": true, "read_scope": ["chronicle/", "facets", "entities", "imports", "health", "identity"] } diff --git a/solstone/talent/weekly_reflection.md b/solstone/talent/weekly_reflection.md index ee2df7540..905395e02 100644 --- a/solstone/talent/weekly_reflection.md +++ b/solstone/talent/weekly_reflection.md @@ -5,6 +5,7 @@ "schedule": "weekly", "priority": 90, "output": "md", + "degradation_check": true, "read_scope_span": 7, "max_turns": 100 } diff --git a/solstone/talent/work.md b/solstone/talent/work.md index c28e9ec3d..e685a8cc2 100644 --- a/solstone/talent/work.md +++ b/solstone/talent/work.md @@ -9,6 +9,7 @@ "output": "json", "schema": "story.schema.json", "hook": {"post": "story"}, + "degradation_check": true, "load": { "transcripts": true, "percepts": true, diff --git a/solstone/think/cortex.py b/solstone/think/cortex.py index 33a44575e..e4864b3eb 100644 --- a/solstone/think/cortex.py +++ b/solstone/think/cortex.py @@ -708,6 +708,7 @@ class CortexService: thinking_count = 0 tool_count = 0 finish_usage = None + degraded = None error_message = None model = None runtime_seconds = None @@ -732,6 +733,7 @@ class CortexService: if event_type == "finish": status = "completed" finish_usage = event.get("usage") + degraded = event.get("degraded") end_ts = event.get("ts", 0) if end_ts and start_ts: runtime_seconds = round((end_ts - start_ts) / 1000.0, 1) @@ -762,6 +764,7 @@ class CortexService: "tool_count": tool_count, "cost": calc_agent_cost(model, finish_usage), "error_message": error_message if status == "error" else None, + "degraded": degraded, "output_file": self._summarize_output_file(request), "prompt": request.get("prompt", ""), } diff --git a/solstone/think/surfaces/health.py b/solstone/think/surfaces/health.py index 68fb0d200..21e205106 100644 --- a/solstone/think/surfaces/health.py +++ b/solstone/think/surfaces/health.py @@ -43,6 +43,7 @@ LEDGER_STALE_DAYS = 14 # 14d mirrors the consumer-signal stale-item threshold so the health surface stays aligned with ledger backlog review. USER_EDIT_ACTOR_PREFIXES = ("cli:", "owner", "user") # These prefixes identify operator- or user-authored corrections without trying to enumerate every internal automation actor string. +DEGRADED_OUTPUT_NOTE_CAP = 10 _DAY_MS = 86_400_000 _HOUR_MS = 3_600_000 _SPEC_POINTER = "cpo/specs/in-flight/consumer-surface-health.md" @@ -370,8 +371,10 @@ def _build_synthesis_health( talent_rows.append(payload) talent_run_failures_24h: int | None + talent_degraded_outputs_24h: int | None if missing_talent_days: talent_run_failures_24h = None + talent_degraded_outputs_24h = None notes.append( HealthNote( severity="info", @@ -387,6 +390,8 @@ def _build_synthesis_health( ) else: talent_run_failures_24h = 0 + talent_degraded_outputs_24h = 0 + degraded_rows: list[tuple[int, dict[str, Any]]] = [] cutoff = generated_at - _DAY_MS for row in talent_rows: try: @@ -398,6 +403,35 @@ def _build_synthesis_health( status = row.get("status") if row.get("error") or status not in ("ok", "completed", None): talent_run_failures_24h += 1 + if row.get("degraded"): + talent_degraded_outputs_24h += 1 + degraded_rows.append((timestamp, row)) + + for _timestamp, row in sorted( + degraded_rows, + key=lambda item: item[0], + reverse=True, + )[:DEGRADED_OUTPUT_NOTE_CAP]: + degraded = row.get("degraded") + if not isinstance(degraded, dict): + continue + output_tokens = degraded.get("output_tokens") + name = row.get("name") + provider = row.get("provider") + model = row.get("model") + day = row.get("day") + notes.append( + HealthNote( + severity="warn", + category="synthesis", + message=( + f"talent '{name}' finished near-empty: {output_tokens} " + f"output tokens ({provider}/{model}) on {day}" + ), + detected_at=generated_at, + detail_pointer=None, + ) + ) indexer_path = Path(get_journal()) / "indexer" / "journal.sqlite" if not indexer_path.exists(): @@ -436,6 +470,7 @@ def _build_synthesis_health( activities_user_edited=aggregate.activities_user_edited, activities_anticipated_unfilled=aggregate.activities_anticipated_unfilled, talent_run_failures_24h=talent_run_failures_24h, + talent_degraded_outputs_24h=talent_degraded_outputs_24h, indexer_last_rebuild_at=indexer_last_rebuild_at, ), notes, diff --git a/solstone/think/surfaces/types.py b/solstone/think/surfaces/types.py index 4be87f80f..bd990cb8b 100644 --- a/solstone/think/surfaces/types.py +++ b/solstone/think/surfaces/types.py @@ -97,6 +97,7 @@ class SynthesisHealth: activities_user_edited: int activities_anticipated_unfilled: int talent_run_failures_24h: int | None + talent_degraded_outputs_24h: int | None indexer_last_rebuild_at: int | None diff --git a/solstone/think/talents.py b/solstone/think/talents.py index 70a573190..9910f82a2 100644 --- a/solstone/think/talents.py +++ b/solstone/think/talents.py @@ -58,6 +58,8 @@ LOG = logging.getLogger("solstone.think.talents") # Minimum content length for transcript-based generation MIN_INPUT_CHARS = 50 +# Minimum model output tokens before a degradation-checked talent run is flagged near-empty +MIN_OUTPUT_TOKENS = 300 def setup_logging(verbose: bool = False) -> logging.Logger: @@ -839,6 +841,25 @@ def _should_fallback(exc: Exception) -> bool: return _is_retryable_error(exc) or isinstance(exc, QuotaExhaustedError) +def _classify_degraded(usage: dict | None, config: dict) -> dict | None: + """Flag an opted-in talent run whose model produced near-zero output. + + Opt-in via the talent's `degradation_check` frontmatter flag. Returns a + marker dict for the finish event, or None when not degraded / not checked / + output-token count unknown (never alarm without a numeric count). + """ + if not config.get("degradation_check"): + return None + if not usage: + return None + tokens = usage.get("output_tokens") + if isinstance(tokens, bool) or not isinstance(tokens, (int, float)): + return None + if tokens < MIN_OUTPUT_TOKENS: + return {"reason": "near_empty", "output_tokens": int(tokens)} + return None + + async def _execute_with_tools( config: dict, emit_event: Callable[[dict], None], @@ -865,8 +886,18 @@ async def _execute_with_tools( if data.get("event") == "finish": result = data.get("result", "") result = _run_post_hooks(result, config) + + updates: dict[str, Any] = {} if result != data.get("result", ""): - data = {**data, "result": result} + updates["result"] = result + + degraded = _classify_degraded(data.get("usage"), config) + if degraded: + updates["degraded"] = degraded + + if updates: + data = {**data, **updates} + if output_path and result: _write_output(output_path, result) @@ -1108,6 +1139,9 @@ async def _execute_generate( finish_event["usage"] = usage_data if "schema_validation" in gen_result: finish_event["schema_validation"] = gen_result["schema_validation"] + degraded = _classify_degraded(usage_data, config) + if degraded: + finish_event["degraded"] = degraded emit_event(finish_event) diff --git a/solstone/think/tools/health.py b/solstone/think/tools/health.py index d872e56e3..e8f46d050 100644 --- a/solstone/think/tools/health.py +++ b/solstone/think/tools/health.py @@ -65,6 +65,10 @@ def _render_summary(report: HealthReport) -> None: " talent_run_failures_24h: " + str(report.synthesis_health.talent_run_failures_24h) ) + typer.echo( + " talent_degraded_outputs_24h: " + + str(report.synthesis_health.talent_degraded_outputs_24h) + ) typer.echo( " indexer_last_rebuild_at: " + str(report.synthesis_health.indexer_last_rebuild_at) diff --git a/tests/baselines/api/stats/stats.json b/tests/baselines/api/stats/stats.json index 4fc2f07bb..059809218 100644 --- a/tests/baselines/api/stats/stats.json +++ b/tests/baselines/api/stats/stats.json @@ -25,6 +25,7 @@ "email" ], "color": "#00796b", + "degradation_check": true, "description": "Generates a conversation story, topics, and structured commitments, closures, and decisions to merge onto the activity record.", "hook": { "post": "story" @@ -70,6 +71,7 @@ }, "documents": { "color": "#5c6bc0", + "degradation_check": true, "description": "Extracts structured intelligence from imported documents", "hook": { "pre": "documents" @@ -150,6 +152,7 @@ "reminder" ], "color": "#ff7043", + "degradation_check": true, "description": "Generates an event story, topics, and structured commitments, closures, and decisions to merge onto the activity record.", "hook": { "post": "story" @@ -332,6 +335,7 @@ "reading" ], "color": "#6d4c41", + "degradation_check": true, "description": "Generates a work story, topics, and structured commitments, closures, and decisions to merge onto the activity record.", "hook": { "post": "story" diff --git a/tests/test_cortex.py b/tests/test_cortex.py index 7c687cecd..20e0e73a1 100644 --- a/tests/test_cortex.py +++ b/tests/test_cortex.py @@ -676,6 +676,50 @@ def test_complete_use_file_no_name(cortex_service, mock_journal): assert not any(path.is_symlink() for path in (mock_journal / "talents").iterdir()) +def test_append_day_index_preserves_degraded_marker(cortex_service, mock_journal): + """Test day-index summaries carry degraded finish markers.""" + use_id = "1234567890000" + completed_path = mock_journal / "talents" / f"{use_id}.jsonl" + completed_path.write_text( + json.dumps( + { + "event": "start", + "ts": 1000, + "model": "claude-haiku-4-5", + } + ) + + "\n" + + json.dumps( + { + "event": "finish", + "ts": 2000, + "result": "x", + "usage": { + "input_tokens": 10, + "output_tokens": 7, + "total_tokens": 17, + }, + "degraded": {"reason": "near_empty", "output_tokens": 7}, + } + ) + + "\n", + encoding="utf-8", + ) + request = { + "name": "morning_briefing", + "day": "20260410", + "ts": 1000, + "provider": "anthropic", + "model": "claude-haiku-4-5", + } + + cortex_service._append_day_index(use_id, request, completed_path) + + day_index_path = mock_journal / "talents" / "20260410.jsonl" + row = json.loads(day_index_path.read_text(encoding="utf-8").strip()) + assert row["degraded"] == {"reason": "near_empty", "output_tokens": 7} + + def test_write_error_and_complete(cortex_service, mock_journal): """Test writing error and completing file.""" use_id = "123456789" diff --git a/tests/test_surfaces_health.py b/tests/test_surfaces_health.py index 707e8e14d..7acbc19a1 100644 --- a/tests/test_surfaces_health.py +++ b/tests/test_surfaces_health.py @@ -572,6 +572,7 @@ def test_missing_talent_day_indexes_emit_info(tmp_path, monkeypatch): report = health_surface.summary("20260410") assert report.synthesis_health.talent_run_failures_24h is None + assert report.synthesis_health.talent_degraded_outputs_24h is None assert any( note.category == "synthesis" and note.severity == "info" @@ -580,6 +581,97 @@ def test_missing_talent_day_indexes_emit_info(tmp_path, monkeypatch): ) +def test_degraded_talent_outputs_count_and_warn_without_failure(tmp_path, monkeypatch): + _configure_env(tmp_path, monkeypatch) + _set_now(monkeypatch, _utc_dt("20260410")) + _minimal_facet_tree(tmp_path) + _write_talent_day( + tmp_path, + "20260410", + { + "use_id": "1", + "name": "morning_briefing", + "day": "20260410", + "facet": None, + "ts": _utc_ms("20260410", 9), + "status": "completed", + "provider": "openai", + "model": "gpt-5", + "degraded": {"reason": "near_empty", "output_tokens": 12}, + }, + ) + _write_talent_day( + tmp_path, + "20260409", + { + "use_id": "2", + "name": "weekly_reflection", + "day": "20260409", + "facet": None, + "ts": _utc_ms("20260409", 15), + "status": "completed", + "provider": "anthropic", + "model": "claude-haiku-4-5", + }, + ) + + report = health_surface.summary("20260410") + + assert report.synthesis_health.talent_degraded_outputs_24h == 1 + assert report.synthesis_health.talent_run_failures_24h == 0 + assert any( + note.category == "synthesis" + and note.severity == "warn" + and "morning_briefing" in note.message + and "12" in note.message + for note in report.notes + ) + + +def test_healthy_talent_outputs_do_not_emit_degraded_notes(tmp_path, monkeypatch): + _configure_env(tmp_path, monkeypatch) + _set_now(monkeypatch, _utc_dt("20260410")) + _minimal_facet_tree(tmp_path) + _write_talent_day( + tmp_path, + "20260410", + { + "use_id": "1", + "name": "morning_briefing", + "day": "20260410", + "facet": None, + "ts": _utc_ms("20260410", 9), + "status": "completed", + "provider": "openai", + "model": "gpt-5", + }, + ) + _write_talent_day( + tmp_path, + "20260409", + { + "use_id": "2", + "name": "weekly_reflection", + "day": "20260409", + "facet": None, + "ts": _utc_ms("20260409", 15), + "status": "completed", + "provider": "anthropic", + "model": "claude-haiku-4-5", + }, + ) + + report = health_surface.summary("20260410") + + assert report.synthesis_health.talent_degraded_outputs_24h == 0 + assert not any( + note.category == "synthesis" + and note.severity == "warn" + and "finished near-empty" in note.message + for note in report.notes + ) + + def test_for_range_defaults_to_last_7_days_ending_today(tmp_path, monkeypatch): _configure_env(tmp_path, monkeypatch) _set_now(monkeypatch, _utc_dt("20260410")) diff --git a/tests/test_talents_degradation.py b/tests/test_talents_degradation.py new file mode 100644 index 000000000..f3ac76276 --- /dev/null +++ b/tests/test_talents_degradation.py @@ -0,0 +1,48 @@ +# SPDX-License-Identifier: AGPL-3.0-only +# Copyright (c) 2026 sol pbc + +from solstone.think.talents import _classify_degraded + + +def test_classify_degraded_marks_opted_in_low_output_tokens(): + assert _classify_degraded( + {"output_tokens": 12}, + {"degradation_check": True}, + ) == {"reason": "near_empty", "output_tokens": 12} + + +def test_classify_degraded_marks_zero_output_tokens(): + assert _classify_degraded( + {"output_tokens": 0}, + {"degradation_check": True}, + ) == {"reason": "near_empty", "output_tokens": 0} + + +def test_classify_degraded_ignores_high_output_tokens(): + assert ( + _classify_degraded( + {"output_tokens": 5000}, + {"degradation_check": True}, + ) + is None + ) + + +def test_classify_degraded_ignores_unchecked_talent(): + assert _classify_degraded({"output_tokens": 12}, {}) is None + + +def test_classify_degraded_ignores_missing_usage(): + assert _classify_degraded(None, {"degradation_check": True}) is None + + +def test_classify_degraded_ignores_missing_output_tokens(): + assert _classify_degraded({"input_tokens": 10}, {"degradation_check": True}) is None + + +def test_classify_degraded_ignores_non_numeric_output_tokens(): + config = {"degradation_check": True} + + assert _classify_degraded({"output_tokens": "12"}, config) is None + assert _classify_degraded({"output_tokens": None}, config) is None + assert _classify_degraded({"output_tokens": True}, config) is None