From a2b3081bee6b347c9e1fc64f6a3a8f95294e6fda Mon Sep 17 00:00:00 2001 From: Jer Miller Date: Mon, 20 Jul 2026 10:16:09 -0600 Subject: [PATCH] fix(transcripts): centralize speaker actionability Emit speaker_actionable from the structured segment route after validating speaker labels, and have the reading view and transcripts CLI consume that authoritative field instead of recomputing route readiness. Cover structurally malformed labels, multi-NPZ source mismatches, the working speakers discovery link, and non-snapshot copy discipline in the follow-up tests. --- solstone/apps/transcripts/call.py | 2 +- solstone/apps/transcripts/routes.py | 9 +- solstone/apps/transcripts/tests/test_call.py | 6 +- solstone/apps/transcripts/tests/test_copy.py | 9 +- .../transcripts/tests/test_segment_routes.py | 107 ++++++++++++++++++ .../tests/test_workspace_html_invariants.py | 4 +- solstone/apps/transcripts/workspace.html | 5 +- .../api/transcripts/segment-detail.json | 84 +++++++++++++- ...st_transcripts_speakers_source_contract.py | 2 +- 9 files changed, 209 insertions(+), 19 deletions(-) diff --git a/solstone/apps/transcripts/call.py b/solstone/apps/transcripts/call.py index a2012da06..c5d9e81b9 100644 --- a/solstone/apps/transcripts/call.py +++ b/solstone/apps/transcripts/call.py @@ -99,7 +99,7 @@ def _speaker_rows(payload: dict) -> list[dict]: "time": chunk.get("time", ""), "text": chunk.get("markdown", ""), "has_embedding": bool(chunk.get("has_embedding")), - "actionable": bool(chunk.get("has_embedding")), + "actionable": bool(chunk.get("speaker_actionable")), "speaker": label, } ) diff --git a/solstone/apps/transcripts/routes.py b/solstone/apps/transcripts/routes.py index b8947eaa6..8bfc26503 100644 --- a/solstone/apps/transcripts/routes.py +++ b/solstone/apps/transcripts/routes.py @@ -977,13 +977,13 @@ def segment_content(day: str, stream: str, segment_key: str) -> Any: labels_data = json.load(f) if not isinstance(labels_data, dict): raise ValueError("speaker labels payload must be an object") - speaker_labels_loaded = True principal = get_journal_principal() principal_id = principal["id"] if principal else None entity_cache: dict[str, dict | None] = {} labels = labels_data.get("labels", []) if not isinstance(labels, list): raise ValueError("speaker labels must be a list") + speaker_labels_loaded = True for label in labels: if not isinstance(label, dict): continue @@ -1118,6 +1118,13 @@ def segment_content(day: str, stream: str, segment_key: str) -> Any: "has_embedding": bool( chunk_sid and chunk_sid in embedding_statement_ids ), + "speaker_actionable": bool( + speaker_labels_present + and speaker_labels_loaded + and labels_source == speaker_source + and chunk_sid + and chunk_sid in embedding_statement_ids + ), "source_ref": { "start": time_str, "source": source.get("source"), diff --git a/solstone/apps/transcripts/tests/test_call.py b/solstone/apps/transcripts/tests/test_call.py index 2e05fb5ce..0821b94be 100644 --- a/solstone/apps/transcripts/tests/test_call.py +++ b/solstone/apps/transcripts/tests/test_call.py @@ -92,6 +92,7 @@ def _speaker_payload() -> dict: "time": "00:00:05", "markdown": "(mic) hello", "has_embedding": True, + "speaker_actionable": True, "speaker_label": { "name": "Romeo Montague", "entity_id": "romeo_montague", @@ -106,7 +107,8 @@ def _speaker_payload() -> dict: "speaker_source": "audio", "time": "00:00:20", "markdown": "(mic) unlabeled", - "has_embedding": False, + "has_embedding": True, + "speaker_actionable": False, }, { "type": "screen", @@ -272,7 +274,7 @@ def test_speakers_json_output_exposes_sentence_ids_and_sources( "speaker_source": "audio", "time": "00:00:20", "text": "(mic) unlabeled", - "has_embedding": False, + "has_embedding": True, "actionable": False, "speaker": None, }, diff --git a/solstone/apps/transcripts/tests/test_copy.py b/solstone/apps/transcripts/tests/test_copy.py index 4da9375ba..2fda44a62 100644 --- a/solstone/apps/transcripts/tests/test_copy.py +++ b/solstone/apps/transcripts/tests/test_copy.py @@ -38,16 +38,11 @@ def test_no_literal_copy_in_templates(): def test_copy_payload_reflects_tr_constants_only(): payload = transcripts_copy_payload() - assert payload["TR_SPEAKER_HEDGE_PROBABLE"] == "probably {name}" - assert payload["TR_SPEAKER_HEDGE_MAYBE"] == "maybe {name}?" - assert payload["TR_SPEAKER_UNKNOWN_CHIP"] == "unknown voice" - assert payload["TR_SPEAKER_PROPAGATION_OFFER"] == ( - "{count} more statements may need this change" - ) - assert payload["TR_SPEAKER_ALREADY_CORRECT"] == "already set" + assert payload assert SPEAKER_LABELS_UNAVAILABLE_MESSAGE not in payload.values() assert SPEAKER_LABEL_SOURCE_AMBIGUOUS_MESSAGE not in payload.values() assert all(name.startswith("TR_") for name in payload) + assert all(isinstance(value, str) for value in payload.values()) def test_all_copy_constants_referenced_by_render_surface(): diff --git a/solstone/apps/transcripts/tests/test_segment_routes.py b/solstone/apps/transcripts/tests/test_segment_routes.py index 180b37ea3..0c97f8ece 100644 --- a/solstone/apps/transcripts/tests/test_segment_routes.py +++ b/solstone/apps/transcripts/tests/test_segment_routes.py @@ -659,6 +659,7 @@ def test_segment_content_adds_speaker_provenance_without_embeddings(client): assert all(chunk["speaker_source"] == "audio" for chunk in audio_chunks) assert all(chunk["source_ref"]["source"] == "mic" for chunk in audio_chunks) assert all(chunk["has_embedding"] is False for chunk in audio_chunks) + assert all(chunk["speaker_actionable"] is False for chunk in audio_chunks) assert audio_chunks[0]["speaker_label"] == { "name": "Romeo Montague", "entity_id": "romeo_montague", @@ -692,9 +693,13 @@ def test_segment_content_marks_seeded_embedding_statement_ids(client, journal_co if chunk["type"] == "audio" } assert audio_by_sid[1]["has_embedding"] is True + assert audio_by_sid[1]["speaker_actionable"] is True assert audio_by_sid[2]["has_embedding"] is False + assert audio_by_sid[2]["speaker_actionable"] is False assert audio_by_sid[4]["has_embedding"] is True + assert audio_by_sid[4]["speaker_actionable"] is True assert audio_by_sid[5]["has_embedding"] is False + assert audio_by_sid[5]["speaker_actionable"] is False def test_segment_content_keeps_missing_confidence_label_as_unknown( @@ -762,6 +767,48 @@ def test_segment_content_malformed_speaker_labels_warns(client, journal_copy): for chunk in data["chunks"] if chunk["type"] == "audio" ) + assert all( + chunk["speaker_actionable"] is False + for chunk in data["chunks"] + if chunk["type"] == "audio" + ) + + +def test_segment_content_structurally_bad_speaker_labels_are_not_loaded( + client, + journal_copy, +): + segment_dir = ( + journal_copy / "chronicle" / FIXTURE_DAY / FIXTURE_STREAM / FIXTURE_SEGMENT + ) + _write_embedding_npz(segment_dir, statement_ids=(1,)) + labels_path = segment_dir / "talents" / "speaker_labels.json" + labels_path.write_text( + json.dumps({"labels": {}, "owner_centroid_last_refreshed_at": "test"}) + "\n", + encoding="utf-8", + ) + + response = client.get( + f"/app/transcripts/api/segment/{FIXTURE_DAY}/{FIXTURE_STREAM}/{FIXTURE_SEGMENT}" + ) + + assert response.status_code == 200 + data = response.get_json() + assert data["speaker_labels"] == { + "present": True, + "loaded": False, + "source": "audio", + "ambiguous": False, + } + assert any( + detail["type"] == "speaker_labels" + and detail["file"] == str(labels_path) + and detail["message"] == SPEAKER_LABELS_UNAVAILABLE_MESSAGE + for detail in data["warning_details"] + ) + audio_chunks = [chunk for chunk in data["chunks"] if chunk["type"] == "audio"] + assert audio_chunks[0]["has_embedding"] is True + assert audio_chunks[0]["speaker_actionable"] is False def test_segment_content_ambiguous_audio_sources_do_not_join_labels( @@ -818,9 +865,68 @@ def test_segment_content_ambiguous_audio_sources_do_not_join_labels( "mic_audio", } assert all(chunk["has_embedding"] is False for chunk in audio_chunks) + assert all(chunk["speaker_actionable"] is False for chunk in audio_chunks) assert all("speaker_label" not in chunk for chunk in audio_chunks) +def test_segment_content_only_labels_source_is_actionable_with_multiple_npz( + client, + journal_copy, +): + day = "20990107" + stream = "default" + segment = "091000_300" + _write_segment(journal_copy, day, stream, segment, screen=False) + segment_dir = journal_copy / "chronicle" / day / stream / segment + _write_jsonl( + segment_dir / "mic_audio.jsonl", + [ + {"raw": "mic_audio.flac"}, + { + "start": "00:00:02", + "source": "mic", + "speaker": 2, + "text": "mic source line", + }, + ], + ) + _write_embedding_npz(segment_dir, source="audio", statement_ids=(1,)) + _write_embedding_npz(segment_dir, source="mic_audio", statement_ids=(1,)) + _write_speaker_labels( + segment_dir, + [ + { + "sentence_id": 1, + "speaker": "romeo_montague", + "confidence": "high", + "method": "owner_centroid", + } + ], + ) + + response = client.get(f"/app/transcripts/api/segment/{day}/{stream}/{segment}") + + assert response.status_code == 200 + data = response.get_json() + assert data["speaker_labels"] == { + "present": True, + "loaded": True, + "source": "audio", + "ambiguous": False, + } + audio_by_source = { + chunk["speaker_source"]: chunk + for chunk in data["chunks"] + if chunk["type"] == "audio" + } + assert audio_by_source["audio"]["has_embedding"] is True + assert audio_by_source["audio"]["speaker_actionable"] is True + assert audio_by_source["audio"]["speaker_label"]["entity_id"] == "romeo_montague" + assert audio_by_source["mic_audio"]["has_embedding"] is True + assert audio_by_source["mic_audio"]["speaker_actionable"] is False + assert "speaker_label" not in audio_by_source["mic_audio"] + + def test_segment_content_invalid_embedding_npz_is_not_route_error( client, journal_copy, @@ -851,6 +957,7 @@ def test_segment_content_invalid_embedding_npz_is_not_route_error( audio_chunks = [chunk for chunk in data["chunks"] if chunk["type"] == "audio"] assert audio_chunks[0]["speaker_label"]["entity_id"] == "romeo_montague" assert audio_chunks[0]["has_embedding"] is False + assert audio_chunks[0]["speaker_actionable"] is False def test_segment_content_merges_browser_between_audio_chunks( diff --git a/solstone/apps/transcripts/tests/test_workspace_html_invariants.py b/solstone/apps/transcripts/tests/test_workspace_html_invariants.py index e8409cb73..95394108b 100644 --- a/solstone/apps/transcripts/tests/test_workspace_html_invariants.py +++ b/solstone/apps/transcripts/tests/test_workspace_html_invariants.py @@ -386,7 +386,8 @@ def test_workspace_html_speaker_picker_markup_and_data_contract(): assert "payload?.speakers" in text assert "payload.success" not in text assert "voices.length > 7" in text - assert 'href="/app/speakers#new-voices"' in text + assert 'href="/app/speakers"' in text + assert "#new-voices" not in text assert "Someone new" not in text @@ -433,6 +434,7 @@ def test_workspace_html_speaker_dispatch_and_local_rerender_contract(): assert "slot.innerHTML = renderSpeakerSlot(chunk" in text assert "renderLoadedSegmentData(data, activeTab)" not in text assert "loadSegmentContent(selectedSegment" not in text + assert "item.speaker_actionable !== true" in text assert "result?.status === 'already_correct'" in text assert "err.reasonCode === 'speaker_voiceprint_busy'" in text assert "err.reasonCode === 'speaker_labels_busy'" in text diff --git a/solstone/apps/transcripts/workspace.html b/solstone/apps/transcripts/workspace.html index 391d52e42..73bf23348 100644 --- a/solstone/apps/transcripts/workspace.html +++ b/solstone/apps/transcripts/workspace.html @@ -4458,6 +4458,9 @@ body.presentation-mode .tr-screen-text { font-size: 16px; padding: 12px 16px; bo if (!item.has_embedding) { return trCopy('TR_SPEAKER_NO_EMBEDDING'); } + if (item.speaker_actionable !== true) { + return trCopy('TR_SPEAKER_ACTION_UNAVAILABLE'); + } return ''; } @@ -4648,7 +4651,7 @@ body.presentation-mode .tr-screen-text { font-size: 16px; padding: 12px 16px; bo
${emptyText ? `
${escapeHtml(emptyText)}
` : ''} - ${escapeHtml(trCopy('TR_SPEAKER_PICKER_DISCOVERY_LINK'))} + ${escapeHtml(trCopy('TR_SPEAKER_PICKER_DISCOVERY_LINK'))}
`; diff --git a/tests/baselines/api/transcripts/segment-detail.json b/tests/baselines/api/transcripts/segment-detail.json index 1a62ae4a0..69a6e1470 100644 --- a/tests/baselines/api/transcripts/segment-detail.json +++ b/tests/baselines/api/transcripts/segment-detail.json @@ -46,86 +46,126 @@ "type": "screen" }, { + "has_embedding": false, "markdown": "(mic) Our first presenter is Juliet Capulet from Capulet Industries, talking about enterprise API architecture.", + "sentence_id": 2, "source_ref": { "source": "mic", "speaker": 1, "start": "00:00:20" }, + "speaker_actionable": false, "speaker_label": { + "acoustic_margin_declined": false, "confidence": "high", + "confidence_state": "high", "entity_id": "romeo_montague", "is_owner": true, - "name": "Romeo Montague" + "method": "owner_centroid", + "name": "Romeo Montague", + "owner_margin_declined": false }, + "speaker_source": "audio", "time": "00:00:20", "timestamp": 1772640020000, "type": "audio" }, { + "has_embedding": false, "markdown": "(mic) Romeo, are you seeing this? Her architecture is almost identical to what we have been building.", + "sentence_id": 5, "source_ref": { "source": "mic", "speaker": 3, "start": "00:01:30" }, + "speaker_actionable": false, "speaker_label": { + "acoustic_margin_declined": false, "confidence": "medium", + "confidence_state": "medium", "entity_id": "mercutio_escalus", "is_owner": false, - "name": "Mercutio Escalus" + "method": "context", + "name": "Mercutio Escalus", + "owner_margin_declined": false }, + "speaker_source": "audio", "time": "00:01:30", "timestamp": 1772640090000, "type": "audio" }, { + "has_embedding": false, "markdown": "(mic) Thank you. Today I want to share our vision for unified API gateways that can serve both enterprise and startup clients.", + "sentence_id": 3, "source_ref": { "source": "mic", "speaker": 2, "start": "00:00:45" }, + "speaker_actionable": false, "speaker_label": { + "acoustic_margin_declined": false, "confidence": "high", + "confidence_state": "high", "entity_id": "juliet_capulet", "is_owner": false, - "name": "Juliet Capulet" + "method": "acoustic", + "name": "Juliet Capulet", + "owner_margin_declined": false }, + "speaker_source": "audio", "time": "00:00:45", "timestamp": 1772640045000, "type": "audio" }, { + "has_embedding": false, "markdown": "(mic) The key insight is that API compatibility does not have to mean lowest common denominator. We can build bridges.", + "sentence_id": 4, "source_ref": { "source": "mic", "speaker": 2, "start": "00:01:10" }, + "speaker_actionable": false, "speaker_label": { + "acoustic_margin_declined": false, "confidence": "medium", + "confidence_state": "medium", "entity_id": "juliet_capulet", "is_owner": false, - "name": "Juliet Capulet" + "method": "acoustic", + "name": "Juliet Capulet", + "owner_margin_declined": false }, + "speaker_source": "audio", "time": "00:01:10", "timestamp": 1772640070000, "type": "audio" }, { + "has_embedding": false, "markdown": "(mic) Welcome everyone to the Denver Tech Summit. We have an incredible lineup today.", + "sentence_id": 1, "source_ref": { "source": "mic", "speaker": 1, "start": "00:00:05" }, + "speaker_actionable": false, "speaker_label": { + "acoustic_margin_declined": false, "confidence": "high", + "confidence_state": "high", "entity_id": "romeo_montague", "is_owner": true, - "name": "Romeo Montague" + "method": "owner_centroid", + "name": "Romeo Montague", + "owner_margin_declined": false }, + "speaker_source": "audio", "time": "00:00:05", "timestamp": 1772640005000, "type": "audio" @@ -157,6 +197,40 @@ "counts": {}, "events": [] }, + "speaker_labels": { + "ambiguous": false, + "loaded": true, + "present": true, + "source": "audio" + }, + "transcripts_copy": { + "TR_SPEAKER_ACTION_UNAVAILABLE": "speaker change unavailable", + "TR_SPEAKER_ALREADY_CORRECT": "already set", + "TR_SPEAKER_ASSIGN_LABEL": "add speaker", + "TR_SPEAKER_CHANGE_LABEL": "change speaker", + "TR_SPEAKER_CONFIDENCE_HIGH": "high confidence", + "TR_SPEAKER_CONFIDENCE_UNKNOWN": "confidence unavailable", + "TR_SPEAKER_CORRECT_BUSY": "speaker files are busy", + "TR_SPEAKER_CORRECT_RETRY": "retry speaker change", + "TR_SPEAKER_HEDGE_MAYBE": "maybe {name}?", + "TR_SPEAKER_HEDGE_PROBABLE": "probably {name}", + "TR_SPEAKER_MARGIN_ACOUSTIC": "close voice match", + "TR_SPEAKER_MARGIN_OWNER": "close owner match", + "TR_SPEAKER_NO_EMBEDDING": "voice sample unavailable", + "TR_SPEAKER_OWNER_IDENTITY_REQUIRED": "set your identity before tagging yourself", + "TR_SPEAKER_OWNER_TOO_CLOSE": "that voice is too close to yours to save there", + "TR_SPEAKER_PICKER_DISCOVERY_LINK": "open speaker discovery", + "TR_SPEAKER_PICKER_EMPTY": "no known voices yet", + "TR_SPEAKER_PICKER_NO_RESULTS": "no matching people", + "TR_SPEAKER_PICKER_OWNER": "this is me", + "TR_SPEAKER_PICKER_SEARCH_PLACEHOLDER": "find a person", + "TR_SPEAKER_PICKER_TITLE": "choose speaker", + "TR_SPEAKER_PROPAGATION_APPLIED": "changes applied", + "TR_SPEAKER_PROPAGATION_APPLY": "apply changes", + "TR_SPEAKER_PROPAGATION_DISMISS": "dismiss", + "TR_SPEAKER_PROPAGATION_OFFER": "{count} more statements may need this change", + "TR_SPEAKER_UNKNOWN_CHIP": "unknown voice" + }, "video_files": {}, "warning_details": [], "warnings": 0 diff --git a/tests/test_transcripts_speakers_source_contract.py b/tests/test_transcripts_speakers_source_contract.py index 8abf58c97..b57546e4e 100644 --- a/tests/test_transcripts_speakers_source_contract.py +++ b/tests/test_transcripts_speakers_source_contract.py @@ -46,7 +46,7 @@ def test_transcripts_speaker_source_resolves_in_speakers_review_cli(journal_copy actionable = [ chunk for chunk in transcripts_payload["chunks"] - if chunk["type"] == "audio" and chunk["has_embedding"] + if chunk["type"] == "audio" and chunk["speaker_actionable"] ] assert actionable -- 2.51.2