diff --git a/solstone/apps/transcripts/call.py b/solstone/apps/transcripts/call.py index a2012da06..c5d9e81b9 100644 --- a/solstone/apps/transcripts/call.py +++ b/solstone/apps/transcripts/call.py @@ -99,7 +99,7 @@ def _speaker_rows(payload: dict) -> list[dict]: "time": chunk.get("time", ""), "text": chunk.get("markdown", ""), "has_embedding": bool(chunk.get("has_embedding")), - "actionable": bool(chunk.get("has_embedding")), + "actionable": bool(chunk.get("speaker_actionable")), "speaker": label, } ) diff --git a/solstone/apps/transcripts/routes.py b/solstone/apps/transcripts/routes.py index b8947eaa6..8bfc26503 100644 --- a/solstone/apps/transcripts/routes.py +++ b/solstone/apps/transcripts/routes.py @@ -977,13 +977,13 @@ def segment_content(day: str, stream: str, segment_key: str) -> Any: labels_data = json.load(f) if not isinstance(labels_data, dict): raise ValueError("speaker labels payload must be an object") - speaker_labels_loaded = True principal = get_journal_principal() principal_id = principal["id"] if principal else None entity_cache: dict[str, dict | None] = {} labels = labels_data.get("labels", []) if not isinstance(labels, list): raise ValueError("speaker labels must be a list") + speaker_labels_loaded = True for label in labels: if not isinstance(label, dict): continue @@ -1118,6 +1118,13 @@ def segment_content(day: str, stream: str, segment_key: str) -> Any: "has_embedding": bool( chunk_sid and chunk_sid in embedding_statement_ids ), + "speaker_actionable": bool( + speaker_labels_present + and speaker_labels_loaded + and labels_source == speaker_source + and chunk_sid + and chunk_sid in embedding_statement_ids + ), "source_ref": { "start": time_str, "source": source.get("source"), diff --git a/solstone/apps/transcripts/tests/test_call.py b/solstone/apps/transcripts/tests/test_call.py index 2e05fb5ce..0821b94be 100644 --- a/solstone/apps/transcripts/tests/test_call.py +++ b/solstone/apps/transcripts/tests/test_call.py @@ -92,6 +92,7 @@ def _speaker_payload() -> dict: "time": "00:00:05", "markdown": "(mic) hello", "has_embedding": True, + "speaker_actionable": True, "speaker_label": { "name": "Romeo Montague", "entity_id": "romeo_montague", @@ -106,7 +107,8 @@ def _speaker_payload() -> dict: "speaker_source": "audio", "time": "00:00:20", "markdown": "(mic) unlabeled", - "has_embedding": False, + "has_embedding": True, + "speaker_actionable": False, }, { "type": "screen", @@ -272,7 +274,7 @@ def test_speakers_json_output_exposes_sentence_ids_and_sources( "speaker_source": "audio", "time": "00:00:20", "text": "(mic) unlabeled", - "has_embedding": False, + "has_embedding": True, "actionable": False, "speaker": None, }, diff --git a/solstone/apps/transcripts/tests/test_copy.py b/solstone/apps/transcripts/tests/test_copy.py index 4da9375ba..2fda44a62 100644 --- a/solstone/apps/transcripts/tests/test_copy.py +++ b/solstone/apps/transcripts/tests/test_copy.py @@ -38,16 +38,11 @@ def test_no_literal_copy_in_templates(): def test_copy_payload_reflects_tr_constants_only(): payload = transcripts_copy_payload() - assert payload["TR_SPEAKER_HEDGE_PROBABLE"] == "probably {name}" - assert payload["TR_SPEAKER_HEDGE_MAYBE"] == "maybe {name}?" - assert payload["TR_SPEAKER_UNKNOWN_CHIP"] == "unknown voice" - assert payload["TR_SPEAKER_PROPAGATION_OFFER"] == ( - "{count} more statements may need this change" - ) - assert payload["TR_SPEAKER_ALREADY_CORRECT"] == "already set" + assert payload assert SPEAKER_LABELS_UNAVAILABLE_MESSAGE not in payload.values() assert SPEAKER_LABEL_SOURCE_AMBIGUOUS_MESSAGE not in payload.values() assert all(name.startswith("TR_") for name in payload) + assert all(isinstance(value, str) for value in payload.values()) def test_all_copy_constants_referenced_by_render_surface(): diff --git a/solstone/apps/transcripts/tests/test_segment_routes.py b/solstone/apps/transcripts/tests/test_segment_routes.py index 180b37ea3..0c97f8ece 100644 --- a/solstone/apps/transcripts/tests/test_segment_routes.py +++ b/solstone/apps/transcripts/tests/test_segment_routes.py @@ -659,6 +659,7 @@ def test_segment_content_adds_speaker_provenance_without_embeddings(client): assert all(chunk["speaker_source"] == "audio" for chunk in audio_chunks) assert all(chunk["source_ref"]["source"] == "mic" for chunk in audio_chunks) assert all(chunk["has_embedding"] is False for chunk in audio_chunks) + assert all(chunk["speaker_actionable"] is False for chunk in audio_chunks) assert audio_chunks[0]["speaker_label"] == { "name": "Romeo Montague", "entity_id": "romeo_montague", @@ -692,9 +693,13 @@ def test_segment_content_marks_seeded_embedding_statement_ids(client, journal_co if chunk["type"] == "audio" } assert audio_by_sid[1]["has_embedding"] is True + assert audio_by_sid[1]["speaker_actionable"] is True assert audio_by_sid[2]["has_embedding"] is False + assert audio_by_sid[2]["speaker_actionable"] is False assert audio_by_sid[4]["has_embedding"] is True + assert audio_by_sid[4]["speaker_actionable"] is True assert audio_by_sid[5]["has_embedding"] is False + assert audio_by_sid[5]["speaker_actionable"] is False def test_segment_content_keeps_missing_confidence_label_as_unknown( @@ -762,6 +767,48 @@ def test_segment_content_malformed_speaker_labels_warns(client, journal_copy): for chunk in data["chunks"] if chunk["type"] == "audio" ) + assert all( + chunk["speaker_actionable"] is False + for chunk in data["chunks"] + if chunk["type"] == "audio" + ) + + +def test_segment_content_structurally_bad_speaker_labels_are_not_loaded( + client, + journal_copy, +): + segment_dir = ( + journal_copy / "chronicle" / FIXTURE_DAY / FIXTURE_STREAM / FIXTURE_SEGMENT + ) + _write_embedding_npz(segment_dir, statement_ids=(1,)) + labels_path = segment_dir / "talents" / "speaker_labels.json" + labels_path.write_text( + json.dumps({"labels": {}, "owner_centroid_last_refreshed_at": "test"}) + "\n", + encoding="utf-8", + ) + + response = client.get( + f"/app/transcripts/api/segment/{FIXTURE_DAY}/{FIXTURE_STREAM}/{FIXTURE_SEGMENT}" + ) + + assert response.status_code == 200 + data = response.get_json() + assert data["speaker_labels"] == { + "present": True, + "loaded": False, + "source": "audio", + "ambiguous": False, + } + assert any( + detail["type"] == "speaker_labels" + and detail["file"] == str(labels_path) + and detail["message"] == SPEAKER_LABELS_UNAVAILABLE_MESSAGE + for detail in data["warning_details"] + ) + audio_chunks = [chunk for chunk in data["chunks"] if chunk["type"] == "audio"] + assert audio_chunks[0]["has_embedding"] is True + assert audio_chunks[0]["speaker_actionable"] is False def test_segment_content_ambiguous_audio_sources_do_not_join_labels( @@ -818,9 +865,68 @@ def test_segment_content_ambiguous_audio_sources_do_not_join_labels( "mic_audio", } assert all(chunk["has_embedding"] is False for chunk in audio_chunks) + assert all(chunk["speaker_actionable"] is False for chunk in audio_chunks) assert all("speaker_label" not in chunk for chunk in audio_chunks) +def test_segment_content_only_labels_source_is_actionable_with_multiple_npz( + client, + journal_copy, +): + day = "20990107" + stream = "default" + segment = "091000_300" + _write_segment(journal_copy, day, stream, segment, screen=False) + segment_dir = journal_copy / "chronicle" / day / stream / segment + _write_jsonl( + segment_dir / "mic_audio.jsonl", + [ + {"raw": "mic_audio.flac"}, + { + "start": "00:00:02", + "source": "mic", + "speaker": 2, + "text": "mic source line", + }, + ], + ) + _write_embedding_npz(segment_dir, source="audio", statement_ids=(1,)) + _write_embedding_npz(segment_dir, source="mic_audio", statement_ids=(1,)) + _write_speaker_labels( + segment_dir, + [ + { + "sentence_id": 1, + "speaker": "romeo_montague", + "confidence": "high", + "method": "owner_centroid", + } + ], + ) + + response = client.get(f"/app/transcripts/api/segment/{day}/{stream}/{segment}") + + assert response.status_code == 200 + data = response.get_json() + assert data["speaker_labels"] == { + "present": True, + "loaded": True, + "source": "audio", + "ambiguous": False, + } + audio_by_source = { + chunk["speaker_source"]: chunk + for chunk in data["chunks"] + if chunk["type"] == "audio" + } + assert audio_by_source["audio"]["has_embedding"] is True + assert audio_by_source["audio"]["speaker_actionable"] is True + assert audio_by_source["audio"]["speaker_label"]["entity_id"] == "romeo_montague" + assert audio_by_source["mic_audio"]["has_embedding"] is True + assert audio_by_source["mic_audio"]["speaker_actionable"] is False + assert "speaker_label" not in audio_by_source["mic_audio"] + + def test_segment_content_invalid_embedding_npz_is_not_route_error( client, journal_copy, @@ -851,6 +957,7 @@ def test_segment_content_invalid_embedding_npz_is_not_route_error( audio_chunks = [chunk for chunk in data["chunks"] if chunk["type"] == "audio"] assert audio_chunks[0]["speaker_label"]["entity_id"] == "romeo_montague" assert audio_chunks[0]["has_embedding"] is False + assert audio_chunks[0]["speaker_actionable"] is False def test_segment_content_merges_browser_between_audio_chunks( diff --git a/solstone/apps/transcripts/tests/test_workspace_html_invariants.py b/solstone/apps/transcripts/tests/test_workspace_html_invariants.py index e8409cb73..95394108b 100644 --- a/solstone/apps/transcripts/tests/test_workspace_html_invariants.py +++ b/solstone/apps/transcripts/tests/test_workspace_html_invariants.py @@ -386,7 +386,8 @@ def test_workspace_html_speaker_picker_markup_and_data_contract(): assert "payload?.speakers" in text assert "payload.success" not in text assert "voices.length > 7" in text - assert 'href="/app/speakers#new-voices"' in text + assert 'href="/app/speakers"' in text + assert "#new-voices" not in text assert "Someone new" not in text @@ -433,6 +434,7 @@ def test_workspace_html_speaker_dispatch_and_local_rerender_contract(): assert "slot.innerHTML = renderSpeakerSlot(chunk" in text assert "renderLoadedSegmentData(data, activeTab)" not in text assert "loadSegmentContent(selectedSegment" not in text + assert "item.speaker_actionable !== true" in text assert "result?.status === 'already_correct'" in text assert "err.reasonCode === 'speaker_voiceprint_busy'" in text assert "err.reasonCode === 'speaker_labels_busy'" in text diff --git a/solstone/apps/transcripts/workspace.html b/solstone/apps/transcripts/workspace.html index 391d52e42..73bf23348 100644 --- a/solstone/apps/transcripts/workspace.html +++ b/solstone/apps/transcripts/workspace.html @@ -4458,6 +4458,9 @@ body.presentation-mode .tr-screen-text { font-size: 16px; padding: 12px 16px; bo if (!item.has_embedding) { return trCopy('TR_SPEAKER_NO_EMBEDDING'); } + if (item.speaker_actionable !== true) { + return trCopy('TR_SPEAKER_ACTION_UNAVAILABLE'); + } return ''; } @@ -4648,7 +4651,7 @@ body.presentation-mode .tr-screen-text { font-size: 16px; padding: 12px 16px; bo
${emptyText ? `
${escapeHtml(emptyText)}
` : ''} - ${escapeHtml(trCopy('TR_SPEAKER_PICKER_DISCOVERY_LINK'))} + ${escapeHtml(trCopy('TR_SPEAKER_PICKER_DISCOVERY_LINK'))}
`; diff --git a/tests/baselines/api/transcripts/segment-detail.json b/tests/baselines/api/transcripts/segment-detail.json index 1a62ae4a0..69a6e1470 100644 --- a/tests/baselines/api/transcripts/segment-detail.json +++ b/tests/baselines/api/transcripts/segment-detail.json @@ -46,86 +46,126 @@ "type": "screen" }, { + "has_embedding": false, "markdown": "(mic) Our first presenter is Juliet Capulet from Capulet Industries, talking about enterprise API architecture.", + "sentence_id": 2, "source_ref": { "source": "mic", "speaker": 1, "start": "00:00:20" }, + "speaker_actionable": false, "speaker_label": { + "acoustic_margin_declined": false, "confidence": "high", + "confidence_state": "high", "entity_id": "romeo_montague", "is_owner": true, - "name": "Romeo Montague" + "method": "owner_centroid", + "name": "Romeo Montague", + "owner_margin_declined": false }, + "speaker_source": "audio", "time": "00:00:20", "timestamp": 1772640020000, "type": "audio" }, { + "has_embedding": false, "markdown": "(mic) Romeo, are you seeing this? Her architecture is almost identical to what we have been building.", + "sentence_id": 5, "source_ref": { "source": "mic", "speaker": 3, "start": "00:01:30" }, + "speaker_actionable": false, "speaker_label": { + "acoustic_margin_declined": false, "confidence": "medium", + "confidence_state": "medium", "entity_id": "mercutio_escalus", "is_owner": false, - "name": "Mercutio Escalus" + "method": "context", + "name": "Mercutio Escalus", + "owner_margin_declined": false }, + "speaker_source": "audio", "time": "00:01:30", "timestamp": 1772640090000, "type": "audio" }, { + "has_embedding": false, "markdown": "(mic) Thank you. Today I want to share our vision for unified API gateways that can serve both enterprise and startup clients.", + "sentence_id": 3, "source_ref": { "source": "mic", "speaker": 2, "start": "00:00:45" }, + "speaker_actionable": false, "speaker_label": { + "acoustic_margin_declined": false, "confidence": "high", + "confidence_state": "high", "entity_id": "juliet_capulet", "is_owner": false, - "name": "Juliet Capulet" + "method": "acoustic", + "name": "Juliet Capulet", + "owner_margin_declined": false }, + "speaker_source": "audio", "time": "00:00:45", "timestamp": 1772640045000, "type": "audio" }, { + "has_embedding": false, "markdown": "(mic) The key insight is that API compatibility does not have to mean lowest common denominator. We can build bridges.", + "sentence_id": 4, "source_ref": { "source": "mic", "speaker": 2, "start": "00:01:10" }, + "speaker_actionable": false, "speaker_label": { + "acoustic_margin_declined": false, "confidence": "medium", + "confidence_state": "medium", "entity_id": "juliet_capulet", "is_owner": false, - "name": "Juliet Capulet" + "method": "acoustic", + "name": "Juliet Capulet", + "owner_margin_declined": false }, + "speaker_source": "audio", "time": "00:01:10", "timestamp": 1772640070000, "type": "audio" }, { + "has_embedding": false, "markdown": "(mic) Welcome everyone to the Denver Tech Summit. We have an incredible lineup today.", + "sentence_id": 1, "source_ref": { "source": "mic", "speaker": 1, "start": "00:00:05" }, + "speaker_actionable": false, "speaker_label": { + "acoustic_margin_declined": false, "confidence": "high", + "confidence_state": "high", "entity_id": "romeo_montague", "is_owner": true, - "name": "Romeo Montague" + "method": "owner_centroid", + "name": "Romeo Montague", + "owner_margin_declined": false }, + "speaker_source": "audio", "time": "00:00:05", "timestamp": 1772640005000, "type": "audio" @@ -157,6 +197,40 @@ "counts": {}, "events": [] }, + "speaker_labels": { + "ambiguous": false, + "loaded": true, + "present": true, + "source": "audio" + }, + "transcripts_copy": { + "TR_SPEAKER_ACTION_UNAVAILABLE": "speaker change unavailable", + "TR_SPEAKER_ALREADY_CORRECT": "already set", + "TR_SPEAKER_ASSIGN_LABEL": "add speaker", + "TR_SPEAKER_CHANGE_LABEL": "change speaker", + "TR_SPEAKER_CONFIDENCE_HIGH": "high confidence", + "TR_SPEAKER_CONFIDENCE_UNKNOWN": "confidence unavailable", + "TR_SPEAKER_CORRECT_BUSY": "speaker files are busy", + "TR_SPEAKER_CORRECT_RETRY": "retry speaker change", + "TR_SPEAKER_HEDGE_MAYBE": "maybe {name}?", + "TR_SPEAKER_HEDGE_PROBABLE": "probably {name}", + "TR_SPEAKER_MARGIN_ACOUSTIC": "close voice match", + "TR_SPEAKER_MARGIN_OWNER": "close owner match", + "TR_SPEAKER_NO_EMBEDDING": "voice sample unavailable", + "TR_SPEAKER_OWNER_IDENTITY_REQUIRED": "set your identity before tagging yourself", + "TR_SPEAKER_OWNER_TOO_CLOSE": "that voice is too close to yours to save there", + "TR_SPEAKER_PICKER_DISCOVERY_LINK": "open speaker discovery", + "TR_SPEAKER_PICKER_EMPTY": "no known voices yet", + "TR_SPEAKER_PICKER_NO_RESULTS": "no matching people", + "TR_SPEAKER_PICKER_OWNER": "this is me", + "TR_SPEAKER_PICKER_SEARCH_PLACEHOLDER": "find a person", + "TR_SPEAKER_PICKER_TITLE": "choose speaker", + "TR_SPEAKER_PROPAGATION_APPLIED": "changes applied", + "TR_SPEAKER_PROPAGATION_APPLY": "apply changes", + "TR_SPEAKER_PROPAGATION_DISMISS": "dismiss", + "TR_SPEAKER_PROPAGATION_OFFER": "{count} more statements may need this change", + "TR_SPEAKER_UNKNOWN_CHIP": "unknown voice" + }, "video_files": {}, "warning_details": [], "warnings": 0 diff --git a/tests/test_transcripts_speakers_source_contract.py b/tests/test_transcripts_speakers_source_contract.py index 8abf58c97..b57546e4e 100644 --- a/tests/test_transcripts_speakers_source_contract.py +++ b/tests/test_transcripts_speakers_source_contract.py @@ -46,7 +46,7 @@ def test_transcripts_speaker_source_resolves_in_speakers_review_cli(journal_copy actionable = [ chunk for chunk in transcripts_payload["chunks"] - if chunk["type"] == "audio" and chunk["has_embedding"] + if chunk["type"] == "audio" and chunk["speaker_actionable"] ] assert actionable