From d73c9ec117498dda37e0affdbce6a0d763fcfd59 Mon Sep 17 00:00:00 2001 From: Jer Miller Date: Mon, 18 May 2026 01:14:10 -0600 Subject: [PATCH] feat(schemas): wrap 4 Class-B root-array schemas + permanent consumer unwrap (req_bfbdbux6 Lode 2/4) WHAT: Wrap extract, detect_transcript_segment, speaker_attribution, and schedule root arrays in strict-portable one-key objects (frame_ids, segments, attributions, events). Apply the strict-portable transform at every nesting depth: strip banned schema keywords, require all object properties, set additionalProperties:false, and preserve enums, patterns, and nullable unions. CONSUMERS: Update the four single parse-sites to unwrap the wrapper key while permanently tolerating bare-list outputs as a defensive shape-compat shim. Relocate the detect_transcript_segment schema $comment to the _SEGMENT_SCHEMA load-site code comment. GUARDS: Promote the four Class-B ids out of PENDING_PORTABILITY. Rename the provider parity list/test generically and extend it to 17 schemas x 3 providers = 51 cases. Reconcile schema-shape and parse-site tests, including dropping the three stripped-constraint negatives by design. SHIP NOTE: Consumer-completeness sweep found 4 single boundaries and no additional raw-list parse sites. No live path was found by which a stale cached bare-list output file is fed into schedule.post_process() or speaker_attribution.post_process(); the permanent bare-list tolerance is a defensive compatibility shim mandated by scope, not justified by a discoverable stale-cache replay path. CI: make ci green (4922 passed, 11 skipped, 8 deselected, 1 unrelated warning). Co-Authored-By: Claude Opus 4.7 (1M context) --- solstone/observe/extract.py | 6 +- solstone/observe/extract.schema.json | 16 +- solstone/talent/schedule.py | 4 + solstone/talent/schedule.schema.json | 217 ++++++++++-------- solstone/talent/speaker_attribution.py | 4 + .../talent/speaker_attribution.schema.json | 38 ++- solstone/think/detect_transcript.py | 7 +- .../detect_transcript_segment.schema.json | 35 ++- .../test_schema_provider_parity.py | 32 ++- tests/test_detect_transcript.py | 16 ++ tests/test_detect_transcript_schema.py | 16 +- tests/test_extract.py | 15 ++ tests/test_extract_schema.py | 19 +- tests/test_schedule_hook.py | 33 +++ tests/test_schedule_schema.py | 35 ++- tests/test_schema_strict_portability.py | 4 - tests/test_speaker_attribution_hook.py | 29 ++- tests/test_speaker_attribution_schema.py | 66 +++--- 18 files changed, 388 insertions(+), 204 deletions(-) diff --git a/solstone/observe/extract.py b/solstone/observe/extract.py index ef7ecb586..436f3ce5a 100644 --- a/solstone/observe/extract.py +++ b/solstone/observe/extract.py @@ -254,8 +254,12 @@ def _ai_select_frames( temperature=0.3, ) - # Parse response - expecting JSON array of frame_ids + # Parse response - expecting wrapped frame_ids or a bare frame_id list selected_ids = json.loads(response) + # Permanent shape-compat: schema emits {"frame_ids": [...]}; tolerate pre-reshape bare-list outputs defensively. + if isinstance(selected_ids, dict): + selected_ids = selected_ids.get("frame_ids", []) + if not isinstance(selected_ids, list): raise ValueError(f"Expected list of frame_ids, got {type(selected_ids)}") diff --git a/solstone/observe/extract.schema.json b/solstone/observe/extract.schema.json index e8fca4e56..12ab36d72 100644 --- a/solstone/observe/extract.schema.json +++ b/solstone/observe/extract.schema.json @@ -1,5 +1,15 @@ { - "$schema": "https://json-schema.org/draft/2020-12/schema", - "type": "array", - "items": {"type": "integer", "minimum": 0} + "type": "object", + "additionalProperties": false, + "required": [ + "frame_ids" + ], + "properties": { + "frame_ids": { + "type": "array", + "items": { + "type": "integer" + } + } + } } diff --git a/solstone/talent/schedule.py b/solstone/talent/schedule.py index 81a14ca29..88e9e8a8b 100644 --- a/solstone/talent/schedule.py +++ b/solstone/talent/schedule.py @@ -52,6 +52,10 @@ def post_process(result: str, context: dict) -> None: logger.error("schedule hook: failed to parse JSON: %s snippet=%r", exc, snippet) return None + # Permanent shape-compat: schema emits {"events": [...]}; tolerate pre-reshape bare-list outputs defensively. + if isinstance(events, dict): + events = events.get("events", []) + if not isinstance(events, list): logger.error("schedule hook: expected top-level array") return None diff --git a/solstone/talent/schedule.schema.json b/solstone/talent/schedule.schema.json index eb74a929d..c0db2632a 100644 --- a/solstone/talent/schedule.schema.json +++ b/solstone/talent/schedule.schema.json @@ -1,102 +1,129 @@ { - "$schema": "https://json-schema.org/draft/2020-12/schema", - "type": "array", - "items": { - "type": "object", - "additionalProperties": false, - "required": [ - "activity", - "target_date", - "start", - "end", - "title", - "description", - "details", - "participation", - "participation_confidence", - "facet", - "cancelled" - ], - "properties": { - "activity": { - "type": "string", - "enum": [ - "meeting", - "call", - "deadline", - "appointment", - "event", - "travel", - "reminder", - "errand", - "celebration", - "doctor_appointment" - ] - }, - "target_date": { - "type": "string", - "pattern": "^\\d{4}-\\d{2}-\\d{2}$" - }, - "start": { - "type": ["string", "null"], - "pattern": "^\\d{2}:\\d{2}:\\d{2}$" - }, - "end": { - "type": ["string", "null"], - "pattern": "^\\d{2}:\\d{2}:\\d{2}$" - }, - "title": { - "type": "string", - "minLength": 1 - }, - "description": { - "type": "string", - "minLength": 1 - }, - "details": { - "type": "string" - }, - "participation": { - "type": "array", - "items": { - "type": "object", - "additionalProperties": false, - "required": ["name", "role", "source", "confidence", "context"], - "properties": { - "name": { - "type": "string", - "minLength": 1 - }, - "role": { - "type": "string", - "enum": ["attendee", "mentioned"] - }, - "source": { - "type": "string", - "enum": ["voice", "speaker_label", "transcript", "screen", "other"] - }, - "confidence": { - "type": "number", - "minimum": 0, - "maximum": 1 - }, - "context": { - "type": "string" + "type": "object", + "additionalProperties": false, + "required": [ + "events" + ], + "properties": { + "events": { + "type": "array", + "items": { + "type": "object", + "additionalProperties": false, + "required": [ + "activity", + "target_date", + "start", + "end", + "title", + "description", + "details", + "participation", + "participation_confidence", + "facet", + "cancelled" + ], + "properties": { + "activity": { + "type": "string", + "enum": [ + "meeting", + "call", + "deadline", + "appointment", + "event", + "travel", + "reminder", + "errand", + "celebration", + "doctor_appointment" + ] + }, + "target_date": { + "type": "string", + "pattern": "^\\d{4}-\\d{2}-\\d{2}$" + }, + "start": { + "type": [ + "string", + "null" + ], + "pattern": "^\\d{2}:\\d{2}:\\d{2}$" + }, + "end": { + "type": [ + "string", + "null" + ], + "pattern": "^\\d{2}:\\d{2}:\\d{2}$" + }, + "title": { + "type": "string" + }, + "description": { + "type": "string" + }, + "details": { + "type": "string" + }, + "participation": { + "type": "array", + "items": { + "type": "object", + "additionalProperties": false, + "required": [ + "name", + "role", + "source", + "confidence", + "context" + ], + "properties": { + "name": { + "type": "string" + }, + "role": { + "type": "string", + "enum": [ + "attendee", + "mentioned" + ] + }, + "source": { + "type": "string", + "enum": [ + "voice", + "speaker_label", + "transcript", + "screen", + "other" + ] + }, + "confidence": { + "type": "number" + }, + "context": { + "type": "string" + } + } } + }, + "participation_confidence": { + "type": [ + "number", + "null" + ] + }, + "facet": { + "type": "string", + "enum": [ + "__RUNTIME_FACETS__" + ] + }, + "cancelled": { + "type": "boolean" } } - }, - "participation_confidence": { - "type": ["number", "null"], - "minimum": 0, - "maximum": 1 - }, - "facet": { - "type": "string", - "enum": ["__RUNTIME_FACETS__"] - }, - "cancelled": { - "type": "boolean" } } } diff --git a/solstone/talent/speaker_attribution.py b/solstone/talent/speaker_attribution.py index 167107740..8f2c56e59 100644 --- a/solstone/talent/speaker_attribution.py +++ b/solstone/talent/speaker_attribution.py @@ -149,6 +149,10 @@ def post_process(result: str, context: dict) -> str | None: if result: try: parsed = json.loads(result) + # Permanent shape-compat: schema emits {"attributions": [...]}; tolerate pre-reshape bare-list outputs defensively. + if isinstance(parsed, dict): + parsed = parsed.get("attributions", []) + if not isinstance(parsed, list): raise TypeError(f"expected JSON array, got {type(parsed).__name__}") items = parsed diff --git a/solstone/talent/speaker_attribution.schema.json b/solstone/talent/speaker_attribution.schema.json index d4b06e882..595a8ecd0 100644 --- a/solstone/talent/speaker_attribution.schema.json +++ b/solstone/talent/speaker_attribution.schema.json @@ -1,14 +1,32 @@ { - "$schema": "https://json-schema.org/draft/2020-12/schema", - "type": "array", - "items": { - "type": "object", - "additionalProperties": false, - "required": ["sentence_id", "speaker", "reasoning"], - "properties": { - "sentence_id": {"type": "integer"}, - "speaker": {"type": "string", "minLength": 1}, - "reasoning": {"type": "string", "minLength": 1} + "type": "object", + "additionalProperties": false, + "required": [ + "attributions" + ], + "properties": { + "attributions": { + "type": "array", + "items": { + "type": "object", + "additionalProperties": false, + "required": [ + "sentence_id", + "speaker", + "reasoning" + ], + "properties": { + "sentence_id": { + "type": "integer" + }, + "speaker": { + "type": "string" + }, + "reasoning": { + "type": "string" + } + } + } } } } diff --git a/solstone/think/detect_transcript.py b/solstone/think/detect_transcript.py index bcb6838a6..265fbcedd 100644 --- a/solstone/think/detect_transcript.py +++ b/solstone/think/detect_transcript.py @@ -17,6 +17,7 @@ _SEGMENT_SCHEMA = json.loads( encoding="utf-8" ) ) +# Output contract for detect_transcript_segment(). Source of truth is think/detect_transcript_segment.md. # Source of truth is think/detect_transcript_json.md. _JSON_SCHEMA = json.loads( (Path(__file__).parent / "detect_transcript_json.schema.json").read_text( @@ -46,7 +47,7 @@ def parse_segment_boundaries(json_text: str, num_lines: int) -> List[dict]: """Validate and return segment boundaries from ``json_text``. Args: - json_text: JSON array of {"start_at": "HH:MM:SS", "line": N} objects + json_text: Wrapped or bare JSON array of boundary objects. num_lines: Total number of lines in the transcript Returns: @@ -58,6 +59,10 @@ def parse_segment_boundaries(json_text: str, num_lines: int) -> List[dict]: logging.error("Failed to parse JSON response") raise ValueError("invalid JSON") from exc + # Permanent shape-compat: schema emits {"segments": [...]}; tolerate pre-reshape bare-list outputs defensively. + if isinstance(data, dict): + data = data.get("segments", []) + if not isinstance(data, list) or not data: logging.error("JSON response is not a non-empty list") raise ValueError("expected non-empty list") diff --git a/solstone/think/detect_transcript_segment.schema.json b/solstone/think/detect_transcript_segment.schema.json index 2b7fab6b3..14337e9cb 100644 --- a/solstone/think/detect_transcript_segment.schema.json +++ b/solstone/think/detect_transcript_segment.schema.json @@ -1,14 +1,29 @@ { - "$schema": "https://json-schema.org/draft/2020-12/schema", - "$comment": "Output contract for detect_transcript_segment(). Source of truth is think/detect_transcript_segment.md.", - "type": "array", - "items": { - "type": "object", - "additionalProperties": false, - "required": ["start_at", "line"], - "properties": { - "start_at": {"type": "string", "pattern": "^\\d{2}:\\d{2}:\\d{2}$"}, - "line": {"type": "integer", "minimum": 1} + "type": "object", + "additionalProperties": false, + "required": [ + "segments" + ], + "properties": { + "segments": { + "type": "array", + "items": { + "type": "object", + "additionalProperties": false, + "required": [ + "start_at", + "line" + ], + "properties": { + "start_at": { + "type": "string", + "pattern": "^\\d{2}:\\d{2}:\\d{2}$" + }, + "line": { + "type": "integer" + } + } + } } } } diff --git a/tests/integration/test_schema_provider_parity.py b/tests/integration/test_schema_provider_parity.py index 78bdbd7ab..2894212f8 100644 --- a/tests/integration/test_schema_provider_parity.py +++ b/tests/integration/test_schema_provider_parity.py @@ -1,7 +1,7 @@ # SPDX-License-Identifier: AGPL-3.0-only # Copyright (c) 2026 sol pbc -"""Raw-SDK strict schema parity tests for req_bfbdbux6 Class-A schemas.""" +"""Raw-SDK strict schema parity tests for req_bfbdbux6 portable schemas.""" from __future__ import annotations @@ -19,12 +19,17 @@ from solstone.think.models import CLAUDE_SONNET_4, GEMINI_FLASH, GPT_5 REPO_ROOT = Path(__file__).resolve().parents[2] FACET_SENTINEL = "__RUNTIME_FACETS__" -CLASS_A = [ +PORTABLE_SCHEMAS = [ ( "describe", "solstone/observe/describe.schema.json", "Categorize a hypothetical code-editor-with-terminal screenshot.", ), + ( + "extract", + "solstone/observe/extract.schema.json", + "Select frame ids 1 and 3 from hypothetical screen frames. Return frame_ids [1, 3].", + ), ( "meeting", "solstone/observe/categories/meeting.schema.json", @@ -54,6 +59,12 @@ CLASS_A = [ "One entry: start '00:00:05', speaker 'Alice', text 'hi'; " "topics/setting short.", ), + ( + "detect_transcript_segment", + "solstone/think/detect_transcript_segment.schema.json", + "Segment a hypothetical transcript into two boundaries: 12:00:00 line 1 " + "and 12:05:00 line 3.", + ), ( "daily_schedule", "solstone/talent/daily_schedule.schema.json", @@ -78,6 +89,19 @@ CLASS_A = [ "Short story body 's', topics ['t'], confidence 0.5, no commitments/" "closures/decisions.", ), + ( + "speaker_attribution", + "solstone/talent/speaker_attribution.schema.json", + "One attribution: sentence_id 1, speaker Alice, reasoning Introduced herself.", + ), + ( + "schedule", + "solstone/talent/schedule.schema.json", + "One future meeting event: target_date 2026-05-20, start 09:00:00, " + "end 09:30:00, title Planning Call, description Planning call, details " + "Google Meet, one attendee Alice from screen confidence 0.9 context " + "calendar invite, participation_confidence 0.8, facet work, cancelled false.", + ), ( "segment_summary", "solstone/apps/timeline/talent/segment_summary.schema.json", @@ -214,9 +238,9 @@ PROVIDERS: tuple[tuple[str, str, Callable[[dict[str, Any], str, str], str]], ... ) @pytest.mark.parametrize( ("schema_name", "schema_path", "prompt"), - [pytest.param(*schema_case, id=schema_case[0]) for schema_case in CLASS_A], + [pytest.param(*schema_case, id=schema_case[0]) for schema_case in PORTABLE_SCHEMAS], ) -def test_class_a_schema_provider_parity( +def test_schema_provider_parity( provider_name: str, api_key_name: str, caller: Callable[[dict[str, Any], str, str], str], diff --git a/tests/test_detect_transcript.py b/tests/test_detect_transcript.py index 76d9259c4..82a01d49e 100644 --- a/tests/test_detect_transcript.py +++ b/tests/test_detect_transcript.py @@ -84,3 +84,19 @@ def test_detect_transcript_segment(monkeypatch): # Returns list of (start_at, text) tuples assert result == [("14:30:00", "a\nb"), ("14:35:00", "c\nd")] + + +def test_detect_transcript_segment_accepts_wrapped_segments(monkeypatch): + mod = importlib.import_module("solstone.think.detect_transcript") + + def mock_generate(**kwargs): + return ( + '{"segments": [{"start_at": "14:30:00", "line": 1}, ' + '{"start_at": "14:35:00", "line": 3}]}' + ) + + monkeypatch.setattr("solstone.think.models.generate", mock_generate) + + result = mod.detect_transcript_segment("a\nb\nc\nd", "14:30:00") + + assert result == [("14:30:00", "a\nb"), ("14:35:00", "c\nd")] diff --git a/tests/test_detect_transcript_schema.py b/tests/test_detect_transcript_schema.py index 6f0f6c8b0..62ad882cc 100644 --- a/tests/test_detect_transcript_schema.py +++ b/tests/test_detect_transcript_schema.py @@ -44,14 +44,16 @@ def test_detect_transcript_json_schema_file_is_valid_draft_2020_12(): def test_detect_transcript_segment_schema_accepts_and_rejects_expected_values(): schema = _load_detect_transcript_segment_schema() validator = Draft202012Validator(schema) - valid = [{"start_at": "12:34:56", "line": 1}] + valid = {"segments": [{"start_at": "12:34:56", "line": 1}]} assert validator.is_valid(valid) - assert not validator.is_valid([{"start_at": "12:34:56"}]) - assert not validator.is_valid([{"start_at": "12:34", "line": 1}]) - assert not validator.is_valid([{"start_at": "12:34:56", "line": "1"}]) - assert not validator.is_valid([{"start_at": "12:34:56", "line": 0}]) - assert not validator.is_valid([{"start_at": "12:34:56", "line": 1, "extra": "x"}]) + assert not validator.is_valid([{"start_at": "12:34:56", "line": 1}]) + assert not validator.is_valid({"segments": [{"start_at": "12:34:56"}]}) + assert not validator.is_valid({"segments": [{"start_at": "12:34", "line": 1}]}) + assert not validator.is_valid({"segments": [{"start_at": "12:34:56", "line": "1"}]}) + assert not validator.is_valid( + {"segments": [{"start_at": "12:34:56", "line": 1, "extra": "x"}]} + ) def test_detect_transcript_json_schema_accepts_and_rejects_expected_values(): @@ -107,7 +109,7 @@ def test_detect_transcript_segment_passes_schema_to_generate(monkeypatch): def fake_generate(**kwargs): captured.update(kwargs) - return '[{"start_at": "12:00:00", "line": 1}]' + return '{"segments": [{"start_at": "12:00:00", "line": 1}]}' monkeypatch.setattr(models, "generate", fake_generate) diff --git a/tests/test_extract.py b/tests/test_extract.py index 1a62e3420..405c2d5c4 100644 --- a/tests/test_extract.py +++ b/tests/test_extract.py @@ -176,6 +176,21 @@ def test_ai_selection_with_categories(): mock_generate.assert_called_once() +def test_ai_selection_accepts_wrapped_frame_ids(): + """Test AI selection accepts the strict-portable wrapper shape.""" + frames = _make_frames(10) + categories = {"code": {"description": "Code editors"}} + + with patch("solstone.think.models.generate") as mock_generate: + mock_generate.return_value = '{"frame_ids": [1, 3, 5]}' + result = select_frames_for_extraction( + frames, max_extractions=5, categories=categories + ) + + assert result == [1, 3, 5] + mock_generate.assert_called_once() + + def test_ai_selection_filters_invalid_ids(): """Test that AI selection filters out invalid frame IDs.""" frames = _make_frames(5) # IDs 1-5 diff --git a/tests/test_extract_schema.py b/tests/test_extract_schema.py index 066fb5c43..06313986a 100644 --- a/tests/test_extract_schema.py +++ b/tests/test_extract_schema.py @@ -28,15 +28,16 @@ def test_extract_schema_file_is_valid_draft_2020_12(): def test_extract_schema_accepts_and_rejects_expected_values(): validator = Draft202012Validator(_SCHEMA) - assert validator.is_valid([]) - assert validator.is_valid([1, 15, 42, 89]) - assert validator.is_valid([1, 0]) - assert not validator.is_valid(["1"]) - assert not validator.is_valid([-1]) - assert not validator.is_valid([1.5]) + assert validator.is_valid({"frame_ids": []}) + assert validator.is_valid({"frame_ids": [1, 15, 42, 89]}) + assert validator.is_valid({"frame_ids": [1, 0]}) + assert not validator.is_valid([1]) + assert not validator.is_valid({"frame_ids": ["1"]}) + assert not validator.is_valid({"frame_ids": [1.5]}) assert not validator.is_valid(42) - assert not validator.is_valid({"ids": [1]}) - assert not validator.is_valid([[1, 2]]) + assert not validator.is_valid({}) + assert not validator.is_valid({"frame_ids": [1], "ids": [1]}) + assert not validator.is_valid({"frame_ids": [[1, 2]]}) def test_ai_select_frames_passes_schema_to_generate(monkeypatch): @@ -44,7 +45,7 @@ def test_ai_select_frames_passes_schema_to_generate(monkeypatch): def fake_generate(**kwargs): captured.update(kwargs) - return "[1]" + return '{"frame_ids": [1]}' monkeypatch.setattr(models, "generate", fake_generate) diff --git a/tests/test_schedule_hook.py b/tests/test_schedule_hook.py index 65b8dc61e..5964af851 100644 --- a/tests/test_schedule_hook.py +++ b/tests/test_schedule_hook.py @@ -106,6 +106,39 @@ def test_schedule_post_process_writes_record_and_resolves_entities( assert record["edits"][-1]["note"] == "created by schedule" +def test_schedule_post_process_accepts_wrapped_events(tmp_path, monkeypatch): + from solstone.talent.schedule import post_process + from solstone.think.activities import load_activity_records + + monkeypatch.setenv("SOLSTONE_JOURNAL", str(tmp_path)) + _write_facet(tmp_path, "work") + + payload = { + "events": [ + { + "activity": "meeting", + "target_date": "2026-04-23", + "start": "09:00:00", + "end": "09:30:00", + "title": "Planning call", + "description": "Planning call with the team.", + "details": "Google Meet", + "participation": [], + "participation_confidence": 0.8, + "facet": "work", + "cancelled": False, + } + ] + } + + post_process(json.dumps(payload), {"day": "20260418"}) + + records = load_activity_records("work", "20260423", include_hidden=True) + assert len(records) == 1 + assert records[0]["id"] == "anticipated_meeting_090000_0423" + assert records[0]["source"] == "anticipated" + + def test_schedule_post_process_marks_cancelled_records_hidden(tmp_path, monkeypatch): from solstone.talent.schedule import post_process from solstone.think.activities import load_activity_records diff --git a/tests/test_schedule_schema.py b/tests/test_schedule_schema.py index 313a3be43..264c3decd 100644 --- a/tests/test_schedule_schema.py +++ b/tests/test_schedule_schema.py @@ -42,15 +42,10 @@ def _load_schedule_schema() -> dict: return schema -def _strip_portability_annotations(value): - if isinstance(value, dict): - for key in ("minLength", "minimum", "maximum"): - value.pop(key, None) - for child in value.values(): - _strip_portability_annotations(child) - elif isinstance(value, list): - for child in value: - _strip_portability_annotations(child) +def _schedule_event_schema(schema: dict | None = None) -> dict: + if schema is None: + schema = _load_schedule_schema() + return schema["properties"]["events"]["items"] def _expected_schedule_activity_ids() -> set[str]: @@ -161,14 +156,14 @@ def test_schedule_talent_loads_schema(): def test_schedule_schema_facet_uses_runtime_sentinel_constant(): schema = _load_schedule_schema() - facet_schema = schema["items"]["properties"]["facet"] + facet_schema = _schedule_event_schema(schema)["properties"]["facet"] assert facet_schema["enum"] == [RUNTIME_FACETS_SENTINEL] def test_schedule_activity_enum_matches_default_activities_drift_detector(): schema = _load_schedule_schema() - item_schema = schema["items"] + item_schema = _schedule_event_schema(schema) assert set(item_schema["properties"]["activity"]["enum"]) == ( _expected_schedule_activity_ids() @@ -189,26 +184,27 @@ def test_schedule_participation_entry_diverges_from_shared_fragment(): ] raw_inline_items = dict( - schedule_schema["items"]["properties"]["participation"]["items"] + _schedule_event_schema(schedule_schema)["properties"]["participation"]["items"] ) assert "entity_id" in fragment["properties"] assert "entity_id" not in raw_inline_items["properties"] assert raw_inline_items != fragment - inline_items = json.loads(json.dumps(raw_inline_items)) - _strip_portability_annotations(inline_items) - - assert inline_items == fragment_without_schema + assert raw_inline_items == fragment_without_schema def test_schedule_schema_mirrors_hook_requirements(): schedule_schema = _load_schedule_schema() - item_schema = schedule_schema["items"] + events_schema = schedule_schema["properties"]["events"] + item_schema = events_schema["items"] properties = item_schema["properties"] participation_items = properties["participation"]["items"] fragment = _load_json(PARTICIPATION_ENTRY_SCHEMA_PATH) - assert schedule_schema["type"] == "array" + assert schedule_schema["type"] == "object" + assert schedule_schema["additionalProperties"] is False + assert schedule_schema["required"] == ["events"] + assert events_schema["type"] == "array" assert set(item_schema["required"]) == SCHEDULE_REQUIRED_FIELDS assert set(properties["activity"]["enum"]) == _expected_schedule_activity_ids() assert ( @@ -229,4 +225,5 @@ def test_schedule_hook_fixtures_validate_against_schema(monkeypatch): validator = Draft202012Validator(hydrate_runtime_enums(_load_schedule_schema())) for payload in _sample_schedule_payloads(): - assert list(validator.iter_errors(payload)) == [] + assert list(validator.iter_errors({"events": payload})) == [] + assert list(validator.iter_errors(payload)) != [] diff --git a/tests/test_schema_strict_portability.py b/tests/test_schema_strict_portability.py index 5061f1c32..17aaf77f9 100644 --- a/tests/test_schema_strict_portability.py +++ b/tests/test_schema_strict_portability.py @@ -35,10 +35,6 @@ BANNED_KEYS = frozenset( # portabilized; Lode 4 deletes this allowlist mechanism entirely. PENDING_PORTABILITY = frozenset( { - "solstone/observe/extract.schema.json", - "solstone/talent/schedule.schema.json", - "solstone/talent/speaker_attribution.schema.json", - "solstone/think/detect_transcript_segment.schema.json", "solstone/talent/chat.schema.json", "build_rollup_schema(3)", } diff --git a/tests/test_speaker_attribution_hook.py b/tests/test_speaker_attribution_hook.py index 3dbd7c57e..7c861f921 100644 --- a/tests/test_speaker_attribution_hook.py +++ b/tests/test_speaker_attribution_hook.py @@ -214,9 +214,17 @@ class TestPostProcess: } accumulate_mock.assert_not_called() - def test_wrapper_shape_yields_zero_merges(self, tmp_path): + def test_wrapped_attributions_merge_layer4_attributions(self, tmp_path): result = json.dumps( - {"attributions": [{"sentence_id": 1, "speaker": "Alice", "reasoning": "x"}]} + { + "attributions": [ + { + "sentence_id": 1, + "speaker": "Alice", + "reasoning": "said her name", + } + ] + } ) context = _post_process_context() @@ -224,15 +232,17 @@ class TestPostProcess: patch( "solstone.apps.speakers.attribution.save_speaker_labels" ) as save_mock, - patch("solstone.apps.speakers.attribution.accumulate_voiceprints"), + patch( + "solstone.apps.speakers.attribution.accumulate_voiceprints" + ) as accumulate_mock, patch( "solstone.think.entities.find_matching_entity", side_effect=_match_entity, - ) as match_mock, + ), patch( "solstone.think.entities.journal.load_all_journal_entities", return_value={"alice": {"id": "alice"}}, - ) as load_mock, + ), patch("solstone.think.utils.segment_path", return_value=tmp_path), ): from solstone.talent.speaker_attribution import post_process @@ -242,9 +252,9 @@ class TestPostProcess: saved_labels = save_mock.call_args[0][1] assert saved_labels[0] == { "sentence_id": 1, - "speaker": None, - "confidence": None, - "method": None, + "speaker": "alice", + "confidence": "medium", + "method": "contextual", } assert saved_labels[1] == { "sentence_id": 2, @@ -252,8 +262,7 @@ class TestPostProcess: "confidence": "high", "method": "owner", } - load_mock.assert_not_called() - match_mock.assert_not_called() + accumulate_mock.assert_not_called() def test_non_list_non_dict_yields_zero_merges_and_warns(self, tmp_path, caplog): context = _post_process_context() diff --git a/tests/test_speaker_attribution_schema.py b/tests/test_speaker_attribution_schema.py index cf2f890d6..e9d4996be 100644 --- a/tests/test_speaker_attribution_schema.py +++ b/tests/test_speaker_attribution_schema.py @@ -32,11 +32,25 @@ def test_speaker_attribution_talent_loads_schema(): @pytest.mark.parametrize( "payload", [ - [{"sentence_id": 1, "speaker": "Alice", "reasoning": "Introduced herself."}], - [ - {"sentence_id": 1, "speaker": "Alice", "reasoning": "Introduced herself."}, - {"sentence_id": 2, "speaker": "Bob", "reasoning": "Replied to Alice."}, - ], + { + "attributions": [ + { + "sentence_id": 1, + "speaker": "Alice", + "reasoning": "Introduced herself.", + } + ] + }, + { + "attributions": [ + { + "sentence_id": 1, + "speaker": "Alice", + "reasoning": "Introduced herself.", + }, + {"sentence_id": 2, "speaker": "Bob", "reasoning": "Replied to Alice."}, + ] + }, ], ) def test_positive_payload_validates(payload): @@ -45,39 +59,27 @@ def test_positive_payload_validates(payload): assert validator.is_valid(payload) -def test_negative_wrapper_object_rejected(): +def test_negative_bare_array_rejected(): validator = Draft202012Validator(_load_schema()) assert not validator.is_valid( - { - "attributions": [ - { - "sentence_id": 1, - "speaker": "Alice", - "reasoning": "Introduced herself.", - } - ] - } + [{"sentence_id": 1, "speaker": "Alice", "reasoning": "Introduced herself."}] ) def test_negative_missing_required_field_rejected(): validator = Draft202012Validator(_load_schema()) - assert not validator.is_valid([{"sentence_id": 1, "speaker": "Alice"}]) - - -def test_negative_empty_string_fields_rejected(): - validator = Draft202012Validator(_load_schema()) - - assert not validator.is_valid([{"sentence_id": 1, "speaker": "", "reasoning": "x"}]) + assert not validator.is_valid( + {"attributions": [{"sentence_id": 1, "speaker": "Alice"}]} + ) def test_negative_non_integer_sentence_id_rejected(): validator = Draft202012Validator(_load_schema()) assert not validator.is_valid( - [{"sentence_id": "1", "speaker": "Alice", "reasoning": "x"}] + {"attributions": [{"sentence_id": "1", "speaker": "Alice", "reasoning": "x"}]} ) @@ -85,12 +87,14 @@ def test_negative_additional_properties_rejected(): validator = Draft202012Validator(_load_schema()) assert not validator.is_valid( - [ - { - "sentence_id": 1, - "speaker": "Alice", - "reasoning": "x", - "confidence": "high", - } - ] + { + "attributions": [ + { + "sentence_id": 1, + "speaker": "Alice", + "reasoning": "x", + "confidence": "high", + } + ] + } ) -- 2.51.2