diff --git a/solstone/apps/entities/talent/entity_observer.schema.json b/solstone/apps/entities/talent/entity_observer.schema.json index 5d54a24af..2203ebf81 100644 --- a/solstone/apps/entities/talent/entity_observer.schema.json +++ b/solstone/apps/entities/talent/entity_observer.schema.json @@ -1,34 +1,40 @@ { - "$schema": "https://json-schema.org/draft/2020-12/schema", "type": "object", "additionalProperties": false, - "required": ["observations", "skipped", "summary"], + "required": [ + "observations", + "skipped", + "summary" + ], "properties": { "observations": { "type": "array", "items": { "type": "object", "additionalProperties": false, - "required": ["entity_id", "items"], + "required": [ + "entity_id", + "items" + ], "properties": { "entity_id": { - "type": "string", - "minLength": 1 + "type": "string" }, "items": { "type": "array", "items": { "type": "object", "additionalProperties": false, - "required": ["content", "reasoning"], + "required": [ + "content", + "reasoning" + ], "properties": { "content": { - "type": "string", - "minLength": 1 + "type": "string" }, "reasoning": { - "type": "string", - "minLength": 1 + "type": "string" } } } @@ -39,13 +45,11 @@ "skipped": { "type": "array", "items": { - "type": "string", - "minLength": 1 + "type": "string" } }, "summary": { - "type": "string", - "minLength": 1 + "type": "string" } } } diff --git a/solstone/apps/timeline/talent/segment_summary.schema.json b/solstone/apps/timeline/talent/segment_summary.schema.json index a34b2db33..744ebe712 100644 --- a/solstone/apps/timeline/talent/segment_summary.schema.json +++ b/solstone/apps/timeline/talent/segment_summary.schema.json @@ -10,5 +10,9 @@ "description": "ONE sentence, MAX 10 words, MAX 60 characters, third person, present-tense, verb-led. Examples: 'Maps the app surface and names rebuild priorities.' 'Restarts display manager to recover the desktop session.' 'Fixes panel placement and terminal artifacts in Plasma.' No first-person. No times. No segment IDs." } }, - "required": ["title", "description"] + "required": [ + "title", + "description" + ], + "additionalProperties": false } diff --git a/solstone/observe/categories/meeting.schema.json b/solstone/observe/categories/meeting.schema.json index 65ab7619d..ea3694c31 100644 --- a/solstone/observe/categories/meeting.schema.json +++ b/solstone/observe/categories/meeting.schema.json @@ -1,56 +1,96 @@ { - "$schema": "https://json-schema.org/draft/2020-12/schema", - "$comment": "Meeting extraction output contract. Source of truth for the shape is observe/categories/meeting.md.", "type": "object", "additionalProperties": false, - "required": ["platform", "participants", "screen_share"], + "required": [ + "platform", + "participants", + "screen_share" + ], "properties": { "platform": { "type": "string", - "enum": ["zoom", "meet", "teams", "slack", "discord", "webex", "other"] + "enum": [ + "zoom", + "meet", + "teams", + "slack", + "discord", + "webex", + "other" + ] }, "participants": { "type": "array", "items": { "type": "object", "additionalProperties": false, - "required": ["name", "status", "video"], + "required": [ + "name", + "status", + "video", + "box_2d" + ], "properties": { - "name": {"type": "string", "minLength": 1}, + "name": { + "type": "string" + }, "status": { "type": "string", - "enum": ["speaking", "muted", "active", "presenting", "unknown"] + "enum": [ + "speaking", + "muted", + "active", + "presenting", + "unknown" + ] + }, + "video": { + "type": "boolean" }, - "video": {"type": "boolean"}, "box_2d": { - "type": "array", - "items": {"type": "integer", "minimum": 0}, - "minItems": 4, - "maxItems": 4 + "type": [ + "array", + "null" + ], + "items": { + "type": "integer" + } } } } }, "screen_share": { - "oneOf": [ - {"type": "null"}, - { - "type": "object", - "additionalProperties": false, - "required": ["box_2d", "presenter", "description", "formatted_text"], - "properties": { - "box_2d": { - "type": "array", - "items": {"type": "integer", "minimum": 0}, - "minItems": 4, - "maxItems": 4 - }, - "presenter": {"type": ["string", "null"]}, - "description": {"type": "string"}, - "formatted_text": {"type": "string"} + "type": [ + "object", + "null" + ], + "additionalProperties": false, + "required": [ + "box_2d", + "presenter", + "description", + "formatted_text" + ], + "properties": { + "box_2d": { + "type": "array", + "items": { + "type": "integer" } + }, + "presenter": { + "type": [ + "string", + "null" + ] + }, + "description": { + "type": "string" + }, + "formatted_text": { + "type": "string" } - ] + } } } } diff --git a/solstone/observe/describe.py b/solstone/observe/describe.py index 9f480f78b..cf412171b 100644 --- a/solstone/observe/describe.py +++ b/solstone/observe/describe.py @@ -113,6 +113,7 @@ def _discover_categories() -> dict[str, dict]: if prompt_content.text.strip(): metadata["prompt"] = prompt_content.text + # Per-category output contract from .schema.json; e.g. meeting.schema.json: Source of truth for the shape is observe/categories/meeting.md schema_path = md_path.with_suffix(".schema.json") if schema_path.exists(): metadata["json_schema"] = json.loads(schema_path.read_text("utf-8")) @@ -181,6 +182,7 @@ CATEGORIES = _discover_categories() # Build categorization prompt from template CATEGORIZATION_PROMPT = _build_categorization_prompt() +# The enums in `primary` and `secondary` MUST match the filenames under observe/categories/*.md. _SCHEMA = json.loads( (Path(__file__).parent / "describe.schema.json").read_text(encoding="utf-8") ) diff --git a/solstone/observe/describe.schema.json b/solstone/observe/describe.schema.json index 4a2c6b5cc..298888c90 100644 --- a/solstone/observe/describe.schema.json +++ b/solstone/observe/describe.schema.json @@ -1,13 +1,47 @@ { - "$schema": "https://json-schema.org/draft/2020-12/schema", - "$comment": "Frame categorization schema for observe/describe.py. The enums in `primary` and `secondary` MUST match the filenames under observe/categories/*.md. Enforced by tests/test_observe_describe_schema.py::test_category_enum_matches_registry. If you add/remove/rename a category file, update both enums here.", "type": "object", "additionalProperties": false, - "required": ["visual_description", "primary", "secondary", "overlap"], + "required": [ + "visual_description", + "primary", + "secondary", + "overlap" + ], "properties": { - "visual_description": {"type": "string", "minLength": 1}, - "primary": {"type": "string", "enum": ["browsing", "code", "gaming", "media", "meeting", "messaging", "productivity", "reading", "terminal"]}, - "secondary": {"type": "string", "enum": ["browsing", "code", "gaming", "media", "meeting", "messaging", "productivity", "reading", "terminal", "none"]}, - "overlap": {"type": "boolean"} + "visual_description": { + "type": "string" + }, + "primary": { + "type": "string", + "enum": [ + "browsing", + "code", + "gaming", + "media", + "meeting", + "messaging", + "productivity", + "reading", + "terminal" + ] + }, + "secondary": { + "type": "string", + "enum": [ + "browsing", + "code", + "gaming", + "media", + "meeting", + "messaging", + "productivity", + "reading", + "terminal", + "none" + ] + }, + "overlap": { + "type": "boolean" + } } } diff --git a/solstone/observe/enrich.schema.json b/solstone/observe/enrich.schema.json index c04a678e4..ae6214f45 100644 --- a/solstone/observe/enrich.schema.json +++ b/solstone/observe/enrich.schema.json @@ -1,23 +1,40 @@ { - "$schema": "https://json-schema.org/draft/2020-12/schema", "type": "object", "additionalProperties": false, - "required": ["statements", "topics", "setting", "warning"], + "required": [ + "statements", + "topics", + "setting", + "warning" + ], "properties": { "statements": { "type": "array", "items": { "type": "object", "additionalProperties": false, - "required": ["corrected", "emotion"], + "required": [ + "corrected", + "emotion" + ], "properties": { - "corrected": {"type": "string"}, - "emotion": {"type": "string"} + "corrected": { + "type": "string" + }, + "emotion": { + "type": "string" + } } } }, - "topics": {"type": "string"}, - "setting": {"type": "string"}, - "warning": {"type": "string"} + "topics": { + "type": "string" + }, + "setting": { + "type": "string" + }, + "warning": { + "type": "string" + } } } diff --git a/solstone/observe/transcribe/gemini.schema.json b/solstone/observe/transcribe/gemini.schema.json index b769e8bd7..e1b465f89 100644 --- a/solstone/observe/transcribe/gemini.schema.json +++ b/solstone/observe/transcribe/gemini.schema.json @@ -1,27 +1,30 @@ { - "$schema": "https://json-schema.org/draft/2020-12/schema", "type": "object", "additionalProperties": false, - "required": ["segments"], + "required": [ + "segments" + ], "properties": { "segments": { "type": "array", "items": { "type": "object", "additionalProperties": false, - "required": ["start", "speaker", "text"], + "required": [ + "start", + "speaker", + "text" + ], "properties": { "start": { "type": "string", "pattern": "^\\d{2}:\\d{2}$" }, "speaker": { - "type": "string", - "minLength": 1 + "type": "string" }, "text": { - "type": "string", - "minLength": 1 + "type": "string" } } } diff --git a/solstone/talent/daily_schedule.schema.json b/solstone/talent/daily_schedule.schema.json index 2669d4e31..497c9179b 100644 --- a/solstone/talent/daily_schedule.schema.json +++ b/solstone/talent/daily_schedule.schema.json @@ -1,8 +1,10 @@ { - "$schema": "https://json-schema.org/draft/2020-12/schema", "type": "object", "additionalProperties": false, - "required": ["primary", "fallback"], + "required": [ + "primary", + "fallback" + ], "properties": { "primary": { "type": "string", diff --git a/solstone/talent/participation.schema.json b/solstone/talent/participation.schema.json index cf12a44e8..23aeb3a59 100644 --- a/solstone/talent/participation.schema.json +++ b/solstone/talent/participation.schema.json @@ -1,32 +1,47 @@ { - "$schema": "https://json-schema.org/draft/2020-12/schema", "type": "object", "additionalProperties": false, - "required": ["participation"], + "required": [ + "participation", + "participation_confidence" + ], "properties": { "participation": { "type": "array", "items": { "type": "object", "additionalProperties": false, - "required": ["name", "role", "source", "confidence", "context", "entity_id"], + "required": [ + "name", + "role", + "source", + "confidence", + "context", + "entity_id" + ], "properties": { "name": { - "type": "string", - "minLength": 1 + "type": "string" }, "role": { "type": "string", - "enum": ["attendee", "mentioned"] + "enum": [ + "attendee", + "mentioned" + ] }, "source": { "type": "string", - "enum": ["voice", "speaker_label", "transcript", "screen", "other"] + "enum": [ + "voice", + "speaker_label", + "transcript", + "screen", + "other" + ] }, "confidence": { - "type": "number", - "minimum": 0, - "maximum": 1 + "type": "number" }, "context": { "type": "string" @@ -38,9 +53,10 @@ } }, "participation_confidence": { - "type": "number", - "minimum": 0, - "maximum": 1 + "type": [ + "number", + "null" + ] } } } diff --git a/solstone/talent/participation_entry.schema.json b/solstone/talent/participation_entry.schema.json index 7d104387c..cf28470f8 100644 --- a/solstone/talent/participation_entry.schema.json +++ b/solstone/talent/participation_entry.schema.json @@ -1,25 +1,37 @@ { - "$schema": "https://json-schema.org/draft/2020-12/schema", "type": "object", "additionalProperties": false, - "required": ["name", "role", "source", "confidence", "context", "entity_id"], + "required": [ + "name", + "role", + "source", + "confidence", + "context", + "entity_id" + ], "properties": { "name": { - "type": "string", - "minLength": 1 + "type": "string" }, "role": { "type": "string", - "enum": ["attendee", "mentioned"] + "enum": [ + "attendee", + "mentioned" + ] }, "source": { "type": "string", - "enum": ["voice", "speaker_label", "transcript", "screen", "other"] + "enum": [ + "voice", + "speaker_label", + "transcript", + "screen", + "other" + ] }, "confidence": { - "type": "number", - "minimum": 0, - "maximum": 1 + "type": "number" }, "context": { "type": "string" diff --git a/solstone/talent/sense.schema.json b/solstone/talent/sense.schema.json index f8c0757d5..f7abc0cee 100644 --- a/solstone/talent/sense.schema.json +++ b/solstone/talent/sense.schema.json @@ -1,53 +1,171 @@ { - "$schema": "https://json-schema.org/draft/2020-12/schema", "type": "object", "additionalProperties": false, - "required": ["density","content_type","activity_summary","entities","facets","meeting_detected","speakers","recommend","emotional_register"], + "required": [ + "density", + "content_type", + "activity_summary", + "entities", + "facets", + "meeting_detected", + "speakers", + "recommend", + "emotional_register" + ], "properties": { - "density": {"type": "string", "enum": ["active","low_change","idle"]}, - "content_type": {"type": "string", "enum": ["meeting","coding","browsing","email","messaging","ai_conversation","writing","reading","video","gaming","social","planning","productivity","terminal","design","music","idle"]}, - "activity_summary": {"type": "string", "minLength": 1}, + "density": { + "type": "string", + "enum": [ + "active", + "low_change", + "idle" + ] + }, + "content_type": { + "type": "string", + "enum": [ + "meeting", + "coding", + "browsing", + "email", + "messaging", + "ai_conversation", + "writing", + "reading", + "video", + "gaming", + "social", + "planning", + "productivity", + "terminal", + "design", + "music", + "idle" + ] + }, + "activity_summary": { + "type": "string" + }, "entities": { "type": "array", "items": { "type": "object", "additionalProperties": false, - "required": ["type","name","role","source","context"], + "required": [ + "type", + "name", + "role", + "source", + "context" + ], "properties": { - "type": {"type": "string", "enum": ["Person","Company","Project","Tool"]}, - "name": {"type": "string", "minLength": 1}, - "role": {"type": "string", "enum": ["attendee","mentioned"]}, - "source": {"type": "string", "enum": ["voice","speaker_label","transcript","screen","other"]}, - "context": {"type": "string", "minLength": 1} + "type": { + "type": "string", + "enum": [ + "Person", + "Company", + "Project", + "Tool" + ] + }, + "name": { + "type": "string" + }, + "role": { + "type": "string", + "enum": [ + "attendee", + "mentioned" + ] + }, + "source": { + "type": "string", + "enum": [ + "voice", + "speaker_label", + "transcript", + "screen", + "other" + ] + }, + "context": { + "type": "string" + } } } }, "facets": { "type": "array", - "minItems": 1, "items": { "type": "object", "additionalProperties": false, - "required": ["facet","activity","level"], + "required": [ + "facet", + "activity", + "level" + ], "properties": { - "facet": {"type": "string", "enum": ["__RUNTIME_FACETS__"]}, - "activity": {"type": "string", "minLength": 1}, - "level": {"type": "string", "enum": ["high","medium","low"]} + "facet": { + "type": "string", + "enum": [ + "__RUNTIME_FACETS__" + ] + }, + "activity": { + "type": "string" + }, + "level": { + "type": "string", + "enum": [ + "high", + "medium", + "low" + ] + } } } }, - "meeting_detected": {"type": "boolean"}, - "speakers": {"type": "array", "items": {"type": "string"}}, + "meeting_detected": { + "type": "boolean" + }, + "speakers": { + "type": "array", + "items": { + "type": "string" + } + }, "recommend": { "type": "object", "additionalProperties": false, - "required": ["screen_record","speaker_attribution","pulse_update"], + "required": [ + "screen_record", + "speaker_attribution", + "pulse_update" + ], "properties": { - "screen_record": {"type": "boolean"}, - "speaker_attribution": {"type": "boolean"}, - "pulse_update": {"type": "boolean"} + "screen_record": { + "type": "boolean" + }, + "speaker_attribution": { + "type": "boolean" + }, + "pulse_update": { + "type": "boolean" + } } }, - "emotional_register": {"type": "string", "enum": ["high_energy","tense","focused","collaborative","flat","celebratory","strained","neutral"]} + "emotional_register": { + "type": "string", + "enum": [ + "high_energy", + "tense", + "focused", + "collaborative", + "flat", + "celebratory", + "strained", + "neutral" + ] + } } } diff --git a/solstone/talent/story.schema.json b/solstone/talent/story.schema.json index 8b377dae8..d32b5eb64 100644 --- a/solstone/talent/story.schema.json +++ b/solstone/talent/story.schema.json @@ -1,28 +1,55 @@ { - "$schema": "https://json-schema.org/draft/2020-12/schema", "type": "object", "additionalProperties": false, - "required": ["body","topics","confidence","commitments","closures","decisions"], + "required": [ + "body", + "topics", + "confidence", + "commitments", + "closures", + "decisions" + ], "properties": { - "body": {"type": "string", "minLength": 1}, + "body": { + "type": "string" + }, "topics": { "type": "array", - "items": {"type": "string"}, - "maxItems": 10 + "items": { + "type": "string" + } + }, + "confidence": { + "type": "number" }, - "confidence": {"type": "number", "minimum": 0, "maximum": 1}, "commitments": { "type": "array", "items": { "type": "object", "additionalProperties": false, - "required": ["owner","action","counterparty","when","context"], + "required": [ + "owner", + "action", + "counterparty", + "when", + "context" + ], "properties": { - "owner": {"type": "string", "minLength": 1}, - "action": {"type": "string", "minLength": 1}, - "counterparty": {"type": "string", "minLength": 1}, - "when": {"type": "string", "minLength": 1}, - "context": {"type": "string", "minLength": 1} + "owner": { + "type": "string" + }, + "action": { + "type": "string" + }, + "counterparty": { + "type": "string" + }, + "when": { + "type": "string" + }, + "context": { + "type": "string" + } } } }, @@ -31,16 +58,36 @@ "items": { "type": "object", "additionalProperties": false, - "required": ["owner","action","counterparty","resolution","context"], + "required": [ + "owner", + "action", + "counterparty", + "resolution", + "context" + ], "properties": { - "owner": {"type": "string", "minLength": 1}, - "action": {"type": "string", "minLength": 1}, - "counterparty": {"type": "string", "minLength": 1}, + "owner": { + "type": "string" + }, + "action": { + "type": "string" + }, + "counterparty": { + "type": "string" + }, "resolution": { "type": "string", - "enum": ["sent","done","signed","dropped","deferred"] + "enum": [ + "sent", + "done", + "signed", + "dropped", + "deferred" + ] }, - "context": {"type": "string", "minLength": 1} + "context": { + "type": "string" + } } } }, @@ -49,11 +96,21 @@ "items": { "type": "object", "additionalProperties": false, - "required": ["owner","action","context"], + "required": [ + "owner", + "action", + "context" + ], "properties": { - "owner": {"type": "string", "minLength": 1}, - "action": {"type": "string", "minLength": 1}, - "context": {"type": "string", "minLength": 1} + "owner": { + "type": "string" + }, + "action": { + "type": "string" + }, + "context": { + "type": "string" + } } } } diff --git a/solstone/think/detect_created.schema.json b/solstone/think/detect_created.schema.json index 42fa3259e..4c8b682aa 100644 --- a/solstone/think/detect_created.schema.json +++ b/solstone/think/detect_created.schema.json @@ -1,8 +1,13 @@ { - "$schema": "https://json-schema.org/draft/2020-12/schema", "type": "object", "additionalProperties": false, - "required": ["day", "time", "confidence", "source", "utc"], + "required": [ + "day", + "time", + "confidence", + "source", + "utc" + ], "properties": { "day": { "type": "string", @@ -14,11 +19,14 @@ }, "confidence": { "type": "string", - "enum": ["high", "medium", "low"] + "enum": [ + "high", + "medium", + "low" + ] }, "source": { - "type": "string", - "minLength": 1 + "type": "string" }, "utc": { "type": "boolean" diff --git a/solstone/think/detect_transcript.py b/solstone/think/detect_transcript.py index b36418738..bcb6838a6 100644 --- a/solstone/think/detect_transcript.py +++ b/solstone/think/detect_transcript.py @@ -17,6 +17,7 @@ _SEGMENT_SCHEMA = json.loads( encoding="utf-8" ) ) +# Source of truth is think/detect_transcript_json.md. _JSON_SCHEMA = json.loads( (Path(__file__).parent / "detect_transcript_json.schema.json").read_text( encoding="utf-8" diff --git a/solstone/think/detect_transcript_json.schema.json b/solstone/think/detect_transcript_json.schema.json index fcbe4bfc5..27612bbec 100644 --- a/solstone/think/detect_transcript_json.schema.json +++ b/solstone/think/detect_transcript_json.schema.json @@ -1,24 +1,41 @@ { - "$schema": "https://json-schema.org/draft/2020-12/schema", - "$comment": "Output contract for detect_transcript_json(). Source of truth is think/detect_transcript_json.md.", "type": "object", "additionalProperties": false, - "required": ["entries", "topics", "setting"], + "required": [ + "entries", + "topics", + "setting" + ], "properties": { "entries": { "type": "array", "items": { "type": "object", "additionalProperties": false, - "required": ["start", "speaker", "text"], + "required": [ + "start", + "speaker", + "text" + ], "properties": { - "start": {"type": "string", "pattern": "^\\d{2}:\\d{2}:\\d{2}$"}, - "speaker": {"type": "string", "minLength": 1}, - "text": {"type": "string", "minLength": 1} + "start": { + "type": "string", + "pattern": "^\\d{2}:\\d{2}:\\d{2}$" + }, + "speaker": { + "type": "string" + }, + "text": { + "type": "string" + } } } }, - "topics": {"type": "string"}, - "setting": {"type": "string"} + "topics": { + "type": "string" + }, + "setting": { + "type": "string" + } } } diff --git a/tests/integration/test_schema_provider_parity.py b/tests/integration/test_schema_provider_parity.py new file mode 100644 index 000000000..c67fdc555 --- /dev/null +++ b/tests/integration/test_schema_provider_parity.py @@ -0,0 +1,233 @@ +# SPDX-License-Identifier: AGPL-3.0-only +# Copyright (c) 2026 sol pbc + +"""Raw-SDK strict schema parity tests for req_bfbdbux6 Class-A schemas.""" + +from __future__ import annotations + +import json +import os +from pathlib import Path +from typing import Any, Callable + +import pytest +from dotenv import load_dotenv +from jsonschema import Draft202012Validator + +from solstone.think.models import CLAUDE_SONNET_4, GEMINI_FLASH, GPT_5 + +REPO_ROOT = Path(__file__).resolve().parents[2] +FACET_SENTINEL = "__RUNTIME_FACETS__" + +CLASS_A = [ + ( + "describe", + "solstone/observe/describe.schema.json", + "Categorize a hypothetical code-editor-with-terminal screenshot.", + ), + ( + "meeting", + "solstone/observe/categories/meeting.schema.json", + "Hypothetical 2-person Zoom: Alice speaking video-on box [10,20,30,40]; " + "Bob muted video-off; no screen share.", + ), + ( + "enrich", + "solstone/observe/enrich.schema.json", + "Enrich a hypothetical transcript: one corrected statement, neutral emotion; " + "topics/setting/warning short.", + ), + ( + "transcribe_gemini", + "solstone/observe/transcribe/gemini.schema.json", + "One transcript segment: start '00:05', speaker 'Alice', text 'hello'.", + ), + ( + "detect_created", + "solstone/think/detect_created.schema.json", + "Detect created date: day '20260518', time '143000', confidence high, " + "source 'header', utc false.", + ), + ( + "detect_transcript_json", + "solstone/think/detect_transcript_json.schema.json", + "One entry: start '00:00:05', speaker 'Alice', text 'hi'; " + "topics/setting short.", + ), + ( + "daily_schedule", + "solstone/talent/daily_schedule.schema.json", + "primary '09:00', fallback '13:30'.", + ), + ( + "entity_observer", + "solstone/apps/entities/talent/entity_observer.schema.json", + "One observation entity_id 'e1' with one item content 'c' reasoning 'r'; " + "skipped []; summary 's'.", + ), + ( + "sense", + "solstone/talent/sense.schema.json", + "Hypothetical idle frame: density idle, content_type idle, summary 'idle', " + "no entities, one facet, meeting_detected false, no speakers, all recommend " + "false, emotional_register neutral.", + ), + ( + "story", + "solstone/talent/story.schema.json", + "Short story body 's', topics ['t'], confidence 0.5, no commitments/" + "closures/decisions.", + ), + ( + "segment_summary", + "solstone/apps/timeline/talent/segment_summary.schema.json", + "title 'Dev Env', description 'Sets up the environment.'", + ), + ( + "participation", + "solstone/talent/participation.schema.json", + "One participant: name 'Alice', role attendee, source voice, confidence 0.9, " + "context 'c', entity_id null; participation_confidence 0.8.", + ), + ( + "participation_entry", + "solstone/talent/participation_entry.schema.json", + "name 'Alice', role attendee, source voice, confidence 0.9, context 'c', " + "entity_id null.", + ), +] + + +def get_fixtures_env(api_key_name: str): + fixtures_env = Path(__file__).parent.parent / "fixtures" / ".env" + if not fixtures_env.exists(): + return None, None, None + load_dotenv(fixtures_env, override=True) + return fixtures_env, os.getenv(api_key_name), os.getenv("SOLSTONE_JOURNAL") + + +def hydrate(node: Any) -> None: + if isinstance(node, dict): + if node.get("enum") == [FACET_SENTINEL]: + node["enum"] = ["work", "personal", "health"] + for value in node.values(): + hydrate(value) + elif isinstance(node, list): + for value in node: + hydrate(value) + + +def load_schema(rel_path: str) -> dict[str, Any]: + schema = json.loads((REPO_ROOT / rel_path).read_text(encoding="utf-8")) + hydrate(schema) + return schema + + +def conforms(schema: dict[str, Any], text: str) -> tuple[bool, str]: + try: + obj = json.loads(text) + except Exception as exc: + return False, f"not JSON: {exc}: {text[:120]!r}" + errors = sorted(Draft202012Validator(schema).iter_errors(obj), key=str) + if errors: + return False, f"NONCONFORM: {errors[0].message[:160]}" + return True, json.dumps(obj)[:140] + + +def call_anthropic(schema: dict[str, Any], prompt: str, api_key: str) -> str: + from anthropic import Anthropic + + message = Anthropic(api_key=api_key).messages.create( + model=CLAUDE_SONNET_4, + max_tokens=2048, + temperature=0.3, + messages=[{"role": "user", "content": prompt + " Respond JSON only."}], + output_config={"format": {"type": "json_schema", "schema": schema}}, + ) + return "".join( + block.text + for block in message.content + if getattr(block, "type", None) == "text" + ) + + +def call_openai(schema: dict[str, Any], prompt: str, api_key: str) -> str: + import openai + + response = openai.OpenAI(api_key=api_key).responses.create( + model=GPT_5, + input=prompt + " Respond JSON only.", + max_output_tokens=2048, + text={ + "format": { + "type": "json_schema", + "name": "r", + "schema": schema, + "strict": True, + } + }, + ) + return response.output_text or "" + + +def call_google(schema: dict[str, Any], prompt: str, api_key: str) -> str: + from google import genai + from google.genai import types + + response = genai.Client(api_key=api_key, vertexai=False).models.generate_content( + model=GEMINI_FLASH, + contents=[prompt + " Respond JSON only."], + config=types.GenerateContentConfig( + temperature=0.3, + max_output_tokens=2048, + response_mime_type="application/json", + response_json_schema=schema, + thinking_config=types.ThinkingConfig(thinking_budget=0), + ), + ) + return response.text or "" + + +PROVIDERS: tuple[tuple[str, str, Callable[[dict[str, Any], str, str], str]], ...] = ( + ("google", "GOOGLE_API_KEY", call_google), + ("openai-strict", "OPENAI_API_KEY", call_openai), + ("anthropic-strict", "ANTHROPIC_API_KEY", call_anthropic), +) + + +@pytest.mark.integration +@pytest.mark.requires_api +@pytest.mark.parametrize( + ("provider_name", "api_key_name", "caller"), + [pytest.param(*provider, id=provider[0]) for provider in PROVIDERS], +) +@pytest.mark.parametrize( + ("schema_name", "schema_path", "prompt"), + [pytest.param(*schema_case, id=schema_case[0]) for schema_case in CLASS_A], +) +def test_class_a_schema_provider_parity( + provider_name: str, + api_key_name: str, + caller: Callable[[dict[str, Any], str, str], str], + schema_name: str, + schema_path: str, + prompt: str, +) -> None: + fixtures_env, api_key, journal_path = get_fixtures_env(api_key_name) + if not fixtures_env: + pytest.skip("tests/fixtures/.env not found") + if not api_key: + pytest.skip(f"{api_key_name} not found in tests/fixtures/.env file") + if not journal_path: + pytest.skip("SOLSTONE_JOURNAL not found in tests/fixtures/.env file") + + schema = load_schema(schema_path) + try: + output = caller(schema, prompt, api_key) + except Exception as exc: + pytest.fail( + f"{provider_name} rejected {schema_name}: {type(exc).__name__}: {exc}" + ) + + ok, detail = conforms(schema, output) + assert ok, f"{provider_name} returned invalid {schema_name}: {detail}" diff --git a/tests/test_anthropic.py b/tests/test_anthropic.py index 033e7bc8d..b3ad9acf5 100644 --- a/tests/test_anthropic.py +++ b/tests/test_anthropic.py @@ -43,10 +43,17 @@ def _decoded_image(b64: str) -> Image.Image: def _load_describe_schema() -> dict: - schema_path = ( - Path(__file__).resolve().parents[1] / "solstone/observe/describe.schema.json" - ) - return json.loads(schema_path.read_text(encoding="utf-8")) + return { + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$comment": f"Inline dirty fixture for {Path('describe.schema.json').name}", + "type": "object", + "additionalProperties": False, + "required": ["visual_description", "primary"], + "properties": { + "visual_description": {"type": "string", "minLength": 1}, + "primary": {"type": "string", "enum": ["browsing", "code"]}, + }, + } def _assert_no_schema_metadata(value): @@ -781,6 +788,7 @@ class TestRunGenerateJsonSchema: assert "$comment" in schema assert json.dumps(schema, sort_keys=True) == original_json assert output_schema["additionalProperties"] is False + assert schema["properties"]["visual_description"]["minLength"] == 1 assert output_schema["properties"]["visual_description"]["minLength"] == 1 assert "code" in output_schema["properties"]["primary"]["enum"] _validate_json_response(result, True) @@ -813,6 +821,7 @@ class TestRunGenerateJsonSchema: assert "$comment" in schema assert json.dumps(schema, sort_keys=True) == original_json assert output_schema["additionalProperties"] is False + assert schema["properties"]["visual_description"]["minLength"] == 1 assert output_schema["properties"]["visual_description"]["minLength"] == 1 assert "code" in output_schema["properties"]["primary"]["enum"] _validate_json_response(result, True) diff --git a/tests/test_detect_created_schema.py b/tests/test_detect_created_schema.py index a7b6f355b..3b8c3957a 100644 --- a/tests/test_detect_created_schema.py +++ b/tests/test_detect_created_schema.py @@ -51,7 +51,7 @@ def test_detect_created_schema_accepts_and_rejects_expected_values(): assert not validator.is_valid({**valid, "time": "14:30:52"}) assert not validator.is_valid({**valid, "confidence": "certain"}) assert not validator.is_valid({**valid, "extra": "x"}) - assert not validator.is_valid({**valid, "source": ""}) + assert not validator.is_valid({**valid, "source": 42}) def test_detect_created_passes_schema_to_generate(monkeypatch): diff --git a/tests/test_detect_transcript_schema.py b/tests/test_detect_transcript_schema.py index 124aedd66..6f0f6c8b0 100644 --- a/tests/test_detect_transcript_schema.py +++ b/tests/test_detect_transcript_schema.py @@ -83,7 +83,7 @@ def test_detect_transcript_json_schema_accepts_and_rejects_expected_values(): assert not validator.is_valid( { **valid, - "entries": [{"start": "12:34:56", "speaker": "Alice", "text": ""}], + "entries": [{"start": "12:34:56", "speaker": "Alice", "text": 7}], } ) assert not validator.is_valid({**valid, "extra": "x"}) diff --git a/tests/test_entity_observer_schema.py b/tests/test_entity_observer_schema.py index d68e17bae..9a01ce0b1 100644 --- a/tests/test_entity_observer_schema.py +++ b/tests/test_entity_observer_schema.py @@ -130,7 +130,7 @@ def test_invalid_empty_content(): "observations": [ { "entity_id": "alice_johnson", - "items": [{"content": "", "reasoning": "Durable preference."}], + "items": [{"content": 7, "reasoning": "Durable preference."}], } ], "skipped": [], diff --git a/tests/test_meeting_schema.py b/tests/test_meeting_schema.py index 24ffca7e6..097ae5dd8 100644 --- a/tests/test_meeting_schema.py +++ b/tests/test_meeting_schema.py @@ -34,7 +34,7 @@ def test_meeting_schema_accepts_and_rejects_expected_values(): { "platform": "zoom", "participants": [ - {"name": "Alice", "status": "active", "video": True}, + {"name": "Alice", "status": "active", "video": True, "box_2d": None}, ], "screen_share": None, } @@ -62,7 +62,7 @@ def test_meeting_schema_accepts_and_rejects_expected_values(): { "platform": "hangouts", "participants": [ - {"name": "Alice", "status": "active", "video": True}, + {"name": "Alice", "status": "active", "video": True, "box_2d": None}, ], "screen_share": None, } @@ -71,7 +71,7 @@ def test_meeting_schema_accepts_and_rejects_expected_values(): { "platform": "zoom", "participants": [ - {"name": "Alice", "status": "talking", "video": True}, + {"name": "Alice", "status": "talking", "video": True, "box_2d": None}, ], "screen_share": None, } @@ -87,7 +87,7 @@ def test_meeting_schema_accepts_and_rejects_expected_values(): { "platform": "zoom", "participants": [ - {"name": "Alice", "status": "active", "video": True}, + {"name": "Alice", "status": "active", "video": True, "box_2d": None}, ], "screen_share": None, "extra": True, @@ -97,7 +97,7 @@ def test_meeting_schema_accepts_and_rejects_expected_values(): { "platform": "zoom", "participants": [ - {"status": "active", "video": True}, + {"status": "active", "video": True, "box_2d": None}, ], "screen_share": None, } @@ -106,7 +106,7 @@ def test_meeting_schema_accepts_and_rejects_expected_values(): { "platform": "zoom", "participants": [ - {"name": "", "status": "active", "video": True}, + {"name": "Alice", "status": "active", "video": True}, ], "screen_share": None, } diff --git a/tests/test_observe_describe_schema.py b/tests/test_observe_describe_schema.py index 2e55f1902..b9cd44384 100644 --- a/tests/test_observe_describe_schema.py +++ b/tests/test_observe_describe_schema.py @@ -85,7 +85,7 @@ def test_describe_schema_accepts_and_rejects_expected_values(): ) assert not validator.is_valid( { - "visual_description": "", + "visual_description": 7, "primary": "productivity", "secondary": "none", "overlap": False, @@ -119,6 +119,7 @@ async def test_describe_batch_call_passes_schema(mock_agenerate): def test_category_enum_matches_registry(): + """The enums in `primary` and `secondary` MUST match the filenames under observe/categories/*.md.""" categories_dir = Path(describe_mod.__file__).resolve().parent / "categories" on_disk = {p.stem for p in categories_dir.glob("*.md")} diff --git a/tests/test_participation_schema.py b/tests/test_participation_schema.py index 769671e18..278106361 100644 --- a/tests/test_participation_schema.py +++ b/tests/test_participation_schema.py @@ -22,8 +22,6 @@ def test_participation_entry_schema_is_valid_draft_2020_12(): Draft202012Validator.check_schema(schema) - assert schema["$schema"] == "https://json-schema.org/draft/2020-12/schema" - def test_participation_schema_is_valid_and_matches_loaded(): schema = _load_json(PARTICIPATION_SCHEMA_PATH) @@ -39,6 +37,5 @@ def test_participation_schema_items_match_fragment(): items = dict(schema["properties"]["participation"]["items"]) fragment_without_schema = dict(fragment) - fragment_without_schema.pop("$schema") assert items == fragment_without_schema diff --git a/tests/test_schedule_schema.py b/tests/test_schedule_schema.py index f0d0cc440..313a3be43 100644 --- a/tests/test_schedule_schema.py +++ b/tests/test_schedule_schema.py @@ -42,6 +42,17 @@ def _load_schedule_schema() -> dict: return schema +def _strip_portability_annotations(value): + if isinstance(value, dict): + for key in ("minLength", "minimum", "maximum"): + value.pop(key, None) + for child in value.values(): + _strip_portability_annotations(child) + elif isinstance(value, list): + for child in value: + _strip_portability_annotations(child) + + def _expected_schedule_activity_ids() -> set[str]: # Why: `meeting` is emitted by both the schedule talent and sense; the # other 9 are schedule-only (their instructions carry the marker). @@ -171,16 +182,21 @@ def test_schedule_participation_entry_diverges_from_shared_fragment(): assert isinstance(fragment, dict) fragment_without_schema = dict(fragment) - fragment_without_schema.pop("$schema") fragment_without_schema["properties"] = dict(fragment_without_schema["properties"]) fragment_without_schema["properties"].pop("entity_id") fragment_without_schema["required"] = [ key for key in fragment_without_schema["required"] if key != "entity_id" ] - inline_items = dict( + raw_inline_items = dict( schedule_schema["items"]["properties"]["participation"]["items"] ) + assert "entity_id" in fragment["properties"] + assert "entity_id" not in raw_inline_items["properties"] + assert raw_inline_items != fragment + + inline_items = json.loads(json.dumps(raw_inline_items)) + _strip_portability_annotations(inline_items) assert inline_items == fragment_without_schema diff --git a/tests/test_schema_strict_portability.py b/tests/test_schema_strict_portability.py new file mode 100644 index 000000000..5061f1c32 --- /dev/null +++ b/tests/test_schema_strict_portability.py @@ -0,0 +1,126 @@ +# SPDX-License-Identifier: AGPL-3.0-only +# Copyright (c) 2026 sol pbc + +"""Offline CI gate for req_bfbdbux6 strict schema portability. + +The allowlist is temporary: later portability lodes remove entries as they +portabilize Class-B/C/D schemas, then delete this allowlist mechanism entirely. +""" + +from __future__ import annotations + +import json +from pathlib import Path +from typing import Any + +import pytest + +from solstone.apps.timeline.rollup import build_rollup_schema + +REPO_ROOT = Path(__file__).resolve().parents[1] +BANNED_KEYS = frozenset( + { + "$schema", + "$comment", + "minLength", + "maxLength", + "minItems", + "maxItems", + "minimum", + "maximum", + } +) + +# Lodes 2-4 of req_bfbdbux6 remove their entries as those classes are +# portabilized; Lode 4 deletes this allowlist mechanism entirely. +PENDING_PORTABILITY = frozenset( + { + "solstone/observe/extract.schema.json", + "solstone/talent/schedule.schema.json", + "solstone/talent/speaker_attribution.schema.json", + "solstone/think/detect_transcript_segment.schema.json", + "solstone/talent/chat.schema.json", + "build_rollup_schema(3)", + } +) + + +def _discover_schemas() -> tuple[tuple[str, dict[str, Any]], ...]: + discovered: list[tuple[str, dict[str, Any]]] = [] + for path in sorted((REPO_ROOT / "solstone").glob("**/*.schema.json")): + schema_id = path.relative_to(REPO_ROOT).as_posix() + discovered.append((schema_id, json.loads(path.read_text(encoding="utf-8")))) + discovered.append(("build_rollup_schema(3)", build_rollup_schema(3))) + return tuple(discovered) + + +SCHEMAS = _discover_schemas() + + +def violations(schema: dict[str, Any]) -> list[str]: + found: list[str] = [] + + root_is_object = schema.get("type") == "object" or ( + "properties" in schema and "type" not in schema + ) + if not root_is_object: + found.append("$: root schema must be an object") + + def walk(node: Any, path: str) -> None: + if isinstance(node, dict): + for key in node: + if key in BANNED_KEYS: + found.append(f"{path}: banned key {key!r}") + if key == "oneOf": + found.append(f"{path}: banned key 'oneOf'") + + if node.get("type") == "object" or "properties" in node: + if node.get("additionalProperties") is not False: + found.append(f"{path}: object missing additionalProperties:false") + properties = node.get("properties") or {} + required = node.get("required") or [] + missing = sorted(set(properties) - set(required)) + if missing: + found.append(f"{path}: properties not required {missing!r}") + + for key, value in node.items(): + walk(value, f"{path}/{key}") + elif isinstance(node, list): + for index, value in enumerate(node): + walk(value, f"{path}[{index}]") + + walk(schema, "$") + return found + + +@pytest.mark.parametrize( + ("schema_id", "schema"), + [pytest.param(schema_id, schema, id=schema_id) for schema_id, schema in SCHEMAS], +) +def test_pending_portability_set_matches_discovery( + schema_id: str, schema: dict[str, Any] +) -> None: + schema_violations = violations(schema) + if schema_id in PENDING_PORTABILITY: + assert schema_violations, f"{schema_id} is still allowlisted but is portable" + else: + assert schema_violations == [], f"{schema_id}: {schema_violations}" + + +@pytest.mark.parametrize( + "schema", + [ + { + "type": "object", + "$comment": "bad", + "properties": { + "a": {"type": "array", "minItems": 1}, + "b": {"type": "string"}, + }, + "required": ["a"], + "additionalProperties": False, + } + ], +) +def test_strict_portability_guard_rejects_bad_schema(schema: dict[str, Any]) -> None: + assert violations(schema) diff --git a/tests/test_sense_schema.py b/tests/test_sense_schema.py index 2ee246d08..4dd68eeea 100644 --- a/tests/test_sense_schema.py +++ b/tests/test_sense_schema.py @@ -93,15 +93,6 @@ def test_hydrate_runtime_enums_replaces_facet_sentinel(monkeypatch): assert hydrated["properties"]["facet"]["enum"] == ["alpha", "valid_one"] -def test_sense_schema_facets_array_requires_minItems_one(): - schema = json.loads(SENSE_SCHEMA_PATH.read_text(encoding="utf-8")) - facets_node = schema["properties"]["facets"] - - assert facets_node["type"] == "array" - assert facets_node.get("minItems") == 1 - Draft202012Validator.check_schema(schema) - - def test_hydrate_runtime_enums_preserves_facet_minItems_when_facets_exist( monkeypatch, ): diff --git a/tests/test_transcribe_gemini_schema.py b/tests/test_transcribe_gemini_schema.py index abb052c13..ce909e57f 100644 --- a/tests/test_transcribe_gemini_schema.py +++ b/tests/test_transcribe_gemini_schema.py @@ -68,10 +68,10 @@ def test_gemini_schema_accepts_and_rejects_expected_values(): } ) assert not validator.is_valid( - {"segments": [{"start": "01:23", "speaker": "Speaker 1", "text": ""}]} + {"segments": [{"start": "01:23", "speaker": "Speaker 1", "text": 7}]} ) assert not validator.is_valid( - {"segments": [{"start": "01:23", "speaker": "", "text": "hi"}]} + {"segments": [{"start": "01:23", "speaker": 7, "text": "hi"}]} ) assert not validator.is_valid( {"segments": [{"start": "1:23", "speaker": "Speaker 1", "text": "hi"}]}