diff --git a/solstone/convey/provider_readiness.py b/solstone/convey/provider_readiness.py index 5fa2030b1..8ae848a90 100644 --- a/solstone/convey/provider_readiness.py +++ b/solstone/convey/provider_readiness.py @@ -253,6 +253,16 @@ _ENTRIES: dict[str, _Entry] = { detail="Try a shorter or more focused request.", recovery_action=None, ), + "incomplete_json_length": _Entry( + klass="generic", + summary="the answer ran out of room before it finished", + detail=( + "The reply hit its length limit before it could finish. On your own " + "machine sol tries once more with different settings; if it still runs " + "long, ask for less at once or choose another provider." + ), + recovery_action=None, + ), "max_turns_exhausted": _Entry( klass="generic", summary="this took too many steps to finish", diff --git a/solstone/convey/static/chat_reasons.js b/solstone/convey/static/chat_reasons.js index 8d95b8dc8..2ac75f090 100644 --- a/solstone/convey/static/chat_reasons.js +++ b/solstone/convey/static/chat_reasons.js @@ -114,6 +114,10 @@ "template": "the conversation grew too long to finish", "action": null }, + "incomplete_json_length": { + "template": "the answer ran out of room before it finished", + "action": null + }, "max_turns_exhausted": { "template": "this took too many steps to finish", "action": null diff --git a/solstone/convey/static/tests/chat-bar-reasons.html b/solstone/convey/static/tests/chat-bar-reasons.html index 02c777a0a..14cce7bcc 100644 --- a/solstone/convey/static/tests/chat-bar-reasons.html +++ b/solstone/convey/static/tests/chat-bar-reasons.html @@ -72,6 +72,7 @@ ['provider_unavailable', 'google', 'Gemini is having trouble right now', false], ['chat_pipeline_unavailable', '', "the chat pipeline isn't ready yet", false], ['chat_timeout', '', 'chat took too long', false], + ['incomplete_json_length', '', 'the answer ran out of room before it finished', false], ['unknown', '', 'chat had trouble', false] ]; diff --git a/solstone/think/models.py b/solstone/think/models.py index c7d385492..27f6c142a 100644 --- a/solstone/think/models.py +++ b/solstone/think/models.py @@ -206,6 +206,9 @@ TYPE_DEFAULTS: Dict[str, Dict[str, Any]] = { # --------------------------------------------------------------------------- +_LENGTH_FINISH_REASONS = frozenset({"length", "max_tokens"}) + + class IncompleteJSONError(ValueError): """Raised when JSON response is truncated due to token limits or other reasons. @@ -217,6 +220,11 @@ class IncompleteJSONError(ValueError): def __init__(self, reason: str, partial_text: str): self.reason = reason self.partial_text = partial_text + # Safety/content-filter/recitation finishes are refusals, not length + # truncations; labeling them incomplete_json_length would be dishonest + # and retrying a refusal would not help. + if str(reason).strip().lower() in _LENGTH_FINISH_REASONS: + self.reason_code = "incomplete_json_length" super().__init__(f"JSON response incomplete (reason: {reason})") diff --git a/solstone/think/providers/shared.py b/solstone/think/providers/shared.py index b36b32691..4b35f7ade 100644 --- a/solstone/think/providers/shared.py +++ b/solstone/think/providers/shared.py @@ -221,6 +221,7 @@ RUNTIME_REASON_CODES = frozenset( "network_unreachable", "provider_unavailable", "provider_response_invalid", + "incomplete_json_length", "unknown", } ) diff --git a/tests/test_chat_reasons.py b/tests/test_chat_reasons.py index 6d7006f8e..cc1b3197a 100644 --- a/tests/test_chat_reasons.py +++ b/tests/test_chat_reasons.py @@ -37,6 +37,7 @@ EXPECTED_CODES = { "chat_pipeline_unavailable", "chat_timeout", "context_window_exceeded", + "incomplete_json_length", "max_turns_exhausted", "no_output", "token_budget_exceeded", diff --git a/tests/test_models.py b/tests/test_models.py index edc23f6f6..3eeb3ecd2 100644 --- a/tests/test_models.py +++ b/tests/test_models.py @@ -132,6 +132,45 @@ def test_get_model_provider_mlx_backend_models_are_local(): assert get_model_provider(QWEN_35_9B) == "local" +@pytest.mark.parametrize("reason", ["length", "max_tokens", "MAX_TOKENS", " Length "]) +def test_incomplete_json_error_sets_length_reason_code(reason): + exc = IncompleteJSONError(reason, "") + + assert exc.reason_code == "incomplete_json_length" + + +@pytest.mark.parametrize("reason", ["safety", "content_filter", "recitation", "error"]) +def test_incomplete_json_error_non_length_reasons_have_no_reason_code(reason): + exc = IncompleteJSONError(reason, "") + + assert not hasattr(exc, "reason_code") + + +def test_incomplete_json_error_preserves_positional_and_keyword_construction(): + positional = IncompleteJSONError("length", "partial") + keyword = IncompleteJSONError(reason="max_tokens", partial_text="body") + + assert positional.reason == "length" + assert positional.partial_text == "partial" + assert keyword.reason == "max_tokens" + assert keyword.partial_text == "body" + assert positional.reason_code == "incomplete_json_length" + assert keyword.reason_code == "incomplete_json_length" + + +def test_classify_provider_error_uses_incomplete_json_reason_code(): + from solstone.think.providers.shared import classify_provider_error + + assert ( + classify_provider_error(IncompleteJSONError("length", ""), "local") + == "incomplete_json_length" + ) + assert ( + classify_provider_error(IncompleteJSONError("safety", ""), "local") + != "incomplete_json_length" + ) + + def test_calc_token_cost_gemma4_zero_cost(): token_data = { "model": GEMMA4_26B_A4B_4BIT, diff --git a/tests/test_provider_state.py b/tests/test_provider_state.py index 1a3c93877..1398edf5f 100644 --- a/tests/test_provider_state.py +++ b/tests/test_provider_state.py @@ -129,6 +129,7 @@ def test_runtime_reason_codes_are_state_reason_codes(): "provider_unavailable", "provider_response_invalid", "context_window_exceeded", + "incomplete_json_length", "max_turns_exhausted", "unknown", }