diff --git a/solstone/convey/provider_readiness.py b/solstone/convey/provider_readiness.py index 98b4d03ca..b75f34c53 100644 --- a/solstone/convey/provider_readiness.py +++ b/solstone/convey/provider_readiness.py @@ -239,6 +239,15 @@ _ENTRIES: dict[str, _Entry] = { detail="Open provider setup and check the saved credentials.", recovery_action=_THINKING_ACTION, ), + "model_not_found": _Entry( + klass="setup", + summary="{provider} doesn't offer this model to this key", + detail=( + "The key reached the provider, but this model is not available to " + "it. Pick a different model in Thinking." + ), + recovery_action=_THINKING_ACTION, + ), "provider_quota_exceeded": _Entry( klass="provider", summary="your {provider} quota is spent", diff --git a/solstone/convey/static/chat_reasons.js b/solstone/convey/static/chat_reasons.js index 6029594fa..d3966eb43 100644 --- a/solstone/convey/static/chat_reasons.js +++ b/solstone/convey/static/chat_reasons.js @@ -102,6 +102,10 @@ "template": "your {provider} key didn't validate", "action": {"label": "Open Thinking", "href": "/app/thinking/#main"} }, + "model_not_found": { + "template": "{provider} doesn't offer this model to this key", + "action": {"label": "Open Thinking", "href": "/app/thinking/#main"} + }, "provider_quota_exceeded": { "template": "your {provider} quota is spent", "action": null diff --git a/solstone/think/cogitate_policy.py b/solstone/think/cogitate_policy.py index 84e6ad2a5..e9d3b6c88 100644 --- a/solstone/think/cogitate_policy.py +++ b/solstone/think/cogitate_policy.py @@ -41,6 +41,7 @@ DETERMINISTIC_FAILURE_REASON_CODES = frozenset( "agent_stuck", "context_window_exceeded", "max_turns_exhausted", + "model_not_found", "no_output", "provider_request_rejected", "schema_invalid", @@ -61,6 +62,7 @@ DETERMINISTIC_FAILURE_CAPS: dict[str, int] = { "agent_stuck": 2, "context_window_exceeded": 2, "max_turns_exhausted": 2, + "model_not_found": 1, "no_output": 2, "provider_request_rejected": 1, "schema_invalid": 3, diff --git a/solstone/think/native/chat/command.rs b/solstone/think/native/chat/command.rs index 9ca384f38..de19c2849 100644 --- a/solstone/think/native/chat/command.rs +++ b/solstone/think/native/chat/command.rs @@ -572,6 +572,7 @@ fn readiness_summary(reason: &str) -> Option<&'static str> { Some("local model setup could not be verified") } "provider_key_invalid" => Some("your {provider} key didn't validate"), + "model_not_found" => Some("{provider} doesn't offer this model to this key"), "provider_quota_exceeded" => Some("your {provider} quota is spent"), "provider_request_rejected" => { Some("the provider refused a request sol sent; this is a defect in sol") diff --git a/solstone/think/providers/openhands.py b/solstone/think/providers/openhands.py index fb1424144..8648aa010 100644 --- a/solstone/think/providers/openhands.py +++ b/solstone/think/providers/openhands.py @@ -68,6 +68,8 @@ from solstone.think.providers.shared import ( GenerateResult, JSONEventCallback, classify_provider_error, + exception_chain, + is_cloud_model_not_found, safe_raw, validate_generate_result_strict, ) @@ -2034,29 +2036,10 @@ async def run_agenerate( ) -def _exception_chain(exc: BaseException) -> list[BaseException]: - chain: list[BaseException] = [] - current: BaseException | None = exc - while current is not None and current not in chain: - chain.append(current) - current = current.__cause__ or current.__context__ - return chain - - -def _model_not_found(exc: BaseException) -> bool: - for item in _exception_chain(exc): - status = getattr(item, "status_code", None) - if status is None: - status = getattr(getattr(item, "response", None), "status_code", None) - if status == 404 or "notfound" in type(item).__name__.lower(): - return True - return False - - def _validation_reason(exc: BaseException, provider: str) -> str: - if _model_not_found(exc): + if is_cloud_model_not_found(exc, provider): return "model_not_found" - for item in _exception_chain(exc): + for item in exception_chain(exc): reason = classify_provider_error(item, provider) if reason != "unknown": return reason diff --git a/solstone/think/providers/shared.py b/solstone/think/providers/shared.py index d095f3ee0..50b754a5f 100644 --- a/solstone/think/providers/shared.py +++ b/solstone/think/providers/shared.py @@ -194,11 +194,30 @@ def _exception_name_matches( return exc_name in names or any(exc_qualname.endswith(f".{name}") for name in names) +def exception_chain(exc: BaseException) -> list[BaseException]: + chain: list[BaseException] = [] + current: BaseException | None = exc + while current is not None and current not in chain: + chain.append(current) + current = current.__cause__ or current.__context__ + return chain + + +# Deliberately class-shape based: real cloud SDK missing-model 404s are +# NotFoundError-shaped classes (litellm, openai, anthropic), while a bare 404 +# can come from an unrelated endpoint. Do not broaden this to status alone. +def is_cloud_model_not_found(exc: BaseException, provider: str) -> bool: + return is_cloud_provider(provider) and any( + "notfound" in type(item).__name__.lower() for item in exception_chain(exc) + ) + + RUNTIME_REASON_CODES = frozenset( { "context_window_exceeded", "context_budget_exceeded", "local_capacity_exhausted", + "model_not_found", "provider_quota_exceeded", "provider_key_invalid", "chat_timeout", @@ -329,6 +348,9 @@ def classify_provider_error(exc: BaseException, provider: str) -> str: ) and (status_code or 0) >= 500: return "provider_unavailable" + if is_cloud_model_not_found(exc, provider): + return "model_not_found" + if isinstance(exc, RuntimeError): if _contains_any(message_lower, _CLI_UNAVAILABLE_PATTERNS): return "provider_unavailable" diff --git a/tests/test_chat_reasons.py b/tests/test_chat_reasons.py index b3380ede0..72559b1c0 100644 --- a/tests/test_chat_reasons.py +++ b/tests/test_chat_reasons.py @@ -34,6 +34,7 @@ EXPECTED_CODES = { "archive_path_traversal", "cuda_runtime_incomplete", "provider_key_invalid", + "model_not_found", "provider_quota_exceeded", "provider_request_rejected", "network_unreachable", diff --git a/tests/test_generate_full.py b/tests/test_generate_full.py index 3d17ab12d..32d15437a 100644 --- a/tests/test_generate_full.py +++ b/tests/test_generate_full.py @@ -304,6 +304,64 @@ def test_execute_generate_provider_blank_records_runtime_failure( ) +def test_generate_model_not_found_records_runtime_failure(tmp_path, monkeypatch): + from litellm.exceptions import NotFoundError + + from solstone.think.providers.brain_state import inspect_brain_state + + mod = importlib.import_module("solstone.think.talents") + copy_day(tmp_path, monkeypatch) + _write_ready_brain_record(tmp_path) + + import solstone.think.talent as talent + + monkeypatch.setattr(talent, "TALENT_DIR", tmp_path) + _write_generator_file( + tmp_path, + "missing_model_day_gen", + { + "type": "generate", + "schedule": "daily", + "priority": 10, + "output": "md", + "load": {"transcripts": True, "percepts": True}, + }, + ) + provider_module = MagicMock() + provider_module.run_generate.side_effect = NotFoundError( + "model not found", + model="gemini-3.5-flash", + llm_provider="gemini", + ) + monkeypatch.setattr( + "solstone.think.providers.get_provider_module", + lambda _provider: provider_module, + ) + monkeypatch.setenv("GOOGLE_API_KEY", "test-key") + + events = run_generator_with_config( + mod, + { + "name": "missing_model_day_gen", + "day": "20240101", + "output": "md", + }, + monkeypatch, + ) + + error_events = [event for event in events if event["event"] == "error"] + assert len(error_events) == 1 + assert error_events[0]["reason_code"] == "model_not_found" + assert error_events[0]["provider"] == "google" + assert [event for event in events if event["event"] == "finish"] == [] + + inspection = inspect_brain_state(datetime.now(timezone.utc), journal_path=tmp_path) + record = inspection["record"] + assert record is not None + assert record["reason_code"] == "model_not_found" + assert record["evidence"]["generate"]["reason_code"] == "model_not_found" + + def test_execute_generate_provider_blank_rejected_when_config_switches_in_flight( tmp_path, monkeypatch, diff --git a/tests/test_openhands_generate.py b/tests/test_openhands_generate.py index 50fce96ff..91b6403f3 100644 --- a/tests/test_openhands_generate.py +++ b/tests/test_openhands_generate.py @@ -443,6 +443,25 @@ def test_validation_uses_runtime_probe_and_classifies_results(monkeypatch): ) +def test_validation_model_not_found_uses_shared_cloud_predicate(monkeypatch): + from litellm.exceptions import NotFoundError + + assert not hasattr(openhands, "_model_not_found") + + def missing(*_args): + raise NotFoundError("model not found", model="m", llm_provider="gemini") + + monkeypatch.setattr(openhands, "_probe", missing) + + assert openhands.validate_key("google", "key") == { + "valid": True, + "probe_reason_code": "model_not_found", + } + assert openhands.validate_model("google", "missing", "key")["reason_code"] == ( + "model_not_found" + ) + + def test_validation_probe_rejects_blank_canned_response_without_retry(monkeypatch): calls: list[dict] = [] diff --git a/tests/test_provider_error_classification.py b/tests/test_provider_error_classification.py index fa1202c03..15f8456e5 100644 --- a/tests/test_provider_error_classification.py +++ b/tests/test_provider_error_classification.py @@ -107,6 +107,159 @@ def test_classifies_openhands_bad_request_google_request_rejected(): assert classify_provider_error(exc, "google") == "provider_request_rejected" +def _litellm_not_found_error(): + from litellm.exceptions import NotFoundError + + return NotFoundError("model not found", model="m", llm_provider="gemini") + + +def _openai_response(status_code: int): + import httpx + + request = httpx.Request("GET", "https://api.openai.example/models/missing") + return httpx.Response(status_code, request=request) + + +def _openai_not_found_error(): + import openai + + return openai.NotFoundError( + "model not found", + response=_openai_response(404), + body=None, + ) + + +def _mapped_provider_exception(exc: Exception) -> Exception: + from openhands.sdk.llm.exceptions.mapping import map_provider_exception + + return map_provider_exception(exc) + + +@pytest.mark.parametrize( + "factory", + [ + _litellm_not_found_error, + _openai_not_found_error, + lambda: _mapped_provider_exception(_litellm_not_found_error()), + lambda: _mapped_provider_exception(_openai_not_found_error()), + ], +) +def test_cloud_not_found_shapes_classify_model_not_found(factory): + for provider in ("google", "openai", "anthropic"): + assert classify_provider_error(factory(), provider) == "model_not_found" + + +@pytest.mark.parametrize( + "factory", + [ + _litellm_not_found_error, + _openai_not_found_error, + lambda: _mapped_provider_exception(_litellm_not_found_error()), + lambda: _mapped_provider_exception(_openai_not_found_error()), + ], +) +def test_local_not_found_shapes_stay_unknown(factory): + assert classify_provider_error(factory(), "local") == "unknown" + + +def test_anthropic_not_found_shape_classifies_model_not_found_when_installed(): + import httpx + + anthropic = pytest.importorskip("anthropic") + (not_found_error,) = _require_attrs(anthropic, "NotFoundError") + request = httpx.Request("GET", "https://api.anthropic.example/v1/models/missing") + response = httpx.Response(404, request=request) + exc = not_found_error("model not found", response=response, body=None) + + for provider in ("google", "openai", "anthropic"): + assert classify_provider_error(exc, provider) == "model_not_found" + assert classify_provider_error(exc, "local") == "unknown" + + +def test_status_only_404_shapes_do_not_classify_model_not_found(): + import httpx + + class StatusOnly: + status_code = 404 + + class ResponseOnly: + response = httpx.Response( + 404, request=httpx.Request("GET", "https://example.invalid/missing") + ) + + unrelated = httpx.HTTPStatusError( + "not found", + request=httpx.Request("GET", "https://example.invalid/unrelated"), + response=httpx.Response( + 404, request=httpx.Request("GET", "https://example.invalid/unrelated") + ), + ) + + for exc in (StatusOnly(), ResponseOnly(), unrelated): + for provider in ("google", "openai", "local"): + assert classify_provider_error(exc, provider) == "unknown" + + +def test_sibling_classifications_stay_ahead_of_model_not_found(): + import httpx + import openai + from litellm.exceptions import BadRequestError + + auth = openai.AuthenticationError( + "bad key", + response=_openai_response(401), + body=None, + ) + quota = openai.RateLimitError( + "rate limit", + response=_openai_response(429), + body=None, + ) + ordinary_400 = BadRequestError( + "Invalid value for parameter 'temperature'", + model="gemini-test", + llm_provider="google", + ) + status_500 = httpx.HTTPStatusError( + "server error", + request=httpx.Request("GET", "https://api.example.invalid/models"), + response=httpx.Response( + 500, request=httpx.Request("GET", "https://api.example.invalid/models") + ), + ) + status_503 = httpx.HTTPStatusError( + "unavailable", + request=httpx.Request("GET", "https://api.example.invalid/models"), + response=httpx.Response( + 503, request=httpx.Request("GET", "https://api.example.invalid/models") + ), + ) + + cases = [ + (auth, "provider_key_invalid"), + (quota, "provider_quota_exceeded"), + (ordinary_400, "provider_request_rejected"), + (status_500, "provider_unavailable"), + (status_503, "provider_unavailable"), + (httpx.ReadTimeout("timeout"), "chat_timeout"), + (httpx.ConnectError("connection failed"), "network_unreachable"), + (Exception("plain failure"), "unknown"), + ] + + for exc, expected in cases: + assert classify_provider_error(exc, "google") == expected + + +@pytest.mark.parametrize("message", _CONTEXT_WINDOW_PATTERNS) +def test_context_window_messages_stay_context_window_before_model_not_found(message): + from litellm.exceptions import BadRequestError + + exc = BadRequestError(message, model="gemini-test", llm_provider="google") + + assert classify_provider_error(exc, "google") == "context_window_exceeded" + + @pytest.mark.parametrize("message", _CONTEXT_WINDOW_PATTERNS) def test_classifies_openhands_bad_request_context_window_messages_before_rejection( message, @@ -231,6 +384,7 @@ def test_runtime_reason_codes_are_registered_with_owner_copy(): for reason_code in ( "context_window_exceeded", "context_budget_exceeded", + "model_not_found", "provider_request_rejected", ): assert reason_code in RUNTIME_REASON_CODES diff --git a/tests/test_talents_ndjson.py b/tests/test_talents_ndjson.py index 204eefd25..78988baa3 100644 --- a/tests/test_talents_ndjson.py +++ b/tests/test_talents_ndjson.py @@ -6,12 +6,21 @@ import asyncio import json import sys +from datetime import datetime, timezone from io import StringIO +from types import SimpleNamespace from unittest.mock import MagicMock, patch import pytest from solstone.think.models import GPT_5 +from solstone.think.providers.brain_state import ( + begin_brain_refresh, + finish_brain_refresh, + inspect_brain_state, +) + +NOW = datetime(2026, 1, 2, 3, 4, 5, tzinfo=timezone.utc) @pytest.fixture @@ -71,6 +80,46 @@ def mock_prepare_config(request: dict) -> dict: return config +def _ok_component() -> dict[str, str]: + return { + "status": "ok", + "observed_at": NOW.isoformat(), + "expires_at": datetime(2026, 1, 3, 3, 4, 5, tzinfo=timezone.utc).isoformat(), + } + + +def _write_ready_brain(journal_path): + config_path = journal_path / "config" / "journal.json" + config_path.parent.mkdir(parents=True, exist_ok=True) + config_path.write_text( + json.dumps( + { + "providers": { + "active": { + "provider": "google", + "model": "gemini-3.5-flash", + } + }, + "env": {"GOOGLE_API_KEY": "test-key"}, + } + ), + encoding="utf-8", + ) + permit = begin_brain_refresh(NOW, journal_path=journal_path) + assert permit is not None + finish_brain_refresh( + permit, + { + "configuration": _ok_component(), + "lane_prerequisites": _ok_component(), + "generate": _ok_component(), + "cogitate": _ok_component(), + }, + NOW, + journal_path=journal_path, + ) + + def mock_all_providers(monkeypatch): """Mock the registered cogitate provider module with mock_run_cogitate. @@ -130,6 +179,45 @@ def test_ndjson_single_request(mock_journal, monkeypatch, capsys): assert finish_events +def test_ndjson_cogitate_model_not_found_records_runtime_failure( + mock_journal, monkeypatch, capsys +): + from litellm.exceptions import NotFoundError + + _write_ready_brain(mock_journal) + + async def run_cogitate(config, on_event=None): + del config, on_event + raise NotFoundError("model not found", model="m", llm_provider="gemini") + + monkeypatch.setattr( + "solstone.think.providers.get_provider_module", + lambda _provider: SimpleNamespace(run_cogitate=run_cogitate), + ) + monkeypatch.setattr("solstone.think.talents.prepare_config", mock_prepare_config) + monkeypatch.setattr("sys.stdin", StringIO(json.dumps({"prompt": "use tools"}))) + mock_args = MagicMock() + mock_args.verbose = False + mock_args.dry_run = False + + from solstone.think.talents import main_async + + with patch("solstone.think.talents.setup_cli", return_value=mock_args): + asyncio.run(main_async()) + + captured = capsys.readouterr() + events = [json.loads(line) for line in captured.out.strip().split("\n") if line] + error_events = [event for event in events if event["event"] == "error"] + assert len(error_events) == 1 + assert error_events[0]["reason_code"] == "model_not_found" + assert error_events[0]["provider"] == "google" + + record = inspect_brain_state(NOW, journal_path=mock_journal)["record"] + assert record is not None + assert record["reason_code"] == "model_not_found" + assert record["evidence"]["cogitate"]["reason_code"] == "model_not_found" + + def test_ndjson_multiple_requests(mock_journal, monkeypatch, capsys): """Test processing multiple NDJSON requests from stdin.""" requests = [ diff --git a/tests/test_think_daily_idempotency.py b/tests/test_think_daily_idempotency.py index 906ca5f81..2b17c1e49 100644 --- a/tests/test_think_daily_idempotency.py +++ b/tests/test_think_daily_idempotency.py @@ -497,6 +497,31 @@ def test_run_daily_prompts_skips_two_deterministic_failures(daily_journal, monke ) +def test_run_daily_prompts_skips_one_model_not_found_failure( + daily_journal, monkeypatch +): + mod = importlib.import_module("solstone.think.thinking") + _write_health( + daily_journal, + DAY, + "001_daily.jsonl", + [_fail("alpha", reason_code="model_not_found")], + ) + dispatched: list[tuple[str, dict]] = [] + _install_daily_mocks(monkeypatch, mod, _single_configs("alpha"), dispatched) + + _run_daily_with_writer(mod, daily_journal, DAY, "002_daily.jsonl") + + assert dispatched == [] + skips = _skip_events(daily_journal, DAY, "002_daily.jsonl") + assert [(event["name"], event["reason"]) for event in skips] == [ + ("alpha", "deterministic_failure_no_retry") + ] + assert skips[0]["detail"] == ( + "1 same-day deterministic failures (model_not_found); not re-dispatching" + ) + + def test_run_daily_prompts_schema_invalid_retries_until_third_failure( daily_journal, monkeypatch ):