diff --git a/pyproject.toml b/pyproject.toml index 48561c01b..d4914c293 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -253,6 +253,7 @@ python_functions = ["test_*"] markers = [ "integration: exercises real local processes, builds, or persisted index contracts outside unit CI", "performance: asserts a wall-clock performance floor and is opt-in on an idle host", + "real_local_backend_probe: opts out of the deterministic bundled local backend probe monkeypatch", "xdist_group: marks tests to run in the same xdist worker group", ] timeout = 15 diff --git a/solstone/think/check.py b/solstone/think/check.py index 6aadd1b45..d31d71d87 100644 --- a/solstone/think/check.py +++ b/solstone/think/check.py @@ -87,13 +87,10 @@ def build_check_report() -> CheckReport: _linux_ram_check(), _disk_check(), ) - choice = local_cuda.select_local_backend( - probe, - local_cuda.CUDA_EMBEDDED_ARCH_SET, - local_cuda.CUDA_MIN_DRIVER_VERSION, - ) recommended_package = ( - "solstone-journal-cuda" if choice.backend == "cuda" else "solstone-journal" + "solstone-journal-cuda" + if arch == "x86_64" and probe.detected + else "solstone-journal" ) return CheckReport( diff --git a/solstone/think/install_provider.py b/solstone/think/install_provider.py index 9fa13224f..684b695fd 100644 --- a/solstone/think/install_provider.py +++ b/solstone/think/install_provider.py @@ -11,6 +11,7 @@ from __future__ import annotations import argparse import json +import logging import sys from typing import Any @@ -32,6 +33,7 @@ PARAKEET_DOWNLOAD_DISCLOSURE = ( "cache before it can run: the parakeet.cpp server binary from github.com " "(MIT) and the speech model from huggingface.co (CC-BY-4.0)." ) +LOG = logging.getLogger(__name__) def _render_fit_report(report: FitReport) -> None: @@ -86,6 +88,25 @@ def _status_exit_code(status: dict[str, Any]) -> int: return 1 if status.get("install_state") == "failed" else 0 +def _handle_install_failure(provider: str, exc: Exception) -> int: + print(str(exc), file=sys.stderr) + try: + status = read_install_status(name=provider) + except Exception as status_exc: + print( + f"could not read persisted {provider} install status: {status_exc}", + file=sys.stderr, + ) + LOG.warning( + "could not read persisted %s install status after failure", + provider, + exc_info=True, + ) + return 1 + print(json.dumps(status, indent=2)) + return 1 + + def _is_mlx_backend() -> bool: return mlx_install.is_mlx_platform_supported() @@ -117,12 +138,8 @@ def _install_mlx_local() -> int: lease=lease, attempt_status=attempt_status, ) - except ( - mlx_install.MLXInstallUnavailableError, - mlx_install.MLXVerificationError, - ) as exc: - print(str(exc), file=sys.stderr) - return 1 + except Exception as exc: + return _handle_install_failure("local", exc) finally: lease.release() print(json.dumps(status, indent=2)) @@ -172,9 +189,8 @@ def main() -> int: lease=lease, attempt_status=attempt_status, ) - except parakeet_install.ParakeetProviderError as exc: - print(str(exc), file=sys.stderr) - return 1 + except Exception as exc: + return _handle_install_failure("parakeet", exc) finally: lease.release() print(json.dumps(status, indent=2)) @@ -204,9 +220,8 @@ def main() -> int: owner={"entry": "install_provider"}, ) status = local_install.install_local(lease=lease, attempt_status=attempt_status) - except local_install.LocalProviderError as exc: - print(str(exc), file=sys.stderr) - return 1 + except Exception as exc: + return _handle_install_failure("local", exc) finally: lease.release() print(json.dumps(status, indent=2)) diff --git a/solstone/think/providers/fit_report.py b/solstone/think/providers/fit_report.py index 88ff913f5..b84d8afa9 100644 --- a/solstone/think/providers/fit_report.py +++ b/solstone/think/providers/fit_report.py @@ -82,6 +82,10 @@ def build_local_fit_report(model_id: str) -> FitReport: probe, local_install.CUDA_SERVER_PIN.embedded_arch_set, local_install.CUDA_SERVER_PIN.cuda_version, + local_install.probe_cuda_runtime_artifact_trust( + local_install.CUDA_SERVER_PIN + ), + persisted_installed_cuda=local_install.has_persisted_installed_cuda_target(), ) unknown_server = ( "CUDA llama-server OCI image" diff --git a/solstone/think/providers/local_cuda.py b/solstone/think/providers/local_cuda.py index ee4329008..6acf54ff5 100644 --- a/solstone/think/providers/local_cuda.py +++ b/solstone/think/providers/local_cuda.py @@ -10,6 +10,7 @@ import re import shutil import subprocess from dataclasses import dataclass +from enum import StrEnum from pathlib import Path from typing import TYPE_CHECKING @@ -42,6 +43,12 @@ class LocalCudaError(RuntimeError): self.reason_code = reason_code +class ArtifactTrust(StrEnum): + TRUSTED = "trusted" + ABSENT = "absent" + UNAVAILABLE = "unavailable" + + @dataclass(frozen=True) class NvidiaProbe: index: int | None @@ -269,7 +276,36 @@ def select_local_backend( probe: NvidiaProbe, arch_set: frozenset[str], cuda_version: int, + trust: ArtifactTrust, + *, + persisted_installed_cuda: bool, ) -> BackendChoice: + hardware_rejection = _hardware_backend_rejection(probe, arch_set, cuda_version) + if hardware_rejection is not None: + return hardware_rejection + + assert probe.compute_cap is not None + assert probe.driver_cuda_version is not None + cuda_reason = ( + f"compute_cap {probe.compute_cap} covered; " + f"driver CUDA {probe.driver_cuda_version} >= {cuda_version}" + ) + if trust == ArtifactTrust.TRUSTED or ( + trust == ArtifactTrust.UNAVAILABLE and persisted_installed_cuda + ): + return BackendChoice("cuda", cuda_reason) + + return BackendChoice( + "vulkan", + (f"{cuda_reason}; no trusted CUDA runtime artifact present"), + ) + + +def _hardware_backend_rejection( + probe: NvidiaProbe, + arch_set: frozenset[str], + cuda_version: int, +) -> BackendChoice | None: if not probe.detected: return BackendChoice("vulkan", "no NVIDIA GPU detected") if probe.compute_cap is None: @@ -289,21 +325,29 @@ def select_local_backend( "vulkan", f"driver CUDA {probe.driver_cuda_version} < required {cuda_version}", ) + return None - return BackendChoice( - "cuda", - ( - f"compute_cap {probe.compute_cap} covered; " - f"driver CUDA {probe.driver_cuda_version} >= {cuda_version}" - ), + +def resolve_local_backend(pin: CudaServerPin) -> BackendChoice: + probe = probe_nvidia_gpu() + hardware_rejection = _hardware_backend_rejection( + probe, + pin.embedded_arch_set, + pin.cuda_version, ) + if hardware_rejection is not None: + return hardware_rejection + from solstone.think.providers import local_install -def resolve_local_backend(pin: CudaServerPin) -> BackendChoice: + trust = local_install.probe_cuda_runtime_artifact_trust(pin) + persisted_installed_cuda = local_install.has_persisted_installed_cuda_target() return select_local_backend( - probe_nvidia_gpu(), + probe, pin.embedded_arch_set, pin.cuda_version, + trust, + persisted_installed_cuda=persisted_installed_cuda, ) @@ -323,6 +367,7 @@ def verify_cuda_pin_arch_set(text: str, declared: frozenset[str]) -> None: __all__ = [ + "ArtifactTrust", "BackendChoice", "CUDA_EMBEDDED_ARCH_SET", "CUDA_MIN_DRIVER_VERSION", diff --git a/solstone/think/providers/local_install.py b/solstone/think/providers/local_install.py index 0866e3c09..a7da1b9ea 100644 --- a/solstone/think/providers/local_install.py +++ b/solstone/think/providers/local_install.py @@ -10,6 +10,7 @@ access at import time. from __future__ import annotations import hashlib +import json import logging import platform import shutil @@ -23,6 +24,7 @@ from typing import Any, Callable from solstone.think.journal_config import read_journal_config from solstone.think.models import LOCAL_MODEL from solstone.think.providers.artifact_proof import ( + ProofResult, ReadinessOutcome, artifact_manifest_path, build_manifest, @@ -39,6 +41,7 @@ from solstone.think.providers.install_lease import ( from solstone.think.providers.install_state import ( IN_FLIGHT_STATES, InstallStatus, + InstallStatusMalformedError, assert_install_attempt_current, begin_or_replace_install_attempt, bump_progress, @@ -56,6 +59,7 @@ from solstone.think.providers.local import ( from solstone.think.providers.local_cuda import ( CUDA_EMBEDDED_ARCH_SET, CUDA_MIN_DRIVER_VERSION, + ArtifactTrust, ) from solstone.think.providers.memory import assess_memory from solstone.think.providers.oci_image import OciSignaturePolicy @@ -355,6 +359,78 @@ def _cuda_pin_identity( } +def _prove_cuda_runtime_artifact( + pin: CudaServerPin, + *, + journal_path: str | Path | None = None, +) -> ProofResult: + from solstone.think.providers import oci_image + + arch = _oci_arch() + wanted_files = pin.wanted_files_for_arch(arch) + return prove_cuda_sidecar( + provider=LOCAL_PROVIDER_NAME, + image_ref=pin.image_ref, + arch=arch, + wanted_files=wanted_files, + target_dir=cuda_binary_dir(), + pin_identity=_cuda_pin_identity(arch, wanted_files), + verifier=oci_image.verify_sidecar_install, + journal_path=journal_path, + ) + + +def probe_cuda_runtime_artifact_trust( + pin: CudaServerPin, + *, + journal_path: str | Path | None = None, +) -> ArtifactTrust: + try: + result = _prove_cuda_runtime_artifact(pin, journal_path=journal_path) + except Exception: + LOG.warning( + "CUDA runtime artifact trust probe failed; treating proof as unavailable", + exc_info=True, + ) + return ArtifactTrust.UNAVAILABLE + if result.status == "ready": + return ArtifactTrust.TRUSTED + if result.status == "missing-or-mismatched": + return ArtifactTrust.ABSENT + if result.status == "proof-unavailable": + return ArtifactTrust.UNAVAILABLE + LOG.warning( + "CUDA runtime artifact proof returned unknown status: %s", result.status + ) + return ArtifactTrust.UNAVAILABLE + + +def has_persisted_installed_cuda_target( + *, + journal_path: str | Path | None = None, +) -> bool: + try: + status = read_install_status( + name=LOCAL_PROVIDER_NAME, + journal_path=journal_path, + ) + target_json = status["target_fingerprint_json"] + if status["install_state"] != "installed" or target_json is None: + return False + target = json.loads(target_json) + except (InstallStatusMalformedError, ValueError): + LOG.warning( + "could not read persisted CUDA install target; not holding CUDA backend", + exc_info=True, + ) + return False + return ( + isinstance(target, dict) + and target.get("provider") == LOCAL_PROVIDER_NAME + and target.get("backend") == "cuda" + ) + + def target_fingerprint(model_id: str = LOCAL_MODEL) -> dict[str, Any]: from solstone.think.providers import local_cuda @@ -869,8 +945,6 @@ def _combined_artifact_status( def inspect_artifacts(model_id: str | None = None) -> dict[str, Any]: - from solstone.think.providers import oci_image - selected_model = normalize_model_id(model_id or LOCAL_MODEL) spec = LOCAL_MODEL_SPECS[selected_model] gguf_path = model_path(selected_model) @@ -885,18 +959,8 @@ def inspect_artifacts(model_id: str | None = None) -> dict[str, Any]: ) vulkan_payload = _proof_result_payload(vulkan_proof) - arch = _oci_arch() cuda_binary = cuda_binary_path() - cuda_wanted_files = CUDA_SERVER_PIN.wanted_files_for_arch(arch) - cuda_proof = prove_cuda_sidecar( - provider=LOCAL_PROVIDER_NAME, - image_ref=CUDA_SERVER_PIN.image_ref, - arch=arch, - wanted_files=cuda_wanted_files, - target_dir=cuda_binary_dir(), - pin_identity=_cuda_pin_identity(arch, cuda_wanted_files), - verifier=oci_image.verify_sidecar_install, - ) + cuda_proof = _prove_cuda_runtime_artifact(CUDA_SERVER_PIN) cuda_payload = _proof_result_payload(cuda_proof) model_proof = prove_manifest( artifact_manifest_path(model_dir(selected_model)), @@ -1042,6 +1106,7 @@ __all__ = [ "LLAMA_SERVER_PINS", "CudaServerPin", "LocalArtifacts", + "has_persisted_installed_cuda_target", "llama_server_artifact_key", "pin_for_current_platform", "binary_path_for_pin", @@ -1053,6 +1118,7 @@ __all__ = [ "install_model", "install_local", "install_hint", + "probe_cuda_runtime_artifact_trust", "probe_binary_runnable", "gpu_device_override", "inspect_readiness", diff --git a/tests/_baseline_harness.py b/tests/_baseline_harness.py index 0fe75d863..202992b11 100644 --- a/tests/_baseline_harness.py +++ b/tests/_baseline_harness.py @@ -137,12 +137,8 @@ def isolated_app_env(journal: Path) -> Iterator[Path]: previous = {key: os.environ.get(key) for key in overrides} os.environ.update(overrides) try: - from solstone.think.providers import local_cuda, local_vulkan + from solstone.think.providers import local_cuda, local_install, local_vulkan - deterministic_backend = local_cuda.BackendChoice( - "vulkan", - "test default: no CUDA host", - ) deterministic_devices = [ local_vulkan.VulkanDevice( 1, @@ -154,8 +150,26 @@ def isolated_app_env(journal: Path) -> Iterator[Path]: with ( patch.object( local_cuda, - "resolve_local_backend", - lambda _pin: deterministic_backend, + "probe_nvidia_gpu", + lambda: local_cuda.NvidiaProbe( + index=None, + compute_cap=None, + driver_cuda_version=None, + vram_mib=None, + tiering_memory_mib=None, + memory_source=local_cuda.MEMORY_SOURCE_UNAVAILABLE, + detected=False, + ), + ), + patch.object( + local_install, + "probe_cuda_runtime_artifact_trust", + lambda _pin, **_kwargs: local_cuda.ArtifactTrust.ABSENT, + ), + patch.object( + local_install, + "has_persisted_installed_cuda_target", + lambda **_kwargs: False, ), patch.object( local_vulkan, diff --git a/tests/conftest.py b/tests/conftest.py index acc7c3e45..7c8b81065 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -190,14 +190,33 @@ def set_test_journal_path(monkeypatch, _isolate_os_environ): @pytest.fixture(autouse=True) -def _default_local_backend_vulkan(monkeypatch): - from solstone.think.providers import local_cuda - - monkeypatch.setattr( - local_cuda, - "resolve_local_backend", - lambda pin: local_cuda.BackendChoice("vulkan", "test default: no CUDA host"), - ) +def _default_local_backend_vulkan(monkeypatch, request): + from solstone.think.providers import local_cuda, local_install + + if request.node.get_closest_marker("real_local_backend_probe") is None: + monkeypatch.setattr( + local_cuda, + "probe_nvidia_gpu", + lambda: local_cuda.NvidiaProbe( + index=None, + compute_cap=None, + driver_cuda_version=None, + vram_mib=None, + tiering_memory_mib=None, + memory_source=local_cuda.MEMORY_SOURCE_UNAVAILABLE, + detected=False, + ), + ) + monkeypatch.setattr( + local_install, + "probe_cuda_runtime_artifact_trust", + lambda _pin, **_kwargs: local_cuda.ArtifactTrust.ABSENT, + ) + monkeypatch.setattr( + local_install, + "has_persisted_installed_cuda_target", + lambda **_kwargs: False, + ) @pytest.fixture(autouse=True) diff --git a/tests/test_artifact_proof.py b/tests/test_artifact_proof.py index 428bcbc2e..107b5ab10 100644 --- a/tests/test_artifact_proof.py +++ b/tests/test_artifact_proof.py @@ -386,10 +386,16 @@ def test_cuda_sidecar_success_is_cached_without_second_verifier( encoding="utf-8", ) calls = 0 + hash_calls: list[Path] = [] + real_hash = artifact_proof._sha256_file - def verifier(_image_ref, _arch, _wanted, _target) -> bool: + def verifier(_image_ref, _arch, wanted, verify_target) -> bool: nonlocal calls calls += 1 + for name in wanted: + path = verify_target / name + hash_calls.append(path) + real_hash(path) return True first = prove_cuda_sidecar( @@ -415,6 +421,7 @@ def test_cuda_sidecar_success_is_cached_without_second_verifier( assert second.ready assert second.cache_hit is True assert calls == 1 + assert hash_calls == [target / "llama-server"] def test_cuda_sidecar_absent_is_repair_needed(tmp_path, monkeypatch) -> None: diff --git a/tests/test_check.py b/tests/test_check.py index 0600b287b..3fa670b09 100644 --- a/tests/test_check.py +++ b/tests/test_check.py @@ -163,7 +163,7 @@ def test_linux_cuda_too_small_blocks( capsys.readouterr() -def test_linux_nvidia_vulkan_backend_recommendation( +def test_linux_x86_64_nvidia_recommends_cuda_package( monkeypatch: pytest.MonkeyPatch, capsys: pytest.CaptureFixture, ) -> None: @@ -176,6 +176,24 @@ def test_linux_nvidia_vulkan_backend_recommendation( result = check.build_check_report() + assert _checks(result)["gpu"].severity == "ok" + assert result.report.overall == "ok" + assert result.recommended_package == "solstone-journal-cuda" + assert check.main([]) == 0 + output = capsys.readouterr().out + assert "solstone-journal-cuda" in output + + +def test_linux_aarch64_nvidia_recommends_base_package( + monkeypatch: pytest.MonkeyPatch, + capsys: pytest.CaptureFixture, +) -> None: + _patch_linux_ok(monkeypatch) + _patch_platform(monkeypatch, arch="aarch64") + monkeypatch.setattr(local_cuda, "probe_nvidia_gpu", lambda: _nvidia_probe()) + + result = check.build_check_report() + assert _checks(result)["gpu"].severity == "ok" assert result.report.overall == "ok" assert result.recommended_package == "solstone-journal" @@ -185,6 +203,22 @@ def test_linux_nvidia_vulkan_backend_recommendation( assert "solstone-journal-cuda" not in output +def test_linux_recommendation_does_not_select_local_backend( + monkeypatch: pytest.MonkeyPatch, +) -> None: + _patch_linux_ok(monkeypatch) + monkeypatch.setattr(local_cuda, "probe_nvidia_gpu", lambda: _nvidia_probe()) + monkeypatch.setattr( + local_cuda, + "select_local_backend", + lambda *_args, **_kwargs: pytest.fail("backend selection is not a check input"), + ) + + result = check.build_check_report() + + assert result.recommended_package == "solstone-journal-cuda" + + def test_linux_nvidia_small_single_discrete_mentions_cpu_transcription( monkeypatch: pytest.MonkeyPatch, ) -> None: diff --git a/tests/test_fit_report.py b/tests/test_fit_report.py index 2a5671d1d..3e354921d 100644 --- a/tests/test_fit_report.py +++ b/tests/test_fit_report.py @@ -170,6 +170,62 @@ def test_local_gpu_check_uses_vulkan_when_nvidia_probe_is_unavailable( ) +def test_build_local_fit_report_uses_cuda_artifact_trust_probe( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + monkeypatch.setattr(fit_report.sys, "platform", "linux") + monkeypatch.setattr( + local_cuda, + "probe_nvidia_gpu", + lambda: local_cuda.NvidiaProbe( + index=0, + compute_cap="sm_86", + driver_cuda_version=14, + vram_mib=24564, + tiering_memory_mib=24564, + memory_source=local_cuda.MEMORY_SOURCE_NVIDIA_VRAM, + detected=True, + ), + ) + monkeypatch.setattr( + local_install, + "probe_cuda_runtime_artifact_trust", + lambda _pin: local_cuda.ArtifactTrust.ABSENT, + ) + monkeypatch.setattr( + local_install, + "has_persisted_installed_cuda_target", + lambda: False, + ) + monkeypatch.setattr( + local_vulkan, + "detect_gpus", + lambda: [_vulkan_device(vram_mib=24564)], + ) + monkeypatch.setattr(local_vulkan, "gpu_probe_ok", lambda: True) + monkeypatch.setattr(local_install, "cache_root", lambda: tmp_path) + monkeypatch.setattr( + fit_report, + "assess_memory", + lambda required, *, block_below_floor: MemoryVerdict( + available_bytes=required, + required_bytes=required, + severity="ok", + ), + ) + monkeypatch.setattr(fit_report, "free_bytes", lambda _path: 500 * 1024**3) + + report = fit_report.build_local_fit_report(local_install.LOCAL_MODEL) + + gpu = next(check for check in report.checks if check.name == "gpu") + assert gpu.severity == "ok" + assert ( + "resolved backend is vulkan: compute_cap sm_86 covered; " + "driver CUDA 14 >= 13; no trusted CUDA runtime artifact present" + ) in gpu.detail + + def test_disk_unknown_size_warns_when_known_size_fits( tmp_path: Path, monkeypatch: pytest.MonkeyPatch, diff --git a/tests/test_install_provider.py b/tests/test_install_provider.py index 4770cf170..1c46184e5 100644 --- a/tests/test_install_provider.py +++ b/tests/test_install_provider.py @@ -320,6 +320,13 @@ def test_install_provider_local_skips_fit_report_when_ready(monkeypatch, capsys) def test_install_provider_local_expected_error_returns_nonzero(monkeypatch, capsys): + persisted = { + "provider": "local", + "install_state": "failed", + "install_error": "blocked detail", + "error_code": "host_unfit", + } + def install_local(**_kwargs): raise install_provider.local_install.LocalProviderError( "host_unfit", "blocked detail" @@ -336,12 +343,188 @@ def test_install_provider_local_expected_error_returns_nonzero(monkeypatch, caps fit_report, "build_local_fit_report", lambda _model: _fit("blocked") ) monkeypatch.setattr(install_provider.local_install, "install_local", install_local) + monkeypatch.setattr( + install_provider, + "read_install_status", + lambda name: persisted if name == "local" else pytest.fail(name), + ) + + assert install_provider.main() == 1 + + captured = capsys.readouterr() + assert lease.released is True + assert "blocked detail" in captured.err + assert json.loads(captured.out) == persisted + + +def test_install_provider_parakeet_expected_error_prints_persisted_status( + monkeypatch, + capsys, +): + persisted = { + "provider": "parakeet", + "install_state": "failed", + "install_error": "blocked detail", + "error_code": "host_unfit", + } + + def install_parakeet(**_kwargs): + raise install_provider.parakeet_install.ParakeetProviderError( + "host_unfit", + "blocked detail", + ) + + monkeypatch.setattr(sys, "argv", ["journal install-provider", "parakeet"]) + monkeypatch.setattr( + install_provider.parakeet_install, + "inspect_readiness", + lambda: _readiness("parakeet", ready=False), + ) + monkeypatch.setattr( + install_provider.parakeet_install, + "target_fingerprint", + lambda: {"provider": "parakeet"}, + ) + lease = _patch_lease_and_attempt(monkeypatch, "parakeet") + monkeypatch.setattr(fit_report, "build_parakeet_fit_report", lambda: _fit("ok")) + monkeypatch.setattr( + install_provider.parakeet_install, + "install_parakeet", + install_parakeet, + ) + monkeypatch.setattr( + install_provider, + "read_install_status", + lambda name: persisted if name == "parakeet" else pytest.fail(name), + ) + + assert install_provider.main() == 1 + + captured = capsys.readouterr() + assert lease.released is True + assert "blocked detail" in captured.err + assert json.loads(captured.out) == persisted + + +def test_install_provider_mlx_expected_error_prints_persisted_status( + monkeypatch, + capsys, +): + persisted = { + "provider": "local", + "install_state": "failed", + "install_error": "mlx unavailable", + "error_code": "mlx_unavailable", + } + + def install_local_mlx(*_args, **_kwargs): + raise install_provider.mlx_install.MLXInstallUnavailableError("mlx unavailable") + + monkeypatch.setattr(sys, "argv", ["journal install-provider", "local"]) + monkeypatch.setattr(install_provider, "_is_mlx_backend", lambda: True) + monkeypatch.setattr( + install_provider.mlx_install, + "resolve_model_spec", + lambda: type("Spec", (), {"name": "model"})(), + ) + monkeypatch.setattr( + install_provider.mlx_install, + "inspect_readiness", + lambda _model: _readiness("local", ready=False), + ) + monkeypatch.setattr( + install_provider.mlx_install, + "target_fingerprint", + lambda _model: {"provider": "local", "runtime": "mlx"}, + ) + lease = _patch_lease_and_attempt(monkeypatch, "local") + monkeypatch.setattr(fit_report, "build_mlx_fit_report", lambda _model: _fit("ok")) + monkeypatch.setattr( + install_provider.mlx_install, + "install_local_mlx", + install_local_mlx, + ) + monkeypatch.setattr( + install_provider, + "read_install_status", + lambda name: persisted if name == "local" else pytest.fail(name), + ) + + assert install_provider.main() == 1 + + captured = capsys.readouterr() + assert lease.released is True + assert "mlx unavailable" in captured.err + assert json.loads(captured.out) == persisted + + +def test_install_provider_unexpected_error_prints_persisted_status( + monkeypatch, + capsys, +): + persisted = { + "provider": "local", + "install_state": "failed", + "install_error": "boom", + "error_code": None, + } + + def install_local(**_kwargs): + raise RuntimeError("boom") + + monkeypatch.setattr(sys, "argv", ["journal install-provider", "local"]) + monkeypatch.setattr( + install_provider.local_install, + "inspect_readiness", + lambda: _readiness("local", ready=False), + ) + lease = _patch_lease_and_attempt(monkeypatch, "local") + monkeypatch.setattr(fit_report, "build_local_fit_report", lambda _model: _fit("ok")) + monkeypatch.setattr(install_provider.local_install, "install_local", install_local) + monkeypatch.setattr( + install_provider, + "read_install_status", + lambda name: persisted if name == "local" else pytest.fail(name), + ) + + assert install_provider.main() == 1 + + captured = capsys.readouterr() + assert lease.released is True + assert "boom" in captured.err + assert json.loads(captured.out) == persisted + + +def test_install_provider_failure_status_read_error_keeps_stdout_empty( + monkeypatch, + capsys, +): + def install_local(**_kwargs): + raise install_provider.local_install.LocalProviderError( + "host_unfit", + "blocked detail", + ) + + def fail_read_status(*_args, **_kwargs): + raise ValueError("malformed status") + + monkeypatch.setattr(sys, "argv", ["journal install-provider", "local"]) + monkeypatch.setattr( + install_provider.local_install, + "inspect_readiness", + lambda: _readiness("local", ready=False), + ) + lease = _patch_lease_and_attempt(monkeypatch, "local") + monkeypatch.setattr(fit_report, "build_local_fit_report", lambda _model: _fit("ok")) + monkeypatch.setattr(install_provider.local_install, "install_local", install_local) + monkeypatch.setattr(install_provider, "read_install_status", fail_read_status) assert install_provider.main() == 1 captured = capsys.readouterr() assert lease.released is True assert "blocked detail" in captured.err + assert "could not read persisted local install status" in captured.err assert captured.out == "" diff --git a/tests/test_install_state.py b/tests/test_install_state.py index af30e0d86..dcbde6577 100644 --- a/tests/test_install_state.py +++ b/tests/test_install_state.py @@ -11,7 +11,13 @@ from typing import get_args import pytest -from solstone.think.providers import install_state +from solstone.think.models import LOCAL_MODEL +from solstone.think.providers import ( + install_state, + local_cuda, + local_install, + local_vulkan, +) from solstone.think.providers.install_state import ( IN_FLIGHT_STATES, TERMINAL_STATES, @@ -254,6 +260,99 @@ def test_migration_api_removes_legacy_status_fields(tmp_path, monkeypatch) -> No assert data["providers"]["local"] == {"vulkan_device_index": "1"} +def test_legacy_local_vulkan_adoption_when_cuda_artifact_absent_on_covered_host( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + _set_journal(tmp_path, monkeypatch) + config_path = tmp_path / "config" / "journal.json" + config_path.parent.mkdir(parents=True) + pin = local_install.pin_for_current_platform() + artifact_key = local_install.llama_server_artifact_key() + binary_path = local_install.binary_path_for_pin(artifact_key, pin) + model_path = local_install.model_path(LOCAL_MODEL) + mmproj_path = local_install.mmproj_path(LOCAL_MODEL) + binary_path.parent.mkdir(parents=True, exist_ok=True) + binary_path.write_text("llama", encoding="utf-8") + binary_path.chmod(0o755) + model_path.parent.mkdir(parents=True, exist_ok=True) + model_path.write_text("model", encoding="utf-8") + if mmproj_path is not None: + mmproj_path.write_text("mmproj", encoding="utf-8") + config_path.write_text( + json.dumps( + { + "providers": { + "bundled": { + "local": { + "install_state": "installed", + "binary_artifact": artifact_key, + "binary_sha256": pin["sha256"], + "binary_path": str(binary_path), + "model_id": LOCAL_MODEL, + "model_path": str(model_path), + "mmproj_path": str(mmproj_path) + if mmproj_path is not None + else None, + } + } + } + } + ) + + "\n", + encoding="utf-8", + ) + monkeypatch.setattr( + local_cuda, + "probe_nvidia_gpu", + lambda: local_cuda.NvidiaProbe( + index=0, + compute_cap="sm_121", + driver_cuda_version=16, + vram_mib=24564, + tiering_memory_mib=24564, + memory_source=local_cuda.MEMORY_SOURCE_NVIDIA_VRAM, + detected=True, + ), + ) + monkeypatch.setattr( + local_install, + "probe_cuda_runtime_artifact_trust", + lambda _pin, **_kwargs: local_cuda.ArtifactTrust.ABSENT, + ) + monkeypatch.setattr( + local_install, + "has_persisted_installed_cuda_target", + lambda **_kwargs: False, + ) + monkeypatch.setattr( + local_vulkan, + "detect_gpus", + lambda: [ + local_vulkan.VulkanDevice( + 0, + "Test Vulkan GPU", + local_vulkan.VK_TYPE_DISCRETE, + 8192, + ) + ], + ) + monkeypatch.setattr(local_vulkan, "gpu_probe_ok", lambda: True) + monkeypatch.setattr(local_install, "_verify_sha256", lambda _path, _sha: None) + + result = install_state.migrate_legacy_provider_artifact_truth(journal_path=tmp_path) + + status = read_install_status(name="local", journal_path=tmp_path) + target = json.loads(str(status["target_fingerprint_json"])) + assert result["actions"][0]["action"] == "promoted" + assert status["install_state"] == "installed" + assert target["backend"] == "vulkan" + assert target["backend_reason"] == ( + "compute_cap sm_121 covered; driver CUDA 16 >= 13; " + "no trusted CUDA runtime artifact present" + ) + + def test_two_process_stale_transition_one_writer_wins(tmp_path, monkeypatch) -> None: _set_journal(tmp_path, monkeypatch) ctx = multiprocessing.get_context("spawn") diff --git a/tests/test_local_cuda.py b/tests/test_local_cuda.py index eb86a6317..3e94bb3f5 100644 --- a/tests/test_local_cuda.py +++ b/tests/test_local_cuda.py @@ -10,6 +10,7 @@ import pytest from solstone.think.providers import local_cuda ARCH_SET = frozenset({"sm_86", "sm_89", "sm_120a", "sm_121a"}) +pytestmark = pytest.mark.real_local_backend_probe def _completed(stdout: str, returncode: int = 0) -> SimpleNamespace: @@ -256,28 +257,149 @@ def test_probe_nvidia_gpu_garbled_output_fails_closed( def test_select_local_backend_matrix() -> None: cases = [ - ("sm_75", 13, "vulkan", "compute_cap sm_75 not in CUDA image arch set"), - ("sm_75", 12, "vulkan", "compute_cap sm_75 not in CUDA image arch set"), - ("sm_86", 13, "cuda", "compute_cap sm_86 covered; driver CUDA 13 >= 13"), - ("sm_86", 12, "vulkan", "driver CUDA 12 < required 13"), - ("sm_121", 13, "cuda", "compute_cap sm_121 covered; driver CUDA 13 >= 13"), - ("sm_121", 12, "vulkan", "driver CUDA 12 < required 13"), - (None, 13, "vulkan", "NVIDIA compute capability unreadable"), - (None, 12, "vulkan", "NVIDIA compute capability unreadable"), + ( + False, + None, + None, + local_cuda.ArtifactTrust.TRUSTED, + True, + "vulkan", + "no NVIDIA GPU detected", + ), + ( + True, + None, + 13, + local_cuda.ArtifactTrust.ABSENT, + False, + "vulkan", + "NVIDIA compute capability unreadable", + ), + ( + True, + "sm_75", + 13, + local_cuda.ArtifactTrust.UNAVAILABLE, + True, + "vulkan", + "compute_cap sm_75 not in CUDA image arch set", + ), + ( + True, + "sm_75", + 12, + local_cuda.ArtifactTrust.ABSENT, + True, + "vulkan", + "compute_cap sm_75 not in CUDA image arch set", + ), + ( + True, + "sm_86", + 13, + local_cuda.ArtifactTrust.TRUSTED, + False, + "cuda", + "compute_cap sm_86 covered; driver CUDA 13 >= 13", + ), + ( + True, + "sm_86", + 12, + local_cuda.ArtifactTrust.ABSENT, + True, + "vulkan", + "driver CUDA 12 < required 13", + ), + ( + True, + "sm_121", + 13, + local_cuda.ArtifactTrust.TRUSTED, + False, + "cuda", + "compute_cap sm_121 covered; driver CUDA 13 >= 13", + ), + ( + True, + "sm_121", + 12, + local_cuda.ArtifactTrust.UNAVAILABLE, + True, + "vulkan", + "driver CUDA 12 < required 13", + ), + ( + True, + None, + 13, + local_cuda.ArtifactTrust.ABSENT, + False, + "vulkan", + "NVIDIA compute capability unreadable", + ), + ( + True, + None, + 12, + local_cuda.ArtifactTrust.UNAVAILABLE, + True, + "vulkan", + "NVIDIA compute capability unreadable", + ), + ( + True, + "sm_121", + 13, + local_cuda.ArtifactTrust.ABSENT, + False, + "vulkan", + ( + "compute_cap sm_121 covered; driver CUDA 13 >= 13; " + "no trusted CUDA runtime artifact present" + ), + ), + ( + True, + "sm_121", + 13, + local_cuda.ArtifactTrust.UNAVAILABLE, + True, + "cuda", + "compute_cap sm_121 covered; driver CUDA 13 >= 13", + ), + ( + True, + "sm_121", + 13, + local_cuda.ArtifactTrust.UNAVAILABLE, + False, + "vulkan", + ( + "compute_cap sm_121 covered; driver CUDA 13 >= 13; " + "no trusted CUDA runtime artifact present" + ), + ), ] - for compute_cap, driver_cuda, backend, reason in cases: + for detected, compute_cap, driver_cuda, trust, persisted, backend, reason in cases: probe = local_cuda.NvidiaProbe( - index=0, + index=0 if detected else None, compute_cap=compute_cap, driver_cuda_version=driver_cuda, vram_mib=24564, tiering_memory_mib=24564, memory_source=local_cuda.MEMORY_SOURCE_NVIDIA_VRAM, - detected=True, + detected=detected, ) - choice = local_cuda.select_local_backend(probe, ARCH_SET, 13) + choice = local_cuda.select_local_backend( + probe, + ARCH_SET, + 13, + trust, + persisted_installed_cuda=persisted, + ) assert choice == local_cuda.BackendChoice(backend, reason) @@ -295,6 +417,8 @@ def test_select_local_backend_no_gpu_detected() -> None: ), ARCH_SET, 13, + local_cuda.ArtifactTrust.TRUSTED, + persisted_installed_cuda=True, ) assert choice == local_cuda.BackendChoice("vulkan", "no NVIDIA GPU detected") @@ -313,6 +437,8 @@ def test_select_local_backend_driver_cuda_unreadable() -> None: ), ARCH_SET, 13, + local_cuda.ArtifactTrust.TRUSTED, + persisted_installed_cuda=False, ) assert choice == local_cuda.BackendChoice( diff --git a/tests/test_local_install.py b/tests/test_local_install.py index b106c0593..2c2d48d7c 100644 --- a/tests/test_local_install.py +++ b/tests/test_local_install.py @@ -23,11 +23,19 @@ from solstone.think.providers import ( oci_image, ) from solstone.think.providers.artifact_proof import ( + ProofResult, ReadinessOutcome, artifact_manifest_path, prove_manifest, ) -from solstone.think.providers.install_state import read_install_status +from solstone.think.providers.install_state import ( + begin_or_replace_install_attempt, + canonical_fingerprint, + fingerprint_sha256, + read_install_status, + transition_state, + write_install_status, +) from solstone.think.providers.local import LOCAL_MODEL_SPECS from solstone.think.providers.local_endpoint import resolve_local_endpoint @@ -153,12 +161,242 @@ def _fit(severity: fit_report.FitSeverity) -> fit_report.FitReport: ) -@pytest.fixture(autouse=True) -def _default_vulkan_backend(monkeypatch: pytest.MonkeyPatch) -> None: +def _covered_nvidia_probe( + *, + compute_cap: str = "sm_121", + driver_cuda_version: int = 13, + vram_mib: int = 24564, +) -> local_cuda.NvidiaProbe: + return local_cuda.NvidiaProbe( + index=0, + compute_cap=compute_cap, + driver_cuda_version=driver_cuda_version, + vram_mib=vram_mib, + tiering_memory_mib=vram_mib, + memory_source=local_cuda.MEMORY_SOURCE_NVIDIA_VRAM, + detected=True, + ) + + +def _patch_backend_inputs( + monkeypatch: pytest.MonkeyPatch, + *, + compute_cap: str, + driver_cuda_version: int, + trust: local_cuda.ArtifactTrust, + persisted_installed_cuda: bool = False, +) -> None: monkeypatch.setattr( local_cuda, - "resolve_local_backend", - lambda _pin: local_cuda.BackendChoice("vulkan", "test vulkan"), + "probe_nvidia_gpu", + lambda: _covered_nvidia_probe( + compute_cap=compute_cap, + driver_cuda_version=driver_cuda_version, + ), + ) + monkeypatch.setattr( + local_install, + "probe_cuda_runtime_artifact_trust", + lambda _pin, **_kwargs: trust, + ) + monkeypatch.setattr( + local_install, + "has_persisted_installed_cuda_target", + lambda **_kwargs: persisted_installed_cuda, + ) + + +def _force_cuda_backend(monkeypatch: pytest.MonkeyPatch) -> None: + _patch_backend_inputs( + monkeypatch, + compute_cap="sm_121", + driver_cuda_version=13, + trust=local_cuda.ArtifactTrust.TRUSTED, + ) + + +@pytest.mark.parametrize( + ("status", "expected"), + [ + ("ready", local_cuda.ArtifactTrust.TRUSTED), + ("missing-or-mismatched", local_cuda.ArtifactTrust.ABSENT), + ("proof-unavailable", local_cuda.ArtifactTrust.UNAVAILABLE), + ], +) +@pytest.mark.real_local_backend_probe +def test_probe_cuda_runtime_artifact_trust_maps_proof_status( + monkeypatch: pytest.MonkeyPatch, + status: str, + expected: local_cuda.ArtifactTrust, +) -> None: + monkeypatch.setattr( + local_install, + "_prove_cuda_runtime_artifact", + lambda _pin, **_kwargs: ProofResult(status, "test"), + ) + + assert ( + local_install.probe_cuda_runtime_artifact_trust(local_install.CUDA_SERVER_PIN) + == expected + ) + + +@pytest.mark.real_local_backend_probe +def test_probe_cuda_runtime_artifact_trust_contains_unexpected_exception( + monkeypatch: pytest.MonkeyPatch, + caplog: pytest.LogCaptureFixture, +) -> None: + def fail_proof(_pin, **_kwargs): + raise ValueError("required artifact is not a regular file") + + monkeypatch.setattr(local_install, "_prove_cuda_runtime_artifact", fail_proof) + + trust = local_install.probe_cuda_runtime_artifact_trust( + local_install.CUDA_SERVER_PIN + ) + + assert trust == local_cuda.ArtifactTrust.UNAVAILABLE + assert "trust probe failed" in caplog.text + + +@pytest.mark.real_local_backend_probe +def test_probe_cuda_runtime_artifact_trust_contains_non_regular_wanted_file( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + _init_journal(tmp_path, monkeypatch) + monkeypatch.setattr(local_install.platform, "machine", lambda: "x86_64") + target = local_install.cuda_binary_dir() + target.mkdir(parents=True) + wanted_files = local_install.CUDA_SERVER_PIN.wanted_files_for_arch("amd64") + (target / wanted_files[0]).mkdir() + for name in wanted_files[1:]: + (target / name).write_text(name, encoding="utf-8") + (target / ".oci-install.json").write_text( + json.dumps( + { + "image_ref": local_install.CUDA_SERVER_PIN.image_ref, + "arch": "amd64", + "files": {name: "0" * 64 for name in wanted_files}, + } + ) + + "\n", + encoding="utf-8", + ) + + trust = local_install.probe_cuda_runtime_artifact_trust( + local_install.CUDA_SERVER_PIN, + journal_path=tmp_path, + ) + + assert trust == local_cuda.ArtifactTrust.UNAVAILABLE + + +@pytest.mark.real_local_backend_probe +def test_has_persisted_installed_cuda_target_reads_installed_backend( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + _init_journal(tmp_path, monkeypatch) + cuda_target = { + "provider": "local", + "runtime": "llama.cpp", + "backend": "cuda", + "model_pin": {"model_id": LOCAL_MODEL}, + } + vulkan_target = {**cuda_target, "backend": "vulkan"} + + status = begin_or_replace_install_attempt("local", cuda_target) + write_install_status(transition_state(status, new_state="installed")) + assert ( + local_install.has_persisted_installed_cuda_target(journal_path=tmp_path) is True + ) + + status = begin_or_replace_install_attempt("local", vulkan_target) + write_install_status(transition_state(status, new_state="installed")) + assert ( + local_install.has_persisted_installed_cuda_target(journal_path=tmp_path) + is False + ) + + +@pytest.mark.real_local_backend_probe +def test_has_persisted_installed_cuda_target_treats_bad_status_as_false( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + _init_journal(tmp_path, monkeypatch) + path = tmp_path / "health" / "providers" / "local.json" + path.parent.mkdir(parents=True) + path.write_text("{not-json", encoding="utf-8") + + assert ( + local_install.has_persisted_installed_cuda_target(journal_path=tmp_path) + is False + ) + + +def test_target_fingerprint_uses_vulkan_when_cuda_artifact_absent_on_covered_host( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + _init_journal(tmp_path, monkeypatch) + _patch_backend_inputs( + monkeypatch, + compute_cap="sm_86", + driver_cuda_version=14, + trust=local_cuda.ArtifactTrust.ABSENT, + ) + + fingerprint = local_install.target_fingerprint(LOCAL_MODEL) + + assert fingerprint["backend"] == "vulkan" + assert fingerprint["backend_reason"] == ( + "compute_cap sm_86 covered; driver CUDA 14 >= 13; " + "no trusted CUDA runtime artifact present" + ) + + +def test_target_fingerprint_holds_cuda_when_trust_unavailable_and_cuda_installed( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + _init_journal(tmp_path, monkeypatch) + _patch_backend_inputs( + monkeypatch, + compute_cap="sm_89", + driver_cuda_version=15, + trust=local_cuda.ArtifactTrust.UNAVAILABLE, + persisted_installed_cuda=True, + ) + + fingerprint = local_install.target_fingerprint(LOCAL_MODEL) + + assert fingerprint["backend"] == "cuda" + assert fingerprint["backend_reason"] == ( + "compute_cap sm_89 covered; driver CUDA 15 >= 13" + ) + + +def test_target_fingerprint_uses_vulkan_when_trust_unavailable_without_cuda_install( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + _init_journal(tmp_path, monkeypatch) + _patch_backend_inputs( + monkeypatch, + compute_cap="sm_121", + driver_cuda_version=16, + trust=local_cuda.ArtifactTrust.UNAVAILABLE, + persisted_installed_cuda=False, + ) + + fingerprint = local_install.target_fingerprint(LOCAL_MODEL) + + assert fingerprint["backend"] == "vulkan" + assert fingerprint["backend_reason"] == ( + "compute_cap sm_121 covered; driver CUDA 16 >= 13; " + "no trusted CUDA runtime artifact present" ) @@ -712,11 +950,7 @@ def test_install_llama_server_cuda_uses_arch_specific_oci_wanted_files( ): _init_journal(tmp_path, monkeypatch) monkeypatch.setattr(local_install.platform, "machine", lambda: machine) - monkeypatch.setattr( - local_cuda, - "resolve_local_backend", - lambda _pin: local_cuda.BackendChoice("cuda", "test cuda"), - ) + _force_cuda_backend(monkeypatch) pull_calls: list[tuple[str, str, tuple[str, ...], Path]] = [] def fake_pull_and_install( @@ -758,11 +992,7 @@ def test_install_llama_server_cuda_uses_arch_specific_oci_wanted_files( def test_install_llama_server_cuda_preserves_oci_failure_reason(tmp_path, monkeypatch): _init_journal(tmp_path, monkeypatch) - monkeypatch.setattr( - local_cuda, - "resolve_local_backend", - lambda _pin: local_cuda.BackendChoice("cuda", "test cuda"), - ) + _force_cuda_backend(monkeypatch) def fail_pull(*_args, **_kwargs): raise oci_image.OciImageError( @@ -1126,6 +1356,88 @@ def test_install_local_reinstalls_runtime_when_binary_record_stale( assert calls == ["llama_server", "model"] +def test_install_local_replaces_failed_cuda_attempt_with_vulkan_target( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + _init_journal(tmp_path, monkeypatch) + _patch_backend_inputs( + monkeypatch, + compute_cap="sm_89", + driver_cuda_version=15, + trust=local_cuda.ArtifactTrust.ABSENT, + ) + stale_cuda = { + "provider": "local", + "runtime": "llama.cpp", + "backend": "cuda", + "backend_reason": "old cuda", + "runtime_pin": local_install._cuda_pin_identity(), + "model_pin": local_install._model_pin_identity(LOCAL_MODEL), + } + stale_status = begin_or_replace_install_attempt( + "local", + stale_cuda, + initial_state="resolving", + ) + write_install_status( + transition_state( + stale_status, + new_state="failed", + error="the pinned image has no matching signature", + error_code="signature_verify_failed", + ) + ) + stale_sha = fingerprint_sha256(canonical_fingerprint(stale_cuda)) + monkeypatch.setattr( + fit_report, + "build_local_fit_report", + lambda model_id: _fit("ok"), + ) + calls: list[str] = [] + readiness_calls = 0 + + def fake_readiness(model_id: str) -> ReadinessOutcome: + nonlocal readiness_calls + readiness_calls += 1 + return _fake_local_readiness( + binary_installed=readiness_calls > 1, + model_installed=readiness_calls > 1, + binary_path=local_install.binary_path_for_pin(), + model_path=local_install.model_path(model_id), + mmproj_path=local_install.mmproj_path(model_id), + ) + + def fake_install_llama_server(**_kwargs): + calls.append("llama_server") + return {"install_state": "verifying"} + + def fake_install_model(model_id: str, **_kwargs): + calls.append("model") + return {"install_state": "verifying", "model_id": model_id} + + monkeypatch.setattr(local_install, "inspect_readiness", fake_readiness) + monkeypatch.setattr( + local_install, + "install_llama_server", + fake_install_llama_server, + ) + monkeypatch.setattr(local_install, "install_model", fake_install_model) + + result = local_install.install_local(LOCAL_MODEL) + + status = _local_status() + target = json.loads(str(status["target_fingerprint_json"])) + assert result["install_state"] == "installed" + assert status["target_fingerprint_sha256"] != stale_sha + assert target["backend"] == "vulkan" + assert target["backend_reason"] == ( + "compute_cap sm_89 covered; driver CUDA 15 >= 13; " + "no trusted CUDA runtime artifact present" + ) + assert calls == ["llama_server", "model"] + + def test_ensure_artifacts_installed_returns_binary_gguf_and_optional_mmproj( tmp_path, monkeypatch ): @@ -1272,11 +1584,7 @@ def test_inspect_readiness_cuda_uses_sidecar_full_set( from solstone.think.providers import oci_image _init_journal(tmp_path, monkeypatch) - monkeypatch.setattr( - local_cuda, - "resolve_local_backend", - lambda _pin: local_cuda.BackendChoice("cuda", "test cuda"), - ) + _force_cuda_backend(monkeypatch) binary = local_install.cuda_binary_path() binary.parent.mkdir(parents=True, exist_ok=True) wanted_files = local_install.CUDA_SERVER_PIN.wanted_files_for_arch( @@ -1320,7 +1628,9 @@ def test_inspect_readiness_cuda_uses_sidecar_full_set( readiness = local_install.inspect_readiness(LOCAL_MODEL) assert readiness.host["backend"] == "cuda" - assert readiness.host["backend_reason"] == "test cuda" + assert readiness.host["backend_reason"] == ( + "compute_cap sm_121 covered; driver CUDA 13 >= 13" + ) assert readiness.artifacts["binary_path"] == str(binary) assert readiness.artifacts["binary_installed"] is sidecar_ok assert readiness.host["gpu_available"] is True @@ -1354,7 +1664,7 @@ def test_inspect_readiness_reports_gpu_available_with_hardware(tmp_path, monkeyp assert readiness.host["gpu_available"] is True assert readiness.host["backend"] == "vulkan" - assert readiness.host["backend_reason"] == "test vulkan" + assert readiness.host["backend_reason"] == "no NVIDIA GPU detected" def test_inspect_readiness_reports_gpu_unavailable_without_hardware(