diff --git a/solstone/think/providers/fit_report.py b/solstone/think/providers/fit_report.py index 0c168401a..ffac1d6c1 100644 --- a/solstone/think/providers/fit_report.py +++ b/solstone/think/providers/fit_report.py @@ -79,11 +79,7 @@ def build_local_fit_report(model_id: str) -> FitReport: if sys.platform.startswith("linux"): probe = local_cuda.probe_nvidia_gpu() choice = local_cuda.resolve_local_backend(local_install.CUDA_SERVER_PIN) - unknown_server = ( - "CUDA llama-server OCI image" - if choice.backend == "cuda" - else "llama-server tarball" - ) + unknown_server = "llama-server tarball" devices = local_vulkan.detect_gpus() try: from solstone.think.models import is_local_provider_needed @@ -100,12 +96,29 @@ def build_local_fit_report(model_id: str) -> FitReport: known_artifacts = [("GGUF model", spec.size_bytes)] if spec.mmproj_size_bytes is not None: known_artifacts.append(("mmproj", spec.mmproj_size_bytes)) + if ( + sys.platform.startswith("linux") + and choice is not None + and choice.backend == "cuda" + ): + cuda_artifact_pin = local_install.cuda_artifact_pin_for_current_platform( + local_install.CUDA_SERVER_PIN + ) + if cuda_artifact_pin is not None: + known_artifacts.append( + ("CUDA llama-server tarball", cuda_artifact_pin.size_bytes) + ) + unknown_artifacts: tuple[str, ...] = () + else: + unknown_artifacts = (unknown_server,) + else: + unknown_artifacts = (unknown_server,) checks.append( _disk_check( "disk", local_install.cache_root(), tuple(known_artifacts), - (unknown_server,), + unknown_artifacts, ) ) diff --git a/solstone/think/providers/local_install.py b/solstone/think/providers/local_install.py index a7da1b9ea..9502c911b 100644 --- a/solstone/think/providers/local_install.py +++ b/solstone/think/providers/local_install.py @@ -13,6 +13,7 @@ import hashlib import json import logging import platform +import re import shutil import stat import sys @@ -28,7 +29,6 @@ from solstone.think.providers.artifact_proof import ( ReadinessOutcome, artifact_manifest_path, build_manifest, - prove_cuda_sidecar, prove_manifest, publish_staged_tree, write_manifest, @@ -62,17 +62,28 @@ from solstone.think.providers.local_cuda import ( ArtifactTrust, ) from solstone.think.providers.memory import assess_memory -from solstone.think.providers.oci_image import OciSignaturePolicy from solstone.think.utils import get_journal LOG = logging.getLogger(__name__) LOCAL_PROVIDER_NAME = "local" _PROBE_TIMEOUT_SECONDS = 10 +_HEX64_RE = re.compile(r"^[0-9a-f]{64}$") +_LEGACY_OCI_SIDECAR_NAME = ".oci-install.json" + + +@dataclass(frozen=True) +class CudaArtifactPin: + url: str + sha256: str + size_bytes: int + release_tag: str + upstream_image_digest: str + llama_cpp_revision: str + repack_revision: str @dataclass(frozen=True) class CudaServerPin: - image_ref: str cuda_version: int embedded_arch_set: frozenset[str] binary_name: str @@ -80,7 +91,7 @@ class CudaServerPin: visible_devices_env: str shared_wanted_files: tuple[str, ...] cpu_wanted_files_by_arch: dict[str, tuple[str, ...]] - signature_policy: OciSignaturePolicy | None = None + artifacts_by_key: dict[str, CudaArtifactPin] def wanted_files_for_arch(self, arch: str) -> tuple[str, ...]: cpu_wanted_files = self.cpu_wanted_files_by_arch.get(arch) @@ -124,12 +135,6 @@ LLAMA_SERVER_PINS: dict[str, dict[str, str]] = { } CUDA_SERVER_PIN = CudaServerPin( - # TODO(AC10): confirm pinned server-cuda13 digest (build b9853) on the - # Spark GB10. - image_ref=( - "ghcr.io/ggml-org/llama.cpp@sha256:" - "bc998878c040cf2095b4c5cf3b1cf56df3984053e2a2650e5c4c66a4953e10cb" - ), cuda_version=CUDA_MIN_DRIVER_VERSION, embedded_arch_set=CUDA_EMBEDDED_ARCH_SET, binary_name="llama-server", @@ -179,15 +184,38 @@ CUDA_SERVER_PIN = CudaServerPin( "libggml-cpu-armv9.2_2.so", ), }, - # TODO(AC10): narrow this to the exact upstream workflow identity after - # validating the pinned CUDA image signature on hardware. - signature_policy=OciSignaturePolicy( - certificate_identity_regexp=( - r"^https://github\.com/ggml-org/llama\.cpp/\.github/workflows/" - r".+@refs/(heads|tags)/.+$" + artifacts_by_key={ + "x86_64-unknown-linux-gnu": CudaArtifactPin( + url=( + "https://updates.solstone.app/runtimes/llama-cuda13/b10068/" + "llama-b10068-bin-linux-cuda13-amd64-sol1.tar.gz" + ), + sha256="3727630e6ac79953f5c652fddcfd7100da98c55d773c0aec115a55f40f3aafea", + size_bytes=550238443, + release_tag="b10068", + upstream_image_digest=( + "sha256:" + "5bd5290bd35cfde893d0dcbd9811723c16d89575927d537b5f21becbfbab2f63" + ), + llama_cpp_revision="571d0d540df04f25298d0e159e520d9fc62ed121", + repack_revision="sol1", ), - oidc_issuer="https://token.actions.githubusercontent.com", - ), + "aarch64-unknown-linux-gnu": CudaArtifactPin( + url=( + "https://updates.solstone.app/runtimes/llama-cuda13/b10068/" + "llama-b10068-bin-linux-cuda13-arm64-sol1.tar.gz" + ), + sha256="6de68319db40e8c0eb45dc4bd3a45a16971dbdc128f2b621b19bef5dae87d064", + size_bytes=654508507, + release_tag="b10068", + upstream_image_digest=( + "sha256:" + "5bd5290bd35cfde893d0dcbd9811723c16d89575927d537b5f21becbfbab2f63" + ), + llama_cpp_revision="571d0d540df04f25298d0e159e520d9fc62ed121", + repack_revision="sol1", + ), + }, ) @@ -216,6 +244,27 @@ def pin_for_current_platform() -> dict[str, str]: return pin +def cuda_artifact_pin_for_current_platform( + pin: CudaServerPin | None = None, +) -> CudaArtifactPin | None: + server_pin = pin or CUDA_SERVER_PIN + return server_pin.artifacts_by_key.get(llama_server_artifact_key()) + + +def require_cuda_artifact_pin_for_current_platform( + pin: CudaServerPin | None = None, +) -> CudaArtifactPin: + key = llama_server_artifact_key() + server_pin = pin or CUDA_SERVER_PIN + artifact_pin = server_pin.artifacts_by_key.get(key) + if artifact_pin is None: + raise LocalProviderError( + "unsupported_platform", + f"No pinned CUDA llama-server artifact for platform {key}", + ) + return artifact_pin + + def cache_root() -> Path: return Path(get_journal()) / "cache" / "providers" / LOCAL_PROVIDER_NAME @@ -249,12 +298,17 @@ def _oci_arch() -> str: ) -def _cuda_digest_hex() -> str: - return CUDA_SERVER_PIN.image_ref.rsplit("@sha256:", 1)[1] +def _cuda_binary_dir_for_pin( + artifact_key: str, + artifact_pin: CudaArtifactPin, +) -> Path: + return cache_root() / "cuda" / artifact_key / artifact_pin.sha256 def cuda_binary_dir() -> Path: - return cache_root() / "cuda" / llama_server_artifact_key() / _cuda_digest_hex() + artifact_key = llama_server_artifact_key() + artifact_pin = require_cuda_artifact_pin_for_current_platform() + return _cuda_binary_dir_for_pin(artifact_key, artifact_pin) def cuda_binary_path() -> Path: @@ -346,13 +400,23 @@ def _vulkan_pin_identity( def _cuda_pin_identity( arch: str | None = None, wanted_files: tuple[str, ...] | None = None, + artifact_key: str | None = None, + artifact_pin: CudaArtifactPin | None = None, ) -> dict[str, Any]: + artifact_key = artifact_key or llama_server_artifact_key() + artifact_pin = artifact_pin or require_cuda_artifact_pin_for_current_platform() arch = arch or _oci_arch() wanted_files = wanted_files or CUDA_SERVER_PIN.wanted_files_for_arch(arch) return { "unit": "llama-server-cuda", - "artifact_key": llama_server_artifact_key(), - "image_ref": CUDA_SERVER_PIN.image_ref, + "artifact_key": artifact_key, + "url": artifact_pin.url, + "sha256": artifact_pin.sha256, + "size_bytes": artifact_pin.size_bytes, + "release_tag": artifact_pin.release_tag, + "upstream_image_digest": artifact_pin.upstream_image_digest, + "llama_cpp_revision": artifact_pin.llama_cpp_revision, + "repack_revision": artifact_pin.repack_revision, "arch": arch, "binary_name": CUDA_SERVER_PIN.binary_name, "wanted_files": list(wanted_files), @@ -364,18 +428,25 @@ def _prove_cuda_runtime_artifact( *, journal_path: str | Path | None = None, ) -> ProofResult: - from solstone.think.providers import oci_image - + artifact_key = llama_server_artifact_key() + artifact_pin = cuda_artifact_pin_for_current_platform(pin) + if artifact_pin is None: + return ProofResult( + status="missing-or-mismatched", + reason_code="cuda_runtime_pin_missing", + cache_hit=False, + ) arch = _oci_arch() wanted_files = pin.wanted_files_for_arch(arch) - return prove_cuda_sidecar( + return prove_manifest( + artifact_manifest_path(_cuda_binary_dir_for_pin(artifact_key, artifact_pin)), provider=LOCAL_PROVIDER_NAME, - image_ref=pin.image_ref, - arch=arch, - wanted_files=wanted_files, - target_dir=cuda_binary_dir(), - pin_identity=_cuda_pin_identity(arch, wanted_files), - verifier=oci_image.verify_sidecar_install, + pin_identity=_cuda_pin_identity( + arch, + wanted_files, + artifact_key=artifact_key, + artifact_pin=artifact_pin, + ), journal_path=journal_path, ) @@ -385,6 +456,8 @@ def probe_cuda_runtime_artifact_trust( *, journal_path: str | Path | None = None, ) -> ArtifactTrust: + if cuda_artifact_pin_for_current_platform(pin) is not None: + return ArtifactTrust.TRUSTED try: result = _prove_cuda_runtime_artifact(pin, journal_path=journal_path) except Exception: @@ -437,7 +510,11 @@ def target_fingerprint(model_id: str = LOCAL_MODEL) -> dict[str, Any]: selected_model = normalize_model_id(model_id) choice = local_cuda.resolve_local_backend(CUDA_SERVER_PIN) runtime_pin = ( - _cuda_pin_identity() if choice.backend == "cuda" else _vulkan_pin_identity() + _cuda_pin_identity( + artifact_pin=require_cuda_artifact_pin_for_current_platform() + ) + if choice.backend == "cuda" + else _vulkan_pin_identity() ) return { "provider": LOCAL_PROVIDER_NAME, @@ -504,6 +581,36 @@ def _write_vulkan_manifest( write_manifest(artifact_manifest_path(install_dir), manifest) +def _write_cuda_manifest( + *, + artifact_key: str, + artifact_pin: CudaArtifactPin, + arch: str, + wanted_files: tuple[str, ...], + attempt_status: InstallStatus | None, + fingerprint: dict[str, Any] | None = None, + root: Path | None = None, +) -> None: + install_dir = root or _cuda_binary_dir_for_pin(artifact_key, artifact_pin) + fingerprint = fingerprint or target_fingerprint() + manifest = build_manifest( + provider=LOCAL_PROVIDER_NAME, + unit="llama-server-cuda", + target_fingerprint_sha256=_manifest_target_sha(attempt_status, fingerprint), + source={ + "pin_identity": _cuda_pin_identity( + arch, + wanted_files, + artifact_key=artifact_key, + artifact_pin=artifact_pin, + ) + }, + inventory=_runtime_inventory(install_dir, exclude_names=set()), + attempt_id=attempt_status["attempt_id"] if attempt_status else None, + ) + write_manifest(artifact_manifest_path(install_dir), manifest) + + def _write_model_manifest( *, model_id: str, @@ -605,6 +712,22 @@ def _safe_extract_tarball(tarball: Path, dest: Path) -> None: archive.extractall(dest, filter="data") +def _download_verify_extract_tarball( + *, + url: str, + filename: str, + sha256: str, + staging: Path, + on_progress: Callable[[int, int | None], None], +) -> None: + tarball = staging / filename + _download_file(url, tarball, on_progress=on_progress) + _write_local_status(transition_state(_read_local_status(), new_state="verifying")) + _verify_sha256(tarball, sha256) + _safe_extract_tarball(tarball, staging) + tarball.unlink(missing_ok=True) + + def _find_extracted_binary(dest: Path, binary_name: str) -> Path: direct = dest / binary_name if direct.exists(): @@ -641,6 +764,79 @@ def _clear_macos_quarantine(path: Path) -> None: return +def _cuda_artifact_filename(artifact_pin: CudaArtifactPin) -> str: + return artifact_pin.url.rsplit("/", 1)[-1] + + +def _verify_cuda_runtime_tree( + root: Path, + *, + wanted_files: tuple[str, ...], +) -> None: + missing: list[str] = [] + for wanted in wanted_files: + if not (root / wanted).is_file(): + missing.append(wanted) + if not (root / "licenses").is_dir(): + missing.append("licenses/") + if not (root / "provenance.json").is_file(): + missing.append("provenance.json") + if missing: + raise LocalProviderError( + "cuda_runtime_incomplete", + "CUDA runtime artifact is missing required paths: " + ", ".join(missing), + ) + + +def _is_legacy_cuda_oci_tree(path: Path) -> bool: + if not path.is_dir() or path.is_symlink(): + return False + if not _HEX64_RE.fullmatch(path.name): + return False + sidecar = path / _LEGACY_OCI_SIDECAR_NAME + if not sidecar.is_file(): + return False + try: + record = json.loads(sidecar.read_text(encoding="utf-8")) + except (OSError, ValueError): + return False + if not isinstance(record, dict): + return False + image_ref = record.get("image_ref") + files = record.get("files") + return ( + isinstance(image_ref, str) + and image_ref.endswith(f"@sha256:{path.name}") + and isinstance(files, dict) + ) + + +def _cleanup_legacy_cuda_oci_dirs( + *, + artifact_key: str, + keep_dir: Path, +) -> None: + root = cache_root() / "cuda" / artifact_key + if not root.is_dir(): + return + keep_resolved = keep_dir.resolve() + for candidate in root.iterdir(): + try: + if not candidate.is_dir(): + continue + if candidate.resolve() == keep_resolved: + continue + if not _is_legacy_cuda_oci_tree(candidate): + continue + shutil.rmtree(candidate) + except Exception: + LOG.warning( + "failed to remove legacy CUDA OCI install tree: %s", + candidate, + exc_info=True, + ) + + def probe_binary_runnable(binary_path: str | Path) -> tuple[bool, str | None]: import subprocess @@ -677,7 +873,10 @@ def install_llama_server( choice = local_cuda.resolve_local_backend(CUDA_SERVER_PIN) if choice.backend == "cuda": - return _install_cuda_llama_server(attempt_status=attempt_status) + return _install_cuda_llama_server( + attempt_status=attempt_status, + fingerprint=fingerprint, + ) artifact_key = llama_server_artifact_key() pin = pin_for_current_platform() @@ -690,18 +889,18 @@ def install_llama_server( staging = Path( tempfile.mkdtemp(prefix=f".{install_dir.name}.staging-", dir=install_dir.parent) ) - tarball = staging / pin["filename"] try: _write_local_status( transition_state(_read_local_status(), new_state="downloading") ) - _download_file(url, tarball, on_progress=_record_local_progress) - _write_local_status( - transition_state(_read_local_status(), new_state="verifying") + _download_verify_extract_tarball( + url=url, + filename=pin["filename"], + sha256=pin["sha256"], + staging=staging, + on_progress=_record_local_progress, ) - _verify_sha256(tarball, pin["sha256"]) - _safe_extract_tarball(tarball, staging) extracted = _find_extracted_binary(staging, pin["binary_name"]) final_path = staging / pin["binary_name"] if attempt_status is not None: @@ -711,7 +910,6 @@ def install_llama_server( for item in inner_dir.iterdir(): shutil.move(str(item), str(staging / item.name)) inner_dir.rmdir() - tarball.unlink(missing_ok=True) _chmod_executable(final_path) _clear_macos_quarantine(staging) _write_vulkan_manifest( @@ -742,30 +940,51 @@ def install_llama_server( def _install_cuda_llama_server( *, attempt_status: InstallStatus | None = None, + fingerprint: dict[str, Any] | None = None, ) -> dict[str, Any]: - from solstone.think.providers import oci_image - + artifact_key = llama_server_artifact_key() + artifact_pin = require_cuda_artifact_pin_for_current_platform(CUDA_SERVER_PIN) + install_dir = _cuda_binary_dir_for_pin(artifact_key, artifact_pin) + install_dir.parent.mkdir(parents=True, exist_ok=True) + staging = Path( + tempfile.mkdtemp(prefix=f".{install_dir.name}.staging-", dir=install_dir.parent) + ) try: arch = _oci_arch() wanted_files = CUDA_SERVER_PIN.wanted_files_for_arch(arch) _write_local_status( transition_state(_read_local_status(), new_state="downloading") ) - oci_image.pull_and_install( - CUDA_SERVER_PIN.image_ref, - arch, - wanted_files, - cuda_binary_dir(), - policy=CUDA_SERVER_PIN.signature_policy, + _download_verify_extract_tarball( + url=artifact_pin.url, + filename=_cuda_artifact_filename(artifact_pin), + sha256=artifact_pin.sha256, + staging=staging, + on_progress=_record_local_progress, ) - _write_local_status( - transition_state(_read_local_status(), new_state="verifying") + _verify_cuda_runtime_tree(staging, wanted_files=wanted_files) + final_path = staging / CUDA_SERVER_PIN.binary_name + if attempt_status is not None: + assert_install_attempt_current(attempt_status) + _chmod_executable(final_path) + _clear_macos_quarantine(staging) + _write_cuda_manifest( + artifact_key=artifact_key, + artifact_pin=artifact_pin, + arch=arch, + wanted_files=wanted_files, + attempt_status=attempt_status, + fingerprint=fingerprint, + root=staging, ) if attempt_status is not None: assert_install_attempt_current(attempt_status) - _chmod_executable(cuda_binary_path()) + publish_staged_tree(staging, install_dir) + _cleanup_legacy_cuda_oci_dirs(artifact_key=artifact_key, keep_dir=install_dir) return _read_local_status() except Exception as exc: + if staging.exists(): + shutil.rmtree(staging, ignore_errors=True) _write_local_status( transition_state( _read_local_status(), @@ -959,7 +1178,13 @@ def inspect_artifacts(model_id: str | None = None) -> dict[str, Any]: ) vulkan_payload = _proof_result_payload(vulkan_proof) - cuda_binary = cuda_binary_path() + cuda_artifact_pin = cuda_artifact_pin_for_current_platform(CUDA_SERVER_PIN) + cuda_binary = ( + _cuda_binary_dir_for_pin(llama_server_artifact_key(), cuda_artifact_pin) + / CUDA_SERVER_PIN.binary_name + if cuda_artifact_pin is not None + else None + ) cuda_proof = _prove_cuda_runtime_artifact(CUDA_SERVER_PIN) cuda_payload = _proof_result_payload(cuda_proof) model_proof = prove_manifest( @@ -977,7 +1202,7 @@ def inspect_artifacts(model_id: str | None = None) -> dict[str, Any]: "vulkan_binary_installed": vulkan_proof.ready, "cuda_binary_installed": cuda_proof.ready, "vulkan_binary_path": str(vulkan_binary_path), - "cuda_binary_path": str(cuda_binary), + "cuda_binary_path": str(cuda_binary) if cuda_binary is not None else None, "binary_path": str(vulkan_binary_path), "model_path": str(gguf_path), "mmproj_path": str(resolved_mmproj) if resolved_mmproj is not None else None, @@ -1104,11 +1329,14 @@ def ensure_artifacts_installed(model_id: str) -> LocalArtifacts: __all__ = [ "CUDA_SERVER_PIN", "LLAMA_SERVER_PINS", + "CudaArtifactPin", "CudaServerPin", "LocalArtifacts", "has_persisted_installed_cuda_target", "llama_server_artifact_key", "pin_for_current_platform", + "cuda_artifact_pin_for_current_platform", + "require_cuda_artifact_pin_for_current_platform", "binary_path_for_pin", "cuda_binary_dir", "cuda_binary_path", diff --git a/tests/test_fit_report.py b/tests/test_fit_report.py index 3e354921d..061a25886 100644 --- a/tests/test_fit_report.py +++ b/tests/test_fit_report.py @@ -191,7 +191,7 @@ def test_build_local_fit_report_uses_cuda_artifact_trust_probe( monkeypatch.setattr( local_install, "probe_cuda_runtime_artifact_trust", - lambda _pin: local_cuda.ArtifactTrust.ABSENT, + lambda _pin: local_cuda.ArtifactTrust.TRUSTED, ) monkeypatch.setattr( local_install, @@ -220,10 +220,16 @@ def test_build_local_fit_report_uses_cuda_artifact_trust_probe( gpu = next(check for check in report.checks if check.name == "gpu") assert gpu.severity == "ok" - assert ( - "resolved backend is vulkan: compute_cap sm_86 covered; " - "driver CUDA 14 >= 13; no trusted CUDA runtime artifact present" - ) in gpu.detail + assert gpu.detail == ( + "CUDA backend selected: compute_cap sm_86 covered; driver CUDA 14 >= 13" + ) + disk = next(check for check in report.checks if check.name == "disk") + assert disk.severity == "ok" + assert "CUDA llama-server tarball" not in disk.detail + assert disk.required_bytes is not None + assert disk.required_bytes >= ( + local_install.require_cuda_artifact_pin_for_current_platform().size_bytes + ) def test_disk_unknown_size_warns_when_known_size_fits( diff --git a/tests/test_install_state.py b/tests/test_install_state.py index dcbde6577..bc1bdbf17 100644 --- a/tests/test_install_state.py +++ b/tests/test_install_state.py @@ -260,7 +260,7 @@ def test_migration_api_removes_legacy_status_fields(tmp_path, monkeypatch) -> No assert data["providers"]["local"] == {"vulkan_device_index": "1"} -def test_legacy_local_vulkan_adoption_when_cuda_artifact_absent_on_covered_host( +def test_legacy_local_vulkan_not_promoted_when_cuda_pin_selected( tmp_path: Path, monkeypatch: pytest.MonkeyPatch, ) -> None: @@ -318,7 +318,7 @@ def test_legacy_local_vulkan_adoption_when_cuda_artifact_absent_on_covered_host( monkeypatch.setattr( local_install, "probe_cuda_runtime_artifact_trust", - lambda _pin, **_kwargs: local_cuda.ArtifactTrust.ABSENT, + lambda _pin, **_kwargs: local_cuda.ArtifactTrust.TRUSTED, ) monkeypatch.setattr( local_install, @@ -343,14 +343,10 @@ def test_legacy_local_vulkan_adoption_when_cuda_artifact_absent_on_covered_host( result = install_state.migrate_legacy_provider_artifact_truth(journal_path=tmp_path) status = read_install_status(name="local", journal_path=tmp_path) - target = json.loads(str(status["target_fingerprint_json"])) - assert result["actions"][0]["action"] == "promoted" - assert status["install_state"] == "installed" - assert target["backend"] == "vulkan" - assert target["backend_reason"] == ( - "compute_cap sm_121 covered; driver CUDA 16 >= 13; " - "no trusted CUDA runtime artifact present" - ) + assert result["actions"][0]["action"] == "not-promoted" + assert result["actions"][0]["reason_code"] == "manifest_missing" + assert status["install_state"] == "idle" + assert status["target_fingerprint_json"] is None def test_two_process_stale_transition_one_writer_wins(tmp_path, monkeypatch) -> None: diff --git a/tests/test_local_cuda.py b/tests/test_local_cuda.py index 78db3d526..60d3624eb 100644 --- a/tests/test_local_cuda.py +++ b/tests/test_local_cuda.py @@ -3,11 +3,12 @@ from __future__ import annotations +from dataclasses import replace from types import SimpleNamespace import pytest -from solstone.think.providers import local_cuda +from solstone.think.providers import local_cuda, local_install ARCH_SET = frozenset({"sm_86", "sm_89", "sm_120a", "sm_121a"}) pytestmark = pytest.mark.real_local_backend_probe @@ -347,6 +348,8 @@ def test_select_local_backend_matrix() -> None: "vulkan", "NVIDIA compute capability unreadable", ), + # These rows pin select_local_backend's raw trust contract; resolver + # tests cover the production pin-present and pin-absent states. ( True, "sm_86", @@ -404,6 +407,62 @@ def test_select_local_backend_matrix() -> None: assert choice == local_cuda.BackendChoice(backend, reason) +def test_resolve_local_backend_uses_pinned_cuda_artifact_when_present( + monkeypatch: pytest.MonkeyPatch, +) -> None: + monkeypatch.setattr( + local_cuda, + "probe_nvidia_gpu", + lambda: local_cuda.NvidiaProbe( + index=0, + compute_cap="sm_86", + driver_cuda_version=13, + vram_mib=24564, + tiering_memory_mib=24564, + memory_source=local_cuda.MEMORY_SOURCE_NVIDIA_VRAM, + detected=True, + ), + ) + monkeypatch.setattr(local_install.platform, "machine", lambda: "x86_64") + monkeypatch.setattr(local_install.sys, "platform", "linux") + + assert local_cuda.resolve_local_backend(local_install.CUDA_SERVER_PIN) == ( + local_cuda.BackendChoice( + "cuda", + "compute_cap sm_86 covered; driver CUDA 13 >= 13", + ) + ) + + +def test_resolve_local_backend_uses_byte_identical_vulkan_reason_without_platform_pin( + monkeypatch: pytest.MonkeyPatch, +) -> None: + monkeypatch.setattr( + local_cuda, + "probe_nvidia_gpu", + lambda: local_cuda.NvidiaProbe( + index=0, + compute_cap="sm_86", + driver_cuda_version=13, + vram_mib=24564, + tiering_memory_mib=24564, + memory_source=local_cuda.MEMORY_SOURCE_NVIDIA_VRAM, + detected=True, + ), + ) + monkeypatch.setattr(local_install.platform, "machine", lambda: "x86_64") + monkeypatch.setattr(local_install.sys, "platform", "linux") + pin = replace(local_install.CUDA_SERVER_PIN, artifacts_by_key={}) + + assert local_cuda.resolve_local_backend(pin) == local_cuda.BackendChoice( + "vulkan", + ( + "compute_cap sm_86 covered; driver CUDA 13 >= 13; " + "no trusted CUDA runtime artifact present" + ), + ) + + def test_select_local_backend_no_gpu_detected() -> None: choice = local_cuda.select_local_backend( local_cuda.NvidiaProbe( diff --git a/tests/test_local_install.py b/tests/test_local_install.py index 2c2d48d7c..01c51bd2d 100644 --- a/tests/test_local_install.py +++ b/tests/test_local_install.py @@ -5,6 +5,7 @@ from __future__ import annotations import json import shutil +import subprocess import tarfile import time from dataclasses import replace @@ -20,7 +21,6 @@ from solstone.think.providers import ( local_install, local_vulkan, memory, - oci_image, ) from solstone.think.providers.artifact_proof import ( ProofResult, @@ -229,6 +229,11 @@ def test_probe_cuda_runtime_artifact_trust_maps_proof_status( status: str, expected: local_cuda.ArtifactTrust, ) -> None: + monkeypatch.setattr( + local_install, + "cuda_artifact_pin_for_current_platform", + lambda _pin=None: None, + ) monkeypatch.setattr( local_install, "_prove_cuda_runtime_artifact", @@ -241,6 +246,21 @@ def test_probe_cuda_runtime_artifact_trust_maps_proof_status( ) +@pytest.mark.real_local_backend_probe +def test_probe_cuda_runtime_artifact_trust_uses_present_pin_without_local_manifest( + monkeypatch: pytest.MonkeyPatch, +) -> None: + def fail_proof(_pin, **_kwargs): + raise AssertionError("present platform pin should short-circuit proof") + + monkeypatch.setattr(local_install, "_prove_cuda_runtime_artifact", fail_proof) + + assert ( + local_install.probe_cuda_runtime_artifact_trust(local_install.CUDA_SERVER_PIN) + == local_cuda.ArtifactTrust.TRUSTED + ) + + @pytest.mark.real_local_backend_probe def test_probe_cuda_runtime_artifact_trust_contains_unexpected_exception( monkeypatch: pytest.MonkeyPatch, @@ -249,6 +269,11 @@ def test_probe_cuda_runtime_artifact_trust_contains_unexpected_exception( def fail_proof(_pin, **_kwargs): raise ValueError("required artifact is not a regular file") + monkeypatch.setattr( + local_install, + "cuda_artifact_pin_for_current_platform", + lambda _pin=None: None, + ) monkeypatch.setattr(local_install, "_prove_cuda_runtime_artifact", fail_proof) trust = local_install.probe_cuda_runtime_artifact_trust( @@ -260,37 +285,19 @@ def test_probe_cuda_runtime_artifact_trust_contains_unexpected_exception( @pytest.mark.real_local_backend_probe -def test_probe_cuda_runtime_artifact_trust_contains_non_regular_wanted_file( +def test_probe_cuda_runtime_artifact_trust_absent_without_platform_pin( tmp_path: Path, monkeypatch: pytest.MonkeyPatch, ) -> None: _init_journal(tmp_path, monkeypatch) monkeypatch.setattr(local_install.platform, "machine", lambda: "x86_64") - target = local_install.cuda_binary_dir() - target.mkdir(parents=True) - wanted_files = local_install.CUDA_SERVER_PIN.wanted_files_for_arch("amd64") - (target / wanted_files[0]).mkdir() - for name in wanted_files[1:]: - (target / name).write_text(name, encoding="utf-8") - (target / ".oci-install.json").write_text( - json.dumps( - { - "image_ref": local_install.CUDA_SERVER_PIN.image_ref, - "arch": "amd64", - "files": {name: "0" * 64 for name in wanted_files}, - } - ) - + "\n", - encoding="utf-8", - ) + pin = replace(local_install.CUDA_SERVER_PIN, artifacts_by_key={}) - trust = local_install.probe_cuda_runtime_artifact_trust( - local_install.CUDA_SERVER_PIN, - journal_path=tmp_path, + assert ( + local_install.probe_cuda_runtime_artifact_trust(pin, journal_path=tmp_path) + == local_cuda.ArtifactTrust.ABSENT ) - assert trust == local_cuda.ArtifactTrust.UNAVAILABLE - @pytest.mark.real_local_backend_probe def test_has_persisted_installed_cuda_target_reads_installed_backend( @@ -336,7 +343,7 @@ def test_has_persisted_installed_cuda_target_treats_bad_status_as_false( ) -def test_target_fingerprint_uses_vulkan_when_cuda_artifact_absent_on_covered_host( +def test_target_fingerprint_uses_cuda_when_platform_pin_present_on_covered_host( tmp_path: Path, monkeypatch: pytest.MonkeyPatch, ) -> None: @@ -345,15 +352,15 @@ def test_target_fingerprint_uses_vulkan_when_cuda_artifact_absent_on_covered_hos monkeypatch, compute_cap="sm_86", driver_cuda_version=14, - trust=local_cuda.ArtifactTrust.ABSENT, + trust=local_cuda.ArtifactTrust.TRUSTED, ) fingerprint = local_install.target_fingerprint(LOCAL_MODEL) - assert fingerprint["backend"] == "vulkan" - assert fingerprint["backend_reason"] == ( - "compute_cap sm_86 covered; driver CUDA 14 >= 13; " - "no trusted CUDA runtime artifact present" + assert fingerprint["backend"] == "cuda" + assert ( + fingerprint["backend_reason"] + == "compute_cap sm_86 covered; driver CUDA 14 >= 13" ) @@ -378,7 +385,7 @@ def test_target_fingerprint_holds_cuda_when_trust_unavailable_and_cuda_installed ) -def test_target_fingerprint_uses_vulkan_when_trust_unavailable_without_cuda_install( +def test_target_fingerprint_uses_cuda_when_runtime_pin_is_trusted( tmp_path: Path, monkeypatch: pytest.MonkeyPatch, ) -> None: @@ -387,16 +394,15 @@ def test_target_fingerprint_uses_vulkan_when_trust_unavailable_without_cuda_inst monkeypatch, compute_cap="sm_121", driver_cuda_version=16, - trust=local_cuda.ArtifactTrust.UNAVAILABLE, + trust=local_cuda.ArtifactTrust.TRUSTED, persisted_installed_cuda=False, ) fingerprint = local_install.target_fingerprint(LOCAL_MODEL) - assert fingerprint["backend"] == "vulkan" + assert fingerprint["backend"] == "cuda" assert fingerprint["backend_reason"] == ( - "compute_cap sm_121 covered; driver CUDA 16 >= 13; " - "no trusted CUDA runtime artifact present" + "compute_cap sm_121 covered; driver CUDA 16 >= 13" ) @@ -566,9 +572,9 @@ def test_oci_arch_unsupported_raises(monkeypatch: pytest.MonkeyPatch): assert exc_info.value.reason_code == "unsupported_platform" -def test_cuda_binary_paths_include_index_digest(tmp_path, monkeypatch): +def test_cuda_binary_paths_include_tarball_sha256(tmp_path, monkeypatch): _init_journal(tmp_path, monkeypatch) - digest = local_install.CUDA_SERVER_PIN.image_ref.split("@sha256:", 1)[1] + artifact_pin = local_install.require_cuda_artifact_pin_for_current_platform() assert local_install.cuda_binary_dir() == ( tmp_path @@ -577,7 +583,7 @@ def test_cuda_binary_paths_include_index_digest(tmp_path, monkeypatch): / "local" / "cuda" / local_install.llama_server_artifact_key() - / digest + / artifact_pin.sha256 ) assert local_install.cuda_binary_path() == ( local_install.cuda_binary_dir() / local_install.CUDA_SERVER_PIN.binary_name @@ -643,6 +649,98 @@ def test_llama_server_pins_are_complete_immutable_artifacts() -> None: assert pins[key] == expected_pin +def test_cuda_server_artifact_pins_are_complete_immutable_artifacts() -> None: + expected = { + "x86_64-unknown-linux-gnu": local_install.CudaArtifactPin( + url=( + "https://updates.solstone.app/runtimes/llama-cuda13/b10068/" + "llama-b10068-bin-linux-cuda13-amd64-sol1.tar.gz" + ), + sha256="3727630e6ac79953f5c652fddcfd7100da98c55d773c0aec115a55f40f3aafea", + size_bytes=550238443, + release_tag="b10068", + upstream_image_digest=( + "sha256:" + "5bd5290bd35cfde893d0dcbd9811723c16d89575927d537b5f21becbfbab2f63" + ), + llama_cpp_revision="571d0d540df04f25298d0e159e520d9fc62ed121", + repack_revision="sol1", + ), + "aarch64-unknown-linux-gnu": local_install.CudaArtifactPin( + url=( + "https://updates.solstone.app/runtimes/llama-cuda13/b10068/" + "llama-b10068-bin-linux-cuda13-arm64-sol1.tar.gz" + ), + sha256="6de68319db40e8c0eb45dc4bd3a45a16971dbdc128f2b621b19bef5dae87d064", + size_bytes=654508507, + release_tag="b10068", + upstream_image_digest=( + "sha256:" + "5bd5290bd35cfde893d0dcbd9811723c16d89575927d537b5f21becbfbab2f63" + ), + llama_cpp_revision="571d0d540df04f25298d0e159e520d9fc62ed121", + repack_revision="sol1", + ), + } + + assert local_install.CUDA_SERVER_PIN.artifacts_by_key == expected + for key, artifact_pin in local_install.CUDA_SERVER_PIN.artifacts_by_key.items(): + assert artifact_pin.url.startswith("https://updates.solstone.app/runtimes/") + assert len(artifact_pin.sha256) == 64 + assert set(artifact_pin.sha256) <= set("0123456789abcdef") + assert artifact_pin.size_bytes > 0 + assert ( + artifact_pin.release_tag + == local_install.LLAMA_SERVER_PINS[key]["release_tag"] + ) + + +def _write_cuda_runtime_tarball( + tmp_path: Path, + *, + arch: str = "amd64", + missing: tuple[str, ...] = (), + traversal: bool = False, +) -> Path: + source = tmp_path / "cuda-source" + source.mkdir() + missing_set = set(missing) + for name in local_install.CUDA_SERVER_PIN.wanted_files_for_arch(arch): + if name in missing_set: + continue + path = source / name + path.write_text(name, encoding="utf-8") + if "licenses/" not in missing_set: + licenses = source / "licenses" + licenses.mkdir() + (licenses / "LICENSE").write_text("license", encoding="utf-8") + if "provenance.json" not in missing_set: + (source / "provenance.json").write_text("{}\n", encoding="utf-8") + + tarball = tmp_path / "cuda-runtime.tar.gz" + with tarfile.open(tarball, "w:gz") as archive: + for path in sorted(source.rglob("*")): + archive.add(path, arcname=path.relative_to(source).as_posix()) + if traversal: + escape = tmp_path / "escape-source" + escape.write_text("bad", encoding="utf-8") + archive.add(escape, arcname="../escape") + return tarball + + +def _patch_tarball_download( + monkeypatch: pytest.MonkeyPatch, + tarball: Path, +) -> None: + def fake_download(_url: str, dest: Path, **kwargs: object) -> None: + shutil.copyfile(tarball, dest) + on_progress = kwargs.get("on_progress") + if on_progress is not None: + on_progress(tarball.stat().st_size, tarball.stat().st_size) + + monkeypatch.setattr(local_install, "_download_file", fake_download) + + def test_install_llama_server_relocates_binary_and_libraries(tmp_path, monkeypatch): _init_journal(tmp_path, monkeypatch) pin = local_install.pin_for_current_platform() @@ -940,7 +1038,7 @@ def test_install_llama_server_writes_canonical_sequence(tmp_path, monkeypatch): ("arm64", "arm64", "libggml-cpu-armv8.0_1.so", "libggml-cpu-haswell.so"), ], ) -def test_install_llama_server_cuda_uses_arch_specific_oci_wanted_files( +def test_install_llama_server_cuda_extracts_flat_tarball_and_writes_manifest( tmp_path, monkeypatch, machine: str, @@ -951,108 +1049,208 @@ def test_install_llama_server_cuda_uses_arch_specific_oci_wanted_files( _init_journal(tmp_path, monkeypatch) monkeypatch.setattr(local_install.platform, "machine", lambda: machine) _force_cuda_backend(monkeypatch) - pull_calls: list[tuple[str, str, tuple[str, ...], Path]] = [] - - def fake_pull_and_install( - image_ref: str, - arch: str, - wanted_files: tuple[str, ...], - target_dir: Path, - *, - policy: oci_image.OciSignaturePolicy | None = None, - ) -> oci_image.OciInstallResult: - assert policy is local_install.CUDA_SERVER_PIN.signature_policy - target_dir.mkdir(parents=True, exist_ok=True) - binary = target_dir / local_install.CUDA_SERVER_PIN.binary_name - binary.write_text("binary", encoding="utf-8") - pull_calls.append((image_ref, arch, wanted_files, target_dir)) - return oci_image.OciInstallResult( - target_dir=target_dir, - files={}, - already_present=False, - ) + tarball = _write_cuda_runtime_tarball(tmp_path, arch=arch) + _patch_tarball_download(monkeypatch, tarball) + monkeypatch.setattr(local_install, "_verify_sha256", lambda _path, _expected: None) - monkeypatch.setattr(oci_image, "pull_and_install", fake_pull_and_install) result = local_install.install_llama_server() wanted_files = local_install.CUDA_SERVER_PIN.wanted_files_for_arch(arch) assert result["install_state"] == "verifying" - assert pull_calls == [ - ( - local_install.CUDA_SERVER_PIN.image_ref, + assert expected_cpu in wanted_files + assert unexpected_cpu not in wanted_files + for name in wanted_files: + assert (local_install.cuda_binary_dir() / name).is_file() + assert (local_install.cuda_binary_dir() / "licenses" / "LICENSE").is_file() + assert (local_install.cuda_binary_dir() / "provenance.json").is_file() + assert not (local_install.cuda_binary_dir() / ".oci-install.json").exists() + assert local_install.cuda_binary_path().stat().st_mode & 0o111 + + artifact_pin = local_install.require_cuda_artifact_pin_for_current_platform() + proof = prove_manifest( + artifact_manifest_path(local_install.cuda_binary_dir()), + provider=local_install.LOCAL_PROVIDER_NAME, + pin_identity=local_install._cuda_pin_identity( arch, wanted_files, - local_install.cuda_binary_dir(), + artifact_pin=artifact_pin, + ), + ) + assert proof.ready + manifest = json.loads( + artifact_manifest_path(local_install.cuda_binary_dir()).read_text( + encoding="utf-8" ) - ] - assert expected_cpu in pull_calls[0][2] - assert unexpected_cpu not in pull_calls[0][2] - assert local_install.cuda_binary_path().stat().st_mode & 0o111 + ) + inventory_paths = {entry["relative_path"] for entry in manifest["inventory"]} + assert set(wanted_files) <= inventory_paths + assert {"licenses/LICENSE", "provenance.json"} <= inventory_paths -def test_install_llama_server_cuda_preserves_oci_failure_reason(tmp_path, monkeypatch): +def test_install_llama_server_cuda_sha256_mismatch_fails_closed(tmp_path, monkeypatch): _init_journal(tmp_path, monkeypatch) _force_cuda_backend(monkeypatch) + tarball = _write_cuda_runtime_tarball(tmp_path) + _patch_tarball_download(monkeypatch, tarball) - def fail_pull(*_args, **_kwargs): - raise oci_image.OciImageError( - "signature_verify_failed", - "the pinned image has no matching signature", - ) - - monkeypatch.setattr(oci_image, "pull_and_install", fail_pull) - - with pytest.raises(oci_image.OciImageError, match="no matching signature"): + with pytest.raises(local_install.LocalProviderError) as exc_info: local_install.install_llama_server() + assert exc_info.value.reason_code == "sha256_mismatch" status = _local_status() assert status["install_state"] == "failed" - assert status["install_error"] == "the pinned image has no matching signature" - assert status["error_code"] == "signature_verify_failed" + assert status["error_code"] == "sha256_mismatch" + assert not local_install.cuda_binary_dir().exists() -def test_install_llama_server_vulkan_choice_does_not_pull_oci(tmp_path, monkeypatch): +@pytest.mark.parametrize( + ("missing", "expected_detail"), + [ + (("libllama.so.0",), "libllama.so.0"), + (("licenses/",), "licenses/"), + ], +) +def test_install_llama_server_cuda_required_files_fail_closed( + tmp_path, + monkeypatch, + missing: tuple[str, ...], + expected_detail: str, +): _init_journal(tmp_path, monkeypatch) - pin = { - "release_tag": "v1", - "filename": "llama.tar.gz", - "sha256": "abc123", - "binary_name": "llama-server", - } - final_path = local_install.binary_path_for_pin("test-platform", pin) - final_path.parent.mkdir(parents=True) - final_path.write_text("binary", encoding="utf-8") + _force_cuda_backend(monkeypatch) + tarball = _write_cuda_runtime_tarball(tmp_path, missing=missing) + _patch_tarball_download(monkeypatch, tarball) + monkeypatch.setattr(local_install, "_verify_sha256", lambda _path, _expected: None) - monkeypatch.setattr( - local_install, "llama_server_artifact_key", lambda: "test-platform" - ) - monkeypatch.setattr(local_install, "pin_for_current_platform", lambda: pin) - monkeypatch.setattr(local_install, "_download_file", lambda *_args, **_kwargs: None) + with pytest.raises(local_install.LocalProviderError) as exc_info: + local_install.install_llama_server() + + assert exc_info.value.reason_code == "cuda_runtime_incomplete" + assert expected_detail in str(exc_info.value) + assert not local_install.cuda_binary_dir().exists() + + +def test_install_llama_server_cuda_rejects_traversal_member(tmp_path, monkeypatch): + _init_journal(tmp_path, monkeypatch) + _force_cuda_backend(monkeypatch) + tarball = _write_cuda_runtime_tarball(tmp_path, traversal=True) + _patch_tarball_download(monkeypatch, tarball) monkeypatch.setattr(local_install, "_verify_sha256", lambda _path, _expected: None) - monkeypatch.setattr( - local_install, "_safe_extract_tarball", lambda _tarball, _dest: None + + with pytest.raises(local_install.LocalProviderError) as exc_info: + local_install.install_llama_server() + + assert exc_info.value.reason_code == "archive_path_traversal" + assert not (tmp_path / "escape").exists() + assert not local_install.cuda_binary_dir().exists() + + +def test_install_llama_server_cuda_removes_legacy_oci_tree_after_publish_only( + tmp_path, + monkeypatch, +): + _init_journal(tmp_path, monkeypatch) + _force_cuda_backend(monkeypatch) + artifact_key = local_install.llama_server_artifact_key() + legacy_digest = "a" * 64 + legacy_dir = ( + tmp_path + / "cache" + / "providers" + / "local" + / "cuda" + / artifact_key + / legacy_digest ) - monkeypatch.setattr( - local_install, "_find_extracted_binary", lambda _dest, _name: final_path + legacy_dir.mkdir(parents=True) + (legacy_dir / ".oci-install.json").write_text( + json.dumps( + { + "image_ref": f"ghcr.io/acme/runtime@sha256:{legacy_digest}", + "arch": "amd64", + "files": {"llama-server": "b" * 64}, + } + ) + + "\n", + encoding="utf-8", ) - monkeypatch.setattr(local_install, "_chmod_executable", lambda _path: None) - monkeypatch.setattr(local_install, "_clear_macos_quarantine", lambda _path: None) - monkeypatch.setattr( - oci_image, - "pull_and_install", - lambda *_args, **_kwargs: (_ for _ in ()).throw( - AssertionError("OCI pull not expected") - ), + (legacy_dir / "llama-server").write_text("legacy", encoding="utf-8") + vulkan_dir = local_install.binary_install_dir() + vulkan_dir.mkdir(parents=True) + (vulkan_dir / "llama-server").write_text("vulkan", encoding="utf-8") + tarball = _write_cuda_runtime_tarball(tmp_path) + _patch_tarball_download(monkeypatch, tarball) + monkeypatch.setattr(local_install, "_verify_sha256", lambda _path, _expected: None) + + local_install.install_llama_server() + + assert not legacy_dir.exists() + assert (vulkan_dir / "llama-server").read_text(encoding="utf-8") == "vulkan" + assert local_install.cuda_binary_path().is_file() + + +def test_install_llama_server_cuda_failure_leaves_legacy_oci_tree( + tmp_path, + monkeypatch, +): + _init_journal(tmp_path, monkeypatch) + _force_cuda_backend(monkeypatch) + artifact_key = local_install.llama_server_artifact_key() + legacy_digest = "a" * 64 + legacy_dir = ( + tmp_path + / "cache" + / "providers" + / "local" + / "cuda" + / artifact_key + / legacy_digest + ) + legacy_dir.mkdir(parents=True) + (legacy_dir / ".oci-install.json").write_text( + json.dumps( + { + "image_ref": f"ghcr.io/acme/runtime@sha256:{legacy_digest}", + "arch": "amd64", + "files": {"llama-server": "b" * 64}, + } + ) + + "\n", + encoding="utf-8", ) + tarball = _write_cuda_runtime_tarball(tmp_path, missing=("licenses/",)) + _patch_tarball_download(monkeypatch, tarball) + monkeypatch.setattr(local_install, "_verify_sha256", lambda _path, _expected: None) - result = local_install.install_llama_server() + with pytest.raises(local_install.LocalProviderError): + local_install.install_llama_server() - assert result["install_state"] == "verifying" - assert prove_manifest( - artifact_manifest_path(final_path.parent), - provider=local_install.LOCAL_PROVIDER_NAME, - pin_identity=local_install._vulkan_pin_identity("test-platform", pin), - ).ready + assert legacy_dir.exists() + + +def test_local_install_owner_path_has_no_oci_registry_or_cosign_entrypoint( + tmp_path, + monkeypatch, +) -> None: + source = Path(local_install.__file__).read_text(encoding="utf-8") + assert "solstone.think.providers import oci_image" not in source + assert "pull_and_install" not in source + assert "verify_image_signature" not in source + assert "cosign" not in source + + def fail_cosign(cmd: list[str], **_kwargs: object) -> SimpleNamespace: + if cmd and cmd[0] == "cosign": + raise AssertionError("cosign must not run in the owner install path") + return SimpleNamespace(returncode=0, stdout="", stderr="") + + _init_journal(tmp_path, monkeypatch) + _force_cuda_backend(monkeypatch) + tarball = _write_cuda_runtime_tarball(tmp_path) + _patch_tarball_download(monkeypatch, tarball) + monkeypatch.setattr(local_install, "_verify_sha256", lambda _path, _expected: None) + monkeypatch.setattr(subprocess, "run", fail_cosign) + + local_install.install_llama_server() def test_probe_binary_runnable_returns_true_for_zero_exit(tmp_path): @@ -1202,11 +1400,6 @@ def test_install_local_blocks_before_downloads(tmp_path, monkeypatch): "_download_file", lambda *_args, **_kwargs: pytest.fail("download should not start"), ) - monkeypatch.setattr( - oci_image, - "pull_and_install", - lambda *_args, **_kwargs: pytest.fail("OCI pull should not start"), - ) with pytest.raises(local_install.LocalProviderError) as exc_info: local_install.install_local(LOCAL_MODEL) @@ -1263,11 +1456,6 @@ def test_install_local_warning_continues_to_download(tmp_path, monkeypatch): ) monkeypatch.setattr(local_install, "_chmod_executable", lambda _path: None) monkeypatch.setattr(local_install, "_clear_macos_quarantine", lambda _path: None) - monkeypatch.setattr( - oci_image, - "pull_and_install", - lambda *_args, **_kwargs: pytest.fail("OCI pull should not start"), - ) assert local_install.install_local(LOCAL_MODEL)["install_state"] == "installed" @@ -1296,11 +1484,6 @@ def test_install_local_ready_short_circuits_before_fit_report(tmp_path, monkeypa "_download_file", lambda *_args, **_kwargs: pytest.fail("download should not start"), ) - monkeypatch.setattr( - oci_image, - "pull_and_install", - lambda *_args, **_kwargs: pytest.fail("OCI pull should not start"), - ) result = local_install.install_local(LOCAL_MODEL) @@ -1356,7 +1539,7 @@ def test_install_local_reinstalls_runtime_when_binary_record_stale( assert calls == ["llama_server", "model"] -def test_install_local_replaces_failed_cuda_attempt_with_vulkan_target( +def test_install_local_replaces_failed_cuda_attempt_with_current_cuda_target( tmp_path: Path, monkeypatch: pytest.MonkeyPatch, ) -> None: @@ -1365,7 +1548,7 @@ def test_install_local_replaces_failed_cuda_attempt_with_vulkan_target( monkeypatch, compute_cap="sm_89", driver_cuda_version=15, - trust=local_cuda.ArtifactTrust.ABSENT, + trust=local_cuda.ArtifactTrust.TRUSTED, ) stale_cuda = { "provider": "local", @@ -1430,11 +1613,8 @@ def test_install_local_replaces_failed_cuda_attempt_with_vulkan_target( target = json.loads(str(status["target_fingerprint_json"])) assert result["install_state"] == "installed" assert status["target_fingerprint_sha256"] != stale_sha - assert target["backend"] == "vulkan" - assert target["backend_reason"] == ( - "compute_cap sm_89 covered; driver CUDA 15 >= 13; " - "no trusted CUDA runtime artifact present" - ) + assert target["backend"] == "cuda" + assert target["backend_reason"] == "compute_cap sm_89 covered; driver CUDA 15 >= 13" assert calls == ["llama_server", "model"] @@ -1575,48 +1755,38 @@ def test_inspect_readiness_reports_ram_sufficient_for_low_or_unknown_memory( assert readiness.host["ram_sufficient"] is True -@pytest.mark.parametrize("sidecar_ok", [True, False]) -def test_inspect_readiness_cuda_uses_sidecar_full_set( +@pytest.mark.parametrize("manifest_ok", [True, False]) +def test_inspect_readiness_cuda_uses_manifest_full_set( tmp_path, monkeypatch, - sidecar_ok, + manifest_ok, ): - from solstone.think.providers import oci_image - _init_journal(tmp_path, monkeypatch) _force_cuda_backend(monkeypatch) binary = local_install.cuda_binary_path() binary.parent.mkdir(parents=True, exist_ok=True) - wanted_files = local_install.CUDA_SERVER_PIN.wanted_files_for_arch( - local_install._oci_arch() - ) + arch = local_install._oci_arch() + wanted_files = local_install.CUDA_SERVER_PIN.wanted_files_for_arch(arch) for name in wanted_files: member = binary.parent / name member.write_text(name, encoding="utf-8") member.chmod(0o755) - (binary.parent / ".oci-install.json").write_text( - json.dumps( - { - "image_ref": local_install.CUDA_SERVER_PIN.image_ref, - "arch": local_install._oci_arch(), - "files": {name: "0" * 64 for name in wanted_files}, - } - ) - + "\n", - encoding="utf-8", + licenses = binary.parent / "licenses" + licenses.mkdir() + (licenses / "LICENSE").write_text("license", encoding="utf-8") + (binary.parent / "provenance.json").write_text("{}\n", encoding="utf-8") + artifact_pin = local_install.require_cuda_artifact_pin_for_current_platform() + local_install._write_cuda_manifest( + artifact_key=local_install.llama_server_artifact_key(), + artifact_pin=artifact_pin, + arch=arch, + wanted_files=wanted_files, + attempt_status=None, + fingerprint=local_install.target_fingerprint(LOCAL_MODEL), + root=binary.parent, ) - verify_calls: list[tuple[str, str, tuple[str, ...], Path]] = [] - - def fake_verify( - image_ref: str, - arch: str, - wanted_files: tuple[str, ...], - target_dir: Path, - ) -> bool: - verify_calls.append((image_ref, arch, wanted_files, target_dir)) - return sidecar_ok - - monkeypatch.setattr(oci_image, "verify_sidecar_install", fake_verify) + if not manifest_ok: + (binary.parent / wanted_files[-1]).unlink() monkeypatch.setattr( local_vulkan, "detect_gpus", @@ -1632,17 +1802,12 @@ def test_inspect_readiness_cuda_uses_sidecar_full_set( "compute_cap sm_121 covered; driver CUDA 13 >= 13" ) assert readiness.artifacts["binary_path"] == str(binary) - assert readiness.artifacts["binary_installed"] is sidecar_ok + assert readiness.artifacts["binary_installed"] is manifest_ok assert readiness.host["gpu_available"] is True assert readiness.host["gpu_probe_ok"] is True - assert verify_calls == [ - ( - local_install.CUDA_SERVER_PIN.image_ref, - local_install._oci_arch(), - wanted_files, - local_install.cuda_binary_dir(), - ) - ] + assert readiness.proof["cuda"]["status"] == ( + "ready" if manifest_ok else "missing-or-mismatched" + ) def test_inspect_readiness_reports_gpu_available_with_hardware(tmp_path, monkeypatch):