diff --git a/studio/install_llama_prebuilt.py b/studio/install_llama_prebuilt.py index 24220a36da..014470c4eb 100755 --- a/studio/install_llama_prebuilt.py +++ b/studio/install_llama_prebuilt.py @@ -205,15 +205,9 @@ class AssetChoice: name: str url: str source_label: str - # Optional paired runtime archive that ships separately from the main - # binary archive. Used on Windows CUDA where upstream publishes - # ``llama-...-bin-win-cuda-X.Y-x64.zip`` (binaries + ggml DLLs) and a - # separate ``cudart-llama-bin-win-cuda-X.Y-x64.zip`` (cudart64_X.dll, - # cublas64_X.dll, cublasLt64_X.dll) — the upstream release notes - # explicitly require both. When set, ``install_from_archives`` - # downloads and overlays the runtime archive on top of the main one - # so the prebuilt binary can find its CUDA runtime DLLs without the - # user having a matching system CUDA toolkit installed. + # Paired runtime archive (Windows CUDA cudart bundle). When set, + # install_from_archives also downloads it and overlays its DLLs on + # top of the main install. See unslothai/unsloth#5106. runtime_name: str | None = None runtime_url: str | None = None runtime_sha256: str | None = None @@ -2889,17 +2883,10 @@ def windows_cuda_attempts( + ",".join(windows_cuda_upstream_asset_names(llama_tag, runtime)) ) continue - # Pair the cudart runtime bundle when upstream ships it alongside - # the main archive. ggml-org's release notes flag this as - # required: the main zip ships only the ggml DLLs and binaries, - # while cudart-llama-bin-win-cuda-X.Y-x64.zip ships - # cudart64_X.dll + cublas64_X.dll + cublasLt64_X.dll. Without - # the cudart pair, the prebuilt loads only when the user has a - # version-matched system CUDA toolkit on PATH (the Windows - # "GPU detected but model on RAM" symptom in unslothai/unsloth#5106). - # We only pair when the main archive is the binary archive -- - # not when we accidentally selected the cudart archive itself - # (legacy alias path). + # Pair the cudart bundle when upstream ships it. Without this + # the binary needs a system CUDA toolkit on PATH at runtime + # (#5106). Only pair when the selected main archive is the + # binary archive, not the cudart archive itself. runtime_archive_name: str | None = None runtime_archive_url: str | None = None if selected_name.startswith(f"llama-"): @@ -2979,9 +2966,7 @@ def published_windows_cuda_attempts( asset_url = release.assets.get(artifact.asset_name) if not asset_url: continue - # See windows_cuda_attempts for the rationale: pair the - # cudart-llama runtime archive when published alongside - # the main binary asset. + # See windows_cuda_attempts: pair the cudart bundle. runtime_archive_name: str | None = None runtime_archive_url: str | None = None if artifact.asset_name.startswith("llama-"): @@ -4039,19 +4024,10 @@ def install_from_archives( try: extract_archive(main_archive, extract_dir) - # Download and extract the paired runtime archive into its own - # temp dir (Windows CUDA cudart bundle: cudart64_X.dll, - # cublas64_X.dll, cublasLt64_X.dll). We extract separately rather - # than into the same dir to avoid copy_globs's "ambiguous archive - # layout" guard tripping on shared filenames like LICENSE.txt - # that appear in both bundles. After extraction we run copy_globs - # against each source dir in turn, so the cudart DLLs end up - # alongside llama-server.exe in install_dir/build/bin/Release. - # Without this overlay, llama-server.exe's LoadLibrary calls - # can't resolve cudart64_X.dll / cublas64_X.dll unless the user - # has a version-matched system CUDA toolkit on PATH -- the root - # cause of the Windows "GPU detected, model on RAM" reports in - # unslothai/unsloth#5106. + # Download the paired runtime archive into its own temp dir to + # avoid copy_globs's ambiguous-layout guard on shared names + # like LICENSE.txt. Two passes of copy_globs land both archives + # in the same overlay dir. Fixes #5106. if choice.runtime_url and choice.runtime_name: runtime_archive = work_dir / choice.runtime_name log( @@ -4074,10 +4050,9 @@ def install_from_archives( source_dir, overlay_dir, runtime_patterns_for_choice(choice), required = True ) if runtime_extract_dir is not None: - # Pull the runtime DLLs into the same overlay dir as the - # main binary. The runtime archive isn't required to contain - # everything matching runtime_patterns_for_choice; it - # contributes a subset (the CUDA DLLs). + # The runtime archive only contributes the CUDA DLLs -- + # not all runtime_patterns_for_choice entries -- so + # required=False. copy_globs( runtime_extract_dir, overlay_dir, @@ -4801,11 +4776,9 @@ def apply_approved_hashes( missing_assets.append(attempt.name) continue attempt.expected_sha256 = approved.sha256 - # Resolve the paired runtime archive's checksum too. If the - # manifest doesn't list the runtime archive (older releases or - # OSS forks that don't track cudart hashes) we drop the pairing - # rather than installing without checksum coverage -- preserves - # the supply-chain guarantee that anything we extract was vetted. + # Resolve the paired runtime archive's hash too. Drop the pair + # if the manifest does not list it -- never install an + # unverified archive. if attempt.runtime_name and attempt.runtime_url: runtime_approved = checksums.artifacts.get(attempt.runtime_name) if runtime_approved is None: diff --git a/tests/studio/install/test_selection_logic.py b/tests/studio/install/test_selection_logic.py index 2a8d723d62..713af51e19 100644 --- a/tests/studio/install/test_selection_logic.py +++ b/tests/studio/install/test_selection_logic.py @@ -1840,13 +1840,8 @@ class TestWindowsCudaAttempts: assert result[1].name == "cudart-llama-bin-win-cuda-12.4-x64.zip" def test_cudart_runtime_archive_is_paired(self, monkeypatch): - # Regression for unslothai/unsloth#5106. Upstream ships - # llama-...-bin-win-cuda-X.Y-x64.zip (binaries + ggml DLLs) AND - # cudart-llama-bin-win-cuda-X.Y-x64.zip (cudart64_X.dll + - # cublas64_X.dll + cublasLt64_X.dll) in the same release. Both - # are required for the prebuilt to actually load CUDA at runtime - # without a system CUDA toolkit. The pairing must surface on - # AssetChoice.runtime_url so install_from_archives downloads it. + # #5106: cudart bundle must surface on runtime_url so + # install_from_archives downloads it. mock_windows_runtime(monkeypatch, ["cuda13", "cuda12"]) host = make_host(system = "Windows", machine = "AMD64", driver_cuda_version = (13, 1)) assets = { @@ -1868,10 +1863,7 @@ class TestWindowsCudaAttempts: assert result[1].runtime_name == "cudart-llama-bin-win-cuda-12.4-x64.zip" def test_no_runtime_archive_when_cudart_absent(self, monkeypatch): - # If upstream stops shipping the cudart bundle (or this is an - # older release tag that pre-dates the split), the install must - # still proceed -- the user falls back to a system CUDA toolkit - # but at least the install doesn't fail. + # Older releases without the cudart split must still install. mock_windows_runtime(monkeypatch, ["cuda12"]) host = make_host(system = "Windows", machine = "AMD64", driver_cuda_version = (12, 4)) assets = { @@ -1883,11 +1875,7 @@ class TestWindowsCudaAttempts: assert result[0].runtime_name is None def test_cudart_only_assets_do_not_self_pair(self, monkeypatch): - # Backwards-compat: the legacy "cudart-only naming" path - # (test_current_upstream_names_are_supported) must keep working, - # and must NOT set runtime_url to its own URL. Self-pairing would - # cause install_from_archives to download the same archive - # twice or hit copy_globs's ambiguous-layout guard. + # Legacy cudart-only naming path must not self-pair. mock_windows_runtime(monkeypatch, ["cuda13", "cuda12"]) host = make_host(system = "Windows", machine = "AMD64", driver_cuda_version = (13, 1)) assets = self._upstream("13.1", "12.4", current_names = True) @@ -1904,10 +1892,7 @@ class TestWindowsCudaAttempts: class TestApplyApprovedHashesRuntimePair: - """When a published bundle pairs a cudart runtime archive with the - main archive, apply_approved_hashes must resolve and pin the runtime - archive's checksum too. If no runtime hash is in the manifest, drop - the pairing rather than installing without checksum coverage.""" + """Runtime archive must inherit a manifest hash, or be dropped.""" TAG = "b8508" @@ -1952,8 +1937,7 @@ class TestApplyApprovedHashesRuntimePair: assert result[0].runtime_name == "cudart-llama-bin-win-cuda-13.1-x64.zip" def test_runtime_pair_dropped_when_hash_missing(self): - # Manifest hashes the main archive but not the runtime archive. - # Drop the pairing rather than installing an unverified runtime. + # Drop the pair rather than install an unverified runtime. attempt = self._runtime_paired_attempt() checksums = ApprovedReleaseChecksums( repo = "unslothai/llama.cpp",