diff --git a/studio/install_llama_prebuilt.py b/studio/install_llama_prebuilt.py index 77d0905020..b0af64fc8d 100644 --- a/studio/install_llama_prebuilt.py +++ b/studio/install_llama_prebuilt.py @@ -237,13 +237,16 @@ _MAX_PROBE_CUDA_MAJOR = 19 # Last ggml-org release whose Windows win-cuda-13 build is still sub-13.3 # (cuda-13.1, b9360, 2026-05-27). Upstream bumped win-cuda-13 to 13.3 at b9365 # and now ships only cuda-12.4 + cuda-13.3. cuda-12.4 predates Blackwell (ggml -# compiles sm_120 only at toolkit >= 12.8), so a Blackwell host on a 13.1/13.2 +# compiles sm_120 only at toolkit >= 12.8), so a Blackwell host on a 13.0/13.1/13.2 # driver is gated off 13.3 and would drop to a CPU-only 12.4 build. b9360 is # immutable, so we pin its cuda-13.1 build (plus paired cudart) as a GPU # fallback for exactly those hosts. See unslothai/unsloth#5887. _PINNED_BLACKWELL_FALLBACK_TAG = "b9360" _PINNED_BLACKWELL_FALLBACK_RUNTIME = "13.1" -_PINNED_BLACKWELL_DRIVER_FLOOR = (13, 1) +# Floor at 13.0: b9360 ships native sm_120a SASS (no PTX/JIT) and a bundled +# cuda-13.1 cudart, both of which run on a CUDA 13.0 r580+ driver via CUDA +# minor-version compatibility, so the mainstream 13.0 Blackwell branch is covered. +_PINNED_BLACKWELL_DRIVER_FLOOR = (13, 0) _BLACKWELL_MIN_SM = 120 # ggml compiles Blackwell sm_120 only at toolkit >= 12.8, so an in-release # windows-cuda build at or above this already covers Blackwell and makes the diff --git a/tests/studio/install/test_selection_logic.py b/tests/studio/install/test_selection_logic.py index 78aa9c480e..8cb53d1aeb 100644 --- a/tests/studio/install/test_selection_logic.py +++ b/tests/studio/install/test_selection_logic.py @@ -2021,7 +2021,7 @@ class TestWindowsCudaAttempts: class TestPinnedBlackwellCudaFallback: - """A Blackwell host on a 13.1/13.2 driver, gated off the in-release 13.3 + """A Blackwell host on a 13.0/13.1/13.2 driver, gated off the in-release 13.3 build, gets the pinned immutable b9360 cuda-13.1 GPU build instead of the CPU-only cuda-12.4 drop. The pin is dormant for everyone else.""" @@ -2072,15 +2072,19 @@ class TestPinnedBlackwellCudaFallback: # Ada/Hopper run the cuda-12.4 build fine; the pin must not fire. assert _pinned_windows_cuda_fallback(self._win_host((13, 1), [sm]), []) is None - def test_pin_not_offered_to_driver_13_0(self): - # 13.0 cannot run the 13.1 build (forward minor); residual CPU gap. + def test_pin_offered_for_driver_13_0(self): + # b9360 is native sm_120a SASS (no JIT) and ships a cuda-13.1 cudart, + # both of which run on a 13.0 r580+ driver via CUDA minor-version + # compatibility. 13.0 is the mainstream Blackwell branch, so it must fire. assert ( - _pinned_windows_cuda_fallback(self._win_host((13, 0), ["120"]), []) is None + _pinned_windows_cuda_fallback(self._win_host((13, 0), ["120"]), []) + is not None ) def test_pin_not_offered_below_floor(self): + # 12.x predates Blackwell entirely; the pin stays dormant below 13.0. assert ( - _pinned_windows_cuda_fallback(self._win_host((12, 8), ["120"]), []) is None + _pinned_windows_cuda_fallback(self._win_host((12, 9), ["120"]), []) is None ) def test_pin_not_offered_without_driver(self):