diff --git a/studio/install_llama_prebuilt.py b/studio/install_llama_prebuilt.py index f32e814713..ee438a362e 100644 --- a/studio/install_llama_prebuilt.py +++ b/studio/install_llama_prebuilt.py @@ -5664,13 +5664,27 @@ _OFFLOADING_COUNT_RE = re.compile( re.IGNORECASE, ) -# install_kind values that ship a real GPU backend and must offload to the GPU. -# A binary launched with --n-gpu-layers 1 under one of these is held to the -# GPU-offload check; CPU kinds (windows-cpu, linux-cpu, ...) are exempt. +# install_kind values that ship a real GPU backend and are launched with +# --n-gpu-layers 1 during validation, so a real Metal / CUDA / HIP backend is +# loaded and a broken GPU dylib/DLL surfaces. CPU kinds (windows-cpu, +# linux-cpu, ...) are exempt. _GPU_INSTALL_KINDS = frozenset( {"linux-cuda", "linux-rocm", "windows-cuda", "windows-hip", "macos-arm64"} ) +# Subset whose CPU-only load is a *fixable* fault -- a missing cudart64_* / +# cublas64_* DLL, a PTX-only build on an older driver, or a HIP backend that +# did not initialize -- that a different bundle or a source build can repair, +# so validation rejects it to advance the resolver (#5807, #5106). macOS Metal +# is excluded on purpose: on Apple Silicon the macos-arm64 prebuilt is already +# the correct artifact, and a CPU-only load means Metal is unavailable in this +# environment (a headless or virtualized host such as a CI runner), which no +# rebuild can fix. Rejecting it would force a source build that also runs on +# CPU and needlessly breaks the install. macOS is still launched with +# --n-gpu-layers (it is in _GPU_INSTALL_KINDS) so a broken libggml-metal.dylib +# still surfaces; it is only exempt from the CPU-only *rejection*. +_GPU_OFFLOAD_REQUIRED_KINDS = _GPU_INSTALL_KINDS - frozenset({"macos-arm64"}) + def server_log_shows_gpu_offload(log_text: str) -> bool | None: """Classify whether llama-server put the model on the GPU. @@ -5810,6 +5824,7 @@ def validate_server( _gpu_kinds = _GPU_INSTALL_KINDS if install_kind is not None: _enable_gpu_layers = install_kind in _gpu_kinds + _require_gpu_offload = install_kind in _GPU_OFFLOAD_REQUIRED_KINDS else: # Older call sites that don't pass install_kind: keep ROCm # hosts in the GPU-validation path so an AMD-only Linux host @@ -5820,6 +5835,10 @@ def validate_server( or host.has_rocm or (host.is_macos and host.is_arm64) ) + # macOS Metal is exercised but never *required* (a CPU-only Metal + # load is an environment limitation, not a fixable binary fault; + # see _GPU_OFFLOAD_REQUIRED_KINDS). + _require_gpu_offload = host.has_usable_nvidia or host.has_rocm if _enable_gpu_layers: command.extend(["--n-gpu-layers", "1"]) @@ -5909,15 +5928,18 @@ def validate_server( + ("\n" + response_body if response_body else "") ) if completion_succeeded: - # The server served a completion. When GPU offload was - # requested, confirm the model actually landed on the GPU -- - # a binary whose GPU backend failed to load (CPU-only build, - # unresolved cudart64_*/cublas64_* DLLs on Windows, or a - # PTX-only build on an older driver) still serves HTTP 200 - # from CPU, and accepting it ships a silently CPU-only - # install (#5807, #5106). Reject so the resolver advances to - # the next bundle / source build instead of stopping here. - if _enable_gpu_layers: + # The server served a completion. When GPU offload is + # *required* (CUDA / ROCm / HIP -- see + # _GPU_OFFLOAD_REQUIRED_KINDS), confirm the model actually + # landed on the GPU: a binary whose GPU backend failed to + # load (CPU-only build, unresolved cudart64_*/cublas64_* + # DLLs on Windows, or a PTX-only build on an older driver) + # still serves HTTP 200 from CPU, and accepting it ships a + # silently CPU-only install (#5807, #5106). Reject so the + # resolver advances to the next bundle / source build. macOS + # Metal is intentionally excluded: a CPU-only Metal load is + # an unfixable environment limitation, not a bad binary. + if _require_gpu_offload: log_handle.flush() offload = server_log_shows_gpu_offload(read_full_log(log_path)) if offload is False: @@ -6736,7 +6758,10 @@ def existing_gpu_install_offloads( "reinstall/restart keeps the silently CPU-only binary" path (#5807): without it, a metadata match short-circuits the new offload validation entirely.""" choice = plan.attempts[0] - if choice.install_kind not in _GPU_INSTALL_KINDS: + # Only re-validate kinds whose CPU-only load is a fixable fault. macOS Metal + # is exempt (a CPU-only load is unfixable, see _GPU_OFFLOAD_REQUIRED_KINDS), + # so it keeps the existing install without a per-restart smoke test. + if choice.install_kind not in _GPU_OFFLOAD_REQUIRED_KINDS: return True server_name = "llama-server.exe" if host.is_windows else "llama-server" try: @@ -6804,12 +6829,15 @@ def install_prebuilt( published_repo, published_release_tag, ) - # Non-GPU match: keep the fast path (no probe download). A GPU match - # must still be smoke-tested below, so it does not short-circuit here. + # Non-(offload-required) match: keep the fast path (no probe + # download). A CUDA/ROCm/HIP match must still be smoke-tested below, + # so it does not short-circuit here; macOS Metal (not offload + # required) keeps the fast path. if ( release_plans and existing_install_matches_plan(install_dir, host, release_plans[0]) - and release_plans[0].attempts[0].install_kind not in _GPU_INSTALL_KINDS + and release_plans[0].attempts[0].install_kind + not in _GPU_OFFLOAD_REQUIRED_KINDS ): current = release_plans[0] log( @@ -6905,9 +6933,9 @@ def install_prebuilt( def resolve_smoke_test_install_kind(host: HostInfo) -> str: """Best-effort install_kind for a post-build GPU smoke test, derived from - the host. setup.sh / setup.ps1 can override with --install-kind. GPU kinds - (in _GPU_INSTALL_KINDS) make the smoke test require real GPU offload; CPU - kinds skip that check.""" + the host. setup.sh / setup.ps1 can override with --install-kind. Offload- + required kinds (in _GPU_OFFLOAD_REQUIRED_KINDS) make the smoke test require + real GPU offload; CPU kinds and macOS Metal skip that check.""" if host.is_macos and host.is_arm64: return "macos-arm64" if host.has_rocm: @@ -6944,9 +6972,11 @@ def smoke_test_server_binary( Path(install_dir).expanduser().resolve() if install_dir else server_path.parent ) resolved_kind = install_kind or resolve_smoke_test_install_kind(host) - # For a GPU kind, the CLI contract is "0 = offload confirmed", so an - # inconclusive (no-signal) log must not pass -- require a positive signal. - require_signal = resolved_kind in _GPU_INSTALL_KINDS + # For an offload-required kind (CUDA/ROCm/HIP), the CLI contract is + # "0 = offload confirmed", so an inconclusive (no-signal) log must not pass + # -- require a positive signal. macOS Metal is exempt (see + # _GPU_OFFLOAD_REQUIRED_KINDS): it only needs to load and serve. + require_signal = resolved_kind in _GPU_OFFLOAD_REQUIRED_KINDS if probe: probe_path = Path(probe).expanduser().resolve() if not probe_path.exists(): @@ -7205,7 +7235,7 @@ def main() -> int: # build rather than downgrading it on uncertain evidence. log(f"smoke-test inconclusive: {exc}") return EXIT_ERROR - if resolved_kind in _GPU_INSTALL_KINDS: + if resolved_kind in _GPU_OFFLOAD_REQUIRED_KINDS: log( f"smoke-test passed: {args.smoke_test} " f"(install_kind={resolved_kind}) offloaded to the GPU" diff --git a/tests/studio/install/run_smoke_spoof.py b/tests/studio/install/run_smoke_spoof.py index b4aa5391f6..da056aaa90 100644 --- a/tests/studio/install/run_smoke_spoof.py +++ b/tests/studio/install/run_smoke_spoof.py @@ -27,10 +27,12 @@ INSTALLER = REPO / "studio" / "install_llama_prebuilt.py" FAKE = HERE / "fake_llama_server.py" IS_WIN = sys.platform == "win32" -GPU_KIND = { - "win32": "windows-cuda", - "darwin": "macos-arm64", -}.get(sys.platform, "linux-cuda") +# Offload-required GPU kind for the rejection contract. install_kind is passed +# explicitly, so the classifier / validate_server logic is exercised +# independent of the (GPU-less) runner OS. macOS Metal is intentionally NOT +# offload-required, so use a CUDA kind even on macOS to drive the CUDA/ROCm +# rejection path; Metal's accept-CPU behavior is covered by the macOS-only case. +GPU_REQ_KIND = "windows-cuda" if IS_WIN else "linux-cuda" CPU_KIND = { "win32": "windows-cpu", "darwin": "macos-cpu", @@ -79,13 +81,25 @@ def main() -> int: probe.write_bytes(b"GGUF\x00fake") cases = [ - ("cpu", GPU_KIND, 2, "CPU-only binary tagged GPU is rejected"), - ("offloaded_zero", GPU_KIND, 2, "offloaded 0/N tagged GPU is rejected"), - ("cuda", GPU_KIND, 0, "GPU binary tagged GPU is accepted"), - ("cuda_buffer", GPU_KIND, 0, "GPU buffer-format binary is accepted"), + ("cpu", GPU_REQ_KIND, 2, "CPU-only binary tagged GPU is rejected"), + ("offloaded_zero", GPU_REQ_KIND, 2, "offloaded 0/N tagged GPU is rejected"), + ("cuda", GPU_REQ_KIND, 0, "GPU binary tagged GPU is accepted"), + ("cuda_buffer", GPU_REQ_KIND, 0, "GPU buffer-format binary is accepted"), ("cpu", CPU_KIND, 0, "CPU binary tagged CPU is not gated"), - ("no_signal", GPU_KIND, 1, "no-signal GPU log is inconclusive (exit 1)"), + ("no_signal", GPU_REQ_KIND, 1, "no-signal GPU log is inconclusive (exit 1)"), ] + if sys.platform == "darwin": + # macOS Metal is not offload-required: a CPU-only Metal load is an + # unfixable environment limitation (headless / virtualized host), so + # the macos-arm64 prebuilt is accepted rather than rejected. + cases.append( + ( + "cpu", + "macos-arm64", + 0, + "macOS Metal CPU-only load is accepted (no rebuild remedy)", + ) + ) for mode, kind, expected, label in cases: rc = run_smoke(wrapper, probe, kind, mode) ok = rc == expected diff --git a/tests/studio/install/test_validate_server_gpu_offload.py b/tests/studio/install/test_validate_server_gpu_offload.py index 6028b98558..cac1b643bb 100644 --- a/tests/studio/install/test_validate_server_gpu_offload.py +++ b/tests/studio/install/test_validate_server_gpu_offload.py @@ -434,6 +434,40 @@ def test_rocm_cpu_only_rejected(patched_server, tmp_path): _run_validate(tmp_path, rocm_host(), "linux-rocm") +def test_macos_metal_not_in_offload_required_kinds(): + # macos-arm64 ships a GPU backend (Metal) and is launched with + # --n-gpu-layers, but a CPU-only Metal load is an unfixable environment + # limitation, so it must NOT be in the offload-required (reject) set. + assert "macos-arm64" in M._GPU_INSTALL_KINDS + assert "macos-arm64" not in M._GPU_OFFLOAD_REQUIRED_KINDS + + +def test_macos_metal_cpu_only_not_rejected(patched_server, tmp_path): + # The macOS regression in #5858 CI: GitHub macOS runners have no usable + # Metal, so the macos-arm64 prebuilt loads on CPU. It must be accepted, not + # rejected into a source build that also runs on CPU and breaks the install. + patched_server(CPU_ONLY_DEVICE_INFO_LOG) + _run_validate(tmp_path, macos_arm_host(), "macos-arm64") # no raise + + +def test_smoke_test_macos_metal_cpu_only_passes(patched_server, tmp_path): + # The smoke-test CLI must also accept a CPU-only macOS Metal load (exit 0), + # so setup.sh does not pointlessly retry a CPU source build on a Mac. + patched_server(CPU_ONLY_DEVICE_INFO_LOG) + server = tmp_path / "llama-server" + server.write_text("#!/bin/sh\n") + probe = tmp_path / "probe.gguf" + probe.write_bytes(b"GGUF") + # No raise -> the CLI maps this to EXIT_SUCCESS. + smoke_test_server_binary( + str(server), + macos_arm_host(), + install_dir = str(tmp_path), + probe = str(probe), + install_kind = "macos-arm64", + ) + + # -- 4. smoke_test_server_binary ---------------------------------------------