diff --git a/.github/workflows/studio-backend-ci.yml b/.github/workflows/studio-backend-ci.yml
index ee5bbe8633..88c7344683 100644
--- a/.github/workflows/studio-backend-ci.yml
+++ b/.github/workflows/studio-backend-ci.yml
@@ -222,6 +222,7 @@ jobs:
for s in \
tests/sh/test_get_torch_index_url.sh \
tests/sh/test_mac_intel_compat.sh \
+ tests/sh/test_nvcc_meets_llama_minimum.sh \
tests/sh/test_tauri_install_exit_order.sh \
tests/sh/test_torch_constraint.sh; do
echo "::group::$s"
diff --git a/studio/Unsloth_Studio_Colab.ipynb b/studio/Unsloth_Studio_Colab.ipynb
index c3aec04820..00eecfe51d 100644
--- a/studio/Unsloth_Studio_Colab.ipynb
+++ b/studio/Unsloth_Studio_Colab.ipynb
@@ -84,26 +84,7 @@
"id": "277e431e"
},
"outputs": [],
- "source": [
- "import sys, time\n",
- "sys.path.insert(0, \"/content/unsloth/studio/backend\")\n",
- "from colab import start\n",
- "start()"
- ]
- },
- {
- "cell_type": "code",
- "source": [
- "from google.colab import output\n",
- "output.serve_kernel_port_as_iframe(8888, height = 1200, width = \"100%\")\n",
- "for _ in range(10000): time.sleep(300), print(\"=\", end = \"\")"
- ],
- "metadata": {
- "id": "wb9UELh--XzX"
- },
- "id": "wb9UELh--XzX",
- "execution_count": null,
- "outputs": []
+ "source": "import sys\nsys.path.insert(0, \"/content/unsloth/studio/backend\")\nfrom colab import start\nstart()"
},
{
"cell_type": "markdown",
@@ -150,4 +131,4 @@
},
"nbformat": 4,
"nbformat_minor": 5
-}
+}
\ No newline at end of file
diff --git a/studio/backend/colab.py b/studio/backend/colab.py
index 7336f8a532..1bca16359b 100644
--- a/studio/backend/colab.py
+++ b/studio/backend/colab.py
@@ -26,30 +26,68 @@ logger = get_logger(__name__)
def get_colab_url(port: int = 8888) -> str:
"""
Get the actual Colab proxy URL for a port.
+
+ Retries up to 3 times and validates that the result is a real HTTPS Colab
+ URL before returning. Falls back to http://localhost:{port} only when all
+ attempts fail.
"""
+ import time as _time
+
+ fallback = f"http://localhost:{port}"
+
try:
from google.colab.output import eval_js
+ except ImportError:
+ return fallback
- # Use Colab's proxy mechanism
- url = eval_js(f"google.colab.kernel.proxyPort({port})", timeout_sec = 5)
- return url if url else f"http://localhost:{port}"
- except Exception as e:
- logger.info(f"Note: Could not get Colab URL ({e})")
- return f"http://localhost:{port}"
+ for attempt in range(3):
+ try:
+ url = eval_js(f"google.colab.kernel.proxyPort({port})", timeout_sec = 10)
+ # A valid Colab proxy URL starts with https:// and embeds the port.
+ if (
+ url
+ and isinstance(url, str)
+ and url.startswith("https://")
+ and str(port) in url
+ ):
+ return url.rstrip("/")
+ except Exception as e:
+ logger.info(f"Note: Could not get Colab URL (attempt {attempt + 1}/3: {e})")
+ if attempt < 2:
+ _time.sleep(1)
+
+ logger.warning(
+ f"Could not get a valid Colab proxy URL after 3 attempts — using localhost fallback. "
+ f"The link/iframe may not work from outside the runtime."
+ )
+ return fallback
-def show_link(port: int = 8888):
- """Display a styled clickable link to the UI."""
+def show_link(port: int = 8888, *, _url: "str | None" = None):
+ """Display a styled clickable link to the UI.
+
+ *_url* is an optional pre-fetched Colab proxy URL. When omitted,
+ ``get_colab_url(port)`` is called internally. Pass it from
+ ``_show_and_embed`` to avoid a second ``eval_js`` round-trip.
+ """
from IPython.display import display, HTML
- # Get real Colab proxy URL
- url = get_colab_url(port)
+ url = _url if _url is not None else get_colab_url(port)
+
+ # Build a truncated display URL. Wrap in try/except so an unexpected URL
+ # shape never prevents the link from rendering.
+ try:
+ port_prefix = f"{port}-"
+ idx = url.index(port_prefix)
+ next_dash = url.index("-", idx + len(port_prefix))
+ short_url = url[: next_dash + 1] + "..."
+ except (ValueError, IndexError):
+ short_url = url
+
+ # Also emit a plain-text line so the URL is visible even if HTML display
+ # is suppressed or fails.
+ logger.info(f"🌐 Unsloth Studio URL: {url}")
- short_url = (
- url[: url.index("-", url.index(f"{port}-") + len(str(port)) + 1) + 1] + "..."
- if f"{port}-" in url
- else url
- )
html = f"""
@@ -59,10 +97,10 @@ def show_link(port: int = 8888):
height="48" style="display:block;">
Unsloth Studio is Ready!
-
+ font-weight: 800; font-size: 16px; cursor: pointer;">
Open Unsloth Studio
@@ -77,6 +115,75 @@ def show_link(port: int = 8888):
display(HTML(html))
+def _is_studio_healthy(port: int, timeout: float = 2.0) -> bool:
+ """Return True if a Studio backend is already answering health checks on *port*."""
+ import urllib.request
+
+ try:
+ with urllib.request.urlopen(
+ f"http://localhost:{port}/api/health", timeout = timeout
+ ):
+ return True
+ except Exception:
+ return False
+
+
+def _show_and_embed(port: int):
+ """Embed the Studio inline for *port* with a branded header bar.
+
+ Fetches the Colab proxy URL once (registering the port with Colab's
+ reverse-proxy at the same time) then renders a header bar + full-height
+ iframe as a single HTML block.
+
+ Falls back to ``serve_kernel_port_as_iframe`` if ``IPython.display.HTML``
+ is unavailable for any reason.
+ """
+ url = get_colab_url(port)
+ logger.info(f"🌐 Unsloth Studio URL: {url}")
+
+ try:
+ from IPython.display import HTML, display
+
+ iframe_id = f"unsloth-studio-{port}"
+
+ # Truncated URL shown in the header — best-effort, falls back to full URL.
+ try:
+ port_prefix = f"{port}-"
+ idx = url.index(port_prefix)
+ next_dash = url.index("-", idx + len(port_prefix))
+ short_url = url[: next_dash + 1] + "..."
+ except (ValueError, IndexError):
+ short_url = url
+
+ display(
+ HTML(f"""
+
+
+
+ Unsloth Studio
+ {short_url}
+
+
+
+""")
+ )
+ except Exception:
+ # Fallback: Colab's built-in helper (less control, but always works)
+ try:
+ from google.colab import output as colab_output
+
+ colab_output.serve_kernel_port_as_iframe(port, height = 900, width = "100%")
+ except ImportError:
+ pass
+
+
def start(port: int = 8888):
"""
Start Unsloth Studio server in Colab and display the URL.
@@ -85,10 +192,26 @@ def start(port: int = 8888):
from colab import start
start()
"""
- import sys
+ import time
logger.info("🦥 Starting Unsloth Studio...")
+ # --- Fast path: Studio is already running (cell re-run) ---
+ # Re-launching would either collide on the port or silently shift to a new
+ # port and confuse the user. Just re-show the link and iframe instead.
+ if _is_studio_healthy(port):
+ logger.info(
+ f" Studio is already running on port {port} — reusing existing server."
+ )
+ _show_and_embed(port)
+ try:
+ for _ in range(10000):
+ time.sleep(300)
+ print("=", end = "", flush = True)
+ except KeyboardInterrupt:
+ logger.info("\nUnsloth Studio keepalive stopped.")
+ return
+
logger.info(" Loading backend...")
from run import run_server
@@ -96,18 +219,63 @@ def start(port: int = 8888):
repo_root = Path(__file__).parent.parent
frontend_path = repo_root / "frontend" / "dist"
- if not frontend_path.exists():
+ if not (frontend_path / "index.html").exists():
logger.info("❌ Frontend not built! Please run the setup cell first.")
return
logger.info(" Starting server...")
- # Start server silently
- run_server(host = "0.0.0.0", port = port, frontend_path = frontend_path, silent = True)
+ try:
+ app = run_server(
+ host = "0.0.0.0", port = port, frontend_path = frontend_path, silent = True
+ )
+ except SystemExit as exc:
+ logger.error(f"❌ Unsloth Studio failed to start: {exc}")
+ return
+ except Exception as exc:
+ logger.error(f"❌ Unsloth Studio failed to start: {exc}")
+ return
- logger.info(" Server started!")
+ # run_server auto-increments the port when the requested one is already in
+ # use (e.g. Jupyter occupying 8888). Read back the actual bound port so the
+ # Colab proxy URL and iframe always point at the right place.
+ actual_port: int = getattr(getattr(app, "state", None), "server_port", None) or port
- # Show the clickable link with real URL
- show_link(port)
+ logger.info(f" Server started on port {actual_port}!")
+
+ # Poll health endpoint to confirm the server is truly reachable before
+ # showing the link and registering the iframe — avoids the race where
+ # ready_event fires but the process hasn't finished binding.
+ import urllib.request
+
+ server_ready = False
+ for _ in range(40):
+ try:
+ with urllib.request.urlopen(
+ f"http://localhost:{actual_port}/api/health", timeout = 1
+ ):
+ server_ready = True
+ break
+ except Exception:
+ time.sleep(0.5)
+
+ if not server_ready:
+ logger.error(
+ f"❌ Unsloth Studio did not become healthy on port {actual_port}. "
+ "Check for errors above."
+ )
+ return
+
+ _show_and_embed(actual_port)
+
+ # Keep kernel alive so the daemon server thread stays running.
+ # Handle KeyboardInterrupt cleanly so the user gets a readable message
+ # rather than a raw traceback when they interrupt the cell.
+ try:
+ for _ in range(10000):
+ time.sleep(300)
+ print("=", end = "", flush = True)
+ except KeyboardInterrupt:
+ logger.info("\nUnsloth Studio keepalive stopped.")
if __name__ == "__main__":
diff --git a/studio/backend/main.py b/studio/backend/main.py
index 1d60bba881..bd007d2e37 100644
--- a/studio/backend/main.py
+++ b/studio/backend/main.py
@@ -329,20 +329,58 @@ from starlette.requests import Request as _StarletteRequest # noqa: E402
_CSP_SCRIPT_NONCE_HEADER = "x-internal-script-nonce"
+# /content is Colab's working directory — more reliable than env vars which
+# aren't always set depending on Colab runtime version.
+import importlib.util as _importlib_util
+
+_IS_COLAB = os.path.isdir("/content") and (
+ bool(os.environ.get("COLAB_BACKEND_URL"))
+ or bool(os.environ.get("COLAB_JUPYTER_IP"))
+ or _importlib_util.find_spec("google.colab") is not None
+)
+
+
def _build_csp(script_nonce: "str | None" = None) -> str:
script_src = "script-src 'self'"
if script_nonce:
script_src += f" 'nonce-{script_nonce}'"
+ # In Colab the parent frame can be colab.research.google.com, a multi-level
+ # *.prod.colab.dev subdomain (e.g. foo.region.prod.colab.dev — note: CSP
+ # wildcards only match one level, so *.prod.colab.dev misses these), or a
+ # sandboxed null-origin output iframe. Use '*' so any ancestor is allowed;
+ # Colab is already a sandboxed single-user environment.
+ frame_ancestors = "*" if _IS_COLAB else "'none'"
+
+ # In Colab the frontend is served over the Colab reverse-proxy at an HTTPS
+ # *.prod.colab.dev URL. Colab's kernel communication layer and the output
+ # iframe scaffolding inject scripts from *.prod.colab.dev and
+ # *.googleusercontent.com, and make fetch/WebSocket connections to those
+ # same origins. Widen script-src and connect-src in Colab mode so those
+ # requests are not blocked. 'unsafe-inline' for scripts is still omitted;
+ # our own inline script uses a nonce.
+ if _IS_COLAB:
+ script_src += " https://*.prod.colab.dev https://*.googleusercontent.com"
+ connect_src = (
+ "'self' blob: data: "
+ "https://huggingface.co https://datasets-server.huggingface.co "
+ "https://*.prod.colab.dev wss://*.prod.colab.dev "
+ "https://*.googleusercontent.com wss://*.googleusercontent.com"
+ )
+ else:
+ connect_src = (
+ "'self' https://huggingface.co https://datasets-server.huggingface.co"
+ )
+
return (
"default-src 'self'; "
"img-src 'self' data: blob: https://t0.gstatic.com "
"https://t1.gstatic.com https://t2.gstatic.com "
"https://t3.gstatic.com https://www.google.com; "
- "connect-src 'self' https://huggingface.co https://datasets-server.huggingface.co; "
+ f"connect-src {connect_src}; "
"style-src 'self' 'unsafe-inline'; "
f"{script_src}; "
"font-src 'self' data:; "
- "frame-ancestors 'none'; "
+ f"frame-ancestors {frame_ancestors}; "
"form-action 'self'; "
"base-uri 'self'"
)
@@ -358,7 +396,10 @@ class SecurityHeadersMiddleware(BaseHTTPMiddleware):
if nonce is not None:
del response.headers[_CSP_SCRIPT_NONCE_HEADER]
response.headers.setdefault("Content-Security-Policy", _build_csp(nonce))
- response.headers.setdefault("X-Frame-Options", "DENY")
+ # Omit X-Frame-Options in Colab — CSP frame-ancestors handles it, and
+ # DENY would block serve_kernel_port_as_iframe regardless of CSP.
+ if not _IS_COLAB:
+ response.headers.setdefault("X-Frame-Options", "DENY")
response.headers.setdefault("X-Content-Type-Options", "nosniff")
response.headers.setdefault("Referrer-Policy", "no-referrer")
response.headers.setdefault(
diff --git a/studio/backend/run.py b/studio/backend/run.py
index 5ddf404370..f401bc2ec1 100644
--- a/studio/backend/run.py
+++ b/studio/backend/run.py
@@ -653,7 +653,7 @@ def run_server(
from threading import Thread, Event
import uvicorn
- from main import app, setup_frontend
+ from main import app, setup_frontend, _IS_COLAB
from utils.paths import ensure_studio_directories
# Create all standard directories on startup
@@ -737,14 +737,22 @@ def run_server(
ready_event.set()
# server_header=False suppresses uvicorn's "Server: uvicorn"; SecurityHeadersMiddleware sets its own.
- config = uvicorn.Config(
- app,
+ config_kwargs = dict(
host = host,
port = port,
log_level = "info",
access_log = False,
server_header = False,
)
+ # Only in Colab: trust X-Forwarded-* from Colab's reverse proxy so the app
+ # sees the real https origin. forwarded_allow_ips="*" is fine inside Colab's
+ # single-user sandbox, but would be an unwanted security relaxation for a
+ # normal local/standalone Studio, so leave uvicorn's safe defaults
+ # (forwarded headers trusted from loopback only) in place there.
+ if _IS_COLAB:
+ config_kwargs["proxy_headers"] = True
+ config_kwargs["forwarded_allow_ips"] = "*"
+ config = uvicorn.Config(app, **config_kwargs)
_server = _ReadyServer(config)
_shutdown_event = Event()
@@ -766,14 +774,21 @@ def run_server(
app.state.trigger_shutdown = _trigger_shutdown
- # Run server in a daemon thread
+ # Run server in a daemon thread.
+ # Use an explicit new_event_loop() + run_until_complete() instead of
+ # asyncio.run() to avoid nest_asyncio's global patches to asyncio.run
+ # interfering when called from a thread while Colab/IPython already has
+ # a running loop on the main thread.
def _run():
+ loop = asyncio.new_event_loop()
+ asyncio.set_event_loop(loop)
try:
- asyncio.run(_server.serve())
+ loop.run_until_complete(_server.serve())
except BaseException as exc:
startup_errors.append(exc)
startup_failed.set()
finally:
+ loop.close()
if not ready_event.is_set():
startup_failed.set()
diff --git a/studio/frontend/src/features/chat/api/chat-adapter.ts b/studio/frontend/src/features/chat/api/chat-adapter.ts
index 982a739589..33afe92d86 100644
--- a/studio/frontend/src/features/chat/api/chat-adapter.ts
+++ b/studio/frontend/src/features/chat/api/chat-adapter.ts
@@ -45,6 +45,7 @@ import {
} from "../provider-capabilities";
import {
type PendingImageEditReference,
+ resolveToolsEnabledOnLoad,
useChatRuntimeStore,
} from "../stores/chat-runtime-store";
import { useExternalProvidersStore } from "../stores/external-providers-store";
@@ -1251,8 +1252,7 @@ async function autoLoadSmallestModel(): Promise<{
supportsPreserveThinking:
loadResp.supports_preserve_thinking ?? false,
supportsTools: loadResp.supports_tools ?? false,
- toolsEnabled: loadResp.supports_tools ?? false,
- codeToolsEnabled: loadResp.supports_tools ?? false,
+ ...resolveToolsEnabledOnLoad(loadResp.supports_tools ?? false),
kvCacheDtype: loadResp.cache_type_kv ?? null,
loadedKvCacheDtype: loadResp.cache_type_kv ?? null,
defaultChatTemplate: loadResp.chat_template ?? null,
@@ -1315,8 +1315,7 @@ async function autoLoadSmallestModel(): Promise<{
sfLoadResp.supports_preserve_thinking ?? false,
supportsTools: sfLoadResp.supports_tools ?? false,
// Parity with the GGUF branch above.
- toolsEnabled: sfLoadResp.supports_tools ?? false,
- codeToolsEnabled: sfLoadResp.supports_tools ?? false,
+ ...resolveToolsEnabledOnLoad(sfLoadResp.supports_tools ?? false),
defaultChatTemplate: sfLoadResp.chat_template ?? null,
chatTemplateOverride: null,
loadedChatTemplateOverride: null,
@@ -1414,8 +1413,7 @@ async function autoLoadSmallestModel(): Promise<{
reasoningStyle: loadResp.reasoning_style ?? "enable_thinking",
supportsPreserveThinking: loadResp.supports_preserve_thinking ?? false,
supportsTools: loadResp.supports_tools ?? false,
- toolsEnabled: loadResp.supports_tools ?? false,
- codeToolsEnabled: loadResp.supports_tools ?? false,
+ ...resolveToolsEnabledOnLoad(loadResp.supports_tools ?? false),
kvCacheDtype: loadResp.cache_type_kv ?? null,
loadedKvCacheDtype: loadResp.cache_type_kv ?? null,
defaultChatTemplate: loadResp.chat_template ?? null,
diff --git a/studio/frontend/src/features/chat/hooks/use-chat-model-runtime.ts b/studio/frontend/src/features/chat/hooks/use-chat-model-runtime.ts
index 6194fc7ef5..41bd04d9e5 100644
--- a/studio/frontend/src/features/chat/hooks/use-chat-model-runtime.ts
+++ b/studio/frontend/src/features/chat/hooks/use-chat-model-runtime.ts
@@ -27,6 +27,7 @@ import {
CHAT_REASONING_ENABLED_KEY,
loadOptionalBool,
type ReasoningEffort,
+ resolveToolsEnabledOnLoad,
useChatRuntimeStore,
} from "../stores/chat-runtime-store";
import {
@@ -698,14 +699,12 @@ export function useChatModelRuntime() {
reasoningEffort: clampedReasoningEffort,
supportsPreserveThinking: loadResponse.supports_preserve_thinking ?? false,
supportsTools,
- toolsEnabled:
- reloadingSameModel && supportsTools
- ? stateBeforeUnload.toolsEnabled
- : supportsTools,
- codeToolsEnabled:
- reloadingSameModel && supportsTools
- ? stateBeforeUnload.codeToolsEnabled
- : supportsTools,
+ ...(reloadingSameModel && supportsTools
+ ? {
+ toolsEnabled: stateBeforeUnload.toolsEnabled,
+ codeToolsEnabled: stateBeforeUnload.codeToolsEnabled,
+ }
+ : resolveToolsEnabledOnLoad(supportsTools)),
kvCacheDtype: loadedKv,
loadedKvCacheDtype: loadedKv,
speculativeType: loadedSpec,
diff --git a/studio/frontend/src/features/chat/stores/chat-runtime-store.ts b/studio/frontend/src/features/chat/stores/chat-runtime-store.ts
index 62c4f53c12..914fc1bcec 100644
--- a/studio/frontend/src/features/chat/stores/chat-runtime-store.ts
+++ b/studio/frontend/src/features/chat/stores/chat-runtime-store.ts
@@ -186,6 +186,23 @@ export function loadOptionalBool(key: string): boolean | null {
}
}
+/**
+ * Resolve the web-search / code-execution pill state to apply when a model
+ * loads. Honors the user's persisted preference so loading a tool-capable
+ * model never silently re-enables a pill the user turned off; falls back to
+ * the model's capability only when no preference has been expressed.
+ */
+export function resolveToolsEnabledOnLoad(supportsTools: boolean): {
+ toolsEnabled: boolean;
+ codeToolsEnabled: boolean;
+} {
+ if (!supportsTools) return { toolsEnabled: false, codeToolsEnabled: false };
+ return {
+ toolsEnabled: loadOptionalBool(CHAT_TOOLS_ENABLED_KEY) ?? true,
+ codeToolsEnabled: loadOptionalBool(CHAT_CODE_TOOLS_ENABLED_KEY) ?? true,
+ };
+}
+
function saveBool(key: string, value: boolean): void {
if (!canUseStorage()) return;
try {
diff --git a/studio/setup.ps1 b/studio/setup.ps1
index 8950627ae8..afa00409fb 100644
--- a/studio/setup.ps1
+++ b/studio/setup.ps1
@@ -2589,6 +2589,17 @@ if (-not $NeedLlamaSourceBuild) {
# CUDA flags -- only if GPU available, otherwise explicitly disable
if ($HasNvidiaSmi -and $NvccPath) {
$CmakeArgs += '-DGGML_CUDA=ON'
+ # Accept a host MSVC newer than nvcc's whitelist; a fresh toolkit
+ # (e.g. CUDA 13.3) otherwise aborts with "#error -- unsupported
+ # Microsoft Visual Studio version!". Mirrors the Linux fix. Via env
+ # (covers the configure probe + build), after Refresh-Environment, idempotent.
+ $nvccAllowFlag = '-allow-unsupported-compiler'
+ if ([string]::IsNullOrEmpty($env:NVCC_PREPEND_FLAGS)) {
+ $env:NVCC_PREPEND_FLAGS = $nvccAllowFlag
+ } elseif ($env:NVCC_PREPEND_FLAGS -notlike "*$nvccAllowFlag*") {
+ $env:NVCC_PREPEND_FLAGS = "$($env:NVCC_PREPEND_FLAGS) $nvccAllowFlag"
+ }
+ substep "NVCC_PREPEND_FLAGS = $env:NVCC_PREPEND_FLAGS"
$CmakeArgs += "-DCUDAToolkit_ROOT=$CudaToolkitRoot"
$CmakeArgs += "-DCUDA_TOOLKIT_ROOT_DIR=$CudaToolkitRoot"
$CmakeArgs += "-DCMAKE_CUDA_COMPILER=$NvccPath"
diff --git a/studio/setup.sh b/studio/setup.sh
index e8fc8f6f13..f2395255c0 100755
--- a/studio/setup.sh
+++ b/studio/setup.sh
@@ -130,6 +130,30 @@ run_quiet_no_exit() {
_run_quiet return "$@"
}
+_nvcc_meets_llama_minimum() {
+ # Echo "ok|too_old|unknown" then the parsed "X.Y" version, one per line.
+ # llama.cpp needs CUDA toolkit >= 12.4 (#4437; setup.ps1 aborts via #4517).
+ _nvcc_bin=$1
+ [ -n "$_nvcc_bin" ] || { echo "unknown"; echo ""; return 0; }
+ _raw=$("$_nvcc_bin" --version 2>/dev/null \
+ | sed -n 's/.*release \([0-9][0-9]*\.[0-9][0-9]*\).*/\1/p' \
+ | head -1)
+ if [ -z "$_raw" ]; then
+ echo "unknown"; echo ""; return 0
+ fi
+ _maj=${_raw%%.*}
+ _min_raw=${_raw#*.}
+ _min=${_min_raw%%.*}
+ if [ "$_maj" -lt 12 ] 2>/dev/null; then
+ echo "too_old"
+ elif [ "$_maj" -eq 12 ] && [ "$_min" -lt 4 ] 2>/dev/null; then
+ echo "too_old"
+ else
+ echo "ok"
+ fi
+ echo "$_raw"
+}
+
print_llama_error_log() {
local log_file=$1
[ -s "$log_file" ] || return 0
@@ -1005,32 +1029,52 @@ else
CPU_FALLBACK_CMAKE_ARGS="$CPU_FALLBACK_CMAKE_ARGS -DGGML_METAL=OFF"
_TRY_METAL_CPU_FALLBACK=true
elif [ -n "$NVCC_PATH" ]; then
- CMAKE_ARGS="$CMAKE_ARGS -DGGML_CUDA=ON"
+ # Returns "ok|too_old|unknown\nX.Y" on stdout.
+ _NVCC_CHECK="$(_nvcc_meets_llama_minimum "$NVCC_PATH")"
+ _NVCC_STATUS="$(printf '%s\n' "$_NVCC_CHECK" | sed -n '1p')"
+ _NVCC_VER="$(printf '%s\n' "$_NVCC_CHECK" | sed -n '2p')"
- CUDA_ARCHS=""
- if command -v nvidia-smi &>/dev/null; then
- _raw_caps=$(nvidia-smi --query-gpu=compute_cap --format=csv,noheader 2>/dev/null || true)
- while IFS= read -r _cap; do
- _cap=$(echo "$_cap" | tr -d '[:space:]')
- if [[ "$_cap" =~ ^([0-9]+)\.([0-9]+)$ ]]; then
- _arch="${BASH_REMATCH[1]}${BASH_REMATCH[2]}"
- # Append if not already present
- case ";$CUDA_ARCHS;" in
- *";$_arch;"*) ;;
- *) CUDA_ARCHS="${CUDA_ARCHS:+$CUDA_ARCHS;}$_arch" ;;
- esac
- fi
- done <<< "$_raw_caps"
- fi
-
- if [ -n "$CUDA_ARCHS" ]; then
- CMAKE_ARGS="$CMAKE_ARGS -DCMAKE_CUDA_ARCHITECTURES=${CUDA_ARCHS}"
- _BUILD_DESC="building (CUDA, sm_${CUDA_ARCHS//;/+sm_})"
+ if [ "$_NVCC_STATUS" = "too_old" ]; then
+ substep "CUDA toolkit $_NVCC_VER is below llama.cpp minimum (12.4)." "$C_ERR"
+ substep "install a newer CUDA toolkit: https://developer.nvidia.com/cuda-toolkit-archive" "$C_WARN"
+ substep "falling back to CPU llama.cpp build for this run." "$C_WARN"
+ NVCC_PATH=""
+ GPU_BACKEND=""
+ _BUILD_DESC="building (CPU, CUDA toolkit < 12.4)"
else
- _BUILD_DESC="building (CUDA)"
- fi
+ CMAKE_ARGS="$CMAKE_ARGS -DGGML_CUDA=ON"
- CMAKE_ARGS="$CMAKE_ARGS -DCMAKE_CUDA_FLAGS=--threads=0"
+ CUDA_ARCHS=""
+ if command -v nvidia-smi &>/dev/null; then
+ _raw_caps=$(nvidia-smi --query-gpu=compute_cap --format=csv,noheader 2>/dev/null || true)
+ while IFS= read -r _cap; do
+ _cap=$(echo "$_cap" | tr -d '[:space:]')
+ if [[ "$_cap" =~ ^([0-9]+)\.([0-9]+)$ ]]; then
+ _arch="${BASH_REMATCH[1]}${BASH_REMATCH[2]}"
+ # Append if not already present
+ case ";$CUDA_ARCHS;" in
+ *";$_arch;"*) ;;
+ *) CUDA_ARCHS="${CUDA_ARCHS:+$CUDA_ARCHS;}$_arch" ;;
+ esac
+ fi
+ done <<< "$_raw_caps"
+ fi
+
+ if [ -n "$CUDA_ARCHS" ]; then
+ CMAKE_ARGS="$CMAKE_ARGS -DCMAKE_CUDA_ARCHITECTURES=${CUDA_ARCHS}"
+ _BUILD_DESC="building (CUDA, sm_${CUDA_ARCHS//;/+sm_})"
+ else
+ _BUILD_DESC="building (CUDA)"
+ fi
+
+ CMAKE_ARGS="$CMAKE_ARGS -DCMAKE_CUDA_FLAGS=--threads=0"
+
+ # Accept a host gcc/clang newer than nvcc's whitelist; a fresh
+ # toolkit (e.g. CUDA 13.3) otherwise aborts with "#error --
+ # unsupported GNU version". Via env, not CMAKE_ARGS, to avoid
+ # word-splitting.
+ export NVCC_PREPEND_FLAGS="${NVCC_PREPEND_FLAGS:+$NVCC_PREPEND_FLAGS }-allow-unsupported-compiler"
+ fi
elif [ "$GPU_BACKEND" = "rocm" ]; then
# Resolve hipcc symlinks to find the real ROCm root
_HIPCC_REAL="$(readlink -f "$ROCM_HIPCC" 2>/dev/null || printf '%s' "$ROCM_HIPCC")"
@@ -1100,14 +1144,29 @@ else
CMAKE_GENERATOR_ARGS="-G Ninja"
fi
- if ! run_quiet_no_exit "cmake llama.cpp" cmake $CMAKE_GENERATOR_ARGS -S "$_BUILD_TMP" -B "$_BUILD_TMP/build" $CMAKE_ARGS; then
+ # GPU label for the CPU-fallback message: Metal, else GPU_BACKEND
+ # (cuda/rocm). Empty on a bare CPU build (nothing to fall back from).
+ _gpu_fallback_label() {
if [ "$_TRY_METAL_CPU_FALLBACK" = true ]; then
+ echo "Metal"
+ elif [ -n "$GPU_BACKEND" ]; then
+ printf '%s' "$GPU_BACKEND" | tr '[:lower:]' '[:upper:]'
+ fi
+ }
+
+ if ! run_quiet_no_exit "cmake llama.cpp" cmake $CMAKE_GENERATOR_ARGS -S "$_BUILD_TMP" -B "$_BUILD_TMP/build" $CMAKE_ARGS; then
+ _FB_LABEL="$(_gpu_fallback_label)"
+ if [ -n "$_FB_LABEL" ]; then
_TRY_METAL_CPU_FALLBACK=false
- substep "Metal configure failed; retrying CPU build..." "$C_WARN"
+ substep "$_FB_LABEL configure failed; retrying CPU build..." "$C_WARN"
rm -rf "$_BUILD_TMP/build"
- run_quiet_no_exit "cmake llama.cpp (cpu fallback)" cmake $CMAKE_GENERATOR_ARGS -S "$_BUILD_TMP" -B "$_BUILD_TMP/build" $CPU_FALLBACK_CMAKE_ARGS || BUILD_OK=false
- if [ "$BUILD_OK" = true ]; then
- _BUILD_DESC="building (CPU fallback)"
+ if run_quiet_no_exit "cmake llama.cpp (cpu fallback)" cmake $CMAKE_GENERATOR_ARGS -S "$_BUILD_TMP" -B "$_BUILD_TMP/build" $CPU_FALLBACK_CMAKE_ARGS; then
+ _BUILD_DESC="building (CPU fallback after $_FB_LABEL configure failed)"
+ # Now configured for CPU; clear GPU_BACKEND so a later
+ # build-step failure won't re-enter fallback on this config.
+ GPU_BACKEND=""
+ else
+ BUILD_OK=false
fi
else
BUILD_OK=false
@@ -1117,12 +1176,14 @@ else
if [ "$BUILD_OK" = true ]; then
if ! run_quiet_no_exit "build llama-server" cmake --build "$_BUILD_TMP/build" --config Release --target llama-server -j"$NCPU"; then
- if [ "$_TRY_METAL_CPU_FALLBACK" = true ]; then
+ _FB_LABEL="$(_gpu_fallback_label)"
+ if [ -n "$_FB_LABEL" ]; then
_TRY_METAL_CPU_FALLBACK=false
- substep "Metal build failed; retrying CPU build..." "$C_WARN"
+ substep "$_FB_LABEL build failed; retrying CPU build..." "$C_WARN"
rm -rf "$_BUILD_TMP/build"
if run_quiet_no_exit "cmake llama.cpp (cpu fallback)" cmake $CMAKE_GENERATOR_ARGS -S "$_BUILD_TMP" -B "$_BUILD_TMP/build" $CPU_FALLBACK_CMAKE_ARGS; then
- _BUILD_DESC="building (CPU fallback)"
+ _BUILD_DESC="building (CPU fallback after $_FB_LABEL build failed)"
+ GPU_BACKEND=""
run_quiet_no_exit "build llama-server (cpu fallback)" cmake --build "$_BUILD_TMP/build" --config Release --target llama-server -j"$NCPU" || BUILD_OK=false
else
BUILD_OK=false
diff --git a/tests/run_all.sh b/tests/run_all.sh
index 6525263d8f..d84c930392 100755
--- a/tests/run_all.sh
+++ b/tests/run_all.sh
@@ -8,6 +8,7 @@ echo "=== Bash tests ==="
sh "$TESTS_DIR/sh/test_get_torch_index_url.sh"
sh "$TESTS_DIR/sh/test_mac_intel_compat.sh"
sh "$TESTS_DIR/sh/test_torch_constraint.sh"
+sh "$TESTS_DIR/sh/test_nvcc_meets_llama_minimum.sh"
echo ""
echo "=== Python tests ==="
diff --git a/tests/sh/test_get_torch_index_url.sh b/tests/sh/test_get_torch_index_url.sh
index a9fafa5359..89ec32fba5 100755
--- a/tests/sh/test_get_torch_index_url.sh
+++ b/tests/sh/test_get_torch_index_url.sh
@@ -320,6 +320,18 @@ _result=$(run_func "$_dir")
assert_eq "CUDA UMD Version 11.8 -> cu118" "https://download.pytorch.org/whl/cu118" "$_result"
rm -rf "$_dir"
+# 32) Driver-reported "CUDA Version: 13.3" (legacy header) -> cu130.
+_dir=$(make_mock_smi "13.3")
+_result=$(run_func "$_dir")
+assert_eq "CUDA Version 13.3 -> cu130" "https://download.pytorch.org/whl/cu130" "$_result"
+rm -rf "$_dir"
+
+# 33) "CUDA Version: 13.7" -> cu130 (until a cu137 wheel index exists).
+_dir=$(make_mock_smi "13.7")
+_result=$(run_func "$_dir")
+assert_eq "CUDA Version 13.7 -> cu130" "https://download.pytorch.org/whl/cu130" "$_result"
+rm -rf "$_dir"
+
rm -f "$_FUNC_FILE"
rm -rf "$_FAKE_SMI_DIR"
rm -rf "$_TOOLS_DIR"
diff --git a/tests/sh/test_nvcc_meets_llama_minimum.sh b/tests/sh/test_nvcc_meets_llama_minimum.sh
new file mode 100755
index 0000000000..f614c0403a
--- /dev/null
+++ b/tests/sh/test_nvcc_meets_llama_minimum.sh
@@ -0,0 +1,121 @@
+#!/bin/bash
+# Unit tests for _nvcc_meets_llama_minimum() from studio/setup.sh.
+# llama.cpp needs CUDA toolkit >= 12.4 (#4437); setup.ps1 aborts via #4517,
+# the Linux side was silent until this fix.
+set -e
+
+SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
+SETUP_SH="$SCRIPT_DIR/../../studio/setup.sh"
+PASS=0
+FAIL=0
+
+# Extract just the helper function. The sed range is the same pattern the
+# install.sh tests use.
+_FUNC_FILE=$(mktemp)
+sed -n '/^_nvcc_meets_llama_minimum()/,/^}/p' "$SETUP_SH" > "$_FUNC_FILE"
+
+assert_eq() {
+ _label="$1"; _expected="$2"; _actual="$3"
+ if [ "$_actual" = "$_expected" ]; then
+ echo " PASS: $_label"
+ PASS=$((PASS + 1))
+ else
+ echo " FAIL: $_label (expected '$_expected', got '$_actual')"
+ FAIL=$((FAIL + 1))
+ fi
+}
+
+# Fake nvcc printing "release X.Y" in the canonical nvcc -V layout (the helper
+# greps for "release X.Y", stable across CUDA 9.x-13.x).
+make_mock_nvcc() {
+ _ver=$1
+ _dir=$(mktemp -d)
+ cat > "$_dir/nvcc" < ok
+_bin=$(make_mock_nvcc "12.4")
+_out=$(run_check "$_bin")
+assert_eq "12.4 status" "ok" "$(echo "$_out" | sed -n '1p')"
+assert_eq "12.4 version" "12.4" "$(echo "$_out" | sed -n '2p')"
+rm -rf "$(dirname "$_bin")"
+
+# 2) CUDA 12.3 is the highest version that should be rejected.
+_bin=$(make_mock_nvcc "12.3")
+_out=$(run_check "$_bin")
+assert_eq "12.3 status" "too_old" "$(echo "$_out" | sed -n '1p')"
+rm -rf "$(dirname "$_bin")"
+
+# 3) CUDA 12.1 (matches the original bug report in #4437).
+_bin=$(make_mock_nvcc "12.1")
+_out=$(run_check "$_bin")
+assert_eq "12.1 status" "too_old" "$(echo "$_out" | sed -n '1p')"
+rm -rf "$(dirname "$_bin")"
+
+# 4) CUDA 11.8 -> too_old (anything < 12.0 is rejected).
+_bin=$(make_mock_nvcc "11.8")
+_out=$(run_check "$_bin")
+assert_eq "11.8 status" "too_old" "$(echo "$_out" | sed -n '1p')"
+rm -rf "$(dirname "$_bin")"
+
+# 5) CUDA 12.8 -> ok (mid-range supported).
+_bin=$(make_mock_nvcc "12.8")
+_out=$(run_check "$_bin")
+assert_eq "12.8 status" "ok" "$(echo "$_out" | sed -n '1p')"
+rm -rf "$(dirname "$_bin")"
+
+# 6) CUDA 13.0 -> ok.
+_bin=$(make_mock_nvcc "13.0")
+_out=$(run_check "$_bin")
+assert_eq "13.0 status" "ok" "$(echo "$_out" | sed -n '1p')"
+rm -rf "$(dirname "$_bin")"
+
+# 7) CUDA 13.3 -> ok (the freshly shipped toolkit this fix targets).
+_bin=$(make_mock_nvcc "13.3")
+_out=$(run_check "$_bin")
+assert_eq "13.3 status" "ok" "$(echo "$_out" | sed -n '1p')"
+assert_eq "13.3 version" "13.3" "$(echo "$_out" | sed -n '2p')"
+rm -rf "$(dirname "$_bin")"
+
+# 8) Future CUDA 14.0 -> ok (no upper bound).
+_bin=$(make_mock_nvcc "14.0")
+_out=$(run_check "$_bin")
+assert_eq "14.0 status" "ok" "$(echo "$_out" | sed -n '1p')"
+rm -rf "$(dirname "$_bin")"
+
+# 9) Empty argument -> unknown (defensive; never block the build on detection).
+_out=$(run_check "")
+assert_eq "empty path status" "unknown" "$(echo "$_out" | sed -n '1p')"
+
+# 10) Mock nvcc that prints garbage -> unknown.
+_dir=$(mktemp -d)
+cat > "$_dir/nvcc" <<'MOCK'
+#!/bin/sh
+echo "totally not nvcc output"
+MOCK
+chmod +x "$_dir/nvcc"
+_out=$(run_check "$_dir/nvcc")
+assert_eq "garbage output status" "unknown" "$(echo "$_out" | sed -n '1p')"
+rm -rf "$_dir"
+
+rm -f "$_FUNC_FILE"
+
+echo ""
+echo "Results: $PASS passed, $FAIL failed"
+[ "$FAIL" -eq 0 ] || exit 1
diff --git a/tests/studio/install/test_pr4562_bugfixes.py b/tests/studio/install/test_pr4562_bugfixes.py
index 4a309036d6..34b144a905 100644
--- a/tests/studio/install/test_pr4562_bugfixes.py
+++ b/tests/studio/install/test_pr4562_bugfixes.py
@@ -727,16 +727,13 @@ class TestSourceCodePatterns:
assert "-DCMAKE_BUILD_WITH_INSTALL_RPATH=ON" in content
def test_setup_sh_macos_metal_configure_has_cpu_fallback(self):
- """If Metal configure or build fails, setup should retry with CPU fallback."""
+ """If Metal/CUDA/ROCm configure or build fails, setup retries a CPU
+ build. PR #5826 generalised the Metal-only wording via $_FB_LABEL; this
+ check stays label-agnostic so new GPU backends don't require edits."""
content = SETUP_SH.read_text()
assert "_TRY_METAL_CPU_FALLBACK=true" in content
- assert (
- 'substep "Metal configure failed; retrying CPU build..." "$C_WARN"'
- in content
- )
- assert (
- 'substep "Metal build failed; retrying CPU build..." "$C_WARN"' in content
- )
+ assert 'configure failed; retrying CPU build..." "$C_WARN"' in content
+ assert 'build failed; retrying CPU build..." "$C_WARN"' in content
assert 'run_quiet_no_exit "cmake llama.cpp (cpu fallback)"' in content
assert "-DGGML_METAL=OFF" in content
# _TRY_METAL_CPU_FALLBACK must be reset to false in both fallback branches
@@ -745,6 +742,55 @@ class TestSourceCodePatterns:
"_TRY_METAL_CPU_FALLBACK=false should appear at least 3 times "
"(init + configure fallback + build fallback)"
)
+ # The fallback helper must exist and Metal must reach it via the
+ # _TRY_METAL_CPU_FALLBACK shortcut so the macOS path stays covered.
+ assert "_gpu_fallback_label()" in content
+ assert 'echo "Metal"' in content
+
+ def test_setup_sh_exports_allow_unsupported_compiler(self):
+ """Headline fix for PR #5826: a fresh CUDA toolkit's host-compiler
+ whitelist lags the distro gcc/clang, so nvcc rejects the host with
+ "#error -- unsupported GNU version". setup.sh exports
+ NVCC_PREPEND_FLAGS=-allow-unsupported-compiler (via env, not CMAKE_ARGS,
+ for word-splitting safety) so the build and compiler-id probe proceed."""
+ content = SETUP_SH.read_text()
+ assert "-allow-unsupported-compiler" in content
+ # Delivered via NVCC_PREPEND_FLAGS (covers the configure-time compiler
+ # probe too), not embedded in the word-split CMAKE_ARGS string.
+ assert "export NVCC_PREPEND_FLAGS=" in content
+ cmake_args_lines = [
+ line for line in content.splitlines() if "CMAKE_ARGS=" in line
+ ]
+ assert all(
+ "-allow-unsupported-compiler" not in line for line in cmake_args_lines
+ ), "flag must stay out of CMAKE_ARGS (bash word-splitting safety)"
+
+ def test_setup_ps1_exports_allow_unsupported_compiler(self):
+ """Windows parity for the PR #5826 fix: a fresh CUDA toolkit's whitelist
+ also lags MSVC, so nvcc can reject the host with "#error -- unsupported
+ Microsoft Visual Studio version!". setup.ps1 sets
+ NVCC_PREPEND_FLAGS=-allow-unsupported-compiler in the CUDA branch (via
+ env, out of $CmakeArgs) so the configure probe + build proceed."""
+ content = SETUP_PS1.read_text()
+ assert "-allow-unsupported-compiler" in content
+ # Delivered via the process environment, not the $CmakeArgs array, so it
+ # reaches both the configure-time compiler probe and `cmake --build`.
+ assert "$env:NVCC_PREPEND_FLAGS" in content
+ cmake_args_lines = [
+ line for line in content.splitlines() if "$CmakeArgs +=" in line
+ ]
+ assert all(
+ "-allow-unsupported-compiler" not in line for line in cmake_args_lines
+ ), "flag must not be pushed into the $CmakeArgs array"
+ # Must be scoped to the CUDA branch (guarded by the GPU/nvcc check),
+ # not set unconditionally for CPU-only builds.
+ flag_idx = content.index("-allow-unsupported-compiler")
+ cuda_guard_idx = content.index("if ($HasNvidiaSmi -and $NvccPath)")
+ cuda_disable_idx = content.index("'-DGGML_CUDA=OFF'")
+ assert cuda_guard_idx < flag_idx < cuda_disable_idx, (
+ "NVCC_PREPEND_FLAGS must be set inside the CUDA-on branch, "
+ "before the GGML_CUDA=OFF (CPU) branch"
+ )
def test_macos_arm64_cpu_fallback_args_exclude_rpath(self):
"""CPU fallback args must NOT contain Metal-only RPATH flags at runtime."""