diff --git a/.github/workflows/consolidated-tests-ci.yml b/.github/workflows/consolidated-tests-ci.yml
index 71c87c6859..7978a200c0 100644
--- a/.github/workflows/consolidated-tests-ci.yml
+++ b/.github/workflows/consolidated-tests-ci.yml
@@ -209,7 +209,7 @@ jobs:
'peft>=0.18,<0.20' 'accelerate>=0.34,<2' \
ipython
# torchvision: unsloth_zoo.vision_utils imports it at module scope.
- pip install --index-url https://download.pytorch.org/whl/cpu \
+ pip install --index-url https://download.pytorch.org/whl/cpu --extra-index-url https://pypi.org/simple \
'torch>=2.4,<2.11' 'torchvision<0.26'
# transformers + trl from the matrix combo.
pip install "$RESOLVED_TRANSFORMERS_SPEC"
@@ -2174,7 +2174,7 @@ jobs:
python -m pip install --upgrade pip
# Match the matrix job's torch path so unsloth_zoo's
# `import torch` resolves to the same CPU build.
- pip install --index-url https://download.pytorch.org/whl/cpu \
+ pip install --index-url https://download.pytorch.org/whl/cpu --extra-index-url https://pypi.org/simple \
'torch>=2.4,<2.11' 'torchvision<0.26'
pip install \
'numpy<3' protobuf sentencepiece \
diff --git a/.github/workflows/mlx-ci.yml b/.github/workflows/mlx-ci.yml
index 864630f9f0..a2f716a93c 100644
--- a/.github/workflows/mlx-ci.yml
+++ b/.github/workflows/mlx-ci.yml
@@ -163,7 +163,7 @@ jobs:
'pytest==9.0.3' \
'pytest-asyncio==1.3.0' \
'httpx==0.28.1'
- pip install --index-url https://download.pytorch.org/whl/cpu \
+ pip install --index-url https://download.pytorch.org/whl/cpu --extra-index-url https://pypi.org/simple \
'torch==2.10.0'
# github.com occasionally 500s on the git fetch; retry the
# zoo install so a single upstream blip does not fail CI.
@@ -231,99 +231,6 @@ jobs:
tests/studio/test_is_mlx_dispatch_gate.py \
tests/studio/test_mlx_training_worker_behaviors.py
- # Studio prebuilt llama.cpp install + GGUF inference. Mirrors the
- # path Studio's setup.sh takes on macOS since #5963: plan against
- # the unslothai/llama.cpp fork's latest release, which ships the
- # bin-macos-arm64 bundle plus the llama-prebuilt-manifest.json the
- # default policy reads. After install, downloads a small published
- # GGUF (unsloth/gemma-3-270m-it-GGUF, Q4_K_M) and validates
- # llama-server /completion end to end. An install failure or a
- # non-zero binary exit is an Unsloth/Studio bug.
- - name: Studio prebuilt llama.cpp install + GGUF inference (Mac M1)
- env:
- # Withheld on PR: this step runs checked-out PR code; public GGUF still downloads.
- HF_TOKEN: ${{ github.event_name != 'pull_request' && secrets.HF_TOKEN || '' }}
- # install_llama_prebuilt.py hits the GitHub releases API to
- # resolve the asset URL. Anonymous calls share the runner-IP
- # rate-limit bucket and 403 quickly -- pass the workflow's
- # automatic GITHUB_TOKEN to bump us to the 5000/hr authenticated
- # bucket.
- GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
- GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
- run: |
- set -euo pipefail
- INSTALL_DIR="$HOME/.unsloth-studio-prebuilt-test/llama.cpp"
- rm -rf "$INSTALL_DIR"
- # Mirror studio/setup.sh on macOS (the install.sh user path):
- # it plans against the unslothai/llama.cpp fork's latest
- # release with no policy or tag flags.
- python studio/install_llama_prebuilt.py \
- --install-dir "$INSTALL_DIR" \
- --published-repo unslothai/llama.cpp
-
- # Studio bundles only llama-server + llama-quantize from the
- # prebuilt (not llama-cli) -- inference goes through
- # llama-server's HTTP /completion endpoint. Validate both:
- # llama-quantize --help proves the dynamic libs link, then
- # spin up llama-server and POST a /completion request on a
- # tiny published GGUF.
- LLAMA_SERVER="$INSTALL_DIR/build/bin/llama-server"
- LLAMA_QUANT="$INSTALL_DIR/build/bin/llama-quantize"
- [ -x "$LLAMA_SERVER" ] || { echo "::error::llama-server missing at $LLAMA_SERVER"; find "$INSTALL_DIR/build" -type f | head -40; exit 1; }
- [ -x "$LLAMA_QUANT" ] || { echo "::error::llama-quantize missing at $LLAMA_QUANT"; exit 1; }
- echo "llama-server : $LLAMA_SERVER"
- echo "llama-quantize: $LLAMA_QUANT"
- "$LLAMA_QUANT" --help >/dev/null && echo " llama-quantize loads OK"
-
- mkdir -p /tmp/ggufs
- bash .github/scripts/hf-download-with-retry.sh \
- 'unsloth/gemma-3-270m-it-GGUF' \
- 'gemma-3-270m-it-Q4_K_M.gguf' \
- /tmp/ggufs
-
- PORT=18080
- echo "=== starting llama-server on 127.0.0.1:$PORT ==="
- "$LLAMA_SERVER" \
- -m /tmp/ggufs/gemma-3-270m-it-Q4_K_M.gguf \
- --host 127.0.0.1 \
- --port "$PORT" \
- -c 256 \
- -n 16 \
- --no-warmup \
- > /tmp/llama-server.log 2>&1 &
- SERVER_PID=$!
- trap 'kill "$SERVER_PID" 2>/dev/null || true' EXIT
-
- # Wait for /health to come up
- for i in $(seq 1 30); do
- if curl -sf "http://127.0.0.1:$PORT/health" >/dev/null 2>&1; then
- echo " server up after ${i}s"
- break
- fi
- sleep 1
- done
- if ! curl -sf "http://127.0.0.1:$PORT/health" >/dev/null 2>&1; then
- echo "::error::llama-server never became healthy"
- tail -40 /tmp/llama-server.log
- exit 1
- fi
-
- PROMPT="Hello, my name is"
- echo "=== POST /completion ==="
- RESP=$(curl -sf -X POST "http://127.0.0.1:$PORT/completion" \
- -H 'Content-Type: application/json' \
- -d "{\"prompt\":\"$PROMPT\",\"n_predict\":16,\"temperature\":0,\"seed\":3407}")
- echo "raw response (head): $(echo "$RESP" | head -c 600)"
- CONTENT=$(echo "$RESP" | python -c "import json,sys; print(json.loads(sys.stdin.read()).get('content',''))")
- echo "completion content: $CONTENT"
-
- if [ -z "$CONTENT" ]; then
- echo "::error::llama-server /completion returned empty content"
- tail -40 /tmp/llama-server.log
- exit 1
- fi
- echo "OK: Studio prebuilt llama.cpp on Mac M1 + GGUF /completion works"
-
# Real MLX training + inference smoke test. Trains
# unsloth/gemma-3-270m-it for 7 deterministic LoRA steps
# (batch_size=2, gradient_accumulation_steps=3) on a single
@@ -338,6 +245,9 @@ jobs:
UNSLOTH_COMPILE_DISABLE: '1'
run: |
mkdir -p mlx_workdir
+ # Authenticate llama.cpp's release-API lookup (anonymous 403s on rate-limit);
+ # read-only GITHUB_TOKEN scoped here only, never to steps that run binaries.
+ GH_TOKEN="${{ secrets.GITHUB_TOKEN }}" GITHUB_TOKEN="${{ secrets.GITHUB_TOKEN }}" \
python tests/studio/run_real_mlx_smoke.py train \
--workdir "$PWD/mlx_workdir"
@@ -406,3 +316,88 @@ jobs:
cat "$f" 2>/dev/null || echo "(missing)"
echo
done
+
+ # Validates the macOS prebuilt path Studio's setup.sh uses (#5963): install the
+ # unslothai/llama.cpp fork's latest release, download a small public GGUF, and
+ # check llama-server /completion end to end. Split and placed last so the
+ # untrusted binary runs only in the final smoke step, after every HF_TOKEN step,
+ # leaving no token-bearing step or shared workspace for a tampered prebuilt to
+ # corrupt. GH_TOKEN: releases API; HF_TOKEN (withheld on PR): probe + GGUF fetch.
+ - name: Studio prebuilt llama.cpp install + GGUF download (Mac M1)
+ env:
+ GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
+ GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
+ HF_TOKEN: ${{ github.event_name != 'pull_request' && secrets.HF_TOKEN || '' }}
+ run: |
+ set -euo pipefail
+ INSTALL_DIR="$HOME/.unsloth-studio-prebuilt-test/llama.cpp"
+ rm -rf "$INSTALL_DIR"
+ # Download only -- no llama-quantize / llama-server launch in this step.
+ python studio/install_llama_prebuilt.py \
+ --install-dir "$INSTALL_DIR" \
+ --published-repo unslothai/llama.cpp
+ mkdir -p /tmp/ggufs
+ bash .github/scripts/hf-download-with-retry.sh \
+ 'unsloth/gemma-3-270m-it-GGUF' \
+ 'gemma-3-270m-it-Q4_K_M.gguf' \
+ /tmp/ggufs
+
+ # Final step: runs the downloaded binaries with no secrets present, and clears
+ # the GitHub Actions command files so a tampered prebuilt cannot influence the job.
+ - name: Studio prebuilt llama.cpp GGUF inference smoke (Mac M1)
+ run: |
+ set -euo pipefail
+ unset GITHUB_ENV GITHUB_PATH GITHUB_OUTPUT GITHUB_STEP_SUMMARY
+ INSTALL_DIR="$HOME/.unsloth-studio-prebuilt-test/llama.cpp"
+ # Studio bundles only llama-server + llama-quantize (not llama-cli);
+ # inference goes through llama-server's HTTP /completion endpoint.
+ LLAMA_SERVER="$INSTALL_DIR/build/bin/llama-server"
+ LLAMA_QUANT="$INSTALL_DIR/build/bin/llama-quantize"
+ [ -x "$LLAMA_SERVER" ] || { echo "::error::llama-server missing at $LLAMA_SERVER"; find "$INSTALL_DIR/build" -type f | head -40; exit 1; }
+ [ -x "$LLAMA_QUANT" ] || { echo "::error::llama-quantize missing at $LLAMA_QUANT"; exit 1; }
+ echo "llama-server : $LLAMA_SERVER"
+ echo "llama-quantize: $LLAMA_QUANT"
+ "$LLAMA_QUANT" --help >/dev/null && echo " llama-quantize loads OK"
+
+ PORT=18080
+ echo "=== starting llama-server on 127.0.0.1:$PORT ==="
+ "$LLAMA_SERVER" \
+ -m /tmp/ggufs/gemma-3-270m-it-Q4_K_M.gguf \
+ --host 127.0.0.1 \
+ --port "$PORT" \
+ -c 256 \
+ -n 16 \
+ --no-warmup \
+ > /tmp/llama-server.log 2>&1 &
+ SERVER_PID=$!
+ trap 'kill "$SERVER_PID" 2>/dev/null || true' EXIT
+
+ # Wait for /health to come up
+ for i in $(seq 1 30); do
+ if curl -sf "http://127.0.0.1:$PORT/health" >/dev/null 2>&1; then
+ echo " server up after ${i}s"
+ break
+ fi
+ sleep 1
+ done
+ if ! curl -sf "http://127.0.0.1:$PORT/health" >/dev/null 2>&1; then
+ echo "::error::llama-server never became healthy"
+ tail -40 /tmp/llama-server.log
+ exit 1
+ fi
+
+ PROMPT="Hello, my name is"
+ echo "=== POST /completion ==="
+ RESP=$(curl -sf -X POST "http://127.0.0.1:$PORT/completion" \
+ -H 'Content-Type: application/json' \
+ -d "{\"prompt\":\"$PROMPT\",\"n_predict\":16,\"temperature\":0,\"seed\":3407}")
+ echo "raw response (head): $(echo "$RESP" | head -c 600)"
+ CONTENT=$(echo "$RESP" | python -c "import json,sys; print(json.loads(sys.stdin.read()).get('content',''))")
+ echo "completion content: $CONTENT"
+
+ if [ -z "$CONTENT" ]; then
+ echo "::error::llama-server /completion returned empty content"
+ tail -40 /tmp/llama-server.log
+ exit 1
+ fi
+ echo "OK: Studio prebuilt llama.cpp on Mac M1 + GGUF /completion works"
diff --git a/.github/workflows/notebooks-ci.yml b/.github/workflows/notebooks-ci.yml
index 2edcae8ab2..0e0b35dd4d 100644
--- a/.github/workflows/notebooks-ci.yml
+++ b/.github/workflows/notebooks-ci.yml
@@ -263,7 +263,7 @@ jobs:
# unsloth_zoo.vision_utils imports PIL at module top, and the
# easiest way to get a torch-compatible PIL on a CPU runner is
# to let torchvision pull the right Pillow version.
- pip install --index-url https://download.pytorch.org/whl/cpu \
+ pip install --index-url https://download.pytorch.org/whl/cpu --extra-index-url https://pypi.org/simple \
'torch>=2.8,<2.11' 'torchvision<0.26'
# Pin to the same versions update_all_notebooks.py installs in
# generated notebooks. Keep these in lockstep with PIN_TRL /
diff --git a/.github/workflows/studio-backend-ci.yml b/.github/workflows/studio-backend-ci.yml
index ea60252cf6..bce355458a 100644
--- a/.github/workflows/studio-backend-ci.yml
+++ b/.github/workflows/studio-backend-ci.yml
@@ -76,7 +76,7 @@ jobs:
# Torch CPU + transformers are required by a chunk of the backend test
# suite (gpu_selection, kv_cache_estimation, utils). CPU-only torch
# keeps the install ~250 MB / ~1 min on a clean runner.
- pip install --index-url https://download.pytorch.org/whl/cpu 'torch>=2.4,<2.11'
+ pip install --index-url https://download.pytorch.org/whl/cpu --extra-index-url https://pypi.org/simple 'torch>=2.4,<2.11'
pip install 'transformers>=4.51,<5.5'
- name: Backend tests
@@ -137,7 +137,7 @@ jobs:
pyyaml jinja2 mammoth unpdf requests typer \
'numpy<3' pytest pytest-asyncio httpx
# torchvision: unsloth_zoo.vision_utils imports it at module scope.
- pip install --index-url https://download.pytorch.org/whl/cpu \
+ pip install --index-url https://download.pytorch.org/whl/cpu --extra-index-url https://pypi.org/simple \
'torch>=2.4,<2.11' 'torchvision<0.26'
pip install 'transformers>=4.51,<5.5'
# bitsandbytes: hard import in unsloth/models/_utils.py. Recent
diff --git a/.github/workflows/studio-mac-ui-smoke.yml b/.github/workflows/studio-mac-ui-smoke.yml
index 512af54d53..20ca247b9f 100644
--- a/.github/workflows/studio-mac-ui-smoke.yml
+++ b/.github/workflows/studio-mac-ui-smoke.yml
@@ -185,13 +185,14 @@ jobs:
# Retry up to 3 times to absorb known macos-14 free-runner
# flakes: (1) Playwright Node 24 pipeTransport.js 'Unexpected
# end of JSON input' crash when the Chromium browser process
- # dies mid-test, and (2) Chromium net::ERR_NO_BUFFER_SPACE
- # when the runner's kernel briefly runs out of socket buffers.
- # The retry FULLY resets Studio (kill, reset-password, reboot,
- # wait /api/health, re-export bootstrap pw) before re-running
- # the script. A real test failure (assertion / timeout) does
- # NOT match either pattern so it bypasses retry and surfaces
- # immediately.
+ # dies mid-test, (2) Chromium net::ERR_NO_BUFFER_SPACE when the
+ # runner's kernel briefly runs out of socket buffers, and (3) a
+ # goto 'interrupted by another navigation' when the SPA auth
+ # guard redirects mid-navigation. The retry FULLY resets Studio
+ # (kill, reset-password, reboot, wait /api/health, re-export
+ # bootstrap pw) before re-running the script. A real test failure
+ # (assertion / timeout) does NOT match any pattern so it bypasses
+ # retry and surfaces immediately.
run: |
mkdir -p logs/playwright
attempt=1
@@ -204,8 +205,9 @@ jobs:
if [ "$rc" -eq 0 ]; then
break
fi
- if { grep -q "Unexpected end of JSON input" logs/playwright_attempt_${attempt}.log \
- || grep -q "ERR_NO_BUFFER_SPACE" logs/playwright_attempt_${attempt}.log; } \
+ if { grep -q "Unexpected end of JSON input" logs/playwright_attempt_${attempt}.log \
+ || grep -q "ERR_NO_BUFFER_SPACE" logs/playwright_attempt_${attempt}.log \
+ || grep -q "interrupted by another navigation" logs/playwright_attempt_${attempt}.log; } \
&& [ "$attempt" -lt "$max_attempts" ]; then
echo "::warning::Playwright flake on attempt ${attempt}; resetting Studio and retrying..."
kill "${STUDIO_PID}" 2>/dev/null || true
@@ -280,8 +282,8 @@ jobs:
STUDIO_UI_TURN_TIMEOUT_MS: '540000'
GGUF_REPO: ${{ env.GGUF_REPO }}
GGUF_VARIANT: ${{ env.GGUF_VARIANT }}
- # Same flake-retry shape as "Drive the chat UI with Playwright"
- # -- catches pipeTransport JSON crash and ERR_NO_BUFFER_SPACE.
+ # Same flake-retry shape as "Drive the chat UI with Playwright" -- catches
+ # pipeTransport JSON crash, ERR_NO_BUFFER_SPACE, and nav interrupts.
run: |
mkdir -p logs/playwright_extra
attempt=1
@@ -294,8 +296,9 @@ jobs:
if [ "$rc" -eq 0 ]; then
break
fi
- if { grep -q "Unexpected end of JSON input" logs/playwright_extra_attempt_${attempt}.log \
- || grep -q "ERR_NO_BUFFER_SPACE" logs/playwright_extra_attempt_${attempt}.log; } \
+ if { grep -q "Unexpected end of JSON input" logs/playwright_extra_attempt_${attempt}.log \
+ || grep -q "ERR_NO_BUFFER_SPACE" logs/playwright_extra_attempt_${attempt}.log \
+ || grep -q "interrupted by another navigation" logs/playwright_extra_attempt_${attempt}.log; } \
&& [ "$attempt" -lt "$max_attempts" ]; then
echo "::warning::Playwright flake on attempt ${attempt}; resetting Studio and retrying..."
kill "${STUDIO_EXTRA_PID}" 2>/dev/null || true
diff --git a/.github/workflows/studio-windows-inference-smoke.yml b/.github/workflows/studio-windows-inference-smoke.yml
index c44c68278d..8186c07211 100644
--- a/.github/workflows/studio-windows-inference-smoke.yml
+++ b/.github/workflows/studio-windows-inference-smoke.yml
@@ -1338,11 +1338,19 @@ jobs:
shell: pwsh
run: |
$ErrorActionPreference = 'Stop'
+ # A Program Files dir can hold a transient handle (Defender / MSBuild node)
+ # so Rename-Item intermittently fails with "Access is denied"; retry to ride it out.
+ function Rename-WithRetry($Path, $NewName) {
+ for ($i = 1; $i -le 6; $i++) {
+ try { Rename-Item -LiteralPath $Path -NewName $NewName -ErrorAction Stop; return }
+ catch { if ($i -eq 6) { throw }; Start-Sleep -Seconds 3 }
+ }
+ }
# Rename the Visual Studio install roots (incl. the Installer that holds
# vswhere.exe) so Find-VsBuildTools' vswhere + filesystem scan both miss.
foreach ($d in @("$env:ProgramFiles\Microsoft Visual Studio", "${env:ProgramFiles(x86)}\Microsoft Visual Studio")) {
if (Test-Path -LiteralPath $d) {
- Rename-Item -LiteralPath $d -NewName ((Split-Path $d -Leaf) + '.vsoff')
+ Rename-WithRetry $d ((Split-Path $d -Leaf) + '.vsoff')
Write-Host "Hid VS: $d"
}
}
@@ -1351,7 +1359,7 @@ jobs:
$hidden = @()
foreach ($c in (Get-Command cmake -All -ErrorAction SilentlyContinue)) {
if ($c.Source -and (Test-Path -LiteralPath $c.Source)) {
- Rename-Item -LiteralPath $c.Source -NewName ((Split-Path $c.Source -Leaf) + '.off')
+ Rename-WithRetry $c.Source ((Split-Path $c.Source -Leaf) + '.off')
$hidden += $c.Source
Write-Host "Hid cmake: $($c.Source)"
}
@@ -1376,7 +1384,7 @@ jobs:
- name: PyTorch CPU wheel installs and imports (no Visual Studio)
run: |
python -m pip install --upgrade pip
- python -m pip install torch --index-url https://download.pytorch.org/whl/cpu
+ python -m pip install torch --index-url https://download.pytorch.org/whl/cpu --extra-index-url https://pypi.org/simple
python -c "import torch; print('torch', torch.__version__, 'cuda?', torch.cuda.is_available())"
- name: Install Studio (--local, --no-torch) with no build tools present
@@ -1536,8 +1544,16 @@ jobs:
shell: pwsh
run: |
$ErrorActionPreference = 'Stop'
+ # Retry the rename: a Program Files dir can hold a transient handle that
+ # makes Rename-Item intermittently fail with "Access is denied".
+ function Rename-WithRetry($Path, $NewName) {
+ for ($i = 1; $i -le 6; $i++) {
+ try { Rename-Item -LiteralPath $Path -NewName $NewName -ErrorAction Stop; return }
+ catch { if ($i -eq 6) { throw }; Start-Sleep -Seconds 3 }
+ }
+ }
foreach ($d in @("$env:ProgramFiles\Microsoft Visual Studio", "${env:ProgramFiles(x86)}\Microsoft Visual Studio")) {
- if (Test-Path -LiteralPath $d) { Rename-Item -LiteralPath $d -NewName ((Split-Path $d -Leaf) + '.vsoff'); Write-Host "Hid VS: $d" }
+ if (Test-Path -LiteralPath $d) { Rename-WithRetry $d ((Split-Path $d -Leaf) + '.vsoff'); Write-Host "Hid VS: $d" }
}
- name: Windows CUDA and ROCm prebuilts exist in unslothai/llama.cpp (what GPU users download, no VS)
diff --git a/.github/workflows/version-compat-ci.yml b/.github/workflows/version-compat-ci.yml
index 599b53df1d..e492d21e99 100644
--- a/.github/workflows/version-compat-ci.yml
+++ b/.github/workflows/version-compat-ci.yml
@@ -242,7 +242,7 @@ jobs:
run: |
python -m pip install --upgrade pip
# CPU torch (vllm/peft/st all depend on it).
- pip install --index-url https://download.pytorch.org/whl/cpu \
+ pip install --index-url https://download.pytorch.org/whl/cpu --extra-index-url https://pypi.org/simple \
'torch>=2.4,<2.11' 'torchvision<0.26' 'torchcodec<0.10'
# torchcodec is a hard requirement on transformers 5.x:
# transformers/audio_utils.py:55 does
diff --git a/README.md b/README.md
index 9162d29b1c..e3fd4e6980 100644
--- a/README.md
+++ b/README.md
@@ -246,6 +246,11 @@ curl -fsSL https://unsloth.ai/install.sh | UNSLOTH_STUDIO_HOME=/abs/path sh
$env:UNSLOTH_STUDIO_HOME='C:\path'; irm https://unsloth.ai/install.ps1 | iex
```
+On macOS, the installer defaults to the system certificate store (`UV_SYSTEM_CERTS=1`) so uv trusts the CAs in your Keychain, needed behind TLS-inspecting proxies (Cisco Umbrella, Zscaler, etc.). Opt out with:
+```bash
+curl -fsSL https://unsloth.ai/install.sh | UV_SYSTEM_CERTS=0 sh
+```
+
Point the frontend build at a corporate npm mirror/proxy with `UNSLOTH_NPM_REGISTRY` (for the developer install behind a firewall that blocks `registry.npmjs.org`):
```bash
UNSLOTH_NPM_REGISTRY=https://artifactory.example.com/api/npm/npm/ ./install.sh --local
diff --git a/install.sh b/install.sh
index 548e6f702a..7a9f0be87f 100755
--- a/install.sh
+++ b/install.sh
@@ -1636,6 +1636,21 @@ export UV_HTTP_RETRIES
: "${UV_HTTP_TIMEOUT:=180}"
export UV_HTTP_TIMEOUT
+# macOS: trust the system Keychain so uv uses SecureTransport instead of rustls.
+# Required behind TLS-inspecting proxies (Cisco Umbrella, Zscaler, etc.) which
+# present their own CA certificate. rustls (uv's default) ignores the Keychain
+# and rejects intercepted connections with "invalid peer certificate: UnknownIssuer".
+# Set both vars: UV_SYSTEM_CERTS is the modern one (uv >= 0.11), UV_NATIVE_TLS the
+# legacy one understood by uv 0.8.16-0.10.x, which the installer keeps if already
+# present (UV_MIN_VERSION) and which ignores UV_SYSTEM_CERTS. Mirror the choice onto
+# both so it works on either uv. Opt out with UV_SYSTEM_CERTS=0.
+if [ "$OS" = "macos" ]; then
+ : "${UV_SYSTEM_CERTS:=1}"
+ : "${UV_NATIVE_TLS:=$UV_SYSTEM_CERTS}"
+fi
+[ -n "${UV_SYSTEM_CERTS:-}" ] && export UV_SYSTEM_CERTS
+[ -n "${UV_NATIVE_TLS:-}" ] && export UV_NATIVE_TLS
+
version_ge() {
# returns 0 if $1 >= $2
_a=$1
diff --git a/pyproject.toml b/pyproject.toml
index 13c421d8ea..844ead2454 100644
--- a/pyproject.toml
+++ b/pyproject.toml
@@ -255,10 +255,6 @@ cu118onlytorch270 = [
"xformers @ https://download.pytorch.org/whl/cu118/xformers-0.0.30-cp310-cp310-manylinux_2_28_x86_64.whl ; python_version=='3.10' and ('linux' in sys_platform)",
"xformers @ https://download.pytorch.org/whl/cu118/xformers-0.0.30-cp311-cp311-manylinux_2_28_x86_64.whl ; python_version=='3.11' and ('linux' in sys_platform)",
"xformers @ https://download.pytorch.org/whl/cu118/xformers-0.0.30-cp312-cp312-manylinux_2_28_x86_64.whl ; python_version=='3.12' and ('linux' in sys_platform)",
- "xformers @ https://download.pytorch.org/whl/cu118/xformers-0.0.30-cp39-cp39-win_amd64.whl ; python_version=='3.9' and (sys_platform == 'win32')",
- "xformers @ https://download.pytorch.org/whl/cu118/xformers-0.0.30-cp310-cp310-win_amd64.whl ; python_version=='3.10' and (sys_platform == 'win32')",
- "xformers @ https://download.pytorch.org/whl/cu118/xformers-0.0.30-cp311-cp311-win_amd64.whl ; python_version=='3.11' and (sys_platform == 'win32')",
- "xformers @ https://download.pytorch.org/whl/cu118/xformers-0.0.30-cp312-cp312-win_amd64.whl ; python_version=='3.12' and (sys_platform == 'win32')",
]
cu126onlytorch270 = [
"xformers @ https://download.pytorch.org/whl/cu126/xformers-0.0.30-cp39-cp39-manylinux_2_28_x86_64.whl ; python_version=='3.9' and ('linux' in sys_platform)",
@@ -282,7 +278,6 @@ cu128onlytorch270 = [
]
cu118onlytorch271 = [
"xformers @ https://download.pytorch.org/whl/cu118/xformers-0.0.31.post1-cp39-abi3-manylinux_2_28_x86_64.whl ; ('linux' in sys_platform)",
- "xformers @ https://download.pytorch.org/whl/cu118/xformers-0.0.31.post1-cp39-abi3-win_amd64.whl ; (sys_platform == 'win32')",
]
cu126onlytorch271 = [
"xformers @ https://download.pytorch.org/whl/cu126/xformers-0.0.31.post1-cp39-abi3-manylinux_2_28_x86_64.whl ; ('linux' in sys_platform)",
@@ -879,14 +874,12 @@ flashattentiontorch240abiFALSEcu12x = [
"flash-attn @ https://github.com/Dao-AILab/flash-attention/releases/download/v2.7.4.post1/flash_attn-2.7.4.post1+cu12torch2.4cxx11abiFALSE-cp310-cp310-linux_x86_64.whl ; ('linux' in sys_platform) and python_version == '3.10'",
"flash-attn @ https://github.com/Dao-AILab/flash-attention/releases/download/v2.7.4.post1/flash_attn-2.7.4.post1+cu12torch2.4cxx11abiFALSE-cp311-cp311-linux_x86_64.whl ; ('linux' in sys_platform) and python_version == '3.11'",
"flash-attn @ https://github.com/Dao-AILab/flash-attention/releases/download/v2.7.4.post1/flash_attn-2.7.4.post1+cu12torch2.4cxx11abiFALSE-cp312-cp312-linux_x86_64.whl ; ('linux' in sys_platform) and python_version == '3.12'",
- "flash-attn @ https://github.com/Dao-AILab/flash-attention/releases/download/v2.7.4.post1/flash_attn-2.7.4.post1+cu12torch2.4cxx11abiFALSE-cp313-cp313-linux_x86_64.whl ; ('linux' in sys_platform) and python_version == '3.13'",
]
flashattentiontorch240abiTRUEcu12x = [
"flash-attn @ https://github.com/Dao-AILab/flash-attention/releases/download/v2.7.4.post1/flash_attn-2.7.4.post1+cu12torch2.4cxx11abiTRUE-cp39-cp39-linux_x86_64.whl ; ('linux' in sys_platform) and python_version == '3.9'",
"flash-attn @ https://github.com/Dao-AILab/flash-attention/releases/download/v2.7.4.post1/flash_attn-2.7.4.post1+cu12torch2.4cxx11abiTRUE-cp310-cp310-linux_x86_64.whl ; ('linux' in sys_platform) and python_version == '3.10'",
"flash-attn @ https://github.com/Dao-AILab/flash-attention/releases/download/v2.7.4.post1/flash_attn-2.7.4.post1+cu12torch2.4cxx11abiTRUE-cp311-cp311-linux_x86_64.whl ; ('linux' in sys_platform) and python_version == '3.11'",
"flash-attn @ https://github.com/Dao-AILab/flash-attention/releases/download/v2.7.4.post1/flash_attn-2.7.4.post1+cu12torch2.4cxx11abiTRUE-cp312-cp312-linux_x86_64.whl ; ('linux' in sys_platform) and python_version == '3.12'",
- "flash-attn @ https://github.com/Dao-AILab/flash-attention/releases/download/v2.7.4.post1/flash_attn-2.7.4.post1+cu12torch2.4cxx11abiTRUE-cp313-cp313-linux_x86_64.whl ; ('linux' in sys_platform) and python_version == '3.13'",
]
intelgputorch260 = [
"unsloth_zoo[intelgpu]",
@@ -1174,14 +1167,14 @@ intelgputorch2120 = [
"unsloth_zoo[intelgpu]",
"unsloth[huggingfacenotorch]",
- "triton-xpu @ https://download.pytorch.org/whl/triton_xpu-3.7.1-cp310-cp310-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl#sha256=844d981cb1b3948085e8cfa62c74de9f100259f6131959aa70be49123b88ae81 ; platform_system == 'Linux' and python_version == '3.10' and platform_machine == 'x86_64'",
- "triton-xpu @ https://download.pytorch.org/whl/triton_xpu-3.7.1-cp311-cp311-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl#sha256=a16b1d00e94ad87d62af3512e390348b8656419598004100c56028bf494f086b ; platform_system == 'Linux' and python_version == '3.11' and platform_machine == 'x86_64'",
- "triton-xpu @ https://download.pytorch.org/whl/triton_xpu-3.7.1-cp312-cp312-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl#sha256=4e46e71e077cf483404a4c17ce40d71c5f0e13a81459139d4346ca427b1dd455 ; platform_system == 'Linux' and python_version == '3.12' and platform_machine == 'x86_64'",
- "triton-xpu @ https://download.pytorch.org/whl/triton_xpu-3.7.1-cp313-cp313-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl#sha256=4fdaed1bafc51d3a2834656a3420a6686a74ea226508765a49bf15d58ff3a930 ; platform_system == 'Linux' and python_version == '3.13' and platform_machine == 'x86_64'",
- "triton-xpu @ https://download.pytorch.org/whl/triton_xpu-3.7.1-cp310-cp310-win_amd64.whl#sha256=2778b46b22e9fa0916398db299a125027a1b2331c1173b3dd2b9e2cab6263a31 ; sys_platform == 'win32' and python_version == '3.10' and (platform_machine == 'AMD64' or platform_machine == 'x86_64')",
- "triton-xpu @ https://download.pytorch.org/whl/triton_xpu-3.7.1-cp311-cp311-win_amd64.whl#sha256=ad5b147d04ee0d40f3d4d32f85f5aa3a3beb6cd5799ca026d3d7f4afa3d9e24f ; sys_platform == 'win32' and python_version == '3.11' and (platform_machine == 'AMD64' or platform_machine == 'x86_64')",
- "triton-xpu @ https://download.pytorch.org/whl/triton_xpu-3.7.1-cp312-cp312-win_amd64.whl#sha256=d9482063af2a308543f23333e32edd738ea87cbb33ade68afda9ae0fd704ccd9 ; sys_platform == 'win32' and python_version == '3.12' and (platform_machine == 'AMD64' or platform_machine == 'x86_64')",
- "triton-xpu @ https://download.pytorch.org/whl/triton_xpu-3.7.1-cp313-cp313-win_amd64.whl#sha256=5d4d67f0deb1e851c01b293e602b8dcddad26ca2be61221cee3dc0e1aa0cdefd ; sys_platform == 'win32' and python_version == '3.13' and (platform_machine == 'AMD64' or platform_machine == 'x86_64')",
+ "triton-xpu @ https://download.pytorch.org/whl/triton_xpu-3.7.1-cp310-cp310-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl#sha256=81ff0eb0c4fc8e19d2510b28c3e1d9382a3c7d6fdaf6a9f9631a93a030d841cf ; platform_system == 'Linux' and python_version == '3.10' and platform_machine == 'x86_64'",
+ "triton-xpu @ https://download.pytorch.org/whl/triton_xpu-3.7.1-cp311-cp311-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl#sha256=55574a68d275b85cd4d5cbf185084bae019ebf09c3f43b0bd2831b14935ec8e7 ; platform_system == 'Linux' and python_version == '3.11' and platform_machine == 'x86_64'",
+ "triton-xpu @ https://download.pytorch.org/whl/triton_xpu-3.7.1-cp312-cp312-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl#sha256=a31c058c5c2e78ebe490a2e69f2f50caec6b1307ac096e944f116fdc06819d9a ; platform_system == 'Linux' and python_version == '3.12' and platform_machine == 'x86_64'",
+ "triton-xpu @ https://download.pytorch.org/whl/triton_xpu-3.7.1-cp313-cp313-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl#sha256=e701a31efa0334775f357c98716f3821775aa944219f7888e13c2dfe2daabe2a ; platform_system == 'Linux' and python_version == '3.13' and platform_machine == 'x86_64'",
+ "triton-xpu @ https://download.pytorch.org/whl/triton_xpu-3.7.1-cp310-cp310-win_amd64.whl#sha256=0d7730651c3e52fbf3a430cc201455f0c6600dc72e681aec495f131ea44f341a ; sys_platform == 'win32' and python_version == '3.10' and (platform_machine == 'AMD64' or platform_machine == 'x86_64')",
+ "triton-xpu @ https://download.pytorch.org/whl/triton_xpu-3.7.1-cp311-cp311-win_amd64.whl#sha256=8f4a63de73e3d632098f93c8f0bd77244958a47d7c5f728b8ff35f8a91fdb983 ; sys_platform == 'win32' and python_version == '3.11' and (platform_machine == 'AMD64' or platform_machine == 'x86_64')",
+ "triton-xpu @ https://download.pytorch.org/whl/triton_xpu-3.7.1-cp312-cp312-win_amd64.whl#sha256=6589ece3adc2b1ab88d90ff1267afc25df5c7b868f0b633e732cac70df36cbde ; sys_platform == 'win32' and python_version == '3.12' and (platform_machine == 'AMD64' or platform_machine == 'x86_64')",
+ "triton-xpu @ https://download.pytorch.org/whl/triton_xpu-3.7.1-cp313-cp313-win_amd64.whl#sha256=2fdf001a9b0575e8b1827127259bb9b13bf36e659882be74c2dfab46597d3e7a ; sys_platform == 'win32' and python_version == '3.13' and (platform_machine == 'AMD64' or platform_machine == 'x86_64')",
"torch @ https://download.pytorch.org/whl/xpu/torch-2.12.0%2Bxpu-cp310-cp310-linux_x86_64.whl#sha256=e8923cd1fe560472904b1461b745d2f1826bb9c1bc0808225d5f28a450e4d553 ; platform_system == 'Linux' and python_version == '3.10' and platform_machine == 'x86_64'",
"torch @ https://download.pytorch.org/whl/xpu/torch-2.12.0%2Bxpu-cp311-cp311-linux_x86_64.whl#sha256=f7c082b2fc9b61def594d30ea57762dc4a8bc7111a9a9593953ed948de242e28 ; platform_system == 'Linux' and python_version == '3.11' and platform_machine == 'x86_64'",
diff --git a/studio/backend/core/inference/llama_cpp.py b/studio/backend/core/inference/llama_cpp.py
index 255fc4140f..8ec11fa79f 100644
--- a/studio/backend/core/inference/llama_cpp.py
+++ b/studio/backend/core/inference/llama_cpp.py
@@ -1271,6 +1271,9 @@ class LlamaCppBackend:
self._cache_type_kv: Optional[str] = None
# Whether --split-mode tensor was applied on the active load.
self._tensor_parallel: bool = False
+ # Layer load kept multi-GPU only to honor a downgraded tensor request, so a
+ # later explicit tensor-off reloads instead of deduping to it (#6659).
+ self._layer_preserves_tensor_intent: bool = False
self._reasoning_default: bool = True
self._speculative_type: Optional[str] = None
# Canonical UI-facing mode the user requested
@@ -1643,6 +1646,11 @@ class LlamaCppBackend:
"""Whether --split-mode tensor is active on the loaded server."""
return self._tensor_parallel
+ @property
+ def layer_preserves_tensor_intent(self) -> bool:
+ """True when a downgraded tensor request kept this layer load multi-GPU."""
+ return self._layer_preserves_tensor_intent
+
@property
def speculative_type(self) -> Optional[str]:
return self._speculative_type
@@ -2430,6 +2438,37 @@ class LlamaCppBackend:
# aborts a --split-mode tensor load, so it's dropped for the tensor attempt.
_TENSOR_PARALLEL_KV_TYPES = frozenset({"f16", "bf16", "f32"})
+ # (binary, mtime, model) that aborted on --split-mode tensor this process (#6415
+ # geometry limit, e.g. MQA n_head_kv=1). Model-keyed so one model's abort doesn't
+ # skip tensor for others; tensor is tried by default, recorded only on a real abort.
+ _tensor_split_abort_keys: set[tuple[str, int, str]] = set()
+
+ @classmethod
+ def _tensor_split_cache_key(
+ cls, binary: Optional[str], model: Optional[str]
+ ) -> Optional[tuple[str, int, str]]:
+ """(path, mtime_ns, model) key; ns mtime re-probes a same-second binary swap."""
+ if not binary or not model:
+ return None
+ try:
+ mtime = Path(binary).stat().st_mtime_ns
+ except OSError:
+ mtime = 0
+ return (binary, mtime, model)
+
+ @classmethod
+ def _tensor_split_aborts(cls, binary: Optional[str], model: Optional[str]) -> bool:
+ """True if (binary, model) aborted on --split-mode tensor this session."""
+ key = cls._tensor_split_cache_key(binary, model)
+ return key is not None and key in cls._tensor_split_abort_keys
+
+ @classmethod
+ def _record_tensor_split_abort(cls, binary: Optional[str], model: Optional[str]) -> None:
+ """Remember a (binary, model) that aborts on --split-mode tensor."""
+ key = cls._tensor_split_cache_key(binary, model)
+ if key is not None:
+ cls._tensor_split_abort_keys.add(key)
+
@staticmethod
def _windows_pip_nvidia_dll_dirs(prefix: str) -> list[str]:
"""Return DLL dirs from pip-installed CUDA wheels under
@@ -2569,9 +2608,13 @@ class LlamaCppBackend:
usable_fraction: Optional[float] = None,
total_by_idx: Optional[dict[int, int]] = None,
per_device_overhead_bytes: int = 0,
+ min_gpus: int = 1,
) -> tuple[Optional[list[int]], bool]:
"""Pick GPU(s) for a model from estimated VRAM and free memory.
+ ``min_gpus`` (default 1, capped at ``len(gpus)``) keeps a downgraded
+ tensor/multi-GPU request spread instead of collapsing to one card.
+
``model_size_bytes`` should include weights and estimated KV cache.
``usable_fraction`` (default ``_GPU_PIN_VRAM_FRACTION``) provides
headroom for compute buffers, CUDA context, and other runtime
@@ -2590,9 +2633,11 @@ class LlamaCppBackend:
if not gpus:
return None, True
+ min_gpus = max(1, min(min_gpus, len(gpus)))
model_size_mib = model_size_bytes / (1024 * 1024)
if usable_fraction is None:
usable_fraction = LlamaCppBackend._GPU_PIN_VRAM_FRACTION
+ overhead_mib = per_device_overhead_bytes / (1024 * 1024)
# Per-GPU usable budget: free - (1-frac)*total when total is known, else
# the legacy free*frac (also covers a total-0 two-column probe).
@@ -2606,19 +2651,26 @@ class LlamaCppBackend:
# card can have less usable room than a less-used small one.
ranked = sorted(gpus, key = lambda g: _usable(g[0], g[1]), reverse = True)
- # Try 1 GPU at the usable-VRAM threshold.
- if _usable(ranked[0][0], ranked[0][1]) >= model_size_mib:
+ # Cap a downgraded multi-GPU request to the usable count so it doesn't pull
+ # in a near-full card to hit min_gpus. No-op for the default min_gpus == 1.
+ usable_count = sum(1 for idx, free_mib in ranked if _usable(idx, free_mib) > overhead_mib)
+ min_gpus = max(1, min(min_gpus, usable_count or 1))
+
+ # Try 1 GPU at the usable-VRAM threshold (only when one device is allowed).
+ if min_gpus <= 1 and _usable(ranked[0][0], ranked[0][1]) >= model_size_mib:
return [ranked[0][0]], False
- # Try N GPUs (accumulate usable memory from most-free). Each GPU past the
- # first adds a fixed per-device overhead the pool must hold.
- overhead_mib = per_device_overhead_bytes / (1024 * 1024)
+ # Try N GPUs (most-free first); each past the first adds per-device overhead.
+ # Require at least min_gpus devices before accepting a fit.
cumulative = 0.0
selected = []
for idx, free_mib in ranked:
selected.append(idx)
cumulative += _usable(idx, free_mib)
- if cumulative >= model_size_mib + (len(selected) - 1) * overhead_mib:
+ if (
+ len(selected) >= min_gpus
+ and cumulative >= model_size_mib + (len(selected) - 1) * overhead_mib
+ ):
return sorted(selected), False
# Too large even for all GPUs; let --fit handle it
@@ -3868,7 +3920,7 @@ class LlamaCppBackend:
logger.debug(f"Could not list repo files for {label}: {e}")
break
logger.debug(
- f"Could not list repo files for {label} " f"(attempt {attempt + 1}/3): {e}"
+ f"Could not list repo files for {label} (attempt {attempt + 1}/3): {e}"
)
if attempt < 2:
self._cancel_event.wait(2**attempt)
@@ -4332,6 +4384,17 @@ class LlamaCppBackend:
)
)
+ @staticmethod
+ def _is_tensor_split_assert(output: str) -> bool:
+ """True only for the #6415 split-axis warmup assert (GGML_BACKEND_SPLIT_AXIS_*),
+ not any ggml assert/abort, so an unrelated invariant isn't cached. stderr is
+ merged into output."""
+ text = (output or "").lower()
+ if "ggml_assert" not in text and "ggml_abort" not in text:
+ return False
+ # the split-axis enum token, unique to this assert (not the source file).
+ return "split_axis" in text
+
@staticmethod
def _is_signal_crash(returncode: Optional[int]) -> bool:
"""True only on a hard fault (SIGSEGV/SIGABRT/SIGILL/SIGFPE/SIGBUS or a
@@ -4344,6 +4407,20 @@ class LlamaCppBackend:
return True
return -returncode in (4, 6, 7, 8, 11) # SIGILL SIGABRT SIGBUS SIGFPE SIGSEGV
+ @staticmethod
+ def _is_abort_exit(returncode: Optional[int]) -> bool:
+ """Windows CRT abort() exit code (3) from GGML_ASSERT on MSVC -- not a POSIX
+ signal or 0xC0000000+ NTSTATUS."""
+ return returncode == 3
+
+ @classmethod
+ def _should_record_tensor_split_abort(cls, returncode: Optional[int], output: str) -> bool:
+ """The #6415 split-axis abort: the marker plus a hard crash (POSIX signal or
+ Windows abort exit). Marker required so a generic crash isn't cached."""
+ return cls._is_tensor_split_assert(output) and (
+ cls._is_signal_crash(returncode) or cls._is_abort_exit(returncode)
+ )
+
@staticmethod
def _with_flash_attn_off(cmd: list[str]) -> Optional[list[str]]:
"""Return cmd with flash attention forced off, or None when its effective
@@ -4488,6 +4565,8 @@ class LlamaCppBackend:
n_gpu_layers: Optional[int] = None, # caller compat, unused
n_parallel: int = 1,
extra_args: Optional[List[str]] = None,
+ # Route-level tensor->layer fallback retry: keep the layer split multi-GPU.
+ preserve_multi_gpu_on_layer: bool = False,
) -> bool:
"""Start llama-server with a GGUF model.
@@ -4518,6 +4597,8 @@ class LlamaCppBackend:
"n_gpu_layers": n_gpu_layers,
"n_parallel": n_parallel,
"extra_args": list(extra_args) if extra_args is not None else None,
+ # Replayed by _respawn_if_dead so a downgraded model stays multi-GPU.
+ "preserve_multi_gpu_on_layer": preserve_multi_gpu_on_layer,
}
# Serialise the whole load so concurrent /load calls never leave two
# llama-server processes alive (#5401 / #5161). Doesn't block /unload.
@@ -4541,6 +4622,7 @@ class LlamaCppBackend:
chat_template_override = chat_template_override,
extra_args = extra_args,
is_vision = is_vision,
+ preserve_multi_gpu_on_layer = preserve_multi_gpu_on_layer,
):
logger.info(
f"load_model: backend already in target state for "
@@ -4626,6 +4708,9 @@ class LlamaCppBackend:
# Block-diffusion GGUFs (DiffusionGemma) cannot run on llama-server;
# serve them with the diffusion runner (same OpenAI-compat interface).
if self._is_diffusion:
+ # Not a tensor/layer GGUF: clear any preserved-fallback flag from a
+ # prior load (this path skips the command builder that clears it).
+ self._layer_preserves_tensor_intent = False
with self._lock:
if self._cancel_event.is_set():
logger.info("Load cancelled before diffusion server start")
@@ -4780,6 +4865,9 @@ class LlamaCppBackend:
"image input will be disabled for this session"
)
model_size = None # set in the fit try; used by the APU RAM guard
+ # Layer-fallback min GPUs; raised below on a tensor downgrade. Bound
+ # before the try so the --fit-on except path still has it (no UnboundLocal).
+ _layer_min_gpus = 1
try:
gguf_size = self._get_gguf_size_bytes(model_path)
# Include GPU-loaded mmproj in the fit budget (#5825).
@@ -5064,10 +5152,8 @@ class LlamaCppBackend:
_apple_budget_mib = self._apple_metal_memory_budget_bytes() // (1024 * 1024)
def _restore_after_tensor_downgrade():
- # Tensor mode dropped a quantized KV and stripped the cache
- # extras (it rejects quantized); layer split supports them, so
- # restore the original type + extras (minus --split-mode) and
- # clear the env flag so the layer launch re-emits them.
+ # Restore the quantized KV + extras tensor dropped (layer
+ # split supports them), minus --split-mode.
nonlocal cache_type_kv, _cache_type_from_env, extra_args
if _tensor_dropped_cache_type_kv is not None:
cache_type_kv = _tensor_dropped_cache_type_kv
@@ -5078,13 +5164,22 @@ class LlamaCppBackend:
else extra_args
)
- if tensor_parallel and effective_is_vision:
+ # The route fallback retry is tensor-off; keep it multi-GPU.
+ if preserve_multi_gpu_on_layer:
+ _layer_min_gpus = max(_layer_min_gpus, len(gpus))
+
+ if tensor_parallel and self._tensor_split_aborts(binary, model_identifier):
+ # Aborted on tensor for this model this session (#6415); skip
+ # tensor upfront, layer split serves it.
logger.info(
- "Tensor parallelism skipped for vision model: "
- "--split-mode tensor is incompatible with --mmproj "
- "in the current llama.cpp build; using layer split."
+ "Tensor parallelism skipped: this llama.cpp build aborted "
+ "on --split-mode tensor for this model earlier this "
+ "session; using layer split across %d GPU(s).",
+ len(gpus),
)
tensor_parallel = False
+ # Keep the multi-GPU request (gated on it, not the cache).
+ _layer_min_gpus = max(_layer_min_gpus, len(gpus))
_restore_after_tensor_downgrade()
# Tensor mode replicates a compute buffer on every GPU, so drop
@@ -5124,6 +5219,11 @@ class LlamaCppBackend:
len(gpus),
)
tensor_parallel = False
+ # GPUs below tensor's compute-buffer reserve can still do layer
+ # split, so keep multi-GPU (mirrors the budget/geometry drops);
+ # _select_gpus caps unusable cards.
+ if len(gpus) >= 2:
+ _layer_min_gpus = max(_layer_min_gpus, len(gpus))
# Layer split supports a quantized KV the tensor attempt
# dropped; restore the original cache type + extras (minus
# --split-mode) so the layer launch re-emits them.
@@ -5160,8 +5260,12 @@ class LlamaCppBackend:
"per-device compute buffers; falling back to layer split."
)
tensor_parallel = False
- # Restore the dropped quantized KV + original cache extras
- # (minus --split-mode); layer split supports them.
+ # Weights needed >1 card, so keep multi-GPU across the
+ # usable tensor GPUs.
+ if len(tp_gpus) >= 2:
+ _layer_min_gpus = max(_layer_min_gpus, len(tp_gpus))
+ # Restore the dropped quantized KV + cache extras (minus
+ # --split-mode); layer split supports them.
_restore_after_tensor_downgrade()
if tensor_parallel and tp_gpus:
@@ -5263,6 +5367,7 @@ class LlamaCppBackend:
usable_fraction = _pin_fraction,
total_by_idx = total_by_idx,
per_device_overhead_bytes = _pipeline_overhead_bytes,
+ min_gpus = _layer_min_gpus,
)
# No silent shrink: effective_ctx stays == requested_ctx.
else:
@@ -5273,7 +5378,22 @@ class LlamaCppBackend:
ranked = sorted(
gpus, key = lambda g: _gpu_usable(g, pin_fraction), reverse = True
)
- for n_gpus in range(1, len(ranked) + 1):
+ # Skips _select_gpus, so apply its cap: count only cards
+ # whose usable VRAM clears the per-device layer overhead.
+ _pipeline_overhead_mib = _pipeline_overhead_bytes / (1024 * 1024)
+ _auto_min_gpus = max(
+ 1,
+ min(
+ _layer_min_gpus,
+ sum(
+ 1
+ for g in ranked
+ if _gpu_usable(g, pin_fraction) > _pipeline_overhead_mib
+ )
+ or 1,
+ ),
+ )
+ for n_gpus in range(_auto_min_gpus, len(ranked) + 1):
subset = ranked[:n_gpus]
pool_budget = _pool_budget_mib(subset, pin_fraction)
_ms = _subset_model_size(n_gpus)
@@ -5303,7 +5423,7 @@ class LlamaCppBackend:
# at 131k may pin fine with a 4096 KV (#5106).
effective_ctx = min(4096, effective_ctx)
if effective_ctx > 0:
- for n_gpus in range(1, len(ranked) + 1):
+ for n_gpus in range(_auto_min_gpus, len(ranked) + 1):
subset = ranked[:n_gpus]
kv = self._estimate_kv_cache_bytes(
effective_ctx,
@@ -5339,6 +5459,7 @@ class LlamaCppBackend:
usable_fraction = _pin_fraction,
total_by_idx = total_by_idx,
per_device_overhead_bytes = _pipeline_overhead_bytes,
+ min_gpus = _layer_min_gpus,
)
if use_fit and not explicit_ctx:
# Weights don't fit on any subset; default UI to 4096
@@ -5476,6 +5597,15 @@ class LlamaCppBackend:
"--no-context-shift",
]
+ # Report a clean public model id (matching GET /v1/models) rather
+ # than the raw -m path in llama-server's own /v1/models and the
+ # "model" field of its chat/completions responses.
+ from core.inference.model_ids import public_model_id
+
+ _alias = public_model_id(self._model_identifier or model_path)
+ if _alias:
+ cmd.extend(["--alias", _alias])
+
fully_gpu_offloaded = False
if use_fit:
cmd.extend(["--fit", "on"])
@@ -5569,12 +5699,15 @@ class LlamaCppBackend:
]
)
self._tensor_parallel = True
+ self._layer_preserves_tensor_intent = False
logger.info(
"Tensor parallelism: --split-mode tensor, --tensor-split %s",
tp_tensor_split,
)
else:
self._tensor_parallel = False
+ # > 1 only when a tensor request was downgraded but kept multi-GPU.
+ self._layer_preserves_tensor_intent = _layer_min_gpus > 1
# Speculative decoding. See _build_speculative_flags for the
# mode resolution, benchmarks, and llama.cpp references.
@@ -5858,7 +5991,17 @@ class LlamaCppBackend:
_startup_crashed = (
self._process.poll() is not None and self._process.returncode != 0
)
- if _spawn_attempt == 0 and _fit_retry_allowed and _startup_crashed:
+ # A split-axis abort (#6415) is fit-independent: skip the
+ # --fit off retry and let the caller latch it.
+ _split_axis_crash = self._is_tensor_split_assert(
+ "\n".join(self._stdout_lines[-50:])
+ )
+ if (
+ _spawn_attempt == 0
+ and _fit_retry_allowed
+ and _startup_crashed
+ and not _split_axis_crash
+ ):
logger.warning(
"llama-server crashed during startup (exit code %s) "
"with the default memory-fit step enabled; Studio "
@@ -5904,6 +6047,21 @@ class LlamaCppBackend:
)
healthy = _spawn_and_wait(cmd)
+ # #6415 split-mode tensor warmup abort. Latch it on THIS first spawn:
+ # the flash-attn-off retry below can't run tensor (needs flash_attn),
+ # so its output drops the marker and recording later would miss it,
+ # looping every load. Record and raise to the route's layer fallback,
+ # skipping the futile flash-attn/MTP retries.
+ if not healthy and self._tensor_parallel and not self._cancel_event.is_set():
+ _ts_out = "\n".join(self._stdout_lines[-50:])
+ _ts_rc = self._process.poll() if self._process is not None else None
+ if self._should_record_tensor_split_abort(_ts_rc, _ts_out):
+ LlamaCppBackend._record_tensor_split_abort(binary, model_identifier)
+ self._kill_process()
+ raise RuntimeError(
+ "llama-server aborted on --split-mode tensor "
+ "(split-axis geometry); retrying with layer split."
+ )
# Flash-attention kernels hard-crash at startup on some ROCm/GPU
# builds (frequently inside the vision tower). Disabling FA keeps
# both vision and MTP, so retry that way before dropping either.
@@ -6048,6 +6206,7 @@ class LlamaCppBackend:
# Read the crash code before _kill_process() clears _process.
_crash_rc = self._process.poll() if self._process is not None else None
self._kill_process()
+ # The #6415 split-axis abort is latched earlier (first spawn).
# Skip if a cancel/unload is pending (mirrors the MTP guard).
if (
launched_with_mmproj
@@ -6479,6 +6638,7 @@ class LlamaCppBackend:
spec_draft_n_max: Optional[int] = None,
tensor_parallel: bool = False,
mtp_draft_path: Optional[str] = None,
+ preserve_multi_gpu_on_layer: bool = False,
) -> bool:
"""True iff the live server already satisfies these load kwargs.
@@ -6521,6 +6681,17 @@ class LlamaCppBackend:
# server. An identical request would downgrade the same way.
if not _tensor_parallel_matches_loaded(extra_args, tensor_parallel, self._tensor_parallel):
return False
+ # Preserved tensor->layer fallback + an EXPLICIT tensor drop: reload so
+ # placement re-selects instead of keeping the all-GPU mask (mirrors the route,
+ # #6659). preserve_multi_gpu_on_layer carries the route's carry-forward decision
+ # (True for an implicit same-settings reload), so those still dedupe -- the HF
+ # auto-pick / local-dir flows skip the route guard and only reach here.
+ if (
+ self._layer_preserves_tensor_intent
+ and not _effective_tensor_parallel(extra_args, tensor_parallel)
+ and not preserve_multi_gpu_on_layer
+ ):
+ return False
# Compare on the canonical requested mode. With --spec-type in
# extra_args the backend stores None; mirror that here.
@@ -6632,6 +6803,7 @@ class LlamaCppBackend:
self._supports_tools = False
self._cache_type_kv = None
self._tensor_parallel = False
+ self._layer_preserves_tensor_intent = False
self._speculative_type = None
self._requested_spec_mode = None
self._spec_draft_n_max = None
diff --git a/studio/backend/core/inference/llama_server_args.py b/studio/backend/core/inference/llama_server_args.py
index b42be5ee0d..f400d2ae40 100644
--- a/studio/backend/core/inference/llama_server_args.py
+++ b/studio/backend/core/inference/llama_server_args.py
@@ -25,6 +25,11 @@ _DENYLIST_GROUPS: tuple[frozenset[str], ...] = (
# Model identity: Studio resolves it from LoadRequest; a second -m would
# load a different model than Studio thinks it loaded.
frozenset({"-m", "--model"}),
+ # Public model id: Studio sets a sanitized --alias so the OpenAI API never
+ # exposes the local .gguf path. A user-supplied alias is appended after
+ # Studio's and, with llama.cpp's last-wins parsing, would reintroduce the
+ # path leak this is meant to prevent.
+ frozenset({"-a", "--alias"}),
frozenset({"-mu", "--model-url"}),
frozenset({"-dr", "--docker-repo"}),
frozenset({"-hf", "-hfr", "--hf-repo"}),
diff --git a/studio/backend/core/inference/model_ids.py b/studio/backend/core/inference/model_ids.py
new file mode 100644
index 0000000000..548cc60f94
--- /dev/null
+++ b/studio/backend/core/inference/model_ids.py
@@ -0,0 +1,71 @@
+# SPDX-License-Identifier: AGPL-3.0-only
+# Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. See /studio/LICENSE.AGPL-3.0
+
+"""Public model identifiers for the OpenAI-compatible API.
+
+The exposed API must report a stable, clean model id rather than the absolute
+on-disk path of a local GGUF. The internal identifier for a direct local load is
+the absolute ``.gguf`` path, which leaks the host filesystem layout and is
+awkward for clients to round-trip. ``public_model_id`` maps such an internal
+identifier to a clean name while leaving Hugging Face repo ids (``org/model``)
+and already-clean names untouched.
+"""
+
+from __future__ import annotations
+
+import os
+from typing import Optional
+
+_GGUF_SUFFIX = ".gguf"
+
+
+def _looks_like_path(identifier: str) -> bool:
+ """True when *identifier* is a local filesystem path, not a HF repo id.
+
+ A repo id is ``org/model`` (a single forward slash, no leading separator, no
+ drive, no ``.gguf``). Anything ending in ``.gguf``, starting with a path
+ separator or a relative/home prefix (``./``, ``../``, ``~``), carrying a
+ Windows drive, or with three or more ``/`` segments is treated as a local
+ path.
+ """
+ if identifier.lower().endswith(_GGUF_SUFFIX):
+ return True
+ if identifier.startswith(("/", "\\", "./", "../", ".\\", "..\\", "~")):
+ return True
+ if len(identifier) >= 2 and identifier[1] == ":": # Windows drive, e.g. C:\
+ return True
+ if identifier.count("/") >= 2 or "\\" in identifier:
+ return True
+ return False
+
+
+def public_model_id(identifier: Optional[str]) -> Optional[str]:
+ """Return a clean, path-free public id for *identifier*.
+
+ - Local GGUF path -> the file stem with ``.gguf`` stripped, e.g.
+ ``/srv/models/Qwen3-30B-A3B-Q4_K_M.gguf`` -> ``Qwen3-30B-A3B-Q4_K_M``.
+ - HF repo id (``org/model``) and already-clean names -> returned unchanged.
+ - ``None`` / empty -> returned unchanged.
+ """
+ if not identifier:
+ return identifier
+ if not _looks_like_path(identifier):
+ return identifier
+ name = os.path.basename(identifier.replace("\\", "/").rstrip("/"))
+ if name.lower().endswith(_GGUF_SUFFIX):
+ name = name[: -len(_GGUF_SUFFIX)]
+ return name or identifier
+
+
+def model_id_matches(requested: Optional[str], internal: Optional[str]) -> bool:
+ """Whether a client-supplied *requested* id refers to *internal*.
+
+ Accepts the clean public id (preferred) and, for backward compatibility, the
+ raw internal identifier (e.g. a legacy absolute path a client cached from an
+ older ``/v1/models`` response).
+ """
+ if requested is None or internal is None:
+ return False
+ if requested == internal:
+ return True
+ return public_model_id(internal) == requested
diff --git a/studio/backend/core/training/training.py b/studio/backend/core/training/training.py
index 9d991c6512..f4233fcf04 100644
--- a/studio/backend/core/training/training.py
+++ b/studio/backend/core/training/training.py
@@ -299,6 +299,7 @@ class TrainingBackend:
# Build config dict for the subprocess
config = {
"model_name": kwargs["model_name"],
+ "project_name": kwargs.get("project_name"),
"training_type": kwargs.get("training_type", "LoRA/QLoRA"),
"hf_token": kwargs.get("hf_token", ""),
"load_in_4bit": kwargs.get("load_in_4bit", True),
diff --git a/studio/backend/core/training/worker.py b/studio/backend/core/training/worker.py
index 3f020c8abc..610af2472e 100644
--- a/studio/backend/core/training/worker.py
+++ b/studio/backend/core/training/worker.py
@@ -44,6 +44,7 @@ if sys.platform.startswith("linux") and "HSA_ENABLE_DXG_DETECTION" not in os.env
logger = get_logger(__name__)
from utils.hardware import apply_gpu_ids
+from utils.training_runs import build_default_output_dir_name
from utils.wheel_utils import (
direct_wheel_url,
flash_attn_wheel_url,
@@ -1787,11 +1788,14 @@ def _run_mlx_training(event_queue, stop_queue, config):
# ── 5. Build output dir ──
# Resolve to ~/.unsloth/studio/outputs/ so the export page finds it
- from utils.paths import resolve_output_dir, ensure_dir, default_run_dir_name
+ from utils.paths import resolve_output_dir, ensure_dir
output_dir = config.get("output_dir", "")
if not output_dir:
- output_dir = f"{default_run_dir_name(model_name)}_{int(time.time())}"
+ output_dir = build_default_output_dir_name(
+ model_name,
+ config.get("project_name"),
+ )
output_dir = str(resolve_output_dir(output_dir))
ensure_dir(Path(output_dir))
@@ -3019,7 +3023,10 @@ def run_training_process(*, event_queue: Any, stop_queue: Any, config: dict) ->
resume_from_checkpoint
)
if not output_dir:
- output_dir = f"{default_run_dir_name(model_name)}_{int(time.time())}"
+ output_dir = build_default_output_dir_name(
+ model_name,
+ config.get("project_name"),
+ )
output_dir = str(resolve_output_dir(output_dir))
ensure_dir(Path(output_dir))
@@ -3500,7 +3507,10 @@ def _run_embedding_training(event_queue: Any, stop_queue: Any, config: dict) ->
resume_from_checkpoint
)
if not output_dir:
- output_dir = f"{default_run_dir_name(model_name)}_{int(time.time())}"
+ output_dir = build_default_output_dir_name(
+ model_name,
+ config.get("project_name"),
+ )
output_dir = str(resolve_output_dir(output_dir))
num_epochs = config.get("num_epochs", 2)
diff --git a/studio/backend/main.py b/studio/backend/main.py
index 731625ca74..0a5b775775 100644
--- a/studio/backend/main.py
+++ b/studio/backend/main.py
@@ -441,9 +441,30 @@ def _start_llama_cpp_probes_if_enabled(app: FastAPI) -> None:
).start()
+def _warm_rag_embedder() -> None:
+ """Warm RAG embeddings without blocking backend readiness."""
+ try:
+ from storage import rag_db
+
+ if not rag_db.RAG_AVAILABLE:
+ return
+ from core.rag import embeddings
+
+ embeddings.warm()
+ except Exception:
+ pass
+
+
@asynccontextmanager
async def lifespan(app: FastAPI):
"""Startup: detect hardware, seed default admin if needed. Shutdown: clean up compiled cache."""
+
+ import time as _time
+
+ _lifespan_started = _time.perf_counter()
+ import structlog as _structlog
+
+ _lifespan_log = _structlog.get_logger(__name__)
clear_unsloth_compiled_cache()
# Remove stale .venv_overlay from old versions; switching now uses .venv_t5/.
@@ -454,6 +475,11 @@ async def lifespan(app: FastAPI):
# Detect hardware first — sets the DEVICE global used everywhere.
detect_hardware()
+ _lifespan_log.info(
+ "lifespan hardware detection completed in %.1fms",
+ (_time.perf_counter() - _lifespan_started) * 1000,
+ )
+
# Apple Silicon with MLX missing => Train/Export are greyed out (chat-only).
# Reinstall mlx by name on a background thread (off the critical path) and
# re-detect, so a reinstall/update that dropped mlx self-heals. No-op
@@ -465,7 +491,13 @@ async def lifespan(app: FastAPI):
import structlog as _structlog
_structlog.get_logger(__name__).debug("mlx autorepair skipped: %s", _mlx_exc)
- # Reap download workers orphaned by a previous crash before new downloads start.
+ # Reap workers/runs orphaned by a previous crash before new work starts.
+ try:
+ from storage.studio_db import cleanup_orphaned_runs
+ cleanup_orphaned_runs()
+ except Exception as exc:
+ _lifespan_log.warning("cleanup_orphaned_runs failed at startup: %s", exc)
+
reap_hub_orphan_workers()
# llama.cpp probes: capability (MTP support) + freshness (release age).
@@ -479,45 +511,23 @@ async def lifespan(app: FastAPI):
app.state.llama_cpp_freshness = None
_start_llama_cpp_probes_if_enabled(app)
- from storage.studio_db import cleanup_orphaned_runs
-
- try:
- cleanup_orphaned_runs()
- except Exception as exc:
- import structlog
- structlog.get_logger(__name__).warning("cleanup_orphaned_runs failed at startup: %s", exc)
-
- # Same for RAG: fail ingestion jobs stranded mid-ingest by a crash.
try:
from storage.rag_db import reconcile_orphaned_ingestion_jobs
reconcile_orphaned_ingestion_jobs()
except Exception as exc:
- import structlog
- structlog.get_logger(__name__).warning(
- "reconcile_orphaned_ingestion_jobs failed at startup: %s", exc
- )
+ _lifespan_log.warning("reconcile_orphaned_ingestion_jobs failed at startup: %s", exc)
_start_helper_precache_if_enabled()
+ threading.Thread(target = _warm_rag_embedder, daemon = True, name = "rag-embedder-warm").start()
- # Warm the RAG embedder so the first upload skips the cold load. Non-fatal.
- def _warm_rag_embedder():
- try:
- from storage import rag_db
-
- if not rag_db.RAG_AVAILABLE:
- return
- from core.rag import embeddings
-
- embeddings.warm()
- except Exception:
- pass
-
- threading.Thread(target = _warm_rag_embedder, daemon = True).start()
-
- # Initialize RSA key pair for API key encryption (external providers)
+ # Initialize RSA key pair for API key encryption (external providers).
from core.inference.key_exchange import init_key_pair
init_key_pair()
+ _lifespan_log.info(
+ "lifespan pre-auth setup completed in %.1fms",
+ (_time.perf_counter() - _lifespan_started) * 1000,
+ )
if storage.ensure_default_admin():
bootstrap_pw = storage.get_bootstrap_password()
@@ -532,6 +542,11 @@ async def lifespan(app: FastAPI):
print("=" * 60 + "\n")
else:
app.state.bootstrap_password = storage.get_bootstrap_password()
+
+ _lifespan_log.info(
+ "lifespan startup completed in %.1fms",
+ (_time.perf_counter() - _lifespan_started) * 1000,
+ )
yield
from core.inference.llama_http import aclose as _close_llama_http
@@ -919,6 +934,21 @@ install_api_error_handlers(app)
# ============ Health and System Endpoints ============
+@app.get("/api/liveness")
+async def liveness_check():
+ """Cheap process liveness for desktop port validation."""
+ return {
+ "status": "alive",
+ "service": "Unsloth UI Backend",
+ "desktop_protocol_version": 1,
+ "desktop_manageability_version": 1,
+ "supports_desktop_auth": True,
+ "supports_desktop_backend_ownership": True,
+ "studio_root_id": _studio_root_id(),
+ **({"desktop_owner": owner} if (owner := _desktop_owner()) else {}),
+ }
+
+
@app.get("/api/health")
async def health_check(request: Request):
"""Liveness plus launcher capability bits; host fingerprint gated on a bearer.
diff --git a/studio/backend/models/inference.py b/studio/backend/models/inference.py
index 26825a472e..4a3162b09e 100644
--- a/studio/backend/models/inference.py
+++ b/studio/backend/models/inference.py
@@ -106,8 +106,7 @@ class LoadRequest(BaseModel):
"Extra arguments forwarded verbatim to llama-server for GGUF models. "
"One token per list entry, e.g. ['--top-k', '20', '--seed', '42']. "
"Studio-managed flags (model identity, port, context length, GPU placement, "
- "auth, --flash-attn, --no-context-shift, --jinja) are rejected. Ignored for "
- "non-GGUF models."
+ "auth, UI/server mode) are rejected. Ignored for non-GGUF models."
),
)
diff --git a/studio/backend/models/training.py b/studio/backend/models/training.py
index e64b6f731a..ff815a2fa9 100644
--- a/studio/backend/models/training.py
+++ b/studio/backend/models/training.py
@@ -9,6 +9,8 @@ import re
from pydantic import BaseModel, ConfigDict, Field, field_validator, model_validator
from typing import Any, Optional, List, Dict, Literal
+from utils.training_runs import normalize_project_name
+
# ASCII integer, optional single sign. Rejects "++512" and Unicode digits
# ("512") that slip through str.isdigit() + int().
@@ -97,6 +99,11 @@ class TrainingStartRequest(BaseModel):
model_name: str = Field(
..., description = "Model identifier (e.g., 'unsloth/llama-3-8b-bnb-4bit')"
)
+ project_name: Optional[str] = Field(
+ None,
+ max_length = 80,
+ description = "Optional user-defined project name appended to run folders and shown in history",
+ )
training_type: Literal["LoRA/QLoRA", "Full Finetuning", "Continued Pretraining"] = Field(
...,
description = "Training type: 'LoRA/QLoRA', 'Full Finetuning', or 'Continued Pretraining'",
@@ -155,6 +162,11 @@ class TrainingStartRequest(BaseModel):
values.setdefault("train_split", values.pop("split"))
return values
+ @field_validator("project_name")
+ @classmethod
+ def _normalize_project_name(cls, value: Optional[str]) -> Optional[str]:
+ return normalize_project_name(value)
+
# NOTE: pydantic runs all `mode="after"` validators in definition order. A
# second one, `_check_steps_or_epochs`, is defined lower in this class; keep
# these cross-field checks order-independent so the two stay decoupled.
@@ -588,6 +600,7 @@ class TrainingRunSummary(BaseModel):
id: str
status: Literal["running", "completed", "stopped", "error"]
model_name: str
+ project_name: Optional[str] = None
dataset_name: str
display_name: Optional[str] = None
started_at: str
diff --git a/studio/backend/routes/data_recipe/seed.py b/studio/backend/routes/data_recipe/seed.py
index 8fb034ea4e..57a291291e 100644
--- a/studio/backend/routes/data_recipe/seed.py
+++ b/studio/backend/routes/data_recipe/seed.py
@@ -481,6 +481,37 @@ async def upload_unstructured_file(
error = "No extractable text found in file",
)
extracted_path.write_text(extracted_text, encoding = "utf-8")
+ except ImportError as e:
+ raw_path.unlink(missing_ok = True)
+ extracted_path.unlink(missing_ok = True)
+ missing = getattr(e, "name", None)
+ expected_missing = {".pdf": "pymupdf4llm", ".docx": "mammoth"}.get(ext)
+ if isinstance(e, ModuleNotFoundError) and missing == expected_missing:
+ logger.error(
+ "data_recipe.seed.text_extraction_dependency_missing",
+ error = str(e),
+ missing = missing,
+ exc_info = True,
+ )
+ return UnstructuredFileUploadResponse(
+ file_id = file_id,
+ filename = original_filename,
+ size_bytes = size_bytes,
+ status = "error",
+ error = f"Cannot read {ext} files: the '{missing}' package is not installed.",
+ )
+ logger.error(
+ "data_recipe.seed.text_extraction_failed",
+ error = str(e),
+ exc_info = True,
+ )
+ return UnstructuredFileUploadResponse(
+ file_id = file_id,
+ filename = original_filename,
+ size_bytes = size_bytes,
+ status = "error",
+ error = "Text extraction failed.",
+ )
except Exception as e:
raw_path.unlink(missing_ok = True)
extracted_path.unlink(missing_ok = True)
diff --git a/studio/backend/routes/inference.py b/studio/backend/routes/inference.py
index a9c85fa99a..d3f56edec5 100644
--- a/studio/backend/routes/inference.py
+++ b/studio/backend/routes/inference.py
@@ -683,7 +683,9 @@ try:
detect_reasoning_flags,
)
from core.inference.llama_server_args import (
+ _effective_tensor_parallel,
_tensor_parallel_matches_loaded,
+ parse_split_mode_override,
resolve_tensor_parallel,
strip_shadowing_flags,
validate_extra_args,
@@ -718,7 +720,9 @@ except ImportError:
detect_reasoning_flags,
)
from core.inference.llama_server_args import (
+ _effective_tensor_parallel,
_tensor_parallel_matches_loaded,
+ parse_split_mode_override,
resolve_tensor_parallel,
strip_shadowing_flags,
validate_extra_args,
@@ -1107,6 +1111,7 @@ from auth.authentication import get_current_subject
from state.tool_approvals import resolve_tool_decision
from core.inference.key_exchange import decrypt_api_key
+from core.inference.model_ids import public_model_id
from core.inference.api_monitor import api_monitor
from core.inference.llama_http import nonstreaming_client
from core.inference.providers import get_base_url
@@ -2077,6 +2082,32 @@ def _should_strip_split_mode(request: LoadRequest, backend_extra: Optional[list[
)
+def _carry_preserved_tensor_intent(
+ *, preserved: bool, same_model: bool, explicit_drop: bool
+) -> bool:
+ """Carry a preserved multi-GPU layer fallback forward only for a reload of the
+ SAME loaded model that doesn't explicitly drop tensor intent, so a fitting model
+ isn't collapsed to one GPU on a ctx-only change -- but an unrelated model switch
+ (without /unload) or an explicit tensor-off doesn't inherit it (#6659)."""
+ return preserved and same_model and not explicit_drop
+
+
+def _is_explicit_tensor_drop(request: LoadRequest) -> bool:
+ """True only when the request explicitly selects a non-tensor --split-mode (e.g.
+ layer/row/none), a deliberate departure from a preserved tensor->layer fallback.
+
+ A bare tensor_parallel field is NOT a drop: the Studio UI always sends it and echoes
+ the /load response's resolved value back, so after a fallback every reload carries
+ tensor_parallel=false even though the user never changed it -- treating that as a drop
+ would collapse the preserved multi-GPU placement on the next ctx/settings reload. An
+ empty clear is not a drop either (a fallback always stores --split-mode layer, never a
+ tensor split mode, so a clear never wipes tensor intent), nor is an unrelated extra
+ (--top-k) or inherit (None). tensor_parallel=true / --split-mode tensor re-engage
+ tensor. Shared by the already-loaded dedup and the load carry-forward (#6659)."""
+ override = parse_split_mode_override(request.llama_extra_args)
+ return override is not None and override.strip().lower() != "tensor"
+
+
def _request_matches_loaded_settings(
request: LoadRequest,
llama_backend: LlamaCppBackend,
@@ -2115,6 +2146,13 @@ def _request_matches_loaded_settings(
effective_extra, request.tensor_parallel, llama_backend.tensor_parallel
):
return False
+ # Preserved tensor->layer fallback (both report tensor=off, so the check above
+ # matches): if the user now explicitly drops tensor intent, reload so placement
+ # re-selects instead of keeping the all-GPU mask (#6659). The effective check
+ # includes the env, so an env-only tensor (LLAMA_ARG_SPLIT_MODE=tensor) that
+ # can't actually be dropped falls through to the env-downgrade match, not a loop.
+ if llama_backend.layer_preserves_tensor_intent and _is_explicit_tensor_drop(request):
+ return False
# Spec decoding works on vision models too (MTP is mmproj-compatible,
# llama.cpp #22673; the old ``not is_vision`` gate is gone), so compare
# the real requested mode -- coercing vision to ``off`` here used to
@@ -2809,6 +2847,48 @@ async def load_model(
hf_variant = config.gguf_variant,
)
+ # Tensor intent for this load: the request itself, or a preserved
+ # multi-GPU layer fallback carried across a reload of the SAME model that
+ # doesn't drop it (e.g. a ctx-only change), so a fitting model doesn't
+ # silently collapse to one GPU. Only an explicit non-tensor --split-mode
+ # override counts as the drop -- the tensor field echo / unrelated extras keep
+ # the preserved placement; the same-model guard stops a switch-without-unload
+ # inheriting the prior model's intent.
+ _explicit_tensor_drop = _is_explicit_tensor_drop(request)
+ # Compare the resolved config.identifier (what load_model stores), not the
+ # raw request id: from_identifier normalizes shorthands (adds unsloth/, fixes
+ # case), so a reload with the shorthand would otherwise miss the match and
+ # drop the carry-forward. #6659
+ _same_model_loaded = (
+ llama_backend.is_loaded
+ and (llama_backend.model_identifier or "").lower()
+ == (config.identifier or "").lower()
+ )
+ # model_identifier is variant-agnostic for HF repos and dir-level for a
+ # local multi-variant directory, so also require the loaded quant to match
+ # (path else variant, mirroring _already_in_target_state) -- otherwise a
+ # different variant inherits the prior one's preserved intent. #6659
+ if _same_model_loaded:
+ if config.gguf_file and llama_backend.gguf_path:
+ try:
+ _same_model_loaded = (
+ Path(llama_backend.gguf_path).resolve()
+ == Path(config.gguf_file).resolve()
+ )
+ except OSError:
+ _same_model_loaded = False
+ else:
+ _same_model_loaded = (llama_backend.hf_variant or "").lower() == (
+ config.gguf_variant or ""
+ ).lower()
+ _tensor_intent_overall = _effective_tensor_parallel(
+ extra_llama_args, request.tensor_parallel
+ ) or _carry_preserved_tensor_intent(
+ preserved = llama_backend.layer_preserves_tensor_intent,
+ same_model = _same_model_loaded,
+ explicit_drop = _explicit_tensor_drop,
+ )
+
# Run a single load attempt with the given tensor flag + extras.
async def _attempt_gguf_load(
tensor_parallel: bool, attempt_extra_args: Optional[list[str]]
@@ -2822,6 +2902,12 @@ async def load_model(
**_source_load_kwargs,
**attempt_kwargs,
tensor_parallel = tensor_parallel,
+ # True on the layer fallback retry (tensor wanted overall but not on
+ # this attempt): keep multi-GPU. Mirrors the fallback's key.
+ preserve_multi_gpu_on_layer = bool(
+ _tensor_intent_overall
+ and not _effective_tensor_parallel(attempt_extra_args, tensor_parallel)
+ ),
)
# Tensor parallelism is arch-gated in llama.cpp and crashes some loads
@@ -3703,7 +3789,7 @@ async def generate_audio(
# Pick backend — both return (wav_bytes, sample_rate)
llama_backend = get_llama_cpp_backend()
if llama_backend.is_loaded and getattr(llama_backend, "_is_audio", False):
- model_name = llama_backend.model_identifier
+ model_name = public_model_id(llama_backend.model_identifier)
gen = lambda: llama_backend.generate_audio_response(
text = text,
audio_type = llama_backend._audio_type,
@@ -3721,7 +3807,7 @@ async def generate_audio(
model_info = backend.models.get(backend.active_model_name, {})
if not model_info.get("is_audio"):
raise HTTPException(status_code = 400, detail = "Active model is not an audio model.")
- model_name = backend.active_model_name
+ model_name = public_model_id(backend.active_model_name)
gen = lambda: backend.generate_audio_response(
text = text,
temperature = payload.temperature,
@@ -4837,7 +4923,8 @@ async def openai_chat_completions(
return response
if using_gguf:
- model_name = llama_backend.model_identifier or payload.model
+ # Echo a clean public id in the response, never the absolute .gguf path.
+ model_name = public_model_id(llama_backend.model_identifier) or payload.model
if getattr(llama_backend, "_is_audio", False):
if _wants_multiple_choices(payload):
_raise_unsupported_n("GGUF audio chat completions")
@@ -4852,7 +4939,9 @@ async def openai_chat_completions(
status_code = 400,
detail = "No model loaded. Call POST /inference/load first.",
)
- model_name = backend.active_model_name or payload.model
+ # Clean public id so the response never echoes a local path; the audio
+ # branch below receives this sanitized label too.
+ model_name = public_model_id(backend.active_model_name) or payload.model
if _wants_multiple_choices(payload):
_raise_unsupported_n("non-GGUF chat completions")
@@ -6387,6 +6476,9 @@ async def serve_sandbox_file(
# OpenAI-Compatible Models Listing (/models → /v1/models)
# =====================================================================
+# `owned_by` marker on every /v1/models entry (loaded and available alike).
+_OWNED_BY = "unsloth-studio"
+
def _openai_model_objects() -> list[dict]:
"""The model objects GET /v1/models exposes (one per loaded local backend).
@@ -6401,10 +6493,12 @@ def _openai_model_objects() -> list[dict]:
llama_backend = get_llama_cpp_backend()
if llama_backend.is_loaded:
entry = {
- "id": llama_backend.model_identifier,
+ # Public id, never the absolute .gguf path (which leaks the host
+ # filesystem layout); see core.inference.model_ids.public_model_id.
+ "id": public_model_id(llama_backend.model_identifier),
"object": "model",
"created": _created,
- "owned_by": "local",
+ "owned_by": _OWNED_BY,
}
_ctx = _positive_int_or_none(getattr(llama_backend, "context_length", None))
if _ctx is not None:
@@ -6422,10 +6516,10 @@ def _openai_model_objects() -> list[dict]:
if backend.active_model_name:
model_info = backend.models.get(backend.active_model_name, {})
entry = {
- "id": backend.active_model_name,
+ "id": public_model_id(backend.active_model_name),
"object": "model",
"created": _created,
- "owned_by": "local",
+ "owned_by": _OWNED_BY,
}
_ctx = _positive_int_or_none(model_info.get("context_length"))
if _ctx is None:
@@ -6443,15 +6537,86 @@ def _openai_model_objects() -> list[dict]:
return models
+# Brief cache for the local-model filesystem scan so repeated /v1/models calls
+# don't rescan the HF cache and models dirs on every request.
+_CATALOG_CACHE: dict = {"at": 0.0, "models": []}
+_CATALOG_TTL_S = 30.0
+_CATALOG_LOCK = asyncio.Lock()
+
+
+async def _cached_local_catalog() -> list:
+ """Locally available models (models dir + HF caches + LM Studio + scan
+ folders), cached for a few seconds. Returns a list of LocalModelInfo.
+
+ The scan walks several directories and stats many files, so it runs in a
+ worker thread (asyncio.to_thread) -- calling it inline would block the event
+ loop and stall every concurrent request and in-flight inference stream. A
+ lock with a double-check collapses a burst of simultaneous /v1/models calls
+ into a single scan instead of one per request."""
+ # Validity is keyed on "at" (set only after a scan), not on list contents, so
+ # an empty/errored scan is still cached instead of rescanning on every poll.
+ now = time.monotonic()
+ if _CATALOG_CACHE["at"] and (now - _CATALOG_CACHE["at"]) <= _CATALOG_TTL_S:
+ return _CATALOG_CACHE["models"]
+ async with _CATALOG_LOCK:
+ now = time.monotonic()
+ if _CATALOG_CACHE["at"] and (now - _CATALOG_CACHE["at"]) <= _CATALOG_TTL_S:
+ return _CATALOG_CACHE["models"]
+ try:
+ from routes.models import collect_local_models
+ _CATALOG_CACHE["models"] = await asyncio.to_thread(
+ collect_local_models, Path("./models").resolve()
+ )
+ except Exception as exc:
+ logger.debug("model catalog scan failed: %s", exc)
+ _CATALOG_CACHE["models"] = []
+ # Stamp after the scan, not the pre-scan "now": a scan slower than the TTL
+ # would otherwise leave the cache already expired, so every waiter rescans.
+ _CATALOG_CACHE["at"] = time.monotonic()
+ return _CATALOG_CACHE["models"]
+
+
+async def _openai_catalog_objects() -> list[dict]:
+ """Every model the server knows about for ``GET /v1/models``: the loaded
+ model(s) plus locally available (downloaded/cached) models discovered by
+ scanning. Loaded entries keep their context fields and are marked
+ ``loaded: true``. All ids are clean public ids (never absolute paths)."""
+ _created = int(time.time())
+ # Loaded models first (clean ids + context fields), marked loaded.
+ by_id: dict[str, dict] = {}
+ for entry in _openai_model_objects():
+ by_id[entry["id"]] = {**entry, "loaded": True}
+
+ # Locally available (downloaded/cached) models that are not already loaded.
+ for info in await _cached_local_catalog():
+ cid = getattr(info, "model_id", None) or public_model_id(getattr(info, "id", None))
+ if not cid or cid in by_id:
+ continue
+ obj = {
+ "id": cid,
+ "object": "model",
+ "created": _created,
+ "owned_by": _OWNED_BY,
+ "loaded": False,
+ }
+ display = getattr(info, "display_name", None)
+ if display:
+ obj["display_name"] = display
+ by_id[cid] = obj
+
+ return list(by_id.values())
+
+
@router.get("/models")
async def openai_list_models(current_subject: str = Depends(get_current_subject)):
"""
- OpenAI-compatible model listing endpoint.
+ OpenAI-compatible model listing endpoint (``GET /v1/models``).
- Returns the currently loaded model in the format expected by
- OpenAI-compatible clients (``GET /v1/models``).
+ Lists every model available on this server -- the loaded model(s) plus
+ locally available (downloaded/cached) models -- not only what is resident in
+ memory. Each entry carries a clean public id and a ``loaded`` flag.
"""
- return {"object": "list", "data": _openai_model_objects()}
+ return {"object": "list", "data": await _openai_catalog_objects()}
@router.get("/models/{model_id:path}")
@@ -6459,13 +6624,37 @@ async def openai_retrieve_model(model_id: str, current_subject: str = Depends(ge
"""
OpenAI-compatible single-model retrieval endpoint (``GET /v1/models/{id}``).
- Returns the bare model object when ``model_id`` matches a loaded local
- model, or 404 model_not_found otherwise. Defined after the LIST route so
- it does not shadow it; ``{model_id:path}`` keeps ids with slashes intact.
+ Returns the bare model object when ``model_id`` matches a known model
+ (loaded or locally available), or 404 model_not_found otherwise. Defined
+ after the LIST route so it does not shadow it; ``{model_id:path}`` keeps ids
+ with slashes intact.
"""
- for model in _openai_model_objects():
+ from core.inference.model_ids import model_id_matches
+
+ # Loaded models resolve without a catalog scan (the common case); only build
+ # the full catalog -- which may hit the filesystem -- for unloaded ids.
+ for entry in _openai_model_objects():
+ if entry["id"] == model_id:
+ return {**entry, "loaded": True}
+
+ objects = await _openai_catalog_objects()
+ for model in objects:
if model["id"] == model_id:
return model
+ # Backward compatibility: a client may still send the legacy raw identifier
+ # (e.g. an absolute .gguf path cached from an older /v1/models). Resolve it to
+ # the clean object so it keeps working, without ever echoing the path back.
+ llama_backend = get_llama_cpp_backend()
+ backend = get_inference_backend()
+ for raw in (
+ llama_backend.model_identifier if llama_backend.is_loaded else None,
+ backend.active_model_name or None,
+ ):
+ if raw and model_id_matches(model_id, raw):
+ clean = public_model_id(raw)
+ for model in objects:
+ if model["id"] == clean:
+ return model
raise HTTPException(
status_code = 404,
detail = openai_error_body(
@@ -7402,6 +7591,15 @@ async def _responses_stream(
target_url = f"{llama_backend.base_url}/v1/chat/completions"
async def event_generator():
+ # Clean public id for every response envelope. Prefer the loaded model's
+ # id so the stream agrees with /v1/models, chat/completions and the
+ # non-streaming twin; fall back to a sanitized payload.model (a legacy
+ # raw .gguf path is stripped, never echoed back).
+ _clean_model = (
+ public_model_id(getattr(llama_backend, "model_identifier", None))
+ or public_model_id(payload.model)
+ or payload.model
+ )
full_text = ""
full_reasoning = ""
input_tokens = 0
@@ -7563,7 +7761,7 @@ async def _responses_stream(
"object": "response",
"created_at": created_at,
"status": "failed",
- "model": payload.model,
+ "model": _clean_model,
"output": _snapshot_output(),
"usage": {
"input_tokens": input_tokens,
@@ -7587,7 +7785,7 @@ async def _responses_stream(
"object": "response",
"created_at": created_at,
"status": "in_progress",
- "model": payload.model,
+ "model": _clean_model,
"output": [],
"usage": {"input_tokens": 0, "output_tokens": 0, "total_tokens": 0},
},
@@ -7627,7 +7825,7 @@ async def _responses_stream(
"object": "response",
"created_at": created_at,
"status": "failed",
- "model": payload.model,
+ "model": _clean_model,
"output": [],
"error": {"code": 502, "message": _friendly_error(e)},
},
@@ -7653,7 +7851,7 @@ async def _responses_stream(
"object": "response",
"created_at": created_at,
"status": "failed",
- "model": payload.model,
+ "model": _clean_model,
"output": [],
"error": {
"code": resp.status_code,
@@ -8002,7 +8200,7 @@ async def _responses_stream(
"object": "response",
"created_at": created_at,
"status": "completed",
- "model": payload.model,
+ "model": _clean_model,
"output": _snapshot_output(),
"usage": {
"input_tokens": input_tokens,
@@ -8274,7 +8472,13 @@ async def anthropic_messages(
),
)
- model_name = getattr(llama_backend, "model_identifier", None) or payload.model
+ # Clean public id so /v1/messages never echoes the local .gguf path (and a
+ # legacy raw path sent as payload.model is sanitized rather than returned).
+ model_name = (
+ public_model_id(getattr(llama_backend, "model_identifier", None))
+ or public_model_id(payload.model)
+ or payload.model
+ )
message_id = f"msg_{uuid.uuid4().hex[:24]}"
# ── Translate Anthropic → OpenAI ──────────────────────────
diff --git a/studio/backend/routes/models.py b/studio/backend/routes/models.py
index 951c2960f3..e22f65751c 100644
--- a/studio/backend/routes/models.py
+++ b/studio/backend/routes/models.py
@@ -722,6 +722,94 @@ def _scan_ollama_dir(ollama_dir: Path, limit: Optional[int] = None) -> List[Loca
return found
+def collect_local_models(models_root: Path) -> List[LocalModelInfo]:
+ """Scan ``models_root``, the HF caches, LM Studio dirs, and user scan folders,
+ returning a deduplicated, hidden-filtered list of discovered local models.
+
+ Shared by ``GET /models/local`` (the model picker) and the OpenAI-compatible
+ catalog (``GET /v1/models``) so the UI and the API never drift. ``models_root``
+ must already be validated/trusted by the caller.
+ """
+ from storage.studio_db import list_scan_folders
+ from utils.paths import (
+ hf_default_cache_dir,
+ legacy_hf_cache_dir,
+ lmstudio_model_dirs,
+ )
+
+ hf_cache_dir = _resolve_hf_cache_dir()
+ legacy_hf = legacy_hf_cache_dir()
+ hf_default = hf_default_cache_dir()
+ lm_dirs = lmstudio_model_dirs()
+
+ local_models = _scan_models_dir(models_root) + _scan_hf_cache(hf_cache_dir)
+
+ # Resolve once; an inaccessible aux cache must skip that scan, not 500.
+ hf_cache_real = _safe_resolve(hf_cache_dir)
+ legacy_real = _safe_resolve(legacy_hf)
+ default_real = _safe_resolve(hf_default)
+
+ # Scan legacy Unsloth HF cache for backward compatibility.
+ if _safe_is_dir(legacy_hf) and legacy_real != hf_cache_real:
+ local_models += _scan_hf_cache(legacy_hf)
+
+ # Scan HF system default cache (may differ under env overrides).
+ if _safe_is_dir(hf_default) and default_real != hf_cache_real and default_real != legacy_real:
+ local_models += _scan_hf_cache(hf_default)
+
+ # Scan LM Studio directories.
+ for lm_dir in lm_dirs:
+ local_models += _scan_lmstudio_dir(lm_dir)
+
+ # Scan user-added custom folders (per-folder cap).
+ _MAX_MODELS_PER_FOLDER = 200
+ try:
+ custom_folders = list_scan_folders()
+ except Exception as e:
+ logger.warning("Could not load custom scan folders: %s", e)
+ custom_folders = []
+ for folder in custom_folders:
+ folder_path = Path(folder["path"])
+ try:
+ # Filter Ollama .studio_links/ from generic scanners to
+ # avoid duplicates and leaking internal paths into the UI.
+ _generic = [
+ m
+ for m in (
+ _scan_models_dir(folder_path, limit = _MAX_MODELS_PER_FOLDER)
+ + _scan_hf_cache(folder_path)
+ + _scan_lmstudio_dir(folder_path)
+ )
+ if not any(p in (".studio_links", "ollama_links") for p in Path(m.path).parts)
+ ]
+ custom_models = _generic
+ if len(custom_models) < _MAX_MODELS_PER_FOLDER:
+ custom_models += _scan_ollama_dir(
+ folder_path,
+ limit = _MAX_MODELS_PER_FOLDER - len(custom_models),
+ )
+ except OSError as e:
+ logger.warning("Skipping unreadable scan folder %s: %s", folder_path, e)
+ continue
+ local_models += [m.model_copy(update = {"source": "custom"}) for m in custom_models]
+
+ # Deduplicate, but always keep custom folder entries (keyed by
+ # (id, source)) so they show in the "Custom Folders" UI section
+ # even when the model is also in the HF cache.
+ deduped: dict[str, LocalModelInfo] = {}
+ for model in local_models:
+ key = f"{model.id}\x00custom" if model.source == "custom" else model.id
+ if key not in deduped:
+ deduped[key] = model
+
+ models = sorted(
+ deduped.values(),
+ key = lambda item: (item.updated_at or 0),
+ reverse = True,
+ )
+ return [m for m in models if not _is_hidden_model(m.id, m.path)]
+
+
@router.get("/local", response_model = LocalModelListResponse)
async def list_local_models(
models_dir: str = Query(
@@ -770,78 +858,7 @@ async def list_local_models(
)
try:
- local_models = _scan_models_dir(models_root) + _scan_hf_cache(hf_cache_dir)
-
- # Resolve once; an inaccessible aux cache must skip that scan, not 500.
- hf_cache_real = _safe_resolve(hf_cache_dir)
- legacy_real = _safe_resolve(legacy_hf)
- default_real = _safe_resolve(hf_default)
-
- # Scan legacy Unsloth HF cache for backward compatibility.
- if _safe_is_dir(legacy_hf) and legacy_real != hf_cache_real:
- local_models += _scan_hf_cache(legacy_hf)
-
- # Scan HF system default cache (may differ under env overrides).
- if (
- _safe_is_dir(hf_default)
- and default_real != hf_cache_real
- and default_real != legacy_real
- ):
- local_models += _scan_hf_cache(hf_default)
-
- # Scan LM Studio directories.
- for lm_dir in lm_dirs:
- local_models += _scan_lmstudio_dir(lm_dir)
-
- # Scan user-added custom folders (per-folder cap).
- from storage.studio_db import list_scan_folders
-
- _MAX_MODELS_PER_FOLDER = 200
- try:
- custom_folders = list_scan_folders()
- except Exception as e:
- logger.warning("Could not load custom scan folders: %s", e)
- custom_folders = []
- for folder in custom_folders:
- folder_path = Path(folder["path"])
- try:
- # Filter Ollama .studio_links/ from generic scanners to
- # avoid duplicates and leaking internal paths into the UI.
- _generic = [
- m
- for m in (
- _scan_models_dir(folder_path, limit = _MAX_MODELS_PER_FOLDER)
- + _scan_hf_cache(folder_path)
- + _scan_lmstudio_dir(folder_path)
- )
- if not any(p in (".studio_links", "ollama_links") for p in Path(m.path).parts)
- ]
- custom_models = _generic
- if len(custom_models) < _MAX_MODELS_PER_FOLDER:
- custom_models += _scan_ollama_dir(
- folder_path,
- limit = _MAX_MODELS_PER_FOLDER - len(custom_models),
- )
- except OSError as e:
- logger.warning("Skipping unreadable scan folder %s: %s", folder_path, e)
- continue
- local_models += [m.model_copy(update = {"source": "custom"}) for m in custom_models]
-
- # Deduplicate, but always keep custom folder entries (keyed by
- # (id, source)) so they show in the "Custom Folders" UI section
- # even when the model is also in the HF cache.
- deduped: dict[str, LocalModelInfo] = {}
- for model in local_models:
- key = f"{model.id}\x00custom" if model.source == "custom" else model.id
- if key not in deduped:
- deduped[key] = model
-
- models = sorted(
- deduped.values(),
- key = lambda item: (item.updated_at or 0),
- reverse = True,
- )
- models = [m for m in models if not _is_hidden_model(m.id, m.path)]
+ models = collect_local_models(models_root)
return LocalModelListResponse(
models_dir = str(models_root),
diff --git a/studio/backend/routes/training.py b/studio/backend/routes/training.py
index 3818fe9f73..4f131ad2f2 100644
--- a/studio/backend/routes/training.py
+++ b/studio/backend/routes/training.py
@@ -255,6 +255,7 @@ async def start_training(
# Convert request to backend kwargs.
training_kwargs = {
"model_name": request.model_name,
+ "project_name": request.project_name,
"training_type": request.training_type,
"hf_token": request.hf_token or "",
"load_in_4bit": request.load_in_4bit,
diff --git a/studio/backend/run.py b/studio/backend/run.py
index d4cbc26b41..6f36f0928a 100644
--- a/studio/backend/run.py
+++ b/studio/backend/run.py
@@ -933,6 +933,9 @@ def run_server(
"""
global _server, _server_thread, _shutdown_event
+ boot_started = time.perf_counter()
+ logger.info("run_server startup begin api_only=%s host=%s port=%s", api_only, host, port)
+
# Reap every child if the parent dies abnormally (terminal close, Task
# Manager kill, SIGKILL); must run before any child can spawn.
from utils.process_lifetime import initialize_parent_lifetime
@@ -984,7 +987,14 @@ def run_server(
from threading import Thread, Event
import uvicorn
+ import_started = time.perf_counter()
+
from main import app, setup_frontend, _IS_COLAB
+
+ logger.info(
+ "Imported FastAPI app in %.1fms",
+ (time.perf_counter() - import_started) * 1000,
+ )
from utils.paths import ensure_studio_directories
# Allow local stdio MCP servers on a loopback bind (the user's own machine),
@@ -997,6 +1007,11 @@ def run_server(
# Create all standard directories on startup.
ensure_studio_directories()
+ logger.info(
+ "Ensured Studio directories in %.1fms",
+ (time.perf_counter() - boot_started) * 1000,
+ )
+
# Auto-find a free port if the requested one is in use.
if not _is_port_free(host, port):
original_port = port
@@ -1060,6 +1075,11 @@ def run_server(
display_host = _resolve_external_ip() if host == "0.0.0.0" else host
_install_uvicorn_startup_log_rewrite(host, display_host)
+ logger.info(
+ "run_server pre-uvicorn setup completed in %.1fms",
+ (time.perf_counter() - boot_started) * 1000,
+ )
+
ready_event = Event()
startup_failed = Event()
startup_errors = []
@@ -1068,6 +1088,10 @@ def run_server(
async def startup(self, *args, **kwargs):
await super().startup(*args, **kwargs)
if getattr(self, "started", False) and not self.should_exit:
+ logger.info(
+ "Uvicorn startup hook completed in %.1fms",
+ (time.perf_counter() - boot_started) * 1000,
+ )
ready_event.set()
# server_header=False suppresses uvicorn's "Server: uvicorn"; SecurityHeadersMiddleware sets its own.
@@ -1150,6 +1174,11 @@ def run_server(
_shutdown_event.set()
raise
+ logger.info(
+ "run_server uvicorn ready after %.1fms",
+ (time.perf_counter() - boot_started) * 1000,
+ )
+
_write_pid_file()
import atexit
diff --git a/studio/backend/storage/studio_db.py b/studio/backend/storage/studio_db.py
index 7421b42b2f..23b90d7002 100644
--- a/studio/backend/storage/studio_db.py
+++ b/studio/backend/storage/studio_db.py
@@ -23,6 +23,16 @@ from typing import Any, Iterable, Optional
from utils.paths import project_workspaces_root, studio_db_path, ensure_dir
+from utils.training_runs import extract_project_name
+
+
+def _extract_project_name_from_config_json(config_json: Optional[str]) -> Optional[str]:
+ if not config_json:
+ return None
+ try:
+ return extract_project_name(json.loads(config_json))
+ except (json.JSONDecodeError, TypeError):
+ return None
def _denied_path_prefixes() -> list[str]:
@@ -680,6 +690,7 @@ def list_runs(limit: int = 50, offset: int = 0) -> dict:
runs = []
for row in rows:
run = dict(row)
+ run["project_name"] = _extract_project_name_from_config_json(run.get("config_json"))
sparkline = run.get("loss_sparkline")
if sparkline:
try:
@@ -719,6 +730,7 @@ def get_run(id: str) -> Optional[dict]:
if row is None:
return None
run = dict(row)
+ run["project_name"] = _extract_project_name_from_config_json(run.get("config_json"))
sparkline = run.get("loss_sparkline")
if sparkline:
try:
diff --git a/studio/backend/tests/test_checkpoints_scan.py b/studio/backend/tests/test_checkpoints_scan.py
new file mode 100644
index 0000000000..6d473146f5
--- /dev/null
+++ b/studio/backend/tests/test_checkpoints_scan.py
@@ -0,0 +1,256 @@
+# SPDX-License-Identifier: AGPL-3.0-only
+# Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. See /studio/LICENSE.AGPL-3.0
+
+import json
+import sqlite3
+import sys
+import types as _types
+from pathlib import Path
+
+_BACKEND_DIR = str(Path(__file__).resolve().parent.parent)
+if _BACKEND_DIR not in sys.path:
+ sys.path.insert(0, _BACKEND_DIR)
+
+_loggers_stub = _types.ModuleType("loggers")
+_loggers_stub.get_logger = lambda name: __import__("logging").getLogger(name)
+sys.modules.setdefault("loggers", _loggers_stub)
+sys.modules.setdefault("structlog", _types.ModuleType("structlog"))
+
+from utils.models import checkpoints as checkpoints_module
+from utils.training_runs import build_default_output_dir_name
+
+
+def _make_history_connection(db_path: Path) -> sqlite3.Connection:
+ conn = sqlite3.connect(str(db_path))
+ conn.row_factory = sqlite3.Row
+ return conn
+
+
+def _setup_training_runs_table(db_path: Path) -> None:
+ conn = _make_history_connection(db_path)
+ try:
+ conn.execute(
+ """
+ CREATE TABLE training_runs (
+ id TEXT PRIMARY KEY,
+ model_name TEXT NOT NULL,
+ config_json TEXT NOT NULL,
+ output_dir TEXT,
+ started_at TEXT NOT NULL
+ )
+ """
+ )
+ conn.commit()
+ finally:
+ conn.close()
+
+
+def _make_outputs_dir(tmp_path, monkeypatch) -> Path:
+ studio_home = tmp_path / "studio-home"
+ outputs_dir = studio_home / "outputs"
+ outputs_dir.mkdir(parents = True)
+ monkeypatch.setenv("UNSLOTH_STUDIO_HOME", str(studio_home))
+ return outputs_dir
+
+
+def test_scan_checkpoints_uses_output_dir_history_for_base_model(tmp_path, monkeypatch):
+ outputs_dir = _make_outputs_dir(tmp_path, monkeypatch)
+ run_dir = outputs_dir / "custom-run"
+ run_dir.mkdir()
+ (run_dir / "config.json").write_text("{}")
+
+ db_path = tmp_path / "studio.db"
+ _setup_training_runs_table(db_path)
+ conn = _make_history_connection(db_path)
+ try:
+ conn.execute(
+ """
+ INSERT INTO training_runs (id, model_name, config_json, output_dir, started_at)
+ VALUES (?, ?, ?, ?, ?)
+ """,
+ (
+ "run-1",
+ "unsloth/Llama-3.2-3B-Instruct",
+ "{}",
+ str(run_dir.resolve()),
+ "2026-04-09T00:00:00Z",
+ ),
+ )
+ conn.commit()
+ finally:
+ conn.close()
+
+ monkeypatch.setattr(
+ checkpoints_module,
+ "get_connection",
+ lambda: _make_history_connection(db_path),
+ )
+
+ models = checkpoints_module.scan_checkpoints(outputs_dir = str(outputs_dir))
+
+ assert models[0][2]["base_model"] == "unsloth/Llama-3.2-3B-Instruct"
+
+
+def test_scan_checkpoints_matches_project_suffixed_default_dir_against_history(
+ tmp_path, monkeypatch
+):
+ outputs_dir = _make_outputs_dir(tmp_path, monkeypatch)
+ run_name = build_default_output_dir_name(
+ "unsloth/Llama-3.2-3B-Instruct",
+ "Customer Support",
+ timestamp = 1771227800,
+ )
+ run_dir = outputs_dir / run_name
+ run_dir.mkdir()
+ (run_dir / "config.json").write_text("{}")
+
+ db_path = tmp_path / "studio.db"
+ _setup_training_runs_table(db_path)
+ conn = _make_history_connection(db_path)
+ try:
+ conn.execute(
+ """
+ INSERT INTO training_runs (id, model_name, config_json, output_dir, started_at)
+ VALUES (?, ?, ?, ?, ?)
+ """,
+ (
+ "run-2",
+ "unsloth/Llama-3.2-3B-Instruct",
+ json.dumps({"project_name": "Customer Support"}),
+ None,
+ "2026-04-09T00:00:00Z",
+ ),
+ )
+ conn.commit()
+ finally:
+ conn.close()
+
+ monkeypatch.setattr(
+ checkpoints_module,
+ "get_connection",
+ lambda: _make_history_connection(db_path),
+ )
+
+ models = checkpoints_module.scan_checkpoints(outputs_dir = str(outputs_dir))
+
+ assert models[0][2]["base_model"] == "unsloth/Llama-3.2-3B-Instruct"
+
+
+def test_scan_checkpoints_strips_project_suffix_without_history(tmp_path, monkeypatch):
+ outputs_dir = _make_outputs_dir(tmp_path, monkeypatch)
+ run_name = build_default_output_dir_name(
+ "unsloth/Llama-3.2-3B-Instruct",
+ "Customer Support",
+ timestamp = 1771227800,
+ )
+ run_dir = outputs_dir / run_name
+ run_dir.mkdir()
+ (run_dir / "config.json").write_text("{}")
+
+ db_path = tmp_path / "studio.db"
+ _setup_training_runs_table(db_path)
+ monkeypatch.setattr(
+ checkpoints_module,
+ "get_connection",
+ lambda: _make_history_connection(db_path),
+ )
+
+ models = checkpoints_module.scan_checkpoints(outputs_dir = str(outputs_dir))
+
+ assert models[0][2]["base_model"] == "unsloth/Llama-3.2-3B-Instruct"
+
+
+def test_scan_checkpoints_preserves_project_marker_in_model_without_history(tmp_path, monkeypatch):
+ outputs_dir = _make_outputs_dir(tmp_path, monkeypatch)
+ run_name = build_default_output_dir_name(
+ "org/foo__project-bar",
+ timestamp = 1771227800,
+ )
+ run_dir = outputs_dir / run_name
+ run_dir.mkdir()
+ (run_dir / "config.json").write_text("{}")
+
+ db_path = tmp_path / "studio.db"
+ _setup_training_runs_table(db_path)
+ monkeypatch.setattr(
+ checkpoints_module,
+ "get_connection",
+ lambda: _make_history_connection(db_path),
+ )
+
+ models = checkpoints_module.scan_checkpoints(outputs_dir = str(outputs_dir))
+
+ assert models[0][2]["base_model"] == "org/foo__project-bar"
+
+
+def test_scan_checkpoints_preserves_legacy_folder_name_fallback(tmp_path, monkeypatch):
+ outputs_dir = _make_outputs_dir(tmp_path, monkeypatch)
+ run_dir = outputs_dir / "unsloth_Llama-3.2-3B-Instruct_1771227800"
+ run_dir.mkdir()
+ (run_dir / "config.json").write_text("{}")
+
+ db_path = tmp_path / "studio.db"
+ _setup_training_runs_table(db_path)
+ monkeypatch.setattr(
+ checkpoints_module,
+ "get_connection",
+ lambda: _make_history_connection(db_path),
+ )
+
+ models = checkpoints_module.scan_checkpoints(outputs_dir = str(outputs_dir))
+
+ assert models[0][2]["base_model"] == "unsloth/Llama-3.2-3B-Instruct"
+
+
+def test_scan_checkpoints_prefers_exact_history_match_over_newer_suffix(tmp_path, monkeypatch):
+ outputs_dir = _make_outputs_dir(tmp_path, monkeypatch)
+ run_dir = outputs_dir / "unsloth_Test_1771227800"
+ run_dir.mkdir()
+ (run_dir / "config.json").write_text("{}")
+
+ copied_dir = tmp_path / "copied" / run_dir.name
+ copied_dir.mkdir(parents = True)
+
+ db_path = tmp_path / "studio.db"
+ _setup_training_runs_table(db_path)
+ conn = _make_history_connection(db_path)
+ try:
+ conn.execute(
+ """
+ INSERT INTO training_runs (id, model_name, config_json, output_dir, started_at)
+ VALUES (?, ?, ?, ?, ?)
+ """,
+ (
+ "run-exact",
+ "correct/base",
+ "{}",
+ str(run_dir.resolve()),
+ "2026-04-09T00:00:00Z",
+ ),
+ )
+ conn.execute(
+ """
+ INSERT INTO training_runs (id, model_name, config_json, output_dir, started_at)
+ VALUES (?, ?, ?, ?, ?)
+ """,
+ (
+ "run-suffix",
+ "wrong/base",
+ "{}",
+ str(copied_dir.resolve()),
+ "2026-04-10T00:00:00Z",
+ ),
+ )
+ conn.commit()
+ finally:
+ conn.close()
+
+ monkeypatch.setattr(
+ checkpoints_module,
+ "get_connection",
+ lambda: _make_history_connection(db_path),
+ )
+
+ models = checkpoints_module.scan_checkpoints(outputs_dir = str(outputs_dir))
+
+ assert models[0][2]["base_model"] == "correct/base"
diff --git a/studio/backend/tests/test_data_recipe_seed.py b/studio/backend/tests/test_data_recipe_seed.py
index 601df8bbfe..09e22116ed 100644
--- a/studio/backend/tests/test_data_recipe_seed.py
+++ b/studio/backend/tests/test_data_recipe_seed.py
@@ -1,12 +1,126 @@
# SPDX-License-Identifier: AGPL-3.0-only
# Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. See /studio/LICENSE.AGPL-3.0
+import asyncio
+import importlib.util
from pathlib import Path
+import pytest
-def test_seed_inspect_load_kwargs_disables_remote_code_execution():
- seed_route = (
+
+def _seed_route_source() -> str:
+ return (
Path(__file__).resolve().parent.parent / "routes" / "data_recipe" / "seed.py"
).read_text()
- assert '"trust_remote_code": False' in seed_route
+
+def test_seed_inspect_load_kwargs_disables_remote_code_execution():
+ assert '"trust_remote_code": False' in _seed_route_source()
+
+
+class _FakeUpload:
+ def __init__(self, filename: str, content: bytes):
+ self.filename = filename
+ self._content = content
+
+ async def read(self) -> bytes:
+ return self._content
+
+
+def _load_seed_route(monkeypatch: pytest.MonkeyPatch, tmp_path: Path):
+ pytest.importorskip("fastapi")
+ pytest.importorskip("multipart")
+ pytest.importorskip("structlog")
+
+ backend_root = Path(__file__).resolve().parent.parent
+ monkeypatch.syspath_prepend(str(backend_root))
+ route_path = backend_root / "routes" / "data_recipe" / "seed.py"
+ spec = importlib.util.spec_from_file_location("seed_under_test", route_path)
+ assert spec is not None and spec.loader is not None
+ seed_route = importlib.util.module_from_spec(spec)
+ spec.loader.exec_module(seed_route)
+ seed_route.UNSTRUCTURED_UPLOAD_ROOT = tmp_path / "unstructured-uploads"
+ return seed_route
+
+
+def _run_upload(
+ seed_route,
+ filename: str,
+ content: bytes,
+ block_id: str = "block",
+):
+ return asyncio.run(
+ seed_route.upload_unstructured_file(_FakeUpload(filename, content), block_id)
+ )
+
+
+def _block_files(seed_route, block_id: str = "block") -> list[str]:
+ block_dir = seed_route.UNSTRUCTURED_UPLOAD_ROOT / block_id
+ if not block_dir.exists():
+ return []
+ return sorted(path.name for path in block_dir.iterdir())
+
+
+def _raise(exc: BaseException):
+ def raise_exc(*args, **kwargs):
+ raise exc
+
+ return raise_exc
+
+
+@pytest.mark.parametrize(
+ ("filename", "package"),
+ [
+ ("paper.pdf", "pymupdf4llm"),
+ ("notes.docx", "mammoth"),
+ ],
+)
+def test_unstructured_upload_names_missing_extractor_dependency(
+ monkeypatch, tmp_path, filename, package
+):
+ seed_route = _load_seed_route(monkeypatch, tmp_path)
+ monkeypatch.setattr(
+ seed_route,
+ "_extract_text_from_file",
+ _raise(ModuleNotFoundError(f"No module named {package!r}", name = package)),
+ )
+
+ result = _run_upload(seed_route, filename, b"%PDF-1.7")
+
+ assert result.status == "error"
+ assert (
+ result.error
+ == f"Cannot read {Path(filename).suffix} files: the '{package}' package is not installed."
+ )
+ assert _block_files(seed_route) == []
+
+
+def test_unstructured_upload_keeps_txt_path_working(monkeypatch, tmp_path):
+ seed_route = _load_seed_route(monkeypatch, tmp_path)
+
+ result = _run_upload(seed_route, "notes.txt", b"hello")
+
+ assert result.status == "ok"
+ assert result.error is None
+ assert any(name.endswith(".txt") for name in _block_files(seed_route))
+ assert any(name.endswith(".extracted.txt") for name in _block_files(seed_route))
+
+
+@pytest.mark.parametrize(
+ "exc",
+ [
+ ImportError("cannot import internal symbol"),
+ ModuleNotFoundError(
+ "No module named 'missing_transitive_pkg'",
+ name = "missing_transitive_pkg",
+ ),
+ ],
+)
+def test_unstructured_upload_import_errors_stay_generic(monkeypatch, tmp_path, exc):
+ seed_route = _load_seed_route(monkeypatch, tmp_path)
+ monkeypatch.setattr(seed_route, "_extract_text_from_file", _raise(exc))
+ result = _run_upload(seed_route, "paper.pdf", b"%PDF-1.7")
+
+ assert result.status == "error"
+ assert result.error == "Text extraction failed."
+ assert _block_files(seed_route) == []
diff --git a/studio/backend/tests/test_model_ids.py b/studio/backend/tests/test_model_ids.py
new file mode 100644
index 0000000000..f9116afec3
--- /dev/null
+++ b/studio/backend/tests/test_model_ids.py
@@ -0,0 +1,62 @@
+# SPDX-License-Identifier: AGPL-3.0-only
+# Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. See /studio/LICENSE.AGPL-3.0
+
+import sys
+from pathlib import Path
+
+_BACKEND = Path(__file__).resolve().parents[1]
+if str(_BACKEND) not in sys.path:
+ sys.path.insert(0, str(_BACKEND))
+
+from core.inference.model_ids import model_id_matches, public_model_id # noqa: E402
+
+
+def test_local_gguf_path_becomes_clean_stem():
+ assert public_model_id("/srv/models/Qwen3-30B-A3B-Q4_K_M.gguf") == "Qwen3-30B-A3B-Q4_K_M"
+ assert public_model_id("/home/u/.cache/models/llama.gguf") == "llama"
+
+
+def test_hf_repo_id_unchanged():
+ assert public_model_id("unsloth/Qwen3-30B-A3B-GGUF") == "unsloth/Qwen3-30B-A3B-GGUF"
+ assert public_model_id("Qwen3-30B-A3B") == "Qwen3-30B-A3B"
+
+
+def test_none_and_empty_passthrough():
+ assert public_model_id(None) is None
+ assert public_model_id("") == ""
+
+
+def test_windows_path():
+ assert public_model_id("C:\\models\\foo.gguf") == "foo"
+ assert public_model_id("models\\sub\\bar.gguf") == "bar"
+
+
+def test_directory_path_uses_basename():
+ assert public_model_id("/opt/models/MyModelDir") == "MyModelDir"
+ # A 3+ segment relative path is a local path, not an org/model repo id.
+ assert public_model_id("a/b/c") == "c"
+
+
+def test_relative_and_home_paths_are_sanitized():
+ # ./ ../ ~ prefixed paths are local and must not be echoed raw.
+ assert public_model_id("./model.gguf") == "model"
+ assert public_model_id("../models/foo.gguf") == "foo"
+ assert public_model_id("~/models/baz.gguf") == "baz"
+ assert public_model_id("./mistral") == "mistral"
+ assert public_model_id("~/mistral") == "mistral"
+ assert public_model_id(".\\models\\foo.gguf") == "foo"
+
+
+def test_dotted_repo_id_not_mistaken_for_relative_path():
+ # A leading dot that is not ./ or ../ is an ordinary clean name.
+ assert public_model_id(".hidden-model") == ".hidden-model"
+ assert public_model_id("org/.config") == "org/.config"
+
+
+def test_matches_clean_and_legacy():
+ path = "/srv/models/Qwen3-Q4.gguf"
+ assert model_id_matches("Qwen3-Q4", path) # clean public id
+ assert model_id_matches(path, path) # legacy raw path
+ assert not model_id_matches("other", path)
+ assert not model_id_matches(None, path)
+ assert not model_id_matches("x", None)
diff --git a/studio/backend/tests/test_openai_catalog.py b/studio/backend/tests/test_openai_catalog.py
new file mode 100644
index 0000000000..f9baf20a66
--- /dev/null
+++ b/studio/backend/tests/test_openai_catalog.py
@@ -0,0 +1,181 @@
+# SPDX-License-Identifier: AGPL-3.0-only
+# Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. See /studio/LICENSE.AGPL-3.0
+
+"""GET /v1/models lists the full server catalog (loaded + locally available)."""
+
+import asyncio
+import json
+import sys
+from pathlib import Path
+
+_BACKEND = Path(__file__).resolve().parents[1]
+if str(_BACKEND) not in sys.path:
+ sys.path.insert(0, str(_BACKEND))
+
+import routes.inference as inf # noqa: E402
+
+
+class _Info:
+ def __init__(
+ self,
+ id,
+ display_name,
+ model_id = None,
+ ):
+ self.id = id
+ self.display_name = display_name
+ self.model_id = model_id
+
+
+class _FakeLlama:
+ is_loaded = True
+ model_identifier = "/srv/models/Qwen3-Q4.gguf"
+ context_length = 4096
+ max_context_length = None
+ native_context_length = None
+
+ def __init__(self, loaded = True):
+ self.is_loaded = loaded
+
+
+class _FakeUnsloth:
+ active_model_name = None
+ models: dict = {}
+ context_length = None
+ max_seq_length = None
+
+
+def test_catalog_lists_loaded_and_available(monkeypatch):
+ monkeypatch.setattr(inf, "get_llama_cpp_backend", lambda: _FakeLlama())
+ monkeypatch.setattr(inf, "get_inference_backend", lambda: _FakeUnsloth())
+
+ async def _fake_catalog():
+ return [
+ _Info("/data/models/Qwen3-Q4.gguf", "Qwen3-Q4"), # same as loaded -> dedup
+ _Info("/data/models/Llama-8B-Q8.gguf", "Llama-8B-Q8"), # available, not loaded
+ _Info("models--org--Foo", "Foo", model_id = "org/Foo"), # hf cache repo id
+ ]
+
+ monkeypatch.setattr(inf, "_cached_local_catalog", _fake_catalog)
+
+ data = asyncio.run(inf._openai_catalog_objects())
+ ids = {m["id"]: m for m in data}
+
+ # Loaded model is present, marked loaded, and keeps context fields.
+ assert ids["Qwen3-Q4"]["loaded"] is True
+ assert ids["Qwen3-Q4"]["context_length"] == 4096
+ # Available-but-not-loaded models are listed too.
+ assert ids["Llama-8B-Q8"]["loaded"] is False
+ assert ids["org/Foo"]["loaded"] is False
+ # The loaded gguf and the on-disk copy collapse to one clean id.
+ assert [m["id"] for m in data].count("Qwen3-Q4") == 1
+ # No absolute paths or .gguf suffixes leak anywhere.
+ blob = json.dumps(data)
+ assert ".gguf" not in blob
+ assert "/srv/" not in blob
+ assert "/data/" not in blob
+
+
+def test_empty_and_errored_scans_are_cached(monkeypatch):
+ # Cache validity is keyed on the timestamp, not list contents, so an empty
+ # (fresh install / no local models) or errored scan is still cached for the
+ # TTL instead of rescanning the filesystem on every /v1/models poll.
+ import routes.models as models_mod
+ for outcome in ("empty", "error"):
+ calls = {"n": 0}
+
+ def _scan(_root, _outcome = outcome):
+ calls["n"] += 1
+ if _outcome == "error":
+ raise RuntimeError("scan blew up")
+ return []
+
+ monkeypatch.setattr(models_mod, "collect_local_models", _scan)
+ monkeypatch.setattr(inf, "_CATALOG_CACHE", {"at": 0.0, "models": []})
+
+ async def _run():
+ return [await inf._cached_local_catalog() for _ in range(3)]
+
+ results = asyncio.run(_run())
+ assert results == [[], [], []], outcome
+ assert calls["n"] == 1, f"{outcome} scan ran {calls['n']}x (TTL not honored)"
+
+
+def test_catalog_ttl_starts_after_scan_completes(monkeypatch):
+ # The cache timestamp must be taken AFTER the scan, not before it. A scan that
+ # outlives the TTL would otherwise leave the cache born-expired, so the next
+ # caller rescans instead of reusing the just-computed catalog.
+ import routes.models as models_mod
+
+ clock = {"t": 1000.0}
+ monkeypatch.setattr(inf.time, "monotonic", lambda: clock["t"])
+ monkeypatch.setattr(inf, "_CATALOG_CACHE", {"at": 0.0, "models": []})
+
+ calls = {"n": 0}
+
+ def _slow_scan(_root):
+ calls["n"] += 1
+ clock["t"] += inf._CATALOG_TTL_S + 10 # the scan itself outlives the TTL
+ return [_Info("/m/A.gguf", "A")]
+
+ monkeypatch.setattr(models_mod, "collect_local_models", _slow_scan)
+
+ async def _run():
+ first = await inf._cached_local_catalog()
+ second = await inf._cached_local_catalog() # clock unchanged since scan end
+ return first, second
+
+ first, second = asyncio.run(_run())
+ assert [i.id for i in first] == ["/m/A.gguf"]
+ assert calls["n"] == 1, "TTL started before the scan -> cache born expired, rescanned"
+
+
+def test_retrieve_loaded_model_skips_catalog_scan(monkeypatch):
+ # Retrieving a loaded id must resolve from the loaded set alone, never paying
+ # for the filesystem scan that _cached_local_catalog drives.
+ monkeypatch.setattr(inf, "get_llama_cpp_backend", lambda: _FakeLlama())
+ monkeypatch.setattr(inf, "get_inference_backend", lambda: _FakeUnsloth())
+
+ async def _boom():
+ raise AssertionError("catalog scan must not run for a loaded id")
+
+ monkeypatch.setattr(inf, "_cached_local_catalog", _boom)
+
+ model = asyncio.run(inf.openai_retrieve_model("Qwen3-Q4", current_subject = "t"))
+ assert model["id"] == "Qwen3-Q4"
+ assert model["loaded"] is True
+
+
+def test_cached_local_catalog_offloads_and_caches(monkeypatch):
+ # The filesystem scan must run off the event loop (asyncio.to_thread) and be
+ # cached, so a burst of /v1/models calls does not re-scan or block.
+ calls = {"scan": 0, "threaded": 0}
+
+ def _fake_collect(_root):
+ calls["scan"] += 1
+ return [_Info("/data/models/A.gguf", "A")]
+
+ import routes.models as models_mod
+
+ monkeypatch.setattr(models_mod, "collect_local_models", _fake_collect)
+
+ real_to_thread = inf.asyncio.to_thread
+
+ async def _counting_to_thread(fn, *a, **k):
+ calls["threaded"] += 1
+ return await real_to_thread(fn, *a, **k)
+
+ monkeypatch.setattr(inf.asyncio, "to_thread", _counting_to_thread)
+ # Fresh cache for a deterministic count.
+ monkeypatch.setattr(inf, "_CATALOG_CACHE", {"at": 0.0, "models": []})
+
+ async def _run():
+ first = await inf._cached_local_catalog()
+ second = await inf._cached_local_catalog() # within TTL -> cached
+ return first, second
+
+ first, second = asyncio.run(_run())
+ assert [i.id for i in first] == ["/data/models/A.gguf"]
+ assert second is first or [i.id for i in second] == [i.id for i in first]
+ assert calls["scan"] == 1 # cached: scanned once for two calls
+ assert calls["threaded"] == 1 # offloaded to a worker thread
diff --git a/studio/backend/tests/test_openai_models_path_leak.py b/studio/backend/tests/test_openai_models_path_leak.py
new file mode 100644
index 0000000000..a84a33f840
--- /dev/null
+++ b/studio/backend/tests/test_openai_models_path_leak.py
@@ -0,0 +1,45 @@
+# SPDX-License-Identifier: AGPL-3.0-only
+# Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. See /studio/LICENSE.AGPL-3.0
+
+"""GET /v1/models must report a clean public id, never the on-disk .gguf path."""
+
+import json
+import sys
+from pathlib import Path
+
+_BACKEND = Path(__file__).resolve().parents[1]
+if str(_BACKEND) not in sys.path:
+ sys.path.insert(0, str(_BACKEND))
+
+import routes.inference as inf # noqa: E402
+
+
+class _FakeLlama:
+ is_loaded = True
+ model_identifier = "/srv/models/Qwen3-30B-A3B-Q4_K_M.gguf"
+ context_length = 4096
+ max_context_length = None
+ native_context_length = None
+
+
+class _FakeUnsloth:
+ active_model_name = None
+ models: dict = {}
+ context_length = None
+ max_seq_length = None
+
+
+def test_openai_models_returns_clean_id_without_path(monkeypatch):
+ monkeypatch.setattr(inf, "get_llama_cpp_backend", lambda: _FakeLlama())
+ monkeypatch.setattr(inf, "get_inference_backend", lambda: _FakeUnsloth())
+
+ objs = inf._openai_model_objects()
+
+ assert len(objs) == 1
+ assert objs[0]["id"] == "Qwen3-30B-A3B-Q4_K_M"
+ # The serialized payload must not leak the absolute path or the .gguf suffix.
+ blob = json.dumps(objs)
+ assert "/srv/models" not in blob
+ assert ".gguf" not in blob
+ # Context fields still flow through.
+ assert objs[0]["context_length"] == 4096
diff --git a/studio/backend/tests/test_tp_vision_regression.py b/studio/backend/tests/test_tp_vision_regression.py
new file mode 100644
index 0000000000..09af876da6
--- /dev/null
+++ b/studio/backend/tests/test_tp_vision_regression.py
@@ -0,0 +1,805 @@
+# SPDX-License-Identifier: AGPL-3.0-only
+# Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. See /studio/LICENSE.AGPL-3.0
+
+"""Regression guards for silent tensor-parallel downgrades in load_model.
+
+PR #6416 blanket-disabled tensor parallelism for vision models to dodge a
+--split-mode tensor + --mmproj GGML_ASSERT (#6415), which silently single-GPU'd
+any mmproj/MTP GGUF that fit on one card. The fix makes the skip self-healing:
+tensor is tried by default and recorded per (binary, model) only on a real abort.
+
+load_model is too entangled to drive end-to-end, so these tests inspect the
+source / drive the pure helpers. The headline test pins the set of TP-drop
+conditions, so a new silent drop fails CI. No GPU; fully deterministic.
+"""
+
+from __future__ import annotations
+
+import ast
+import importlib.util
+import inspect
+import os
+import sys
+import textwrap
+import types as _types
+from pathlib import Path
+
+_BACKEND_DIR = str(Path(__file__).resolve().parent.parent)
+if _BACKEND_DIR not in sys.path:
+ sys.path.insert(0, _BACKEND_DIR)
+
+# External-dep stubs so importing the backend doesn't require structlog / httpx /
+# loggers -- but only when the real module is missing, so a lightweight stub never
+# shadows the real package (or `loggers.handlers` submodule) for tests collected
+# later in the same pytest process.
+try:
+ import structlog # noqa: F401
+except ImportError:
+ _structlog_stub = _types.ModuleType("structlog")
+ _structlog_stub.get_logger = lambda *a, **k: __import__("logging").getLogger("stub")
+ sys.modules["structlog"] = _structlog_stub
+try:
+ import loggers # noqa: F401
+except ImportError:
+ _loggers_stub = _types.ModuleType("loggers")
+ _loggers_stub.get_logger = lambda name: __import__("logging").getLogger(name)
+ sys.modules["loggers"] = _loggers_stub
+try:
+ import httpx as _httpx_real # noqa: F401
+except ImportError:
+ _httpx_stub = _types.ModuleType("httpx")
+ for _exc in (
+ "ConnectError",
+ "TimeoutException",
+ "ReadTimeout",
+ "ReadError",
+ "RemoteProtocolError",
+ "CloseError",
+ "HTTPError",
+ "RequestError",
+ ):
+ setattr(_httpx_stub, _exc, type(_exc, (Exception,), {}))
+ _httpx_stub.Timeout = type("T", (), {"__init__": lambda s, *a, **k: None})
+ _httpx_stub.Response = type("Response", (), {})
+ _httpx_stub.Client = type(
+ "C",
+ (),
+ {
+ "__init__": lambda s, **kw: None,
+ "__enter__": lambda s: s,
+ "__exit__": lambda s, *a: None,
+ },
+ )
+ sys.modules["httpx"] = _httpx_stub
+
+from core.inference.llama_cpp import LlamaCppBackend # noqa: E402
+
+_GB = 1024**3
+
+
+def _load_inference_routes_module():
+ """Load routes/inference.py directly, bypassing routes/__init__.py (which imports
+ every router, dragging in unrelated deps like python-multipart) (Codex #6659)."""
+ route_path = Path(_BACKEND_DIR) / "routes" / "inference.py"
+ spec = importlib.util.spec_from_file_location(
+ "tp_vision_regression_inference_routes", route_path
+ )
+ assert spec is not None and spec.loader is not None
+ module = importlib.util.module_from_spec(spec)
+ sys.modules[spec.name] = module
+ spec.loader.exec_module(module)
+ return module
+
+
+def _load_model_ast() -> ast.FunctionDef:
+ """Parse load_model into an AST FunctionDef (no import side effects)."""
+ src = textwrap.dedent(inspect.getsource(LlamaCppBackend.load_model))
+ return ast.parse(src).body[0]
+
+
+def _tensor_parallel_false_drop_guards() -> list[str]:
+ """Source of the guard expression for every `if ...: tensor_parallel = False`
+ (the LOCAL variable, not self._tensor_parallel) inside load_model."""
+ fn = _load_model_ast()
+
+ def _body_drops_tp(body) -> bool:
+ for n in body:
+ if (
+ isinstance(n, ast.Assign)
+ and any(isinstance(t, ast.Name) and t.id == "tensor_parallel" for t in n.targets)
+ and isinstance(n.value, ast.Constant)
+ and n.value.value is False
+ ):
+ return True
+ return False
+
+ return [
+ ast.unparse(node.test)
+ for node in ast.walk(fn)
+ if isinstance(node, ast.If) and _body_drops_tp(node.body)
+ ]
+
+
+# Every condition that may flip a requested tensor_parallel back to False. Adding
+# one must be conscious: update this allowlist and keep multi-GPU where possible.
+_ALLOWED_TP_DROP_GUARDS = {
+ # Capability: --split-mode tensor aborted for this (binary, model) (#6415).
+ # Self-healing -- tried by default, skipped only after a real abort (vs #6416).
+ "tensor_parallel and self._tensor_split_aborts(binary, model_identifier)",
+ # Capacity: tensor needs >= 2 GPUs clearing the compute-buffer reserve.
+ "tensor_parallel and len(tp_gpus) < 2",
+ # Capacity: pooled usable VRAM can't hold weights + MTP reserve -> layer split.
+ "_tp_weight_budget_mib <= _tp_required_mib",
+}
+
+
+def test_tensor_parallel_drop_sites_match_allowlist():
+ """The set of reasons a requested TP can be dropped is fixed and reviewed: a new
+ drop site fails this set-equality until consciously allowlisted (would catch #6416)."""
+ found = set(_tensor_parallel_false_drop_guards())
+ assert found == _ALLOWED_TP_DROP_GUARDS, (
+ "tensor_parallel drop sites changed.\n"
+ f" unexpected (new) : {sorted(found - _ALLOWED_TP_DROP_GUARDS)}\n"
+ f" missing (removed): {sorted(_ALLOWED_TP_DROP_GUARDS - found)}\n"
+ "A new drop means a user's TP request is ignored for a new reason -- "
+ "review it, keep multi-GPU where possible, surface it, then update "
+ "_ALLOWED_TP_DROP_GUARDS."
+ )
+
+
+def test_every_tp_drop_is_logged_not_silent():
+ """Each tensor_parallel downgrade must log why, so it never disappears silently."""
+ fn = _load_model_ast()
+
+ def _body_drops_tp(body):
+ return any(
+ isinstance(n, ast.Assign)
+ and any(isinstance(t, ast.Name) and t.id == "tensor_parallel" for t in n.targets)
+ and isinstance(n.value, ast.Constant)
+ and n.value.value is False
+ for n in body
+ )
+
+ def _body_logs(body) -> bool:
+ for n in ast.walk(ast.Module(body = list(body), type_ignores = [])):
+ if (
+ isinstance(n, ast.Call)
+ and isinstance(n.func, ast.Attribute)
+ and isinstance(n.func.value, ast.Name)
+ and n.func.value.id == "logger"
+ ):
+ return True
+ return False
+
+ for node in ast.walk(fn):
+ if isinstance(node, ast.If) and _body_drops_tp(node.body):
+ assert _body_logs(node.body), (
+ f"TP drop under `{ast.unparse(node.test)}` has no logger call -- "
+ "downgrades must explain themselves."
+ )
+
+
+def test_tensor_split_gate_is_self_healing_not_blanket():
+ """Skip is conditional on a recorded (binary, model) abort, not a blanket
+ is_vision disable (the #6416 regression)."""
+ src = inspect.getsource(LlamaCppBackend.load_model)
+ assert "self._tensor_split_aborts(binary, model_identifier)" in src
+ assert "if tensor_parallel and is_vision:" not in src
+ assert "if tensor_parallel and effective_is_vision:" not in src
+
+
+def test_tensor_split_skip_documents_layer_split_fallback():
+ """When the skip fires (known-bad binary+model), it states the fallback."""
+ src = inspect.getsource(LlamaCppBackend.load_model)
+ gate = src.find("self._tensor_split_aborts(binary, model_identifier)")
+ assert gate != -1
+ block = src[gate : gate + 600]
+ assert "layer split" in block, "the skip should state it falls back to layer split"
+
+
+def test_tensor_split_abort_recorded_early_on_first_spawn():
+ """Recorded on the first spawn showing the marker, before the flash-attn-off
+ retry (which can't run tensor so drops the marker) -- else it loops (oobabooga, #6659)."""
+ src = inspect.getsource(LlamaCppBackend.load_model)
+ idx = src.find("_record_tensor_split_abort(binary, model_identifier)")
+ assert idx != -1, "load_model must record a (binary, model) tensor-split abort"
+ guard = src[max(0, idx - 600) : idx]
+ assert "self._tensor_parallel" in guard
+ assert (
+ "_should_record_tensor_split_abort" in guard
+ ), "record must be gated on the marker-plus-hard-crash decision helper"
+ # Recorded before the flash-attn-off retry, not after the full ladder.
+ fa_off = src.find("_with_flash_attn_off")
+ assert 0 <= idx < fa_off, "recording must latch on the first spawn, before flash-off"
+
+
+def test_vision_downgrade_preserves_multi_gpu_intent():
+ """The vision downgrade raises _layer_min_gpus and threads it into both the
+ _select_gpus and auto-context layer paths, so a fitting model still spreads."""
+ src = inspect.getsource(LlamaCppBackend.load_model)
+ assert "_layer_min_gpus = max(_layer_min_gpus, len(gpus))" in src
+ assert src.count("min_gpus = _layer_min_gpus") >= 2
+ assert "range(_auto_min_gpus, len(ranked) + 1)" in src
+ auto = src.find("_auto_min_gpus = max(")
+ assert auto != -1 and "_layer_min_gpus" in src[auto : auto + 200]
+
+
+# ── per-binary capability cache (pure) ───────────────────────────────
+
+
+def test_tensor_attempted_by_default_for_unknown_binary():
+ """A (binary, model) not seen to abort -> tensor is attempted (not skipped)."""
+ assert LlamaCppBackend._tensor_split_aborts("/never/seen/llama-server", "m") is False
+ assert LlamaCppBackend._tensor_split_aborts(None, "m") is False
+ assert LlamaCppBackend._tensor_split_aborts("/x", None) is False
+
+
+def test_recorded_tensor_abort_is_per_model():
+ """A recorded (binary, model) abort trips the gate for that model only -- a
+ different model on the same binary still attempts tensor (oobabooga, #6659)."""
+ b = f"/tmp/llama-server-{id(object())}"
+ try:
+ assert LlamaCppBackend._tensor_split_aborts(b, "model-a") is False
+ LlamaCppBackend._record_tensor_split_abort(b, "model-a")
+ assert LlamaCppBackend._tensor_split_aborts(b, "model-a") is True
+ # a different model on the same binary is unaffected
+ assert LlamaCppBackend._tensor_split_aborts(b, "model-b") is False
+ finally:
+ LlamaCppBackend._tensor_split_abort_keys.discard(
+ LlamaCppBackend._tensor_split_cache_key(b, "model-a")
+ )
+
+
+# ── _select_gpus: single-GPU collapse vs honored multi-GPU intent (pure) ──
+
+
+def test_select_gpus_collapses_to_single_gpu_when_model_fits():
+ """Default (min_gpus=1): a 39 GB model on four 183 GB GPUs pins ONE GPU -- the
+ 'single GPU' symptom once TP drops, and why the downgrade needs min_gpus."""
+ gpus = [(0, 180000), (1, 180000), (2, 180000), (3, 180000)] # (idx, free MiB)
+ gpu_indices, _use_fit = LlamaCppBackend._select_gpus(int(39 * _GB), gpus)
+ assert gpu_indices is not None and len(gpu_indices) == 1
+
+
+def test_select_gpus_min_gpus_keeps_multi_gpu_for_fitting_model():
+ """min_gpus>=2 must NOT collapse to one GPU for a model that fits on one."""
+ gpus = [(0, 180000), (1, 180000), (2, 180000), (3, 180000)]
+ gpu_indices, _ = LlamaCppBackend._select_gpus(int(39 * _GB), gpus, min_gpus = 2)
+ assert gpu_indices is not None and len(gpu_indices) >= 2
+
+
+def test_select_gpus_min_gpus_capped_to_available():
+ """min_gpus larger than the GPU count is capped, not an error."""
+ gpus = [(0, 180000), (1, 180000)]
+ gi, _ = LlamaCppBackend._select_gpus(int(10 * _GB), gpus, min_gpus = 8)
+ assert gi is not None and len(gi) == 2
+
+
+def test_select_gpus_uses_multiple_gpus_when_model_does_not_fit():
+ """Sanity: selection spreads across GPUs when one card can't hold the model."""
+ gpus = [(0, 40000), (1, 40000), (2, 40000), (3, 40000)] # 40 GB free each
+ gpu_indices, _use_fit = LlamaCppBackend._select_gpus(int(120 * _GB), gpus)
+ assert gpu_indices is not None and len(gpu_indices) >= 2
+
+
+def test_select_gpus_min_gpus_excludes_unusable_gpu():
+ """min_gpus caps to usable cards: 2 free + 1 nearly-full -> 2-GPU split, not
+ forcing the full card (OOM) or tripping --fit (#6659)."""
+ gpus = [(0, 180000), (1, 180000), (2, 500)] # GPU 2 is nearly full
+ total = {0: 180000, 1: 180000, 2: 180000}
+ gi, _ = LlamaCppBackend._select_gpus(
+ int(39 * _GB),
+ gpus,
+ min_gpus = 3,
+ total_by_idx = total,
+ per_device_overhead_bytes = int(1 * _GB),
+ )
+ assert gi is not None
+ assert 2 not in gi, "a nearly-full GPU must not be forced in to satisfy min_gpus"
+ assert len(gi) == 2
+
+
+def test_tensor_abort_cache_invalidated_on_binary_mtime_change(tmp_path):
+ """Cache keys on (path, mtime, model), so a binary swapped in place (in-app
+ update, no restart) is re-probed instead of inheriting the old abort (#6659)."""
+ binp = tmp_path / "llama-server"
+ binp.write_text("v1")
+ p = str(binp)
+ try:
+ LlamaCppBackend._record_tensor_split_abort(p, "m")
+ assert LlamaCppBackend._tensor_split_aborts(p, "m") is True
+ # Simulate an in-place update bumping the binary's mtime.
+ st = binp.stat()
+ os.utime(p, (st.st_atime, st.st_mtime + 10))
+ assert (
+ LlamaCppBackend._tensor_split_aborts(p, "m") is False
+ ), "a binary swapped in place (new mtime) must be re-probed"
+ # A same-second replacement (sub-second mtime bump) must also re-probe:
+ # second-resolution mtime would inherit the stale abort (reviewer.py P2).
+ sec_ns = (binp.stat().st_mtime_ns // 1_000_000_000) * 1_000_000_000
+ os.utime(p, ns = (sec_ns, sec_ns))
+ LlamaCppBackend._record_tensor_split_abort(p, "m")
+ binp.write_text("v2")
+ os.utime(p, ns = (sec_ns, sec_ns + 1))
+ assert (
+ LlamaCppBackend._tensor_split_aborts(p, "m") is False
+ ), "a same-second in-place swap (ns mtime bump) must be re-probed"
+ finally:
+ for key in list(LlamaCppBackend._tensor_split_abort_keys):
+ if key and key[0] == p:
+ LlamaCppBackend._tensor_split_abort_keys.discard(key)
+
+
+def test_tensor_split_abort_raises_early_to_layer_fallback():
+ """The first-spawn abort raises to the route's layer fallback (not the text-only
+ mmproj strip), before the flash-attn-off retry, preserving the projector (#6659)."""
+ src = inspect.getsource(LlamaCppBackend.load_model)
+ raise_idx = src.find("(split-axis geometry); retrying with layer split")
+ assert raise_idx != -1, "the split-axis abort must raise to trigger a layer retry"
+ # raises before both the flash-attn-off retry and the text-only mmproj strip
+ assert raise_idx < src.find("_with_flash_attn_off")
+ assert raise_idx < src.find("_strip_mmproj_args(_last_spawn_cmd)")
+ # gated on the marker-plus-crash helper, which also drives the record just above
+ guard = src[max(0, raise_idx - 600) : raise_idx]
+ assert "_should_record_tensor_split_abort" in guard
+ rec_idx = src.find("_record_tensor_split_abort(binary, model_identifier)")
+ assert rec_idx != -1 and rec_idx < raise_idx
+
+
+def test_budget_downgrade_preserves_multi_gpu_intent():
+ """The pooled-VRAM downgrade raises _layer_min_gpus from the usable tensor GPUs
+ too, symmetric with the vision downgrade (reviewer.py asymmetric fix, #6659)."""
+ src = inspect.getsource(LlamaCppBackend.load_model)
+ budget = src.find("_tp_weight_budget_mib <= _tp_required_mib")
+ assert budget != -1
+ block = src[budget : budget + 1000]
+ assert "tensor_parallel = False" in block
+ assert (
+ "_layer_min_gpus = max(_layer_min_gpus, len(tp_gpus))" in block
+ ), "the budget downgrade must preserve multi-GPU intent like the vision gate"
+
+
+def test_compute_buffer_downgrade_preserves_multi_gpu_intent():
+ """The len(tp_gpus) < 2 compute-buffer downgrade raises _layer_min_gpus from the
+ full GPU set too, so it is symmetric with the budget/geometry downgrades and
+ doesn't collapse a multi-GPU layer load to one card (reviewer.py P1 on #6659)."""
+ src = inspect.getsource(LlamaCppBackend.load_model)
+ gate = src.find("tensor_parallel and len(tp_gpus) < 2")
+ assert gate != -1
+ # Bound to exactly this block: from its gate to the next (budget) downgrade.
+ nxt = src.find("_tp_weight_budget_mib <= _tp_required_mib", gate)
+ assert nxt != -1
+ block = src[gate:nxt]
+ assert "tensor_parallel = False" in block
+ assert (
+ "_layer_min_gpus = max(_layer_min_gpus, len(gpus))" in block
+ ), "the compute-buffer downgrade must preserve multi-GPU intent like the others"
+
+
+def test_tensor_split_layer_min_gpus_bump_requires_tensor_request():
+ """Every guard that bumps _layer_min_gpus off the abort cache also tests
+ tensor_parallel, so a non-tensor load on a known-bad binary doesn't grab every
+ GPU for a fitting model (#6659)."""
+ fn = _load_model_ast()
+ checked = 0
+ for node in ast.walk(fn):
+ if isinstance(node, ast.If):
+ test_src = ast.unparse(node.test)
+ if "self._tensor_split_aborts(binary, model_identifier)" not in test_src:
+ continue
+ body = "\n".join(ast.unparse(n) for n in node.body)
+ if "_layer_min_gpus" in body:
+ checked += 1
+ assert "tensor_parallel" in test_src, (
+ "the cached _layer_min_gpus bump must require a current tensor "
+ f"request, but fires under `{test_src}`"
+ )
+ assert checked >= 1, "expected an abort-cache guard that bumps _layer_min_gpus"
+
+
+# ── round-2 follow-up: route-fallback retry + auto-context cap + assert marker ──
+
+
+def test_layer_fallback_retry_preserves_multi_gpu_intent():
+ """load_model takes a preserve_multi_gpu_on_layer hint and raises _layer_min_gpus
+ for it, so the tensor-off fallback retry still spreads a fitting model (#6659)."""
+ sig = inspect.signature(LlamaCppBackend.load_model)
+ assert "preserve_multi_gpu_on_layer" in sig.parameters
+ assert sig.parameters["preserve_multi_gpu_on_layer"].default is False
+ fn = _load_model_ast()
+ found = any(
+ isinstance(n, ast.If)
+ and "preserve_multi_gpu_on_layer" in ast.unparse(n.test)
+ and "_layer_min_gpus" in "\n".join(ast.unparse(b) for b in n.body)
+ for n in ast.walk(fn)
+ )
+ assert found, "preserve_multi_gpu_on_layer must raise _layer_min_gpus"
+
+
+def test_auto_context_layer_loops_capped_to_usable_gpus():
+ """The auto-context loops bypass _select_gpus, so they apply its cap: a card
+ counts only if usable VRAM clears the per-device layer overhead (#6659)."""
+ src = inspect.getsource(LlamaCppBackend.load_model)
+ assert (
+ "range(max(1, _layer_min_gpus), len(ranked) + 1)" not in src
+ ), "auto-context loops must cap _layer_min_gpus to usable GPUs, not use it raw"
+ assert "_auto_min_gpus" in src
+ assert "range(_auto_min_gpus, len(ranked) + 1)" in src
+ # the eligibility threshold is the per-device layer overhead, not bare > 0
+ auto = src.find("_auto_min_gpus = max(")
+ assert auto != -1
+ block = src[auto : auto + 400]
+ assert "_pipeline_overhead_mib" in block, (
+ "a card must clear the per-device layer overhead to count, mirroring "
+ "_select_gpus, so a nearly-full GPU is not exposed and OOMs"
+ )
+
+
+def test_fallback_hint_uses_effective_tensor_request_not_just_toggle():
+ """Tensor intent keys off _effective_tensor_parallel (toggle + extras + env), not
+ just the toggle, so extra/env-driven tensor users keep multi-GPU (#6659)."""
+ route = Path(_BACKEND_DIR) / "routes" / "inference.py"
+ src = route.read_text()
+ idx = src.find("_tensor_intent_overall = _effective_tensor_parallel(")
+ assert idx != -1, "the GGUF load closure must compute tensor intent"
+ block = src[idx : idx + 300]
+ assert "extra_llama_args, request.tensor_parallel" in block
+ pres = src.find("preserve_multi_gpu_on_layer = bool(")
+ assert (
+ "_effective_tensor_parallel(attempt_extra_args, tensor_parallel)" in src[pres : pres + 200]
+ )
+ # not the toggle-only form this replaced
+ assert (
+ "bool(\n request.tensor_parallel and not tensor_parallel" not in src
+ )
+
+
+def test_carry_preserved_tensor_intent_truth_table():
+ """Behavioral check of the carry-forward decision: carried only for the SAME
+ model, preserved, and not an explicit drop. Catches a `not` inversion (ctx-only
+ collapse) and a missing same-model guard (cross-model leak) (#6659)."""
+ inference_routes = _load_inference_routes_module()
+ f = inference_routes._carry_preserved_tensor_intent
+ assert f(preserved = True, same_model = True, explicit_drop = False) is True
+ assert f(preserved = True, same_model = True, explicit_drop = True) is False # explicit drop
+ assert f(preserved = True, same_model = False, explicit_drop = False) is False # model switch
+ assert f(preserved = False, same_model = True, explicit_drop = False) is False # not a fallback
+
+
+def test_preserved_fallback_carried_across_non_drop_reload():
+ """The hint carries the preserved fallback via _carry_preserved_tensor_intent,
+ gated on the same model loaded, so a ctx-only reload keeps multi-GPU but a model
+ switch / explicit drop doesn't inherit it (#6659)."""
+ route = Path(_BACKEND_DIR) / "routes" / "inference.py"
+ src = route.read_text()
+ idx = src.find("_tensor_intent_overall = _effective_tensor_parallel(")
+ assert idx != -1
+ block = src[idx : idx + 400]
+ assert "_carry_preserved_tensor_intent(" in block
+ assert "preserved = llama_backend.layer_preserves_tensor_intent" in block
+ assert "same_model = _same_model_loaded" in block
+ assert "explicit_drop = _explicit_tensor_drop" in block
+
+
+def test_same_model_guard_checks_path_and_variant():
+ """The same-model guard matches the resolved config.identifier (what load_model
+ stores, after from_identifier normalizes shorthands) -- not the raw request id --
+ and also matches the loaded quant by path (local multi-variant dir) else variant (HF
+ repo), so a reload keeps the carry-forward and a different variant doesn't inherit
+ the prior one's preserved tensor intent (#6659)."""
+ route = Path(_BACKEND_DIR) / "routes" / "inference.py"
+ src = route.read_text()
+ idx = src.find("_same_model_loaded = (")
+ assert idx != -1
+ block = src[idx : idx + 1300]
+ # Identity compares the normalized config.identifier, not the raw model_identifier.
+ head = src[idx : idx + 200]
+ assert "config.identifier" in head and "== (model_identifier" not in head
+ assert "llama_backend.gguf_path" in block and "config.gguf_file" in block
+ assert "llama_backend.hf_variant" in block and "config.gguf_variant" in block
+
+
+def test_diffusion_load_clears_preserved_tensor_flag():
+ """The diffusion early-return path (skips the command builder) clears the
+ preserved-fallback flag, so a prior tensor fallback doesn't churn it (#6659)."""
+ src = inspect.getsource(LlamaCppBackend.load_model)
+ diff = src.find("if self._is_diffusion:")
+ assert diff != -1
+ start = src.find("return self._start_diffusion_server", diff)
+ assert start != -1
+ assert "self._layer_preserves_tensor_intent = False" in src[diff:start]
+
+
+def test_is_tensor_split_assert_marker():
+ """Matches the specific #6415 split-axis assert, not any ggml assert/abort, so
+ an unrelated invariant a corrupt GGUF/projector trips isn't cached (#6659)."""
+ f = LlamaCppBackend._is_tensor_split_assert
+ # the real #6415 warmup assert (split-axis enum, in ggml-backend-meta)
+ assert (
+ f(
+ "ggml-backend-meta.cpp:541: GGML_ASSERT(src_ss[0].axis != "
+ "GGML_BACKEND_SPLIT_AXIS_0) failed"
+ )
+ is True
+ )
+ # the split-axis token alone (file path elided / reworded) still matches
+ assert f("GGML_ASSERT(x.axis != GGML_BACKEND_SPLIT_AXIS_1) failed") is True
+ # UNRELATED asserts must NOT match -- including a different invariant from the
+ # same multi-assert source file (matched on the token, not the file name).
+ assert f("ggml-backend-meta.cpp:99: GGML_ASSERT(buf != NULL) failed") is False
+ assert f("/x/ggml.c:1234: GGML_ASSERT(ne == 1) failed") is False
+ assert f("ggml_abort: something else entirely") is False
+ assert f("Segmentation fault (core dumped)") is False
+ assert f("") is False
+ assert f(None) is False
+
+
+def test_layer_preserve_hint_replayed_on_respawn():
+ """The preserve hint is in the replay snapshot (_pending_load_kwargs), so a
+ respawn keeps the downgraded model multi-GPU (Codex review on #6659)."""
+ src = inspect.getsource(LlamaCppBackend.load_model)
+ pend = src.find("_pending_load_kwargs = {")
+ assert pend != -1
+ block = src[pend : src.find("}", pend) + 1]
+ assert '"preserve_multi_gpu_on_layer": preserve_multi_gpu_on_layer' in block, (
+ "the layer-preserve hint must be in the replay snapshot so _respawn_if_dead "
+ "keeps the multi-GPU placement"
+ )
+
+
+def test_should_record_tensor_split_abort_decision():
+ """Behavioral check of marker AND (signal crash OR Windows abort), so an
+ or->and typo or caching a generic crash fails here, not just the source pins."""
+ f = LlamaCppBackend._should_record_tensor_split_abort
+ marker = "ggml-backend-meta.cpp:541: GGML_ASSERT(x.axis != GGML_BACKEND_SPLIT_AXIS_0) failed"
+ # marker + a hard crash records, across every platform's abort encoding
+ assert f(-6, marker) is True # POSIX SIGABRT
+ assert f(-11, marker) is True # POSIX SIGSEGV
+ assert f(3, marker) is True # Windows CRT abort() exit (not a signal)
+ assert f(0xC0000005, marker) is True # Windows NTSTATUS access violation
+ # marker present but no hard crash -> not recorded
+ assert f(0, marker) is False # clean exit
+ assert f(-9, marker) is False # SIGKILL (OOM / unload), not a fault
+ assert f(None, marker) is False # still running
+ # hard crash but not the split-axis marker -> not recorded (no over-caching)
+ assert f(3, "some other failure") is False
+ assert f(-6, "GGML_ASSERT(buf != NULL) failed") is False
+ assert f(0xC0000005, "") is False
+
+
+def test_fit_off_retry_skipped_on_split_axis_abort():
+ """The fit-independent --fit off retry is skipped on the split-axis marker, else
+ the model crashes a second time before the latch records it (reviewer.py, #6659)."""
+ src = inspect.getsource(LlamaCppBackend.load_model)
+ retry = src.find('run_cmd = [*run_cmd, "--fit", "off"]')
+ assert retry != -1
+ guard = src[max(0, retry - 1000) : retry]
+ assert "_fit_retry_allowed" in guard and "_startup_crashed" in guard
+ assert (
+ "not _split_axis_crash" in guard
+ ), "the fit-off retry must be skipped when the crash is a split-axis abort"
+
+
+def test_is_abort_exit_recognizes_windows_crt_abort():
+ """exit code 3 (MSVC abort()) counts as a crash; signals / clean exits do not."""
+ f = LlamaCppBackend._is_abort_exit
+ assert f(3) is True
+ assert f(0) is False
+ assert f(-6) is False # POSIX SIGABRT is handled by _is_signal_crash, not here
+ assert f(None) is False
+
+
+# ── tensor-off after a multi-GPU fallback forces a reload (route dedup) ─
+
+
+class _NoopProcess:
+ """Stand-in for Popen so is_loaded is True and atexit cleanup doesn't crash."""
+
+ def terminate(self):
+ pass
+
+ def wait(self, timeout = None):
+ return 0
+
+ def kill(self):
+ pass
+
+ def poll(self):
+ return 0
+
+
+def _fallback_loaded_backend(layer_preserves_tensor_intent: bool) -> LlamaCppBackend:
+ """A loaded backend in the tensor->layer fallback state (tensor off, --split-mode
+ layer stored), differing only in the preserved-multi-GPU flag."""
+ b = LlamaCppBackend()
+ b._model_identifier = "owner/repo"
+ b._requested_n_ctx = 0
+ b._cache_type_kv = None
+ b._tensor_parallel = False
+ b._layer_preserves_tensor_intent = layer_preserves_tensor_intent
+ b._extra_args = ["--split-mode", "layer"]
+ b._requested_spec_mode = "auto"
+ b._chat_template_override = None
+ b._gguf_path = None
+ return b
+
+
+def test_tensor_off_echo_preserves_multi_gpu_fallback():
+ """The Studio UI always sends tensor_parallel and echoes the /load response's
+ resolved value, so after a fallback a ctx/settings reload carries tensor_parallel=
+ false even though the user never changed it. That echo must NOT collapse the
+ preserved multi-GPU placement -- it dedupes (Codex #6659)."""
+ from models.inference import LoadRequest
+
+ inference_routes = _load_inference_routes_module()
+
+ req = LoadRequest(model_path = "owner/repo", tensor_parallel = False)
+ assert "tensor_parallel" in req.model_fields_set, "the UI always sends the field"
+
+ # Preserved fallback + bare tensor=false echo: dedupe, keep multi-GPU (no collapse).
+ assert (
+ inference_routes._request_matches_loaded_settings(
+ req, _fallback_loaded_backend(layer_preserves_tensor_intent = True)
+ )
+ is True
+ )
+ # A genuine layer load (no preserved intent): tensor-off also dedupes, no churn.
+ assert (
+ inference_routes._request_matches_loaded_settings(
+ req, _fallback_loaded_backend(layer_preserves_tensor_intent = False)
+ )
+ is True
+ )
+
+
+def test_explicit_split_mode_layer_extras_reloads_after_multi_gpu_fallback():
+ """Tensor intent can be dropped via extras too: an explicit --split-mode layer
+ matches the stored fallback extras but must still reload (reviewer.py P1, #6659)."""
+ from models.inference import LoadRequest
+
+ inference_routes = _load_inference_routes_module()
+
+ req = LoadRequest(model_path = "owner/repo", llama_extra_args = ["--split-mode", "layer"])
+ assert "llama_extra_args" in req.model_fields_set
+ assert (
+ inference_routes._request_matches_loaded_settings(
+ req, _fallback_loaded_backend(layer_preserves_tensor_intent = True)
+ )
+ is False
+ )
+
+
+def test_tensor_off_reload_requires_explicit_toggle():
+ """An Apply that doesn't touch the toggle (e.g. a context change) isn't churned
+ by the preserved-fallback reload -- the working server is kept (Codex #6659)."""
+ from models.inference import LoadRequest
+
+ inference_routes = _load_inference_routes_module()
+
+ req = LoadRequest(model_path = "owner/repo") # tensor_parallel left unset
+ assert "tensor_parallel" not in req.model_fields_set
+ assert (
+ inference_routes._request_matches_loaded_settings(
+ req, _fallback_loaded_backend(layer_preserves_tensor_intent = True)
+ )
+ is True
+ )
+
+
+def test_tensor_off_under_env_tensor_does_not_reload_loop(monkeypatch):
+ """With LLAMA_ARG_SPLIT_MODE=tensor set, a tensor-off request can't drop tensor
+ intent, so the env-aware guard dedupes instead of reload-looping (Codex #6659)."""
+ from models.inference import LoadRequest
+
+ inference_routes = _load_inference_routes_module()
+ monkeypatch.setenv("LLAMA_ARG_SPLIT_MODE", "tensor")
+
+ req = LoadRequest(model_path = "owner/repo", tensor_parallel = False)
+ assert "tensor_parallel" in req.model_fields_set
+ # env still forces tensor -> not a real drop -> dedupe (no reload loop).
+ assert (
+ inference_routes._request_matches_loaded_settings(
+ req, _fallback_loaded_backend(layer_preserves_tensor_intent = True)
+ )
+ is True
+ )
+
+
+def test_is_explicit_tensor_drop_truth_table():
+ """Only an explicit non-tensor --split-mode override is a drop. A bare
+ tensor_parallel field (the UI always sends it and echoes the fallback's false), an
+ empty clear, an unrelated extra (--top-k), or inherit (None) must NOT collapse a
+ preserved fallback; --split-mode tensor / tensor_parallel=true re-engage (Codex
+ #6659)."""
+ from models.inference import LoadRequest
+
+ f = _load_inference_routes_module()._is_explicit_tensor_drop
+ # A non-tensor split-mode override is the one deliberate departure -> drop.
+ assert (
+ f(LoadRequest(model_path = "owner/repo", llama_extra_args = ["--split-mode", "layer"])) is True
+ )
+ # tensor / retry re-engages, never a drop.
+ assert (
+ f(LoadRequest(model_path = "owner/repo", llama_extra_args = ["--split-mode", "tensor"]))
+ is False
+ )
+ # A bare tensor_parallel field is the UI echo, not a drop (would collapse on reload).
+ assert f(LoadRequest(model_path = "owner/repo", tensor_parallel = False)) is False
+ assert f(LoadRequest(model_path = "owner/repo", tensor_parallel = True)) is False
+ # Unrelated extra / empty clear / inherit all keep the preserved placement.
+ assert f(LoadRequest(model_path = "owner/repo", llama_extra_args = ["--top-k", "20"])) is False
+ assert f(LoadRequest(model_path = "owner/repo", llama_extra_args = [])) is False
+ assert f(LoadRequest(model_path = "owner/repo")) is False
+
+
+def test_explicit_tensor_drop_uses_shared_helper_in_both_readers():
+ """Both the already-loaded dedup and the load carry-forward derive the drop from
+ _is_explicit_tensor_drop, so they agree on what counts as a drop -- a reload for
+ an unrelated extra still carries the preserved intent rather than collapsing to one
+ GPU (Codex #6659)."""
+ src = (Path(_BACKEND_DIR) / "routes" / "inference.py").read_text()
+ # Dedup reader (the preserved-fallback reload guard).
+ assert "layer_preserves_tensor_intent and _is_explicit_tensor_drop(request)" in src
+ # Load carry-forward reader feeds the same decision into the carry-forward.
+ assert "_explicit_tensor_drop = _is_explicit_tensor_drop(request)" in src
+
+
+def test_layer_preserves_tensor_intent_set_only_on_preserved_downgrade():
+ """load_model latches the flag from _layer_min_gpus (raised only when a tensor
+ request is downgraded but kept multi-GPU), and clears it when tensor stays on."""
+ src = inspect.getsource(LlamaCppBackend.load_model)
+ on = src.find("self._tensor_parallel = True")
+ off = src.find("self._tensor_parallel = False")
+ assert 0 <= on and 0 <= off
+ assert "self._layer_preserves_tensor_intent = False" in src[on : on + 120]
+ assert "self._layer_preserves_tensor_intent = _layer_min_gpus > 1" in src[off : off + 400]
+
+
+def test_layer_min_gpus_bound_before_gpu_selection_try():
+ """_layer_min_gpus is bound before the GPU-selection try, so the --fit-on except
+ path can't UnboundLocalError when the command builder reads it (Codex #6659)."""
+ src = inspect.getsource(LlamaCppBackend.load_model)
+ assert src.count("_layer_min_gpus = 1") == 1, "exactly one init, before the try"
+ init = src.find("_layer_min_gpus = 1")
+ try_body = src.find("gguf_size = self._get_gguf_size_bytes")
+ fit_except = src.find("GPU selection failed")
+ use_after = src.find("self._layer_preserves_tensor_intent = _layer_min_gpus > 1")
+ assert (
+ -1 < init < try_body < fit_except < use_after
+ ), "the init must precede the try body, the except, and the command-builder use"
+
+
+def test_already_in_target_state_reloads_on_tensor_off_after_fallback():
+ """The backend fast path mirrors the route dedup: a preserved fallback reloads on
+ an EXPLICIT tensor-off request, but an implicit same-settings reload (carry-forward
+ preserve_multi_gpu_on_layer=True) still dedupes (Codex #6659)."""
+
+ def _backend(layer_preserves: bool) -> LlamaCppBackend:
+ b = _fallback_loaded_backend(layer_preserves_tensor_intent = layer_preserves)
+ b._process = _NoopProcess()
+ b._healthy = True
+ return b
+
+ kwargs = dict(
+ gguf_path = None,
+ mtp_draft_path = None,
+ model_identifier = "owner/repo",
+ hf_variant = None,
+ n_ctx = 0,
+ cache_type_kv = None,
+ speculative_type = None,
+ spec_draft_n_max = None,
+ tensor_parallel = False,
+ chat_template_override = None,
+ extra_args = ["--split-mode", "layer"],
+ is_vision = False,
+ )
+ # Preserved fallback + EXPLICIT tensor drop -> reload (not already in target state).
+ assert _backend(True)._already_in_target_state(**kwargs) is False
+ # Same preserved fallback but an implicit reload that carries the intent forward
+ # (HF auto-pick / local-dir flows skip the route guard and reach here) -> dedupe.
+ assert (
+ _backend(True)._already_in_target_state(**kwargs, preserve_multi_gpu_on_layer = True) is True
+ )
+ # A genuine layer load (no preserved intent) -> dedupe, no churn.
+ assert _backend(False)._already_in_target_state(**kwargs) is True
diff --git a/studio/backend/tests/test_training_runs.py b/studio/backend/tests/test_training_runs.py
new file mode 100644
index 0000000000..fd0d6d380f
--- /dev/null
+++ b/studio/backend/tests/test_training_runs.py
@@ -0,0 +1,103 @@
+# SPDX-License-Identifier: AGPL-3.0-only
+# Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. See /studio/LICENSE.AGPL-3.0
+
+import json
+
+from storage.studio_db import _extract_project_name_from_config_json
+from utils.training_runs import (
+ build_default_output_dir_name,
+ model_segment_from_default_output_dir_name,
+ normalize_project_name,
+ slugify_project_name,
+)
+
+
+def test_normalize_project_name_trims_and_collapses_whitespace():
+ assert normalize_project_name(" Customer Support LoRA ") == "Customer Support LoRA"
+
+
+def test_normalize_project_name_returns_none_for_empty_or_invalid_values():
+ assert normalize_project_name(" ") is None
+ assert normalize_project_name(None) is None
+
+
+def test_slugify_project_name_makes_safe_suffix():
+ assert slugify_project_name("Customer Support / LoRA v2") == "customer-support-lora-v2"
+
+
+def test_slugify_project_name_rejects_path_only_or_separator_only_values():
+ assert slugify_project_name("..") is None
+ assert slugify_project_name("///") is None
+
+
+def test_build_default_output_dir_name_appends_project_slug():
+ output_dir = build_default_output_dir_name(
+ "unsloth/Llama-3.2-3B-Instruct",
+ "Customer Support",
+ timestamp = 1771227800,
+ )
+
+ assert output_dir == "unsloth_Llama-3.2-3B-Instruct__project-customer-support_1771227800"
+
+
+def test_build_default_output_dir_name_caps_final_component(tmp_path):
+ output_dir = build_default_output_dir_name(
+ "a" * 240,
+ "b" * 80,
+ timestamp = 1771227800,
+ )
+
+ assert len(output_dir.encode()) <= 255
+ (tmp_path / output_dir).mkdir()
+
+
+def test_build_default_output_dir_name_skips_invalid_project_slug():
+ output_dir = build_default_output_dir_name(
+ "unsloth/Llama-3.2-3B-Instruct",
+ "..",
+ timestamp = 1771227800,
+ )
+
+ assert output_dir == "unsloth_Llama-3.2-3B-Instruct_1771227800"
+
+
+def test_model_segment_from_default_output_dir_name_strips_project_slug():
+ assert (
+ model_segment_from_default_output_dir_name(
+ "unsloth_Llama-3.2-3B-Instruct__project-customer-support_1771227800"
+ )
+ == "unsloth_Llama-3.2-3B-Instruct"
+ )
+
+
+def test_model_segment_preserves_project_marker_text_in_model_name():
+ output_dir = build_default_output_dir_name(
+ "org/foo__project-bar",
+ timestamp = 1771227800,
+ )
+
+ assert output_dir == "org_foo__project--bar_1771227800"
+ assert model_segment_from_default_output_dir_name(output_dir) == "org_foo__project-bar"
+
+
+def test_model_segment_strips_project_slug_after_escaped_model_marker():
+ output_dir = build_default_output_dir_name(
+ "org/foo__project-bar",
+ "Customer Support",
+ timestamp = 1771227800,
+ )
+
+ assert output_dir == "org_foo__project--bar__project-customer-support_1771227800"
+ assert model_segment_from_default_output_dir_name(output_dir) == "org_foo__project-bar"
+
+
+def test_extract_project_name_from_config_json_returns_normalized_name():
+ config_json = json.dumps({"project_name": " Sales Assistant "})
+
+ assert _extract_project_name_from_config_json(config_json) == "Sales Assistant"
+
+
+def test_extract_project_name_from_config_json_handles_missing_or_invalid_payload():
+ assert _extract_project_name_from_config_json(None) is None
+ assert _extract_project_name_from_config_json("not-json") is None
+ assert _extract_project_name_from_config_json(json.dumps({"project_name": " "})) is None
diff --git a/studio/backend/tests/test_training_streaming.py b/studio/backend/tests/test_training_streaming.py
index 8ff016d3bf..70b2d6fdcc 100644
--- a/studio/backend/tests/test_training_streaming.py
+++ b/studio/backend/tests/test_training_streaming.py
@@ -195,6 +195,16 @@ def test_hf_dataset_rejects_unsafe_values(bad_hf_dataset):
)
+def test_project_name_rejects_values_over_ui_limit():
+ with pytest.raises(ValidationError):
+ TrainingStartRequest(
+ model_name = "unsloth/test",
+ project_name = "x" * 81,
+ training_type = "LoRA/QLoRA",
+ format_type = "alpaca",
+ )
+
+
# --- Start-route streaming compatibility guards ---
diff --git a/studio/backend/utils/models/checkpoints.py b/studio/backend/utils/models/checkpoints.py
index d174f6677b..90e26d45d0 100644
--- a/studio/backend/utils/models/checkpoints.py
+++ b/studio/backend/utils/models/checkpoints.py
@@ -9,6 +9,12 @@ import structlog
from loggers import get_logger
from pathlib import Path
from typing import List, Optional, Tuple
+from storage.studio_db import get_connection
+from utils.training_runs import (
+ build_default_output_dir_name,
+ extract_project_name,
+ model_segment_from_default_output_dir_name,
+)
from utils.paths import outputs_root, resolve_output_dir
logger = get_logger(__name__)
@@ -30,6 +36,93 @@ def _checkpoint_sort_key(checkpoint_path: Path) -> tuple[int, int, str]:
return (1, 0, str(checkpoint_path))
+def _infer_base_model_from_history(checkpoint_dir: Path) -> Optional[str]:
+ """Best-effort base-model lookup using persisted Studio run metadata."""
+ checkpoint_name = checkpoint_dir.name
+ resolved_checkpoint_dir = str(checkpoint_dir.resolve())
+
+ try:
+ conn = get_connection()
+ except Exception:
+ return None
+
+ try:
+ exact_rows = conn.execute(
+ """
+ SELECT model_name
+ FROM training_runs
+ WHERE output_dir IN (?, ?)
+ ORDER BY started_at DESC
+ """,
+ (
+ resolved_checkpoint_dir,
+ str(checkpoint_dir),
+ ),
+ ).fetchall()
+ for row in exact_rows:
+ model_name = row["model_name"]
+ if model_name:
+ return model_name
+
+ suffix_rows = conn.execute(
+ """
+ SELECT model_name, output_dir
+ FROM training_runs
+ WHERE output_dir IS NOT NULL
+ ORDER BY started_at DESC
+ """
+ ).fetchall()
+ for row in suffix_rows:
+ output_dir = str(row["output_dir"] or "").rstrip("/\\")
+ if not (
+ output_dir.endswith(f"/{checkpoint_name}")
+ or output_dir.endswith(f"\\{checkpoint_name}")
+ ):
+ continue
+ model_name = row["model_name"]
+ if model_name:
+ return model_name
+
+ parts = checkpoint_name.rsplit("_", 1)
+ if len(parts) != 2 or not parts[1].isdigit():
+ return None
+
+ timestamp = int(parts[1])
+ generated_rows = conn.execute(
+ """
+ SELECT model_name, config_json
+ FROM training_runs
+ ORDER BY started_at DESC
+ """
+ ).fetchall()
+ for row in generated_rows:
+ model_name = row["model_name"]
+ if not model_name:
+ continue
+
+ project_name = None
+ config_json = row["config_json"]
+ if config_json:
+ try:
+ project_name = extract_project_name(json.loads(config_json))
+ except (TypeError, json.JSONDecodeError):
+ project_name = None
+
+ expected_dir_name = build_default_output_dir_name(
+ model_name,
+ project_name,
+ timestamp = timestamp,
+ )
+ if expected_dir_name == checkpoint_name:
+ return model_name
+ except Exception:
+ return None
+ finally:
+ conn.close()
+
+ return None
+
+
def _read_checkpoint_loss(checkpoint_path: Path) -> Optional[float]:
"""Read loss from the last log_history entry of trainer_state.json, or None."""
trainer_state = checkpoint_path / "trainer_state.json"
@@ -106,9 +199,11 @@ def scan_checkpoints(
# Fallback: extract base model name from the folder name, e.g.
# "unsloth_Llama-3.2-3B-Instruct_1771227800" → "unsloth/Llama-3.2-3B-Instruct"
if not metadata.get("base_model"):
- parts = item.name.rsplit("_", 1)
- if len(parts) == 2 and parts[1].isdigit():
- name_part = parts[0]
+ metadata["base_model"] = _infer_base_model_from_history(item)
+
+ if not metadata.get("base_model"):
+ name_part = model_segment_from_default_output_dir_name(item.name)
+ if name_part:
idx = name_part.find("_")
if idx > 0:
metadata["base_model"] = name_part[:idx] + "/" + name_part[idx + 1 :]
diff --git a/studio/backend/utils/training_runs.py b/studio/backend/utils/training_runs.py
new file mode 100644
index 0000000000..dc2535e570
--- /dev/null
+++ b/studio/backend/utils/training_runs.py
@@ -0,0 +1,104 @@
+# SPDX-License-Identifier: AGPL-3.0-only
+# Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. See /studio/LICENSE.AGPL-3.0
+
+"""Helpers for naming and describing Studio training runs."""
+
+from __future__ import annotations
+
+import re
+import time
+from typing import Any, Optional
+
+_INVALID_SEGMENT_CHARS = re.compile(r"[^A-Za-z0-9._-]+")
+_MAX_RUN_DIR_NAME_CHARS = 255
+_PROJECT_MARKER = "__project-"
+_PROJECT_MARKER_ESCAPE = f"{_PROJECT_MARKER}-"
+
+
+def _trim_segment(segment: str, max_chars: int) -> str:
+ if max_chars <= 0:
+ return ""
+ return segment[:max_chars].strip("._-")
+
+
+def _escape_project_marker(segment: str) -> str:
+ return segment.replace(_PROJECT_MARKER, _PROJECT_MARKER_ESCAPE)
+
+
+def _unescape_project_marker(segment: str) -> str:
+ return segment.replace(_PROJECT_MARKER_ESCAPE, _PROJECT_MARKER)
+
+
+def _appended_project_marker_index(segment: str) -> int:
+ marker_index = segment.rfind(_PROJECT_MARKER)
+ while marker_index >= 0 and segment.startswith(_PROJECT_MARKER_ESCAPE, marker_index):
+ marker_index = segment.rfind(_PROJECT_MARKER, 0, marker_index)
+ return marker_index
+
+
+def normalize_project_name(project_name: Any) -> Optional[str]:
+ """Return a trimmed project name, or None when empty/invalid."""
+ if not isinstance(project_name, str):
+ return None
+ normalized = " ".join(project_name.strip().split())
+ return normalized or None
+
+
+def slugify_project_name(project_name: Any) -> Optional[str]:
+ """Convert a project name into a filesystem-safe suffix."""
+ normalized = normalize_project_name(project_name)
+ if normalized is None:
+ return None
+
+ slug = _INVALID_SEGMENT_CHARS.sub("-", normalized).strip("-._")
+ if not slug:
+ return None
+ return slug.lower()
+
+
+def build_default_output_dir_name(
+ model_name: str,
+ project_name: Any = None,
+ *,
+ timestamp: Optional[int] = None,
+) -> str:
+ """Build the default training output folder name."""
+ from utils.paths import default_run_dir_name
+
+ timestamp_part = str(int(time.time() if timestamp is None else timestamp))
+ timestamp_suffix = f"_{timestamp_part}"
+ model_segment = _escape_project_marker(default_run_dir_name(model_name))
+ project_slug = slugify_project_name(project_name)
+ if not project_slug:
+ max_model_chars = _MAX_RUN_DIR_NAME_CHARS - len(timestamp_suffix)
+ model_segment = _trim_segment(model_segment, max_model_chars) or "model"
+ return f"{model_segment}{timestamp_suffix}"
+
+ max_project_chars = (
+ _MAX_RUN_DIR_NAME_CHARS - len("model") - len(_PROJECT_MARKER) - len(timestamp_suffix)
+ )
+ project_slug = _trim_segment(project_slug, max_project_chars) or "project"
+ project_suffix = f"{_PROJECT_MARKER}{project_slug}{timestamp_suffix}"
+ max_model_chars = _MAX_RUN_DIR_NAME_CHARS - len(project_suffix)
+ model_segment = _trim_segment(model_segment, max_model_chars) or "model"
+ return f"{model_segment}{project_suffix}"
+
+
+def model_segment_from_default_output_dir_name(output_dir_name: str) -> Optional[str]:
+ """Return the encoded model segment from a default run folder name."""
+ parts = str(output_dir_name or "").rsplit("_", 1)
+ if len(parts) != 2 or not parts[1].isdigit():
+ return None
+ model_segment = parts[0]
+ marker_index = _appended_project_marker_index(model_segment)
+ if marker_index >= 0:
+ model_segment = model_segment[:marker_index]
+ model_segment = _unescape_project_marker(model_segment)
+ return model_segment or None
+
+
+def extract_project_name(config: Any) -> Optional[str]:
+ """Read and normalize a project name from a stored config dict."""
+ if not isinstance(config, dict):
+ return None
+ return normalize_project_name(config.get("project_name"))
diff --git a/studio/frontend/package-lock.json b/studio/frontend/package-lock.json
index 6202d3ce22..80db64553a 100644
--- a/studio/frontend/package-lock.json
+++ b/studio/frontend/package-lock.json
@@ -1704,6 +1704,7 @@
"os": [
"android"
],
+ "peer": true,
"engines": {
"node": ">= 10"
},
@@ -1724,6 +1725,7 @@
"os": [
"darwin"
],
+ "peer": true,
"engines": {
"node": ">= 10"
},
@@ -1744,6 +1746,7 @@
"os": [
"darwin"
],
+ "peer": true,
"engines": {
"node": ">= 10"
},
@@ -1764,6 +1767,7 @@
"os": [
"linux"
],
+ "peer": true,
"engines": {
"node": ">= 10"
},
@@ -1784,6 +1788,7 @@
"os": [
"linux"
],
+ "peer": true,
"engines": {
"node": ">= 10"
},
@@ -1804,6 +1809,7 @@
"os": [
"linux"
],
+ "peer": true,
"engines": {
"node": ">= 10"
},
@@ -1824,6 +1830,7 @@
"os": [
"linux"
],
+ "peer": true,
"engines": {
"node": ">= 10"
},
@@ -1844,6 +1851,7 @@
"os": [
"linux"
],
+ "peer": true,
"engines": {
"node": ">= 10"
},
@@ -1864,6 +1872,7 @@
"os": [
"linux"
],
+ "peer": true,
"engines": {
"node": ">= 10"
},
@@ -1884,6 +1893,7 @@
"os": [
"win32"
],
+ "peer": true,
"engines": {
"node": ">= 10"
},
@@ -1904,6 +1914,7 @@
"os": [
"win32"
],
+ "peer": true,
"engines": {
"node": ">= 10"
},
@@ -5669,9 +5680,6 @@
"cpu": [
"arm64"
],
- "libc": [
- "glibc"
- ],
"license": "MIT",
"optional": true,
"os": [
@@ -5688,9 +5696,6 @@
"cpu": [
"arm64"
],
- "libc": [
- "musl"
- ],
"license": "MIT",
"optional": true,
"os": [
@@ -5707,9 +5712,6 @@
"cpu": [
"ppc64"
],
- "libc": [
- "glibc"
- ],
"license": "MIT",
"optional": true,
"os": [
@@ -5726,9 +5728,6 @@
"cpu": [
"s390x"
],
- "libc": [
- "glibc"
- ],
"license": "MIT",
"optional": true,
"os": [
@@ -5745,9 +5744,6 @@
"cpu": [
"x64"
],
- "libc": [
- "glibc"
- ],
"license": "MIT",
"optional": true,
"os": [
@@ -5764,9 +5760,6 @@
"cpu": [
"x64"
],
- "libc": [
- "musl"
- ],
"license": "MIT",
"optional": true,
"os": [
@@ -10282,9 +10275,9 @@
}
},
"node_modules/hono": {
- "version": "4.12.21",
- "resolved": "https://registry.npmjs.org/hono/-/hono-4.12.21.tgz",
- "integrity": "sha512-uV63apnb0kyPtAUwoWgaGh9HyIFcv8lgmzPZSiTBQAFOFGIzka5EZ1dZocmGnn0XdX0+XTqJ6Tqv7selMuGLRQ==",
+ "version": "4.12.25",
+ "resolved": "https://registry.npmjs.org/hono/-/hono-4.12.25.tgz",
+ "integrity": "sha512-2NFaIyNVgJmBs/ecmtGzlmluTFs5cHEWGTdu0t1HBwYzoGXOL5nUQBRMXsXWla5i4KkG//QMzVP88m1+I3fdAQ==",
"license": "MIT",
"engines": {
"node": ">=16.9.0"
diff --git a/studio/frontend/package.json b/studio/frontend/package.json
index 0956c710f5..a2eddecda3 100644
--- a/studio/frontend/package.json
+++ b/studio/frontend/package.json
@@ -86,7 +86,7 @@
"@tanstack/router-core": "1.169.2",
"@tanstack/history": "1.161.6",
"mermaid": "11.15.0",
- "hono": "4.12.21",
+ "hono": "4.12.25",
"qs": "6.15.2",
"ip-address": "10.1.1",
"brace-expansion@5.0.5": "5.0.6"
diff --git a/studio/frontend/src/app/provider.tsx b/studio/frontend/src/app/provider.tsx
index 914abbbf1d..176665769d 100644
--- a/studio/frontend/src/app/provider.tsx
+++ b/studio/frontend/src/app/provider.tsx
@@ -354,12 +354,11 @@ function TauriWrapper({ children }: { children: ReactNode }) {
);
}
- const showApp = status === "running" && desktopAuthReady;
+ const showApp = status === "running";
+ const desktopBooting = status === "running" && !desktopAuthReady;
+ const showInteractiveApp = showApp && desktopAuthReady;
const startupStatus = status === "running" ? "starting" : status;
- const startupProgressDetail =
- status === "running" && !desktopAuthReady
- ? "Signing in to desktop session..."
- : progressDetail;
+ const startupProgressDetail = progressDetail;
const usesCustomTitlebar = shouldUseCustomWindowTitlebar();
const usesNativeMacTitlebar = shouldUseNativeMacWindowTitlebar();
const hidesTitlebarSidebar = HIDDEN_TITLEBAR_SIDEBAR_ROUTES.has(pathname);
@@ -369,12 +368,23 @@ function TauriWrapper({ children }: { children: ReactNode }) {