From aa85fa8eb1993795af69a2ea1da5cc8fafc13d7f Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Fri, 27 Mar 2026 19:39:37 +0000 Subject: [PATCH 01/26] add Docker support: skip venv, install only missing deps --- studio/install_python_stack.py | 241 +++++++++++++++++++-------------- studio/setup.sh | 70 +++++----- unsloth_cli/commands/studio.py | 14 +- 3 files changed, 187 insertions(+), 138 deletions(-) diff --git a/studio/install_python_stack.py b/studio/install_python_stack.py index 3fd1e6af66..9273c70272 100644 --- a/studio/install_python_stack.py +++ b/studio/install_python_stack.py @@ -907,9 +907,9 @@ def install_python_stack() -> int: base_total += 3 _TOTAL = (base_total - 1) if skip_base else base_total - # 1. Try to use uv for faster installs (must happen before pip upgrade - # because uv venvs don't include pip by default) - USE_UV = _bootstrap_uv() + # # 1. Try to use uv for faster installs (must happen before pip upgrade + # # because uv venvs don't include pip by default) + # USE_UV = _bootstrap_uv() # 2. Ensure pip is available (uv venvs created by install.sh don't include pip) _progress("pip bootstrap") @@ -938,17 +938,115 @@ def install_python_stack() -> int: ).returncode == 0 ) + # # 2. Ensure pip is available (uv venvs created by install.sh don't include pip) + # _progress("pip bootstrap") + # if USE_UV: + # run( + # "Bootstrapping pip via uv", + # [ + # "uv", + # "pip", + # "install", + # "--python", + # sys.executable, + # "pip", + # ], + # ) + # else: + # # pip may not exist yet (uv-created venvs omit it). Try ensurepip + # # first, then upgrade. Only fall back to a direct upgrade when pip + # # is already present. + # _has_pip = ( + # subprocess.run( + # [sys.executable, "-m", "pip", "--version"], + # stdout = subprocess.DEVNULL, + # stderr = subprocess.DEVNULL, + # ).returncode + # == 0 + # ) + # + # if not _has_pip: + # run( + # "Bootstrapping pip via ensurepip", + # [sys.executable, "-m", "ensurepip", "--upgrade"], + # ) + # else: + # run( + # "Upgrading pip", + # [sys.executable, "-m", "pip", "install", "--upgrade", "pip"], + # ) - if not _has_pip: - run( - "Bootstrapping pip via ensurepip", - [sys.executable, "-m", "ensurepip", "--upgrade"], - ) - else: - run( - "Upgrading pip", - [sys.executable, "-m", "pip", "install", "--upgrade", "pip"], - ) + # # 3. Core packages: unsloth-zoo + unsloth (or custom package name) + # if skip_base: + # print(_green(f"✅ {package_name} already installed — skipping base packages")) + # elif NO_TORCH: + # # No-torch update path: install unsloth + unsloth-zoo with --no-deps + # # (current PyPI metadata still declares torch as a hard dep), then + # # runtime deps with --no-deps (avoids transitive torch). + # _progress("base packages (no torch)") + # pip_install( + # f"Updating {package_name} + unsloth-zoo (no-torch mode)", + # "--no-cache-dir", + # "--no-deps", + # "--upgrade-package", + # package_name, + # "--upgrade-package", + # "unsloth-zoo", + # package_name, + # "unsloth-zoo", + # ) + # pip_install( + # "Installing no-torch runtime deps", + # "--no-cache-dir", + # "--no-deps", + # req = REQ_ROOT / "no-torch-runtime.txt", + # ) + # if local_repo: + # pip_install( + # "Overlaying local repo (editable)", + # "--no-cache-dir", + # "--no-deps", + # "-e", + # local_repo, + # constrain = False, + # ) + # elif local_repo: + # _progress("base packages") + # pip_install( + # "Updating base packages", + # "--no-cache-dir", + # "--upgrade-package", + # "unsloth", + # "--upgrade-package", + # "unsloth-zoo", + # req = REQ_ROOT / "base.txt", + # ) + # pip_install( + # "Overlaying local repo (editable)", + # "--no-cache-dir", + # "--no-deps", + # "-e", + # local_repo, + # constrain = False, + # ) + # elif package_name != "unsloth": + # _progress("base packages") + # pip_install( + # f"Installing {package_name}", + # "--no-cache-dir", + # package_name, + # ) + # else: + # _progress("base packages") + # pip_install( + # "Updating base packages", + # "--no-cache-dir", + # "--upgrade-package", + # "unsloth", + # "--upgrade-package", + # "unsloth-zoo", + # req = REQ_ROOT / "base.txt", + # ) # 3. Core packages: unsloth-zoo + unsloth (or custom package name) if skip_base: @@ -1048,96 +1146,39 @@ def install_python_stack() -> int: "unsloth-zoo", req = REQ_ROOT / "base.txt", ) + # pip_install( + # "Installing additional unsloth dependencies", + # "--no-cache-dir", + # req = REQ_ROOT / "extras.txt", + # ) - # 2b. AMD ROCm: reinstall torch with HIP wheels if the host has ROCm but the - # venv received CPU-only torch (common when pip resolves torch from PyPI). - # Must come immediately after base packages so torch is present for inspection. - if not IS_WINDOWS and not IS_MACOS and not NO_TORCH: - _progress("ROCm torch check") - _ensure_rocm_torch() + # pip_install( + # "Installing extras (no-deps)", + # "--no-deps", + # "--no-cache-dir", + # req = REQ_ROOT / "extras-no-deps.txt", + # ) - # Windows + AMD GPU: PyTorch does not publish ROCm wheels for Windows. - # Detect and warn so users know manual steps are needed for GPU training. - if IS_WINDOWS and not NO_TORCH and not _has_usable_nvidia_gpu(): - # Validate actual AMD GPU presence (not just tool existence) - import re as _re_win + # # 4. Overrides (torchao, transformers) -- force-reinstall + # _progress("dependency overrides") + # pip_install( + # "Installing dependency overrides", + # "--force-reinstall", + # "--no-cache-dir", + # req = REQ_ROOT / "overrides.txt", + # ) - def _win_amd_smi_has_gpu(stdout: str) -> bool: - return bool(_re_win.search(r"(?im)^gpu\s*[:\[]\s*\d", stdout)) - - _win_amd_gpu = False - for _wcmd, _check_fn in ( - (["hipinfo"], lambda out: "gcnarchname" in out.lower()), - (["amd-smi", "list"], _win_amd_smi_has_gpu), - ): - _wexe = shutil.which(_wcmd[0]) - if not _wexe: - continue - try: - _wr = subprocess.run( - [_wexe, *_wcmd[1:]], - stdout = subprocess.PIPE, - stderr = subprocess.DEVNULL, - text = True, - timeout = 10, - ) - except Exception: - continue - if _wr.returncode == 0 and _check_fn(_wr.stdout): - _win_amd_gpu = True - break - if _win_amd_gpu: - _safe_print( - _dim(" Note:"), - "AMD GPU detected on Windows. ROCm-enabled PyTorch must be", - ) - _safe_print( - " " * 8, - "installed manually. See: https://docs.unsloth.ai/get-started/install-and-update/amd", - ) - - # 3. Extra dependencies - _progress("unsloth extras") - pip_install( - "Installing additional unsloth dependencies", - "--no-cache-dir", - req = REQ_ROOT / "extras.txt", - ) - - # 3b. Extra dependencies (no-deps) -- audio model support etc. - _progress("extra codecs") - pip_install( - "Installing extras (no-deps)", - "--no-deps", - "--no-cache-dir", - req = REQ_ROOT / "extras-no-deps.txt", - ) - - # 4. Overrides (torchao, transformers) -- force-reinstall - # Skip entirely when torch is unavailable (e.g. Intel Mac GGUF-only mode) - # because overrides.txt contains torchao which requires torch. - if NO_TORCH: - _progress("dependency overrides (skipped, no torch)") - else: - _progress("dependency overrides") - pip_install( - "Installing dependency overrides", - "--force-reinstall", - "--no-cache-dir", - req = REQ_ROOT / "overrides.txt", - ) - - # 5. Triton kernels (no-deps, from source) - # Skip on Windows (no support) and macOS (no support). - if not IS_WINDOWS and not IS_MACOS: - _progress("triton kernels") - pip_install( - "Installing triton kernels", - "--no-deps", - "--no-cache-dir", - req = REQ_ROOT / "triton-kernels.txt", - constrain = False, - ) + # # 5. Triton kernels (no-deps, from source) + # # Skip on Windows (no support) and macOS (no support). + # if not IS_WINDOWS and not IS_MACOS: + # _progress("triton kernels") + # pip_install( + # "Installing triton kernels", + # "--no-deps", + # "--no-cache-dir", + # req = REQ_ROOT / "triton-kernels.txt", + # constrain = False, + # ) if not IS_WINDOWS and not IS_MACOS and not NO_TORCH: _progress("flash-attn") diff --git a/studio/setup.sh b/studio/setup.sh index c5beb7ebd3..c48542894f 100755 --- a/studio/setup.sh +++ b/studio/setup.sh @@ -417,36 +417,7 @@ if [ -d "$SCRIPT_DIR/backend/core/data_recipe/oxc-validator" ] && command -v npm fi # ── Python venv + deps ── -# UNSLOTH_STUDIO_HOME (or STUDIO_HOME alias) overrides the install root -# (mirrors install.sh). UNSLOTH_STUDIO_HOME wins when both are set. -_studio_override_var="" -_studio_override="${UNSLOTH_STUDIO_HOME:-}" -if [ -n "$_studio_override" ]; then - _studio_override_var="UNSLOTH_STUDIO_HOME" -else - _studio_override="${STUDIO_HOME:-}" - [ -n "$_studio_override" ] && _studio_override_var="STUDIO_HOME" -fi -# Strip whitespace so " " is treated as unset (matches Python .strip()). -_studio_override=$(printf '%s' "$_studio_override" | sed -e 's/^[[:space:]]*//' -e 's/[[:space:]]*$//') -case "$_studio_override" in - "~") _studio_override="$HOME" ;; - "~/"*) _studio_override="$HOME/${_studio_override#'~/'}" ;; -esac -if [ -n "$_studio_override" ]; then - # setup.sh runs against an existing install (via 'unsloth studio update'); - # a typo in the override must fail fast instead of materializing an - # empty workspace dir. Mirrors setup.ps1 behavior. - if [ ! -d "$_studio_override" ]; then - echo "ERROR: $_studio_override_var=$_studio_override does not exist." >&2 - echo " Run install.sh to create the install root before 'unsloth studio update'." >&2 - exit 1 - fi - [ -w "$_studio_override" ] || { echo "ERROR: $_studio_override_var=$_studio_override is not writable." >&2; exit 1; } - STUDIO_HOME="$(CDPATH= cd -P -- "$_studio_override" && pwd -P)" || exit 1 -else - STUDIO_HOME="$HOME/.unsloth/studio" -fi +STUDIO_HOME="${UNSLOTH_STUDIO_HOME:-$HOME/.unsloth/studio}" VENV_DIR="$STUDIO_HOME/unsloth_studio" VENV_T5_530_DIR="$STUDIO_HOME/.venv_t5_530" VENV_T5_550_DIR="$STUDIO_HOME/.venv_t5_550" @@ -459,7 +430,12 @@ VENV_T5_550_DIR="$STUDIO_HOME/.venv_t5_550" # Note: do NOT delete $STUDIO_HOME/.venv here — install.sh handles migration _COLAB_NO_VENV=false -if [ ! -x "$VENV_DIR/bin/python" ]; then +_DOCKER_NO_VENV=false +if [ -n "$UNSLOTH_DOCKER" ]; then + # Docker: packages already in /opt/conda — skip venv entirely. + # Only pre-install .venv_t5 for transformers 5.x switching (handled below). + _DOCKER_NO_VENV=true +elif [ ! -x "$VENV_DIR/bin/python" ]; then if [ "$IS_COLAB" = true ]; then # On Colab there is no Studio venv -- install backend deps into system Python. # Strip all version constraints so pip keeps Colab's pre-installed @@ -524,6 +500,38 @@ if [ "$_COLAB_NO_VENV" = true ]; then substep "continuing to llama.cpp install for GGUF inference support" fi +# In Docker, packages are pre-installed in /opt/conda — only install missing +# studio/data-designer deps and pre-install .venv_t5 for transformers 5.x. +if [ "$_DOCKER_NO_VENV" = true ]; then + echo " Docker detected — skipping venv activation." + + # Install branch's unsloth/unsloth_cli/studio into /opt/conda + # (overwrites PyPI version with Docker-aware code) + echo " Installing local unsloth from branch..." + pip install --force-reinstall --no-deps "$REPO_ROOT" + + # Install only missing deps (studio, data-designer, plugin, metadata patch). + # Heavy packages (torch, unsloth, vllm, etc.) are already in /opt/conda. + # install_python_stack.py has steps 1-5 commented out for this branch. + python "$SCRIPT_DIR/install_python_stack.py" + + # Pre-install transformers 5.x into .venv_t5 + echo "" + echo " Pre-installing transformers 5.x for newer model support..." + mkdir -p "$VENV_T5_DIR" + pip install --target "$VENV_T5_DIR" --no-deps "transformers==5.3.0" 2>/dev/null + pip install --target "$VENV_T5_DIR" --no-deps "huggingface_hub==1.7.1" 2>/dev/null + pip install --target "$VENV_T5_DIR" --no-deps "hf_xet==1.4.2" 2>/dev/null + pip install --target "$VENV_T5_DIR" "tiktoken" 2>/dev/null + echo "✅ Transformers 5.x pre-installed to $VENV_T5_DIR/" + + echo "" + echo "╔══════════════════════════════════════╗" + echo "║ Docker Studio Setup Complete! ║" + echo "╚══════════════════════════════════════╝" + exit 0 +fi + # ── Check if Python deps need updating ── # Compare installed package version against PyPI latest. # Skip all Python dependency work if versions match (fast update path). diff --git a/unsloth_cli/commands/studio.py b/unsloth_cli/commands/studio.py index 714ace4533..c804ae8e54 100644 --- a/unsloth_cli/commands/studio.py +++ b/unsloth_cli/commands/studio.py @@ -518,9 +518,11 @@ def studio_default( if ctx.invoked_subcommand is not None: return - # Always use the studio venv if it exists and we're not already in it - studio_venv_dir = STUDIO_HOME / "unsloth_studio" - in_studio_venv = sys.prefix.startswith(str(studio_venv_dir)) + # In Docker, packages live in /opt/conda — skip venv re-exec entirely. + if not os.environ.get("UNSLOTH_DOCKER"): + # Always use the studio venv if it exists and we're not already in it + studio_venv_dir = STUDIO_HOME / "unsloth_studio" + in_studio_venv = sys.prefix.startswith(str(studio_venv_dir)) if not in_studio_venv: studio_python = _studio_venv_python() @@ -566,10 +568,8 @@ def studio_default( ) raise typer.Exit(rc) else: - os.execvp(str(studio_python), args) - else: - typer.echo("Studio not set up. Run install.sh first.") - raise typer.Exit(1) + typer.echo("Studio not set up. Run install.sh first.") + raise typer.Exit(1) from studio.backend.run import run_server From 07c378a5e18136a57f86fe846786d056955f5aba Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Thu, 2 Apr 2026 23:24:47 +0000 Subject: [PATCH 02/26] fix: patch PEFT for Gemma4ClippableLinear in loader checkpoint path The same Gemma4ClippableLinear monkey-patch that exists in vision.py for training is needed in loader.py for loading existing checkpoints (used by export and inference). Gemma4ClippableLinear wraps nn.Linear but does not subclass it, so PEFT's LoRA injection fails with "Target module not supported". The patch redirects PEFT to target the inner .linear child instead. Applied only to the vision model PeftModel.from_pretrained path. Temporary fix until PEFT adds native support (peft#3129). --- unsloth/models/loader.py | 3 +++ 1 file changed, 3 insertions(+) diff --git a/unsloth/models/loader.py b/unsloth/models/loader.py index fc91178d88..a2453dd8d0 100644 --- a/unsloth/models/loader.py +++ b/unsloth/models/loader.py @@ -1559,7 +1559,10 @@ class FastModel(FastBaseModel): if _clippable_linear_cls is not None: from peft.tuners.lora.model import LoraModel as _LoraModel +<<<<<<< HEAD +======= +>>>>>>> 35ebf398 (fix: patch PEFT for Gemma4ClippableLinear in loader checkpoint path) _original_car = _LoraModel._create_and_replace def _patched_car( From 5a3b288460f213636896130d45317d4491c79273 Mon Sep 17 00:00:00 2001 From: "pre-commit-ci[bot]" <66853113+pre-commit-ci[bot]@users.noreply.github.com> Date: Thu, 2 Apr 2026 23:37:48 +0000 Subject: [PATCH 03/26] [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci --- unsloth/models/loader.py | 4 ---- 1 file changed, 4 deletions(-) diff --git a/unsloth/models/loader.py b/unsloth/models/loader.py index a2453dd8d0..c413de95dc 100644 --- a/unsloth/models/loader.py +++ b/unsloth/models/loader.py @@ -1559,10 +1559,6 @@ class FastModel(FastBaseModel): if _clippable_linear_cls is not None: from peft.tuners.lora.model import LoraModel as _LoraModel -<<<<<<< HEAD - -======= ->>>>>>> 35ebf398 (fix: patch PEFT for Gemma4ClippableLinear in loader checkpoint path) _original_car = _LoraModel._create_and_replace def _patched_car( From 9a7fc993306cf593eadcd4c2af1a57f3651736da Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Sun, 5 Apr 2026 04:50:43 +0000 Subject: [PATCH 04/26] Skip llama.cpp install in Docker mode --- studio/setup.sh | 12 +++++------- 1 file changed, 5 insertions(+), 7 deletions(-) diff --git a/studio/setup.sh b/studio/setup.sh index c48542894f..18d90d7adc 100755 --- a/studio/setup.sh +++ b/studio/setup.sh @@ -644,13 +644,10 @@ fi fi # ── 7. Prefer prebuilt llama.cpp bundles before any source build path ── -# Nest llama.cpp under $STUDIO_HOME only for real env-overrides; legacy -# default keeps ~/.unsloth/llama.cpp so pre-PR builds are still discovered. -if [ "$_STUDIO_HOME_IS_CUSTOM" = true ]; then - UNSLOTH_HOME="$STUDIO_HOME" -else - UNSLOTH_HOME="$HOME/.unsloth" -fi +if [ "$_DOCKER_NO_VENV" = true ]; then + step "llama.cpp" "skipped (Docker)" +else # begin non-Docker llama.cpp block +UNSLOTH_HOME="$HOME/.unsloth" mkdir -p "$UNSLOTH_HOME" LLAMA_CPP_DIR="$UNSLOTH_HOME/llama.cpp" LLAMA_SERVER_BIN="$LLAMA_CPP_DIR/build/bin/llama-server" @@ -1163,6 +1160,7 @@ else fi } fi # end _SKIP_GGUF_BUILD check +fi # end non-Docker llama.cpp block # ── Footer ── if [ "$_LLAMA_ONLY" = "1" ]; then From 7d51af985cabf07fec14482d34804c4a83ef8264 Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Mon, 6 Apr 2026 16:25:07 +0000 Subject: [PATCH 05/26] split venv_t5 into venv_t5_530 and venv_t5_550 for tiered transformers 5.x support --- studio/backend/core/export/worker.py | 3 - studio/backend/core/inference/worker.py | 2 - studio/backend/core/training/worker.py | 17 +++++ studio/setup.ps1 | 83 +++++++++++++++++++++++++ studio/setup.sh | 23 +++++++ 5 files changed, 123 insertions(+), 5 deletions(-) diff --git a/studio/backend/core/export/worker.py b/studio/backend/core/export/worker.py index f77b1966c4..1be456d4d4 100644 --- a/studio/backend/core/export/worker.py +++ b/studio/backend/core/export/worker.py @@ -181,11 +181,8 @@ def _setup_log_capture(resp_queue: Any) -> None: def _activate_transformers_version(model_name: str) -> None: """Activate the correct transformers version BEFORE any ML imports.""" - # Ensure backend is on path for utils imports - backend_path = str(Path(__file__).resolve().parent.parent.parent) if backend_path not in sys.path: sys.path.insert(0, backend_path) - from utils.transformers_version import activate_transformers_for_subprocess activate_transformers_for_subprocess(model_name) diff --git a/studio/backend/core/inference/worker.py b/studio/backend/core/inference/worker.py index 085a1ab899..9829bfe232 100644 --- a/studio/backend/core/inference/worker.py +++ b/studio/backend/core/inference/worker.py @@ -35,8 +35,6 @@ from utils.hardware import apply_gpu_ids def _activate_transformers_version(model_name: str) -> None: """Activate the correct transformers version BEFORE any ML imports.""" - # Ensure backend is on path for utils imports - backend_path = str(Path(__file__).resolve().parent.parent.parent) if backend_path not in sys.path: sys.path.insert(0, backend_path) diff --git a/studio/backend/core/training/worker.py b/studio/backend/core/training/worker.py index ef5cafb175..dbd3628ce9 100644 --- a/studio/backend/core/training/worker.py +++ b/studio/backend/core/training/worker.py @@ -1014,6 +1014,7 @@ def run_training_process( ) return +<<<<<<< HEAD # ── 1a. Auto-enable trust_remote_code for NemotronH/Nano models ── # NemotronH has config parsing bugs in transformers that require # trust_remote_code=True as a workaround. Other transformers 5.x models @@ -1025,6 +1026,22 @@ def run_training_process( if ( any(sub in _lowered for sub in _NEMOTRON_TRUST_SUBSTRINGS) and (_lowered.startswith("unsloth/") or _lowered.startswith("nvidia/")) +======= + # ── 1a. Auto-enable trust_remote_code for unsloth/* transformers 5.x models ── + # Some newer architectures (e.g. NemotronH) have config parsing bugs in + # transformers that require trust_remote_code=True as a workaround. + # Only auto-enable for unsloth/* prefixed models (trusted source). + # Exclude Gemma 4 since it is a native transformers 5.5 model and + # trust_remote_code=True would bypass the compiler (disabling fused CE). + from utils.transformers_version import get_transformers_tier + + _lowered = model_name.lower() + _tier = get_transformers_tier(model_name) + if ( + _tier != "default" + and _lowered.startswith("unsloth/") + and _tier != "550" # Gemma 4 is native t5.5 — trust_remote_code bypasses compiler +>>>>>>> 970219a3 (split venv_t5 into venv_t5_530 and venv_t5_550 for tiered transformers 5.x support) and not config.get("trust_remote_code", False) ): config["trust_remote_code"] = True diff --git a/studio/setup.ps1 b/studio/setup.ps1 index 40788a0ecb..3ffeda69ba 100644 --- a/studio/setup.ps1 +++ b/studio/setup.ps1 @@ -1854,6 +1854,89 @@ if ($stackExit -ne 0) { exit 1 } +# ── Pre-install transformers 5.x into .venv_t5_530/ and .venv_t5_550/ ── +# Models like GLM-4.7-Flash, Qwen3 MoE need transformers>=5.3.0. +# Gemma 4 models need transformers>=5.5.0. +# Pre-install into separate directories to avoid runtime pip overhead. +# The training subprocess prepends the appropriate dir to sys.path. +Write-Host "" + +# Clean up legacy single .venv_t5 directory +$VenvT5Legacy = Join-Path $env:USERPROFILE ".unsloth\studio\.venv_t5" +if (Test-Path $VenvT5Legacy) { Remove-Item -Recurse -Force $VenvT5Legacy } + +$prevEAP_t5 = $ErrorActionPreference +$ErrorActionPreference = "Continue" + +# --- .venv_t5_530 (transformers 5.3.0) --- +substep "pre-installing transformers 5.3.0 for newer model support..." +$VenvT5_530Dir = Join-Path $env:USERPROFILE ".unsloth\studio\.venv_t5_530" +if (Test-Path $VenvT5_530Dir) { Remove-Item -Recurse -Force $VenvT5_530Dir } +New-Item -ItemType Directory -Path $VenvT5_530Dir -Force | Out-Null +foreach ($pkg in @("transformers==5.3.0", "huggingface_hub==1.8.0", "hf_xet==1.4.2")) { + if ($script:UnslothVerbose) { + Fast-Install --target $VenvT5_530Dir --no-deps $pkg + $t5PkgExit = $LASTEXITCODE + $output = "" + } else { + $output = Fast-Install --target $VenvT5_530Dir --no-deps $pkg | Out-String + $t5PkgExit = $LASTEXITCODE + } + if ($t5PkgExit -ne 0) { + Write-Host "[FAIL] Could not install $pkg into .venv_t5_530/" -ForegroundColor Red + Write-Host $output -ForegroundColor Red + $ErrorActionPreference = $prevEAP_t5 + exit 1 + } +} +if ($script:UnslothVerbose) { + Fast-Install --target $VenvT5_530Dir tiktoken + $tiktokenInstallExit = $LASTEXITCODE + $output = "" +} else { + $output = Fast-Install --target $VenvT5_530Dir tiktoken | Out-String + $tiktokenInstallExit = $LASTEXITCODE +} +if ($tiktokenInstallExit -ne 0) { + substep "Could not install tiktoken into .venv_t5_530/ -- Qwen tokenizers may fail" "Yellow" +} +step "transformers" "5.3.0 pre-installed" + +# --- .venv_t5_550 (transformers 5.5.0) --- +substep "pre-installing transformers 5.5.0 for Gemma 4 support..." +$VenvT5_550Dir = Join-Path $env:USERPROFILE ".unsloth\studio\.venv_t5_550" +if (Test-Path $VenvT5_550Dir) { Remove-Item -Recurse -Force $VenvT5_550Dir } +New-Item -ItemType Directory -Path $VenvT5_550Dir -Force | Out-Null +foreach ($pkg in @("transformers==5.5.0", "huggingface_hub==1.8.0", "hf_xet==1.4.2")) { + if ($script:UnslothVerbose) { + Fast-Install --target $VenvT5_550Dir --no-deps $pkg + $t5PkgExit = $LASTEXITCODE + $output = "" + } else { + $output = Fast-Install --target $VenvT5_550Dir --no-deps $pkg | Out-String + $t5PkgExit = $LASTEXITCODE + } + if ($t5PkgExit -ne 0) { + Write-Host "[FAIL] Could not install $pkg into .venv_t5_550/" -ForegroundColor Red + Write-Host $output -ForegroundColor Red + $ErrorActionPreference = $prevEAP_t5 + exit 1 + } +} +if ($script:UnslothVerbose) { + Fast-Install --target $VenvT5_550Dir tiktoken + $tiktokenInstallExit = $LASTEXITCODE + $output = "" +} else { + $output = Fast-Install --target $VenvT5_550Dir tiktoken | Out-String + $tiktokenInstallExit = $LASTEXITCODE +} +if ($tiktokenInstallExit -ne 0) { + substep "Could not install tiktoken into .venv_t5_550/ -- Qwen tokenizers may fail" "Yellow" +} +$ErrorActionPreference = $prevEAP_t5 +step "transformers" "5.5.0 pre-installed" + } else { step "python" "dependencies up to date" # Restore ErrorActionPreference (was lowered for pip/python section) diff --git a/studio/setup.sh b/studio/setup.sh index 18d90d7adc..9a5ec2be14 100755 --- a/studio/setup.sh +++ b/studio/setup.sh @@ -566,6 +566,29 @@ fi if [ "$_SKIP_PYTHON_DEPS" = false ]; then install_python_stack + + # ── 6b. Pre-install transformers 5.x into .venv_t5_530/ and .venv_t5_550/ ── + # Models like GLM-4.7-Flash, Qwen3 MoE need transformers>=5.3.0. + # Gemma 4 models need transformers>=5.5.0. + # Pre-install into separate directories to avoid runtime pip overhead. + # The training subprocess prepends the appropriate dir to sys.path. + + # Clean up legacy single .venv_t5 directory + [ -d "$STUDIO_HOME/.venv_t5" ] && rm -rf "$STUDIO_HOME/.venv_t5" + + mkdir -p "$VENV_T5_530_DIR" + run_quiet "install transformers 5.3.0" fast_install --target "$VENV_T5_530_DIR" --no-deps "transformers==5.3.0" + run_quiet "install huggingface_hub for t5_530" fast_install --target "$VENV_T5_530_DIR" --no-deps "huggingface_hub==1.8.0" + run_quiet "install hf_xet for t5_530" fast_install --target "$VENV_T5_530_DIR" --no-deps "hf_xet==1.4.2" + run_quiet "install tiktoken for t5_530" fast_install --target "$VENV_T5_530_DIR" "tiktoken" + step "transformers" "5.3.0 pre-installed" + + mkdir -p "$VENV_T5_550_DIR" + run_quiet "install transformers 5.5.0" fast_install --target "$VENV_T5_550_DIR" --no-deps "transformers==5.5.0" + run_quiet "install huggingface_hub for t5_550" fast_install --target "$VENV_T5_550_DIR" --no-deps "huggingface_hub==1.8.0" + run_quiet "install hf_xet for t5_550" fast_install --target "$VENV_T5_550_DIR" --no-deps "hf_xet==1.4.2" + run_quiet "install tiktoken for t5_550" fast_install --target "$VENV_T5_550_DIR" "tiktoken" + step "transformers" "5.5.0 pre-installed" else step "python" "dependencies up to date" verbose_substep "python deps check: installed=$_PKG_NAME@${INSTALLED_VER:-unknown} latest=${LATEST_VER:-unknown}" From 8b4bffca4f9113805f192e485919fee8366293bd Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Mon, 6 Apr 2026 17:28:55 +0000 Subject: [PATCH 06/26] fix bfloat16 crash on T4 for FORCE_FLOAT32 models and disable trust_remote_code auto-enable for native t5 models --- studio/backend/core/training/worker.py | 17 ----------------- unsloth/models/loader.py | 4 +++- 2 files changed, 3 insertions(+), 18 deletions(-) diff --git a/studio/backend/core/training/worker.py b/studio/backend/core/training/worker.py index dbd3628ce9..ef5cafb175 100644 --- a/studio/backend/core/training/worker.py +++ b/studio/backend/core/training/worker.py @@ -1014,7 +1014,6 @@ def run_training_process( ) return -<<<<<<< HEAD # ── 1a. Auto-enable trust_remote_code for NemotronH/Nano models ── # NemotronH has config parsing bugs in transformers that require # trust_remote_code=True as a workaround. Other transformers 5.x models @@ -1026,22 +1025,6 @@ def run_training_process( if ( any(sub in _lowered for sub in _NEMOTRON_TRUST_SUBSTRINGS) and (_lowered.startswith("unsloth/") or _lowered.startswith("nvidia/")) -======= - # ── 1a. Auto-enable trust_remote_code for unsloth/* transformers 5.x models ── - # Some newer architectures (e.g. NemotronH) have config parsing bugs in - # transformers that require trust_remote_code=True as a workaround. - # Only auto-enable for unsloth/* prefixed models (trusted source). - # Exclude Gemma 4 since it is a native transformers 5.5 model and - # trust_remote_code=True would bypass the compiler (disabling fused CE). - from utils.transformers_version import get_transformers_tier - - _lowered = model_name.lower() - _tier = get_transformers_tier(model_name) - if ( - _tier != "default" - and _lowered.startswith("unsloth/") - and _tier != "550" # Gemma 4 is native t5.5 — trust_remote_code bypasses compiler ->>>>>>> 970219a3 (split venv_t5 into venv_t5_530 and venv_t5_550 for tiered transformers 5.x support) and not config.get("trust_remote_code", False) ): config["trust_remote_code"] = True diff --git a/unsloth/models/loader.py b/unsloth/models/loader.py index c413de95dc..8aa1161b8c 100644 --- a/unsloth/models/loader.py +++ b/unsloth/models/loader.py @@ -1371,7 +1371,9 @@ class FastModel(FastBaseModel): or disable_name.lower() in model_types_all ) and ((dtype == torch.float16) or not SUPPORTS_BFLOAT16): os.environ["UNSLOTH_FORCE_FLOAT32"] = "1" - dtype = torch.bfloat16 # Change to bfloat16 loading + # Use bfloat16 storage where supported; fall back to float32 on + # older GPUs (e.g. T4) that lack native bfloat16 support. + dtype = torch.bfloat16 if SUPPORTS_BFLOAT16 else torch.float32 break # Apply gradient checkpointing with smart heuristics use_gradient_checkpointing = apply_unsloth_gradient_checkpointing( From 300d1d4f610ffa56b00e6457c649c6f3a36e90a7 Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Mon, 6 Apr 2026 17:29:22 +0000 Subject: [PATCH 07/26] revert FORCE_FLOAT32 dtype change --- unsloth/models/loader.py | 4 +--- 1 file changed, 1 insertion(+), 3 deletions(-) diff --git a/unsloth/models/loader.py b/unsloth/models/loader.py index 8aa1161b8c..c413de95dc 100644 --- a/unsloth/models/loader.py +++ b/unsloth/models/loader.py @@ -1371,9 +1371,7 @@ class FastModel(FastBaseModel): or disable_name.lower() in model_types_all ) and ((dtype == torch.float16) or not SUPPORTS_BFLOAT16): os.environ["UNSLOTH_FORCE_FLOAT32"] = "1" - # Use bfloat16 storage where supported; fall back to float32 on - # older GPUs (e.g. T4) that lack native bfloat16 support. - dtype = torch.bfloat16 if SUPPORTS_BFLOAT16 else torch.float32 + dtype = torch.bfloat16 # Change to bfloat16 loading break # Apply gradient checkpointing with smart heuristics use_gradient_checkpointing = apply_unsloth_gradient_checkpointing( From 047d7d24dfbecac7c335cd08c58f6b2e7995edf7 Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Mon, 6 Apr 2026 18:20:08 +0000 Subject: [PATCH 08/26] restrict trust_remote_code auto-enable to Nemotron models only --- studio/backend/core/inference/worker.py | 52 +++++++-- studio/backend/core/training/worker.py | 139 ++++++++++++++++++++++-- 2 files changed, 173 insertions(+), 18 deletions(-) diff --git a/studio/backend/core/inference/worker.py b/studio/backend/core/inference/worker.py index 9829bfe232..336db7f5b7 100644 --- a/studio/backend/core/inference/worker.py +++ b/studio/backend/core/inference/worker.py @@ -34,13 +34,50 @@ from utils.hardware import apply_gpu_ids def _activate_transformers_version(model_name: str) -> None: - """Activate the correct transformers version BEFORE any ML imports.""" + """Activate the correct transformers version BEFORE any ML imports. + + Uses get_transformers_tier() to decide between .venv_t5_550/ (5.5.0), + .venv_t5_530/ (5.3.0), or the default 4.57.x. + """ + # Ensure backend is on path for utils imports + backend_path = str(Path(__file__).resolve().parent.parent.parent) if backend_path not in sys.path: sys.path.insert(0, backend_path) - from utils.transformers_version import activate_transformers_for_subprocess + from utils.transformers_version import ( + get_transformers_tier, + _resolve_base_model, + _ensure_venv_t5_530_exists, + _ensure_venv_t5_550_exists, + _VENV_T5_530_DIR, + _VENV_T5_550_DIR, + ) - activate_transformers_for_subprocess(model_name) + resolved = _resolve_base_model(model_name) + tier = get_transformers_tier(resolved) + + if tier == "550": + if not _ensure_venv_t5_550_exists(): + raise RuntimeError( + f"Cannot activate transformers 5.5.0: .venv_t5_550 missing at {_VENV_T5_550_DIR}" + ) + if _VENV_T5_550_DIR not in sys.path: + sys.path.insert(0, _VENV_T5_550_DIR) + logger.info("Activated transformers 5.5.0 from %s", _VENV_T5_550_DIR) + _pp = os.environ.get("PYTHONPATH", "") + os.environ["PYTHONPATH"] = _VENV_T5_550_DIR + (os.pathsep + _pp if _pp else "") + elif tier == "530": + if not _ensure_venv_t5_530_exists(): + raise RuntimeError( + f"Cannot activate transformers 5.3.0: .venv_t5_530 missing at {_VENV_T5_530_DIR}" + ) + if _VENV_T5_530_DIR not in sys.path: + sys.path.insert(0, _VENV_T5_530_DIR) + logger.info("Activated transformers 5.3.0 from %s", _VENV_T5_530_DIR) + _pp = os.environ.get("PYTHONPATH", "") + os.environ["PYTHONPATH"] = _VENV_T5_530_DIR + (os.pathsep + _pp if _pp else "") + else: + logger.info("Using default transformers (4.57.x) for %s", model_name) def _decode_image(image_base64: str): @@ -285,18 +322,13 @@ def _handle_load(backend, config: dict, resp_queue: Any) -> None: except Exception as e: logger.warning("Could not read adapter_config.json: %s", e) - # Auto-enable trust_remote_code for NemotronH/Nano models only. + # Auto-enable trust_remote_code for Nemotron models only. # NemotronH has config parsing bugs requiring trust_remote_code=True. # Other transformers 5.x models are native and do NOT need it. - # NOTE: Must NOT match Llama-Nemotron (standard Llama architecture). - _NEMOTRON_TRUST_SUBSTRINGS = ("nemotron_h", "nemotron-h", "nemotron-3-nano") trust_remote_code = config.get("trust_remote_code", False) if not trust_remote_code: model_name = config["model_name"] - _mn_lower = model_name.lower() - if any(sub in _mn_lower for sub in _NEMOTRON_TRUST_SUBSTRINGS) and ( - _mn_lower.startswith("unsloth/") or _mn_lower.startswith("nvidia/") - ): + if "nemotron" in model_name.lower(): trust_remote_code = True logger.info( "Auto-enabled trust_remote_code for Nemotron model: %s", diff --git a/studio/backend/core/training/worker.py b/studio/backend/core/training/worker.py index ef5cafb175..2225e93f41 100644 --- a/studio/backend/core/training/worker.py +++ b/studio/backend/core/training/worker.py @@ -76,6 +76,97 @@ def _model_wants_causal_conv1d(model_name: str) -> bool: ) +def _causal_conv1d_platform_tag() -> str | None: + machine = platform.machine().lower() + if sys.platform.startswith("linux"): + if machine in {"x86_64", "amd64"}: + return "linux_x86_64" + if machine in {"aarch64", "arm64"}: + return "linux_aarch64" + return None + # No prebuilt wheels published for macOS or Windows + return None + + +def _probe_causal_conv1d_env() -> dict[str, str] | None: + try: + probe = _sp.run( + [ + sys.executable, + "-c", + ( + "import json, sys, re, torch; " + "parts = torch.__version__.split('+', 1)[0].split('.')[:2]; " + "minor = re.sub(r'[^0-9].*', '', parts[1]) if len(parts) > 1 else '0'; " + "torch_mm = parts[0] + '.' + minor; " + "print(json.dumps({" + "'python_tag': f'cp{sys.version_info.major}{sys.version_info.minor}', " + "'torch_mm': torch_mm, " + "'cuda_major': str(int(str(torch.version.cuda).split('.', 1)[0])) if torch.version.cuda else '', " + "'cxx11abi': str(torch._C._GLIBCXX_USE_CXX11_ABI).upper()" + "}))" + ), + ], + stdout = _sp.PIPE, + stderr = _sp.PIPE, + text = True, + timeout = 30, + ) + except _sp.TimeoutExpired: + logger.warning("Torch environment probe timed out after 30s") + return None + if probe.returncode != 0: + logger.warning( + "Failed to probe torch environment for causal-conv1d wheel:\n%s", + probe.stdout, + ) + return None + + try: + return json.loads(probe.stdout.strip()) + except json.JSONDecodeError: + logger.warning( + "Failed to parse torch environment probe output: %s", probe.stdout + ) + return None + + +def _direct_wheel_url( + *, + filename_prefix: str, + package_version: str, + release_tag: str, + release_base_url: str, + env: dict[str, str] | None = None, +) -> str | None: + env = env or _probe_causal_conv1d_env() + platform_tag = _causal_conv1d_platform_tag() + if env is None or platform_tag is None or not env.get("cuda_major"): + return None + + filename = ( + f"{filename_prefix}-{package_version}" + f"+cu{env['cuda_major']}torch{env['torch_mm']}" + f"cxx11abi{env['cxx11abi']}-{env['python_tag']}-{env['python_tag']}-{platform_tag}.whl" + ) + return f"{release_base_url}/{release_tag}/{filename}" + + +def _url_exists(url: str) -> bool: + try: + request = urllib.request.Request(url, method = "HEAD") + with urllib.request.urlopen(request, timeout = 10): + return True + except urllib.error.HTTPError as exc: + if exc.code == 404: + return False + logger.warning("Unexpected HTTP error while probing %s: %s", url, exc) + return False + except Exception as exc: + logger.warning("Failed to probe %s: %s", url, exc) + return False + + def _install_package_wheel_first( *, event_queue: Any, @@ -328,15 +419,50 @@ def _ensure_flash_attn_for_long_context(event_queue: Any, max_seq_length: int) - def _activate_transformers_version(model_name: str) -> None: - """Activate the correct transformers version BEFORE any ML imports.""" + """Activate the correct transformers version BEFORE any ML imports. + + Uses get_transformers_tier() to decide between .venv_t5_550/ (5.5.0), + .venv_t5_530/ (5.3.0), or the default 4.57.x. + """ # Ensure backend is on path for utils imports backend_path = str(Path(__file__).resolve().parent.parent.parent) if backend_path not in sys.path: sys.path.insert(0, backend_path) - from utils.transformers_version import activate_transformers_for_subprocess + from utils.transformers_version import ( + get_transformers_tier, + _resolve_base_model, + _ensure_venv_t5_530_exists, + _ensure_venv_t5_550_exists, + _VENV_T5_530_DIR, + _VENV_T5_550_DIR, + ) - activate_transformers_for_subprocess(model_name) + resolved = _resolve_base_model(model_name) + tier = get_transformers_tier(resolved) + + if tier == "550": + if not _ensure_venv_t5_550_exists(): + raise RuntimeError( + f"Cannot activate transformers 5.5.0: .venv_t5_550 missing at {_VENV_T5_550_DIR}" + ) + if _VENV_T5_550_DIR not in sys.path: + sys.path.insert(0, _VENV_T5_550_DIR) + logger.info("Activated transformers 5.5.0 from %s", _VENV_T5_550_DIR) + _pp = os.environ.get("PYTHONPATH", "") + os.environ["PYTHONPATH"] = _VENV_T5_550_DIR + (os.pathsep + _pp if _pp else "") + elif tier == "530": + if not _ensure_venv_t5_530_exists(): + raise RuntimeError( + f"Cannot activate transformers 5.3.0: .venv_t5_530 missing at {_VENV_T5_530_DIR}" + ) + if _VENV_T5_530_DIR not in sys.path: + sys.path.insert(0, _VENV_T5_530_DIR) + logger.info("Activated transformers 5.3.0 from %s", _VENV_T5_530_DIR) + _pp = os.environ.get("PYTHONPATH", "") + os.environ["PYTHONPATH"] = _VENV_T5_530_DIR + (os.pathsep + _pp if _pp else "") + else: + logger.info("Using default transformers (4.57.x) for %s", model_name) def _adapt_for_mlx_vlm(items): @@ -1014,17 +1140,14 @@ def run_training_process( ) return - # ── 1a. Auto-enable trust_remote_code for NemotronH/Nano models ── + # ── 1a. Auto-enable trust_remote_code for Nemotron models ── # NemotronH has config parsing bugs in transformers that require # trust_remote_code=True as a workaround. Other transformers 5.x models # (Qwen3.5, Gemma 4, etc.) are native and do NOT need it — enabling it # bypasses the compiler (disabling fused CE). - # NOTE: Must NOT match Llama-Nemotron (standard Llama architecture). - _NEMOTRON_TRUST_SUBSTRINGS = ("nemotron_h", "nemotron-h", "nemotron-3-nano") _lowered = model_name.lower() if ( - any(sub in _lowered for sub in _NEMOTRON_TRUST_SUBSTRINGS) - and (_lowered.startswith("unsloth/") or _lowered.startswith("nvidia/")) + "nemotron" in _lowered and not config.get("trust_remote_code", False) ): config["trust_remote_code"] = True From 09afc5082ea303a90e2b6ab984e9b69867166afb Mon Sep 17 00:00:00 2001 From: "pre-commit-ci[bot]" <66853113+pre-commit-ci[bot]@users.noreply.github.com> Date: Mon, 6 Apr 2026 19:09:39 +0000 Subject: [PATCH 09/26] [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci --- studio/backend/core/training/worker.py | 5 +---- 1 file changed, 1 insertion(+), 4 deletions(-) diff --git a/studio/backend/core/training/worker.py b/studio/backend/core/training/worker.py index 2225e93f41..096e4577be 100644 --- a/studio/backend/core/training/worker.py +++ b/studio/backend/core/training/worker.py @@ -1146,10 +1146,7 @@ def run_training_process( # (Qwen3.5, Gemma 4, etc.) are native and do NOT need it — enabling it # bypasses the compiler (disabling fused CE). _lowered = model_name.lower() - if ( - "nemotron" in _lowered - and not config.get("trust_remote_code", False) - ): + if "nemotron" in _lowered and not config.get("trust_remote_code", False): config["trust_remote_code"] = True logger.info( "Auto-enabled trust_remote_code for Nemotron model: %s", From 826a91190b2f9e7c411218dc35c953ff463c1ee6 Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Mon, 6 Apr 2026 19:40:11 +0000 Subject: [PATCH 10/26] use config.json model_type for tier detection, add unsloth/nvidia namespace guard --- studio/backend/core/inference/worker.py | 6 +- studio/backend/core/training/worker.py | 6 +- .../tests/test_transformers_version.py | 25 ++- studio/backend/utils/transformers_version.py | 161 ++++++++---------- 4 files changed, 99 insertions(+), 99 deletions(-) diff --git a/studio/backend/core/inference/worker.py b/studio/backend/core/inference/worker.py index 336db7f5b7..4af5a4fd86 100644 --- a/studio/backend/core/inference/worker.py +++ b/studio/backend/core/inference/worker.py @@ -328,7 +328,11 @@ def _handle_load(backend, config: dict, resp_queue: Any) -> None: trust_remote_code = config.get("trust_remote_code", False) if not trust_remote_code: model_name = config["model_name"] - if "nemotron" in model_name.lower(): + _mn_lower = model_name.lower() + if ( + "nemotron" in _mn_lower + and (_mn_lower.startswith("unsloth/") or _mn_lower.startswith("nvidia/")) + ): trust_remote_code = True logger.info( "Auto-enabled trust_remote_code for Nemotron model: %s", diff --git a/studio/backend/core/training/worker.py b/studio/backend/core/training/worker.py index 096e4577be..d36597d600 100644 --- a/studio/backend/core/training/worker.py +++ b/studio/backend/core/training/worker.py @@ -1146,7 +1146,11 @@ def run_training_process( # (Qwen3.5, Gemma 4, etc.) are native and do NOT need it — enabling it # bypasses the compiler (disabling fused CE). _lowered = model_name.lower() - if "nemotron" in _lowered and not config.get("trust_remote_code", False): + if ( + "nemotron" in _lowered + and (_lowered.startswith("unsloth/") or _lowered.startswith("nvidia/")) + and not config.get("trust_remote_code", False) + ): config["trust_remote_code"] = True logger.info( "Auto-enabled trust_remote_code for Nemotron model: %s", diff --git a/studio/backend/tests/test_transformers_version.py b/studio/backend/tests/test_transformers_version.py index c031c2fea3..609a154a9a 100644 --- a/studio/backend/tests/test_transformers_version.py +++ b/studio/backend/tests/test_transformers_version.py @@ -32,8 +32,9 @@ from utils.transformers_version import ( _resolve_base_model, _check_tokenizer_config_needs_v5, _check_config_needs_550, + _get_config_json, _tokenizer_class_cache, - _config_needs_550_cache, + _config_json_cache, needs_transformers_5, get_transformers_tier, ) @@ -202,7 +203,7 @@ class TestCheckConfigNeeds550: """Tests for _check_config_needs_550() local config.json checks.""" def setup_method(self): - _config_needs_550_cache.clear() + _config_json_cache.clear() def test_gemma4_architecture(self, tmp_path: Path): """config.json with Gemma4ForConditionalGeneration should return True.""" @@ -242,8 +243,8 @@ class TestCheckConfigNeeds550: key = str(tmp_path) _check_config_needs_550(key) - assert key in _config_needs_550_cache - assert _config_needs_550_cache[key] is True + assert key in _config_json_cache + assert _config_json_cache[key] is not None def test_local_file_skips_network(self, tmp_path: Path): """When local config.json exists, no network request should be made.""" @@ -265,7 +266,7 @@ class TestGetTransformersTier: def setup_method(self): _tokenizer_class_cache.clear() - _config_needs_550_cache.clear() + _config_json_cache.clear() def test_gemma4_substring_returns_550(self): assert get_transformers_tier("google/gemma-4-E2B-it") == "550" @@ -317,6 +318,20 @@ class TestGetTransformersTier: # This shouldn't happen in practice, but verifies priority assert get_transformers_tier("gemma-4-model") == "550" + def test_config_json_model_type_530(self, tmp_path: Path): + """Local checkpoint with qwen3_moe model_type → 530.""" + cfg = {"model_type": "qwen3_moe", "architectures": ["Qwen3MoeForCausalLM"]} + (tmp_path / "config.json").write_text(json.dumps(cfg)) + + assert get_transformers_tier(str(tmp_path)) == "530" + + def test_config_json_model_type_glm4_moe(self, tmp_path: Path): + """Local checkpoint with glm4_moe model_type → 530.""" + cfg = {"model_type": "glm4_moe", "architectures": ["Glm4MoeForCausalLM"]} + (tmp_path / "config.json").write_text(json.dumps(cfg)) + + assert get_transformers_tier(str(tmp_path)) == "530" + def test_needs_transformers_5_compat(self): """needs_transformers_5 should return True for both 530 and 550 models.""" assert needs_transformers_5("google/gemma-4-E2B-it") is True diff --git a/studio/backend/utils/transformers_version.py b/studio/backend/utils/transformers_version.py index 9075c590ca..a0395b3e54 100644 --- a/studio/backend/utils/transformers_version.py +++ b/studio/backend/utils/transformers_version.py @@ -67,14 +67,25 @@ TRANSFORMERS_550_MODEL_SUBSTRINGS: tuple[str, ...] = ( "qwen3.6", ) -# Architecture classes / model_type values that require transformers 5.5.0. -# Checked via config.json (local or HuggingFace). +# Architecture classes that require transformers 5.5.0. _TRANSFORMERS_550_ARCHITECTURES: set[str] = { "Gemma4ForConditionalGeneration", } + +# model_type values (from config.json) → tier mapping. _TRANSFORMERS_550_MODEL_TYPES: set[str] = { "gemma4", } +_TRANSFORMERS_530_MODEL_TYPES: set[str] = { + "qwen3_moe", + "qwen3_5_moe", + "qwen3_vl_moe", + "qwen3_next", + "deepseek_v3_moe", + "glm4_moe", + "glm4_moe_lite", + "ministral", +} # Tokenizer classes that only exist in transformers>=5.x _TRANSFORMERS_5_TOKENIZER_CLASSES: set[str] = { @@ -84,15 +95,14 @@ _TRANSFORMERS_5_TOKENIZER_CLASSES: set[str] = { # Cache for dynamic tokenizer_config.json lookups to avoid repeated fetches _tokenizer_class_cache: dict[str, bool] = {} -# Cache for dynamic config.json lookups (architecture/model_type checks) -_config_needs_550_cache: dict[str, bool] = {} +# Cache for config.json lookups (returns the parsed dict or None) +_config_json_cache: dict[str, dict | None] = {} # Versions TRANSFORMERS_550_VERSION = "5.5.0" TRANSFORMERS_530_VERSION = "5.3.0" TRANSFORMERS_DEFAULT_VERSION = "4.57.6" -# Backwards-compat alias — points to 5.5.0 (the highest 5.x tier). -# Consumers should prefer TRANSFORMERS_530_VERSION / TRANSFORMERS_550_VERSION. +# Backwards-compat alias used by other modules TRANSFORMERS_5_VERSION = TRANSFORMERS_550_VERSION # Pre-installed directories — created by setup.sh / setup.ps1. @@ -104,45 +114,6 @@ _VENV_T5_550_DIR = str(_studio_root() / ".venv_t5_550") _VENV_T5_DIR = _VENV_T5_550_DIR -def activate_transformers_for_subprocess(model_name: str) -> None: - """Activate the correct transformers version in a subprocess worker. - - Call this BEFORE any ML imports. Resolves LoRA adapters to their base - model, determines the required tier, and prepends the appropriate - ``.venv_t5_*`` directory to ``sys.path``. Also propagates the path - via ``PYTHONPATH`` for child processes (e.g. GGUF converter). - - Used by training, inference, and export workers. - """ - resolved = _resolve_base_model(model_name) - tier = get_transformers_tier(resolved) - - if tier == "550": - if not _ensure_venv_t5_550_exists(): - raise RuntimeError( - f"Cannot activate transformers 5.5.0: " - f".venv_t5_550 missing at {_VENV_T5_550_DIR}" - ) - if _VENV_T5_550_DIR not in sys.path: - sys.path.insert(0, _VENV_T5_550_DIR) - logger.info("Activated transformers 5.5.0 from %s", _VENV_T5_550_DIR) - _pp = os.environ.get("PYTHONPATH", "") - os.environ["PYTHONPATH"] = _VENV_T5_550_DIR + (os.pathsep + _pp if _pp else "") - elif tier == "530": - if not _ensure_venv_t5_530_exists(): - raise RuntimeError( - f"Cannot activate transformers 5.3.0: " - f".venv_t5_530 missing at {_VENV_T5_530_DIR}" - ) - if _VENV_T5_530_DIR not in sys.path: - sys.path.insert(0, _VENV_T5_530_DIR) - logger.info("Activated transformers 5.3.0 from %s", _VENV_T5_530_DIR) - _pp = os.environ.get("PYTHONPATH", "") - os.environ["PYTHONPATH"] = _VENV_T5_530_DIR + (os.pathsep + _pp if _pp else "") - else: - logger.info("Using default transformers (4.57.x) for %s", model_name) - - def _resolve_base_model(model_name: str) -> str: """If *model_name* points to a LoRA adapter, return its base model. @@ -268,43 +239,28 @@ def _check_tokenizer_config_needs_v5(model_name: str) -> bool: return False -def _check_config_needs_550(model_name: str) -> bool: - """Check ``config.json`` for architectures or model_type that require - transformers 5.5.0 (e.g. Gemma 4). +_SENTINEL = object() # distinguishes "not cached" from "cached as None" - Checks locally first, then falls back to fetching from HuggingFace. - Results are cached in ``_config_needs_550_cache``. - Returns False on any error (fail-open to lower tier). + +def _get_config_json(model_name: str) -> dict | None: + """Read and cache ``config.json`` for *model_name*. + + Checks local path first, then fetches from HuggingFace. + Returns the parsed dict, or ``None`` on any error (fail-open). + The result is cached in ``_config_json_cache``. """ - if model_name in _config_needs_550_cache: - return _config_needs_550_cache[model_name] - - def _check_cfg(cfg: dict) -> bool: - archs = cfg.get("architectures", []) - if any(a in _TRANSFORMERS_550_ARCHITECTURES for a in archs): - return True - if cfg.get("model_type") in _TRANSFORMERS_550_MODEL_TYPES: - return True - return False + cached = _config_json_cache.get(model_name, _SENTINEL) + if cached is not _SENTINEL: + return cached # --- Check local config.json first ------------------------------------ - local_path = Path(model_name) - local_cfg = local_path / "config.json" + local_cfg = Path(model_name) / "config.json" if local_cfg.is_file(): try: with open(local_cfg) as f: cfg = json.load(f) - result = _check_cfg(cfg) - if result: - logger.info( - "Local config.json check: %s needs transformers 5.5.0 " - "(architectures=%s, model_type=%s)", - model_name, - cfg.get("architectures", []), - cfg.get("model_type"), - ) - _config_needs_550_cache[model_name] = result - return result + _config_json_cache[model_name] = cfg + return cfg except Exception as exc: logger.debug("Could not read %s: %s", local_cfg, exc) @@ -316,21 +272,26 @@ def _check_config_needs_550(model_name: str) -> bool: req = urllib.request.Request(url, headers = {"User-Agent": "unsloth-studio"}) with urllib.request.urlopen(req, timeout = 10) as resp: cfg = json.loads(resp.read().decode()) - result = _check_cfg(cfg) - if result: - logger.info( - "Dynamic config.json check: %s needs transformers 5.5.0 " - "(architectures=%s, model_type=%s)", - model_name, - cfg.get("architectures", []), - cfg.get("model_type"), - ) - _config_needs_550_cache[model_name] = result - return result + _config_json_cache[model_name] = cfg + return cfg except Exception as exc: - logger.debug("Could not fetch config.json for '%s': %s", model_name, exc) - _config_needs_550_cache[model_name] = False + logger.debug( + "Could not fetch config.json for '%s': %s", model_name, exc + ) + _config_json_cache[model_name] = None + return None + + +def _check_config_needs_550(model_name: str) -> bool: + """Check ``config.json`` for architectures or model_type that require + transformers 5.5.0. Uses the shared ``_get_config_json`` cache.""" + cfg = _get_config_json(model_name) + if cfg is None: return False + archs = cfg.get("architectures", []) + if any(a in _TRANSFORMERS_550_ARCHITECTURES for a in archs): + return True + return cfg.get("model_type") in _TRANSFORMERS_550_MODEL_TYPES def get_transformers_tier(model_name: str) -> str: @@ -340,19 +301,35 @@ def get_transformers_tier(model_name: str) -> str: ``"530"`` for models needing transformers 5.3.0 (e.g. Ministral-3, Qwen3 MoE), or ``"default"`` for everything else (4.57.x). - The 5.5.0 check runs first, then 5.3.0. + Fast path: substring checks (no I/O) for both tiers run first. + Slow path: single config.json fetch (cached) checks model_type for + both tiers, then tokenizer_config.json as final fallback. """ lowered = model_name.lower() - # --- Fast substring checks (no I/O) ------------------------------------ + # --- Fast substring checks (no I/O) ----------------------------------- if any(sub in lowered for sub in TRANSFORMERS_550_MODEL_SUBSTRINGS): return "550" if any(sub in lowered for sub in TRANSFORMERS_5_MODEL_SUBSTRINGS): return "530" - # --- Slow config fallbacks (local file first, then network) ----------- - if _check_config_needs_550(model_name): - return "550" + # --- config.json model_type / architecture check (single fetch) ------- + cfg = _get_config_json(model_name) + if cfg is not None: + model_type = cfg.get("model_type", "") + archs = cfg.get("architectures", []) + + # Check 5.5.0 first + if model_type in _TRANSFORMERS_550_MODEL_TYPES: + return "550" + if any(a in _TRANSFORMERS_550_ARCHITECTURES for a in archs): + return "550" + + # Check 5.3.0 + if model_type in _TRANSFORMERS_530_MODEL_TYPES: + return "530" + + # --- Final fallback: tokenizer_config.json for 5.3.0 ------------------ if _check_tokenizer_config_needs_v5(model_name): return "530" From ae5e57d1f8cd6090f8c2ea9ad3194bdb002dba2a Mon Sep 17 00:00:00 2001 From: "pre-commit-ci[bot]" <66853113+pre-commit-ci[bot]@users.noreply.github.com> Date: Mon, 6 Apr 2026 19:42:48 +0000 Subject: [PATCH 11/26] [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci --- studio/backend/core/inference/worker.py | 5 ++--- studio/backend/utils/transformers_version.py | 4 +--- 2 files changed, 3 insertions(+), 6 deletions(-) diff --git a/studio/backend/core/inference/worker.py b/studio/backend/core/inference/worker.py index 4af5a4fd86..e242844689 100644 --- a/studio/backend/core/inference/worker.py +++ b/studio/backend/core/inference/worker.py @@ -329,9 +329,8 @@ def _handle_load(backend, config: dict, resp_queue: Any) -> None: if not trust_remote_code: model_name = config["model_name"] _mn_lower = model_name.lower() - if ( - "nemotron" in _mn_lower - and (_mn_lower.startswith("unsloth/") or _mn_lower.startswith("nvidia/")) + if "nemotron" in _mn_lower and ( + _mn_lower.startswith("unsloth/") or _mn_lower.startswith("nvidia/") ): trust_remote_code = True logger.info( diff --git a/studio/backend/utils/transformers_version.py b/studio/backend/utils/transformers_version.py index a0395b3e54..14f1f770d9 100644 --- a/studio/backend/utils/transformers_version.py +++ b/studio/backend/utils/transformers_version.py @@ -275,9 +275,7 @@ def _get_config_json(model_name: str) -> dict | None: _config_json_cache[model_name] = cfg return cfg except Exception as exc: - logger.debug( - "Could not fetch config.json for '%s': %s", model_name, exc - ) + logger.debug("Could not fetch config.json for '%s': %s", model_name, exc) _config_json_cache[model_name] = None return None From 6be0585b5714bb7b11c90bd2a667e9b72d313f07 Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Mon, 6 Apr 2026 20:26:32 +0000 Subject: [PATCH 12/26] Revert "[pre-commit.ci] auto fixes from pre-commit.com hooks" This reverts commit fb43d468e25379f28dd2477e6c24dd60cf55c099. --- studio/backend/core/inference/worker.py | 5 +++-- studio/backend/utils/transformers_version.py | 4 +++- 2 files changed, 6 insertions(+), 3 deletions(-) diff --git a/studio/backend/core/inference/worker.py b/studio/backend/core/inference/worker.py index e242844689..4af5a4fd86 100644 --- a/studio/backend/core/inference/worker.py +++ b/studio/backend/core/inference/worker.py @@ -329,8 +329,9 @@ def _handle_load(backend, config: dict, resp_queue: Any) -> None: if not trust_remote_code: model_name = config["model_name"] _mn_lower = model_name.lower() - if "nemotron" in _mn_lower and ( - _mn_lower.startswith("unsloth/") or _mn_lower.startswith("nvidia/") + if ( + "nemotron" in _mn_lower + and (_mn_lower.startswith("unsloth/") or _mn_lower.startswith("nvidia/")) ): trust_remote_code = True logger.info( diff --git a/studio/backend/utils/transformers_version.py b/studio/backend/utils/transformers_version.py index 14f1f770d9..a0395b3e54 100644 --- a/studio/backend/utils/transformers_version.py +++ b/studio/backend/utils/transformers_version.py @@ -275,7 +275,9 @@ def _get_config_json(model_name: str) -> dict | None: _config_json_cache[model_name] = cfg return cfg except Exception as exc: - logger.debug("Could not fetch config.json for '%s': %s", model_name, exc) + logger.debug( + "Could not fetch config.json for '%s': %s", model_name, exc + ) _config_json_cache[model_name] = None return None From bfe72f660fb8c656f57d00deed0278500d877a15 Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Mon, 6 Apr 2026 20:26:32 +0000 Subject: [PATCH 13/26] Revert "use config.json model_type for tier detection, add unsloth/nvidia namespace guard" This reverts commit fc49ae24531780a658049e6c238f146960f466b6. --- studio/backend/core/inference/worker.py | 6 +- studio/backend/core/training/worker.py | 6 +- .../tests/test_transformers_version.py | 25 +--- studio/backend/utils/transformers_version.py | 121 ++++++++---------- 4 files changed, 59 insertions(+), 99 deletions(-) diff --git a/studio/backend/core/inference/worker.py b/studio/backend/core/inference/worker.py index 4af5a4fd86..336db7f5b7 100644 --- a/studio/backend/core/inference/worker.py +++ b/studio/backend/core/inference/worker.py @@ -328,11 +328,7 @@ def _handle_load(backend, config: dict, resp_queue: Any) -> None: trust_remote_code = config.get("trust_remote_code", False) if not trust_remote_code: model_name = config["model_name"] - _mn_lower = model_name.lower() - if ( - "nemotron" in _mn_lower - and (_mn_lower.startswith("unsloth/") or _mn_lower.startswith("nvidia/")) - ): + if "nemotron" in model_name.lower(): trust_remote_code = True logger.info( "Auto-enabled trust_remote_code for Nemotron model: %s", diff --git a/studio/backend/core/training/worker.py b/studio/backend/core/training/worker.py index d36597d600..096e4577be 100644 --- a/studio/backend/core/training/worker.py +++ b/studio/backend/core/training/worker.py @@ -1146,11 +1146,7 @@ def run_training_process( # (Qwen3.5, Gemma 4, etc.) are native and do NOT need it — enabling it # bypasses the compiler (disabling fused CE). _lowered = model_name.lower() - if ( - "nemotron" in _lowered - and (_lowered.startswith("unsloth/") or _lowered.startswith("nvidia/")) - and not config.get("trust_remote_code", False) - ): + if "nemotron" in _lowered and not config.get("trust_remote_code", False): config["trust_remote_code"] = True logger.info( "Auto-enabled trust_remote_code for Nemotron model: %s", diff --git a/studio/backend/tests/test_transformers_version.py b/studio/backend/tests/test_transformers_version.py index 609a154a9a..c031c2fea3 100644 --- a/studio/backend/tests/test_transformers_version.py +++ b/studio/backend/tests/test_transformers_version.py @@ -32,9 +32,8 @@ from utils.transformers_version import ( _resolve_base_model, _check_tokenizer_config_needs_v5, _check_config_needs_550, - _get_config_json, _tokenizer_class_cache, - _config_json_cache, + _config_needs_550_cache, needs_transformers_5, get_transformers_tier, ) @@ -203,7 +202,7 @@ class TestCheckConfigNeeds550: """Tests for _check_config_needs_550() local config.json checks.""" def setup_method(self): - _config_json_cache.clear() + _config_needs_550_cache.clear() def test_gemma4_architecture(self, tmp_path: Path): """config.json with Gemma4ForConditionalGeneration should return True.""" @@ -243,8 +242,8 @@ class TestCheckConfigNeeds550: key = str(tmp_path) _check_config_needs_550(key) - assert key in _config_json_cache - assert _config_json_cache[key] is not None + assert key in _config_needs_550_cache + assert _config_needs_550_cache[key] is True def test_local_file_skips_network(self, tmp_path: Path): """When local config.json exists, no network request should be made.""" @@ -266,7 +265,7 @@ class TestGetTransformersTier: def setup_method(self): _tokenizer_class_cache.clear() - _config_json_cache.clear() + _config_needs_550_cache.clear() def test_gemma4_substring_returns_550(self): assert get_transformers_tier("google/gemma-4-E2B-it") == "550" @@ -318,20 +317,6 @@ class TestGetTransformersTier: # This shouldn't happen in practice, but verifies priority assert get_transformers_tier("gemma-4-model") == "550" - def test_config_json_model_type_530(self, tmp_path: Path): - """Local checkpoint with qwen3_moe model_type → 530.""" - cfg = {"model_type": "qwen3_moe", "architectures": ["Qwen3MoeForCausalLM"]} - (tmp_path / "config.json").write_text(json.dumps(cfg)) - - assert get_transformers_tier(str(tmp_path)) == "530" - - def test_config_json_model_type_glm4_moe(self, tmp_path: Path): - """Local checkpoint with glm4_moe model_type → 530.""" - cfg = {"model_type": "glm4_moe", "architectures": ["Glm4MoeForCausalLM"]} - (tmp_path / "config.json").write_text(json.dumps(cfg)) - - assert get_transformers_tier(str(tmp_path)) == "530" - def test_needs_transformers_5_compat(self): """needs_transformers_5 should return True for both 530 and 550 models.""" assert needs_transformers_5("google/gemma-4-E2B-it") is True diff --git a/studio/backend/utils/transformers_version.py b/studio/backend/utils/transformers_version.py index a0395b3e54..bd8998157c 100644 --- a/studio/backend/utils/transformers_version.py +++ b/studio/backend/utils/transformers_version.py @@ -67,25 +67,14 @@ TRANSFORMERS_550_MODEL_SUBSTRINGS: tuple[str, ...] = ( "qwen3.6", ) -# Architecture classes that require transformers 5.5.0. +# Architecture classes / model_type values that require transformers 5.5.0. +# Checked via config.json (local or HuggingFace). _TRANSFORMERS_550_ARCHITECTURES: set[str] = { "Gemma4ForConditionalGeneration", } - -# model_type values (from config.json) → tier mapping. _TRANSFORMERS_550_MODEL_TYPES: set[str] = { "gemma4", } -_TRANSFORMERS_530_MODEL_TYPES: set[str] = { - "qwen3_moe", - "qwen3_5_moe", - "qwen3_vl_moe", - "qwen3_next", - "deepseek_v3_moe", - "glm4_moe", - "glm4_moe_lite", - "ministral", -} # Tokenizer classes that only exist in transformers>=5.x _TRANSFORMERS_5_TOKENIZER_CLASSES: set[str] = { @@ -95,8 +84,8 @@ _TRANSFORMERS_5_TOKENIZER_CLASSES: set[str] = { # Cache for dynamic tokenizer_config.json lookups to avoid repeated fetches _tokenizer_class_cache: dict[str, bool] = {} -# Cache for config.json lookups (returns the parsed dict or None) -_config_json_cache: dict[str, dict | None] = {} +# Cache for dynamic config.json lookups (architecture/model_type checks) +_config_needs_550_cache: dict[str, bool] = {} # Versions TRANSFORMERS_550_VERSION = "5.5.0" @@ -239,28 +228,43 @@ def _check_tokenizer_config_needs_v5(model_name: str) -> bool: return False -_SENTINEL = object() # distinguishes "not cached" from "cached as None" +def _check_config_needs_550(model_name: str) -> bool: + """Check ``config.json`` for architectures or model_type that require + transformers 5.5.0 (e.g. Gemma 4). - -def _get_config_json(model_name: str) -> dict | None: - """Read and cache ``config.json`` for *model_name*. - - Checks local path first, then fetches from HuggingFace. - Returns the parsed dict, or ``None`` on any error (fail-open). - The result is cached in ``_config_json_cache``. + Checks locally first, then falls back to fetching from HuggingFace. + Results are cached in ``_config_needs_550_cache``. + Returns False on any error (fail-open to lower tier). """ - cached = _config_json_cache.get(model_name, _SENTINEL) - if cached is not _SENTINEL: - return cached + if model_name in _config_needs_550_cache: + return _config_needs_550_cache[model_name] + + def _check_cfg(cfg: dict) -> bool: + archs = cfg.get("architectures", []) + if any(a in _TRANSFORMERS_550_ARCHITECTURES for a in archs): + return True + if cfg.get("model_type") in _TRANSFORMERS_550_MODEL_TYPES: + return True + return False # --- Check local config.json first ------------------------------------ - local_cfg = Path(model_name) / "config.json" + local_path = Path(model_name) + local_cfg = local_path / "config.json" if local_cfg.is_file(): try: with open(local_cfg) as f: cfg = json.load(f) - _config_json_cache[model_name] = cfg - return cfg + result = _check_cfg(cfg) + if result: + logger.info( + "Local config.json check: %s needs transformers 5.5.0 " + "(architectures=%s, model_type=%s)", + model_name, + cfg.get("architectures", []), + cfg.get("model_type"), + ) + _config_needs_550_cache[model_name] = result + return result except Exception as exc: logger.debug("Could not read %s: %s", local_cfg, exc) @@ -272,26 +276,21 @@ def _get_config_json(model_name: str) -> dict | None: req = urllib.request.Request(url, headers = {"User-Agent": "unsloth-studio"}) with urllib.request.urlopen(req, timeout = 10) as resp: cfg = json.loads(resp.read().decode()) - _config_json_cache[model_name] = cfg - return cfg + result = _check_cfg(cfg) + if result: + logger.info( + "Dynamic config.json check: %s needs transformers 5.5.0 " + "(architectures=%s, model_type=%s)", + model_name, + cfg.get("architectures", []), + cfg.get("model_type"), + ) + _config_needs_550_cache[model_name] = result + return result except Exception as exc: - logger.debug( - "Could not fetch config.json for '%s': %s", model_name, exc - ) - _config_json_cache[model_name] = None - return None - - -def _check_config_needs_550(model_name: str) -> bool: - """Check ``config.json`` for architectures or model_type that require - transformers 5.5.0. Uses the shared ``_get_config_json`` cache.""" - cfg = _get_config_json(model_name) - if cfg is None: + logger.debug("Could not fetch config.json for '%s': %s", model_name, exc) + _config_needs_550_cache[model_name] = False return False - archs = cfg.get("architectures", []) - if any(a in _TRANSFORMERS_550_ARCHITECTURES for a in archs): - return True - return cfg.get("model_type") in _TRANSFORMERS_550_MODEL_TYPES def get_transformers_tier(model_name: str) -> str: @@ -301,35 +300,19 @@ def get_transformers_tier(model_name: str) -> str: ``"530"`` for models needing transformers 5.3.0 (e.g. Ministral-3, Qwen3 MoE), or ``"default"`` for everything else (4.57.x). - Fast path: substring checks (no I/O) for both tiers run first. - Slow path: single config.json fetch (cached) checks model_type for - both tiers, then tokenizer_config.json as final fallback. + The 5.5.0 check runs first, then 5.3.0. """ lowered = model_name.lower() - # --- Fast substring checks (no I/O) ----------------------------------- + # --- Check 5.5.0 first ------------------------------------------------ if any(sub in lowered for sub in TRANSFORMERS_550_MODEL_SUBSTRINGS): return "550" + if _check_config_needs_550(model_name): + return "550" + + # --- Check 5.3.0 ------------------------------------------------------ if any(sub in lowered for sub in TRANSFORMERS_5_MODEL_SUBSTRINGS): return "530" - - # --- config.json model_type / architecture check (single fetch) ------- - cfg = _get_config_json(model_name) - if cfg is not None: - model_type = cfg.get("model_type", "") - archs = cfg.get("architectures", []) - - # Check 5.5.0 first - if model_type in _TRANSFORMERS_550_MODEL_TYPES: - return "550" - if any(a in _TRANSFORMERS_550_ARCHITECTURES for a in archs): - return "550" - - # Check 5.3.0 - if model_type in _TRANSFORMERS_530_MODEL_TYPES: - return "530" - - # --- Final fallback: tokenizer_config.json for 5.3.0 ------------------ if _check_tokenizer_config_needs_v5(model_name): return "530" From 74a9f52746be05ed53399fae2063cb5b9d8b1b7d Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Mon, 6 Apr 2026 20:27:15 +0000 Subject: [PATCH 14/26] add unsloth/nvidia namespace guard to Nemotron trust_remote_code auto-enable --- studio/backend/core/inference/worker.py | 6 +++++- studio/backend/core/training/worker.py | 6 +++++- 2 files changed, 10 insertions(+), 2 deletions(-) diff --git a/studio/backend/core/inference/worker.py b/studio/backend/core/inference/worker.py index 336db7f5b7..4af5a4fd86 100644 --- a/studio/backend/core/inference/worker.py +++ b/studio/backend/core/inference/worker.py @@ -328,7 +328,11 @@ def _handle_load(backend, config: dict, resp_queue: Any) -> None: trust_remote_code = config.get("trust_remote_code", False) if not trust_remote_code: model_name = config["model_name"] - if "nemotron" in model_name.lower(): + _mn_lower = model_name.lower() + if ( + "nemotron" in _mn_lower + and (_mn_lower.startswith("unsloth/") or _mn_lower.startswith("nvidia/")) + ): trust_remote_code = True logger.info( "Auto-enabled trust_remote_code for Nemotron model: %s", diff --git a/studio/backend/core/training/worker.py b/studio/backend/core/training/worker.py index 096e4577be..d36597d600 100644 --- a/studio/backend/core/training/worker.py +++ b/studio/backend/core/training/worker.py @@ -1146,7 +1146,11 @@ def run_training_process( # (Qwen3.5, Gemma 4, etc.) are native and do NOT need it — enabling it # bypasses the compiler (disabling fused CE). _lowered = model_name.lower() - if "nemotron" in _lowered and not config.get("trust_remote_code", False): + if ( + "nemotron" in _lowered + and (_lowered.startswith("unsloth/") or _lowered.startswith("nvidia/")) + and not config.get("trust_remote_code", False) + ): config["trust_remote_code"] = True logger.info( "Auto-enabled trust_remote_code for Nemotron model: %s", From 82e0a53cbc0a7528845c1d931919f9f30b67551e Mon Sep 17 00:00:00 2001 From: "pre-commit-ci[bot]" <66853113+pre-commit-ci[bot]@users.noreply.github.com> Date: Mon, 6 Apr 2026 20:27:25 +0000 Subject: [PATCH 15/26] [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci --- studio/backend/core/inference/worker.py | 5 ++--- 1 file changed, 2 insertions(+), 3 deletions(-) diff --git a/studio/backend/core/inference/worker.py b/studio/backend/core/inference/worker.py index 4af5a4fd86..e242844689 100644 --- a/studio/backend/core/inference/worker.py +++ b/studio/backend/core/inference/worker.py @@ -329,9 +329,8 @@ def _handle_load(backend, config: dict, resp_queue: Any) -> None: if not trust_remote_code: model_name = config["model_name"] _mn_lower = model_name.lower() - if ( - "nemotron" in _mn_lower - and (_mn_lower.startswith("unsloth/") or _mn_lower.startswith("nvidia/")) + if "nemotron" in _mn_lower and ( + _mn_lower.startswith("unsloth/") or _mn_lower.startswith("nvidia/") ): trust_remote_code = True logger.info( From dbe3087b6554bd65914265fdca4b19b0897f9242 Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Mon, 6 Apr 2026 20:31:06 +0000 Subject: [PATCH 16/26] reorder tier checks: all substring matches before config.json fetches --- studio/backend/utils/transformers_version.py | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/studio/backend/utils/transformers_version.py b/studio/backend/utils/transformers_version.py index bd8998157c..e692584110 100644 --- a/studio/backend/utils/transformers_version.py +++ b/studio/backend/utils/transformers_version.py @@ -304,15 +304,15 @@ def get_transformers_tier(model_name: str) -> str: """ lowered = model_name.lower() - # --- Check 5.5.0 first ------------------------------------------------ + # --- Fast substring checks (no I/O) ------------------------------------ if any(sub in lowered for sub in TRANSFORMERS_550_MODEL_SUBSTRINGS): return "550" - if _check_config_needs_550(model_name): - return "550" - - # --- Check 5.3.0 ------------------------------------------------------ if any(sub in lowered for sub in TRANSFORMERS_5_MODEL_SUBSTRINGS): return "530" + + # --- Slow config fallbacks (local file first, then network) ----------- + if _check_config_needs_550(model_name): + return "550" if _check_tokenizer_config_needs_v5(model_name): return "530" From bf73a2a85e04dbbc55ecf6aca075924e03d34c56 Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Mon, 6 Apr 2026 22:35:50 +0000 Subject: [PATCH 17/26] extract shared activate_transformers_for_subprocess into transformers_version.py --- studio/backend/core/export/worker.py | 16 ++------ studio/backend/core/inference/worker.py | 41 ++----------------- studio/backend/core/training/worker.py | 41 ++----------------- studio/backend/utils/transformers_version.py | 43 ++++++++++++++++++++ 4 files changed, 52 insertions(+), 89 deletions(-) diff --git a/studio/backend/core/export/worker.py b/studio/backend/core/export/worker.py index 1be456d4d4..80b05aca79 100644 --- a/studio/backend/core/export/worker.py +++ b/studio/backend/core/export/worker.py @@ -181,8 +181,11 @@ def _setup_log_capture(resp_queue: Any) -> None: def _activate_transformers_version(model_name: str) -> None: """Activate the correct transformers version BEFORE any ML imports.""" + # Ensure backend is on path for utils imports + backend_path = str(Path(__file__).resolve().parent.parent.parent) if backend_path not in sys.path: sys.path.insert(0, backend_path) + from utils.transformers_version import activate_transformers_for_subprocess activate_transformers_for_subprocess(model_name) @@ -203,19 +206,6 @@ def _handle_load(backend, cmd: dict, resp_queue: Any) -> None: load_in_4bit = cmd.get("load_in_4bit", True) trust_remote_code = cmd.get("trust_remote_code", False) - # Auto-enable trust_remote_code for NemotronH/Nano models. - if not trust_remote_code: - _NEMOTRON_TRUST_SUBSTRINGS = ("nemotron_h", "nemotron-h", "nemotron-3-nano") - _cp_lower = checkpoint_path.lower() - if any(sub in _cp_lower for sub in _NEMOTRON_TRUST_SUBSTRINGS) and ( - _cp_lower.startswith("unsloth/") or _cp_lower.startswith("nvidia/") - ): - trust_remote_code = True - logger.info( - "Auto-enabled trust_remote_code for Nemotron model: %s", - checkpoint_path, - ) - try: _send_response( resp_queue, diff --git a/studio/backend/core/inference/worker.py b/studio/backend/core/inference/worker.py index e242844689..65cdb9b939 100644 --- a/studio/backend/core/inference/worker.py +++ b/studio/backend/core/inference/worker.py @@ -34,50 +34,15 @@ from utils.hardware import apply_gpu_ids def _activate_transformers_version(model_name: str) -> None: - """Activate the correct transformers version BEFORE any ML imports. - - Uses get_transformers_tier() to decide between .venv_t5_550/ (5.5.0), - .venv_t5_530/ (5.3.0), or the default 4.57.x. - """ + """Activate the correct transformers version BEFORE any ML imports.""" # Ensure backend is on path for utils imports backend_path = str(Path(__file__).resolve().parent.parent.parent) if backend_path not in sys.path: sys.path.insert(0, backend_path) - from utils.transformers_version import ( - get_transformers_tier, - _resolve_base_model, - _ensure_venv_t5_530_exists, - _ensure_venv_t5_550_exists, - _VENV_T5_530_DIR, - _VENV_T5_550_DIR, - ) + from utils.transformers_version import activate_transformers_for_subprocess - resolved = _resolve_base_model(model_name) - tier = get_transformers_tier(resolved) - - if tier == "550": - if not _ensure_venv_t5_550_exists(): - raise RuntimeError( - f"Cannot activate transformers 5.5.0: .venv_t5_550 missing at {_VENV_T5_550_DIR}" - ) - if _VENV_T5_550_DIR not in sys.path: - sys.path.insert(0, _VENV_T5_550_DIR) - logger.info("Activated transformers 5.5.0 from %s", _VENV_T5_550_DIR) - _pp = os.environ.get("PYTHONPATH", "") - os.environ["PYTHONPATH"] = _VENV_T5_550_DIR + (os.pathsep + _pp if _pp else "") - elif tier == "530": - if not _ensure_venv_t5_530_exists(): - raise RuntimeError( - f"Cannot activate transformers 5.3.0: .venv_t5_530 missing at {_VENV_T5_530_DIR}" - ) - if _VENV_T5_530_DIR not in sys.path: - sys.path.insert(0, _VENV_T5_530_DIR) - logger.info("Activated transformers 5.3.0 from %s", _VENV_T5_530_DIR) - _pp = os.environ.get("PYTHONPATH", "") - os.environ["PYTHONPATH"] = _VENV_T5_530_DIR + (os.pathsep + _pp if _pp else "") - else: - logger.info("Using default transformers (4.57.x) for %s", model_name) + activate_transformers_for_subprocess(model_name) def _decode_image(image_base64: str): diff --git a/studio/backend/core/training/worker.py b/studio/backend/core/training/worker.py index d36597d600..fbfc3f99be 100644 --- a/studio/backend/core/training/worker.py +++ b/studio/backend/core/training/worker.py @@ -419,50 +419,15 @@ def _ensure_flash_attn_for_long_context(event_queue: Any, max_seq_length: int) - def _activate_transformers_version(model_name: str) -> None: - """Activate the correct transformers version BEFORE any ML imports. - - Uses get_transformers_tier() to decide between .venv_t5_550/ (5.5.0), - .venv_t5_530/ (5.3.0), or the default 4.57.x. - """ + """Activate the correct transformers version BEFORE any ML imports.""" # Ensure backend is on path for utils imports backend_path = str(Path(__file__).resolve().parent.parent.parent) if backend_path not in sys.path: sys.path.insert(0, backend_path) - from utils.transformers_version import ( - get_transformers_tier, - _resolve_base_model, - _ensure_venv_t5_530_exists, - _ensure_venv_t5_550_exists, - _VENV_T5_530_DIR, - _VENV_T5_550_DIR, - ) + from utils.transformers_version import activate_transformers_for_subprocess - resolved = _resolve_base_model(model_name) - tier = get_transformers_tier(resolved) - - if tier == "550": - if not _ensure_venv_t5_550_exists(): - raise RuntimeError( - f"Cannot activate transformers 5.5.0: .venv_t5_550 missing at {_VENV_T5_550_DIR}" - ) - if _VENV_T5_550_DIR not in sys.path: - sys.path.insert(0, _VENV_T5_550_DIR) - logger.info("Activated transformers 5.5.0 from %s", _VENV_T5_550_DIR) - _pp = os.environ.get("PYTHONPATH", "") - os.environ["PYTHONPATH"] = _VENV_T5_550_DIR + (os.pathsep + _pp if _pp else "") - elif tier == "530": - if not _ensure_venv_t5_530_exists(): - raise RuntimeError( - f"Cannot activate transformers 5.3.0: .venv_t5_530 missing at {_VENV_T5_530_DIR}" - ) - if _VENV_T5_530_DIR not in sys.path: - sys.path.insert(0, _VENV_T5_530_DIR) - logger.info("Activated transformers 5.3.0 from %s", _VENV_T5_530_DIR) - _pp = os.environ.get("PYTHONPATH", "") - os.environ["PYTHONPATH"] = _VENV_T5_530_DIR + (os.pathsep + _pp if _pp else "") - else: - logger.info("Using default transformers (4.57.x) for %s", model_name) + activate_transformers_for_subprocess(model_name) def _adapt_for_mlx_vlm(items): diff --git a/studio/backend/utils/transformers_version.py b/studio/backend/utils/transformers_version.py index e692584110..d38b4ada28 100644 --- a/studio/backend/utils/transformers_version.py +++ b/studio/backend/utils/transformers_version.py @@ -103,6 +103,49 @@ _VENV_T5_550_DIR = str(_studio_root() / ".venv_t5_550") _VENV_T5_DIR = _VENV_T5_550_DIR +def activate_transformers_for_subprocess(model_name: str) -> None: + """Activate the correct transformers version in a subprocess worker. + + Call this BEFORE any ML imports. Resolves LoRA adapters to their base + model, determines the required tier, and prepends the appropriate + ``.venv_t5_*`` directory to ``sys.path``. Also propagates the path + via ``PYTHONPATH`` for child processes (e.g. GGUF converter). + + Used by training, inference, and export workers. + """ + resolved = _resolve_base_model(model_name) + tier = get_transformers_tier(resolved) + + if tier == "550": + if not _ensure_venv_t5_550_exists(): + raise RuntimeError( + f"Cannot activate transformers 5.5.0: " + f".venv_t5_550 missing at {_VENV_T5_550_DIR}" + ) + if _VENV_T5_550_DIR not in sys.path: + sys.path.insert(0, _VENV_T5_550_DIR) + logger.info("Activated transformers 5.5.0 from %s", _VENV_T5_550_DIR) + _pp = os.environ.get("PYTHONPATH", "") + os.environ["PYTHONPATH"] = ( + _VENV_T5_550_DIR + (os.pathsep + _pp if _pp else "") + ) + elif tier == "530": + if not _ensure_venv_t5_530_exists(): + raise RuntimeError( + f"Cannot activate transformers 5.3.0: " + f".venv_t5_530 missing at {_VENV_T5_530_DIR}" + ) + if _VENV_T5_530_DIR not in sys.path: + sys.path.insert(0, _VENV_T5_530_DIR) + logger.info("Activated transformers 5.3.0 from %s", _VENV_T5_530_DIR) + _pp = os.environ.get("PYTHONPATH", "") + os.environ["PYTHONPATH"] = ( + _VENV_T5_530_DIR + (os.pathsep + _pp if _pp else "") + ) + else: + logger.info("Using default transformers (4.57.x) for %s", model_name) + + def _resolve_base_model(model_name: str) -> str: """If *model_name* points to a LoRA adapter, return its base model. From 45e994c8e83b413bcd04323b35cbc0c5e74edd40 Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Mon, 6 Apr 2026 22:45:22 +0000 Subject: [PATCH 18/26] narrow Nemotron trust_remote_code to nemotron_h/nemotron-3-nano, add to export worker --- studio/backend/core/export/worker.py | 13 +++++++++++++ studio/backend/core/inference/worker.py | 6 ++++-- studio/backend/core/training/worker.py | 6 ++++-- 3 files changed, 21 insertions(+), 4 deletions(-) diff --git a/studio/backend/core/export/worker.py b/studio/backend/core/export/worker.py index 80b05aca79..f77b1966c4 100644 --- a/studio/backend/core/export/worker.py +++ b/studio/backend/core/export/worker.py @@ -206,6 +206,19 @@ def _handle_load(backend, cmd: dict, resp_queue: Any) -> None: load_in_4bit = cmd.get("load_in_4bit", True) trust_remote_code = cmd.get("trust_remote_code", False) + # Auto-enable trust_remote_code for NemotronH/Nano models. + if not trust_remote_code: + _NEMOTRON_TRUST_SUBSTRINGS = ("nemotron_h", "nemotron-h", "nemotron-3-nano") + _cp_lower = checkpoint_path.lower() + if any(sub in _cp_lower for sub in _NEMOTRON_TRUST_SUBSTRINGS) and ( + _cp_lower.startswith("unsloth/") or _cp_lower.startswith("nvidia/") + ): + trust_remote_code = True + logger.info( + "Auto-enabled trust_remote_code for Nemotron model: %s", + checkpoint_path, + ) + try: _send_response( resp_queue, diff --git a/studio/backend/core/inference/worker.py b/studio/backend/core/inference/worker.py index 65cdb9b939..085a1ab899 100644 --- a/studio/backend/core/inference/worker.py +++ b/studio/backend/core/inference/worker.py @@ -287,14 +287,16 @@ def _handle_load(backend, config: dict, resp_queue: Any) -> None: except Exception as e: logger.warning("Could not read adapter_config.json: %s", e) - # Auto-enable trust_remote_code for Nemotron models only. + # Auto-enable trust_remote_code for NemotronH/Nano models only. # NemotronH has config parsing bugs requiring trust_remote_code=True. # Other transformers 5.x models are native and do NOT need it. + # NOTE: Must NOT match Llama-Nemotron (standard Llama architecture). + _NEMOTRON_TRUST_SUBSTRINGS = ("nemotron_h", "nemotron-h", "nemotron-3-nano") trust_remote_code = config.get("trust_remote_code", False) if not trust_remote_code: model_name = config["model_name"] _mn_lower = model_name.lower() - if "nemotron" in _mn_lower and ( + if any(sub in _mn_lower for sub in _NEMOTRON_TRUST_SUBSTRINGS) and ( _mn_lower.startswith("unsloth/") or _mn_lower.startswith("nvidia/") ): trust_remote_code = True diff --git a/studio/backend/core/training/worker.py b/studio/backend/core/training/worker.py index fbfc3f99be..9ce6cc2187 100644 --- a/studio/backend/core/training/worker.py +++ b/studio/backend/core/training/worker.py @@ -1105,14 +1105,16 @@ def run_training_process( ) return - # ── 1a. Auto-enable trust_remote_code for Nemotron models ── + # ── 1a. Auto-enable trust_remote_code for NemotronH/Nano models ── # NemotronH has config parsing bugs in transformers that require # trust_remote_code=True as a workaround. Other transformers 5.x models # (Qwen3.5, Gemma 4, etc.) are native and do NOT need it — enabling it # bypasses the compiler (disabling fused CE). + # NOTE: Must NOT match Llama-Nemotron (standard Llama architecture). + _NEMOTRON_TRUST_SUBSTRINGS = ("nemotron_h", "nemotron-h", "nemotron-3-nano") _lowered = model_name.lower() if ( - "nemotron" in _lowered + any(sub in _lowered for sub in _NEMOTRON_TRUST_SUBSTRINGS) and (_lowered.startswith("unsloth/") or _lowered.startswith("nvidia/")) and not config.get("trust_remote_code", False) ): From 3f8b4be6ec447397039c5d184fcccd4c99d3cfd4 Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Mon, 6 Apr 2026 22:46:14 +0000 Subject: [PATCH 19/26] clean venv_t5 dirs before re-install in setup.sh, clarify version alias comment --- studio/backend/utils/transformers_version.py | 3 ++- studio/setup.sh | 2 ++ 2 files changed, 4 insertions(+), 1 deletion(-) diff --git a/studio/backend/utils/transformers_version.py b/studio/backend/utils/transformers_version.py index d38b4ada28..a60b017a3a 100644 --- a/studio/backend/utils/transformers_version.py +++ b/studio/backend/utils/transformers_version.py @@ -91,7 +91,8 @@ _config_needs_550_cache: dict[str, bool] = {} TRANSFORMERS_550_VERSION = "5.5.0" TRANSFORMERS_530_VERSION = "5.3.0" TRANSFORMERS_DEFAULT_VERSION = "4.57.6" -# Backwards-compat alias used by other modules +# Backwards-compat alias — points to 5.5.0 (the highest 5.x tier). +# Consumers should prefer TRANSFORMERS_530_VERSION / TRANSFORMERS_550_VERSION. TRANSFORMERS_5_VERSION = TRANSFORMERS_550_VERSION # Pre-installed directories — created by setup.sh / setup.ps1. diff --git a/studio/setup.sh b/studio/setup.sh index 9a5ec2be14..df7c0cbf87 100755 --- a/studio/setup.sh +++ b/studio/setup.sh @@ -576,6 +576,7 @@ if [ "$_SKIP_PYTHON_DEPS" = false ]; then # Clean up legacy single .venv_t5 directory [ -d "$STUDIO_HOME/.venv_t5" ] && rm -rf "$STUDIO_HOME/.venv_t5" + [ -d "$VENV_T5_530_DIR" ] && rm -rf "$VENV_T5_530_DIR" mkdir -p "$VENV_T5_530_DIR" run_quiet "install transformers 5.3.0" fast_install --target "$VENV_T5_530_DIR" --no-deps "transformers==5.3.0" run_quiet "install huggingface_hub for t5_530" fast_install --target "$VENV_T5_530_DIR" --no-deps "huggingface_hub==1.8.0" @@ -583,6 +584,7 @@ if [ "$_SKIP_PYTHON_DEPS" = false ]; then run_quiet "install tiktoken for t5_530" fast_install --target "$VENV_T5_530_DIR" "tiktoken" step "transformers" "5.3.0 pre-installed" + [ -d "$VENV_T5_550_DIR" ] && rm -rf "$VENV_T5_550_DIR" mkdir -p "$VENV_T5_550_DIR" run_quiet "install transformers 5.5.0" fast_install --target "$VENV_T5_550_DIR" --no-deps "transformers==5.5.0" run_quiet "install huggingface_hub for t5_550" fast_install --target "$VENV_T5_550_DIR" --no-deps "huggingface_hub==1.8.0" From d7550cc5e1445ec4eb422330804bd5d76377c946 Mon Sep 17 00:00:00 2001 From: "pre-commit-ci[bot]" <66853113+pre-commit-ci[bot]@users.noreply.github.com> Date: Mon, 6 Apr 2026 22:46:58 +0000 Subject: [PATCH 20/26] [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci --- studio/backend/utils/transformers_version.py | 8 ++------ 1 file changed, 2 insertions(+), 6 deletions(-) diff --git a/studio/backend/utils/transformers_version.py b/studio/backend/utils/transformers_version.py index a60b017a3a..9075c590ca 100644 --- a/studio/backend/utils/transformers_version.py +++ b/studio/backend/utils/transformers_version.py @@ -127,9 +127,7 @@ def activate_transformers_for_subprocess(model_name: str) -> None: sys.path.insert(0, _VENV_T5_550_DIR) logger.info("Activated transformers 5.5.0 from %s", _VENV_T5_550_DIR) _pp = os.environ.get("PYTHONPATH", "") - os.environ["PYTHONPATH"] = ( - _VENV_T5_550_DIR + (os.pathsep + _pp if _pp else "") - ) + os.environ["PYTHONPATH"] = _VENV_T5_550_DIR + (os.pathsep + _pp if _pp else "") elif tier == "530": if not _ensure_venv_t5_530_exists(): raise RuntimeError( @@ -140,9 +138,7 @@ def activate_transformers_for_subprocess(model_name: str) -> None: sys.path.insert(0, _VENV_T5_530_DIR) logger.info("Activated transformers 5.3.0 from %s", _VENV_T5_530_DIR) _pp = os.environ.get("PYTHONPATH", "") - os.environ["PYTHONPATH"] = ( - _VENV_T5_530_DIR + (os.pathsep + _pp if _pp else "") - ) + os.environ["PYTHONPATH"] = _VENV_T5_530_DIR + (os.pathsep + _pp if _pp else "") else: logger.info("Using default transformers (4.57.x) for %s", model_name) From 0c799210b601a155d5d600eaad55744b4dd24f7c Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Tue, 7 Apr 2026 07:24:47 +0000 Subject: [PATCH 21/26] update --- studio/setup.sh | 28 +++++++++++++++++++++------- 1 file changed, 21 insertions(+), 7 deletions(-) diff --git a/studio/setup.sh b/studio/setup.sh index df7c0cbf87..25909b1f2f 100755 --- a/studio/setup.sh +++ b/studio/setup.sh @@ -515,15 +515,29 @@ if [ "$_DOCKER_NO_VENV" = true ]; then # install_python_stack.py has steps 1-5 commented out for this branch. python "$SCRIPT_DIR/install_python_stack.py" - # Pre-install transformers 5.x into .venv_t5 + # Pre-install transformers 5.x into .venv_t5_530/ and .venv_t5_550/ echo "" echo " Pre-installing transformers 5.x for newer model support..." - mkdir -p "$VENV_T5_DIR" - pip install --target "$VENV_T5_DIR" --no-deps "transformers==5.3.0" 2>/dev/null - pip install --target "$VENV_T5_DIR" --no-deps "huggingface_hub==1.7.1" 2>/dev/null - pip install --target "$VENV_T5_DIR" --no-deps "hf_xet==1.4.2" 2>/dev/null - pip install --target "$VENV_T5_DIR" "tiktoken" 2>/dev/null - echo "✅ Transformers 5.x pre-installed to $VENV_T5_DIR/" + + # Clean up legacy single .venv_t5 directory + [ -d "$STUDIO_HOME/.venv_t5" ] && rm -rf "$STUDIO_HOME/.venv_t5" + + [ -d "$VENV_T5_530_DIR" ] && rm -rf "$VENV_T5_530_DIR" + mkdir -p "$VENV_T5_530_DIR" + pip install --target "$VENV_T5_530_DIR" --no-deps "transformers==5.3.0" 2>/dev/null + pip install --target "$VENV_T5_530_DIR" --no-deps "huggingface_hub==1.8.0" 2>/dev/null + pip install --target "$VENV_T5_530_DIR" --no-deps "hf_xet==1.4.2" 2>/dev/null + pip install --target "$VENV_T5_530_DIR" "tiktoken" 2>/dev/null + + [ -d "$VENV_T5_550_DIR" ] && rm -rf "$VENV_T5_550_DIR" + mkdir -p "$VENV_T5_550_DIR" + pip install --target "$VENV_T5_550_DIR" --no-deps "transformers==5.5.0" 2>/dev/null + pip install --target "$VENV_T5_550_DIR" --no-deps "huggingface_hub==1.8.0" 2>/dev/null + pip install --target "$VENV_T5_550_DIR" --no-deps "hf_xet==1.4.2" 2>/dev/null + pip install --target "$VENV_T5_550_DIR" "tiktoken" 2>/dev/null + + echo "✅ Transformers 5.3.0 pre-installed to $VENV_T5_530_DIR/" + echo "✅ Transformers 5.5.0 pre-installed to $VENV_T5_550_DIR/" echo "" echo "╔══════════════════════════════════════╗" From bdf4d9abb54def91375a4bf8134fdb9fe5cf5646 Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Sun, 5 Apr 2026 05:22:43 +0000 Subject: [PATCH 22/26] fix: derive HF_HUB_CACHE from HF_HOME when set Previously HF_HUB_CACHE always defaulted to ~/.cache/huggingface/hub even when HF_HOME was explicitly set (e.g. in Docker). This caused models to download to the wrong location instead of the configured HF_HOME path. --- studio/backend/utils/paths/storage_roots.py | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/studio/backend/utils/paths/storage_roots.py b/studio/backend/utils/paths/storage_roots.py index 58a4d7967c..3e79f0de74 100644 --- a/studio/backend/utils/paths/storage_roots.py +++ b/studio/backend/utils/paths/storage_roots.py @@ -235,10 +235,11 @@ def _setup_cache_env() -> None: os.environ.get("XDG_CACHE_HOME", Path.home() / ".cache") ).expanduser() hf_default = xdg_cache / "huggingface" + hf_home = Path(os.environ.get("HF_HOME", str(hf_default))) defaults: dict[str, str] = { "HF_HOME": str(hf_default), - "HF_HUB_CACHE": str(hf_default / "hub"), - "HF_XET_CACHE": str(hf_default / "xet"), + "HF_HUB_CACHE": str(hf_home / "hub"), + "HF_XET_CACHE": str(hf_home / "xet"), "UV_CACHE_DIR": str(root / "uv"), "VLLM_CACHE_ROOT": str(root / "vllm"), } From 03633a10df026a80371373d26b786355e181e955 Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Thu, 16 Apr 2026 00:48:14 +0400 Subject: [PATCH 23/26] spark studio v0.13.6 2026.4.5 --- .worktreeinclude | 1 + 1 file changed, 1 insertion(+) create mode 100644 .worktreeinclude diff --git a/.worktreeinclude b/.worktreeinclude new file mode 100644 index 0000000000..ceb2b988dc --- /dev/null +++ b/.worktreeinclude @@ -0,0 +1 @@ +CLAUDE.md From c495b98337a3b10fc36052fb5822f148abc1ff14 Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Wed, 15 Apr 2026 21:03:05 +0000 Subject: [PATCH 24/26] docker container v0.13.6 beta 2026.4.6 --- studio/install_python_stack.py | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/studio/install_python_stack.py b/studio/install_python_stack.py index 9273c70272..fd3115e0ad 100644 --- a/studio/install_python_stack.py +++ b/studio/install_python_stack.py @@ -1180,9 +1180,9 @@ def install_python_stack() -> int: # constrain = False, # ) - if not IS_WINDOWS and not IS_MACOS and not NO_TORCH: - _progress("flash-attn") - _ensure_flash_attn() + #if not IS_WINDOWS and not IS_MACOS and not NO_TORCH: + # _progress("flash-attn") + # _ensure_flash_attn() # # 6. Patch: override llama_cpp.py with fix from unsloth-zoo feature/llama-cpp-windows-support branch # patch_package_file( From b979c1c35229ec0c2303cec2947da9340ee42af7 Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Thu, 23 Apr 2026 20:48:58 +0000 Subject: [PATCH 25/26] fix failed to start on docker --- unsloth_cli/commands/studio.py | 346 ++++++--------------------------- 1 file changed, 61 insertions(+), 285 deletions(-) diff --git a/unsloth_cli/commands/studio.py b/unsloth_cli/commands/studio.py index c804ae8e54..f18d3784df 100644 --- a/unsloth_cli/commands/studio.py +++ b/unsloth_cli/commands/studio.py @@ -1,19 +1,11 @@ # SPDX-License-Identifier: AGPL-3.0-only # Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. See /studio/LICENSE.AGPL-3.0 -import importlib.util -import hashlib -import json import os import platform -import secrets -import sqlite3 import subprocess import sys -import tempfile import time -import types -from datetime import datetime, timezone from pathlib import Path from typing import List, Optional import typer @@ -95,6 +87,7 @@ API_KEY_PBKDF2_SALT_KEY = "api_key_pbkdf2_salt" DESKTOP_SECRET_HASH_KEY = "desktop_secret_hash" DESKTOP_SECRET_CREATED_AT_KEY = "desktop_secret_created_at" PBKDF2_ITERATIONS = 100_000 +STUDIO_HOME = Path.home() / ".unsloth" / "studio" # __file__ is unsloth_cli/commands/studio.py -- two parents up is the package root # (either site-packages or the repo root for editable installs). @@ -231,228 +224,12 @@ def _create_api_key_inprocess(name: str) -> str: ``POST /api/auth/api-keys`` on fresh installs. Safe because the CLI already has filesystem access to ``~/.unsloth/studio``. """ - storage = _load_backend_auth_storage() + from auth.storage import create_api_key, DEFAULT_ADMIN_USERNAME - raw_key, _row = storage.create_api_key( - username = storage.DEFAULT_ADMIN_USERNAME, - name = name, - ) + raw_key, _row = create_api_key(username = DEFAULT_ADMIN_USERNAME, name = name) return raw_key -def _load_backend_auth_storage(): - run_py = _find_run_py() - backend_dir = ( - run_py.parent if run_py is not None else _PACKAGE_ROOT / "studio" / "backend" - ) - if backend_dir.is_dir() and str(backend_dir) not in sys.path: - sys.path.insert(0, str(backend_dir)) - - auth_dir = backend_dir / "auth" - storage_py = auth_dir / "storage.py" - loaded = sys.modules.get("auth.storage") - loaded_path = Path(getattr(loaded, "__file__", "")).resolve() - if loaded is not None and loaded_path == storage_py: - return loaded - - package = sys.modules.get("auth") - package_paths = [Path(path).resolve() for path in getattr(package, "__path__", [])] - if package is None or auth_dir.resolve() not in package_paths: - package = types.ModuleType("auth") - package.__path__ = [str(auth_dir)] - package.__package__ = "auth" - package.__file__ = str(auth_dir / "__init__.py") - sys.modules["auth"] = package - - spec = importlib.util.spec_from_file_location("auth.storage", storage_py) - if spec is None or spec.loader is None: - raise ImportError(f"Could not load backend auth storage from {storage_py}") - storage = importlib.util.module_from_spec(spec) - sys.modules["auth.storage"] = storage - spec.loader.exec_module(storage) - - return storage - - -def _write_auth_secret(path: Path, secret: str) -> None: - path.parent.mkdir(parents = True, exist_ok = True) - fd, tmp_name = tempfile.mkstemp(prefix = f".{path.name}.", dir = path.parent) - tmp_path = Path(tmp_name) - try: - try: - os.chmod(tmp_path, 0o600) - except OSError: - pass - with os.fdopen(fd, "w") as f: - fd = -1 - f.write(secret) - os.replace(tmp_path, path) - except Exception: - if fd >= 0: - os.close(fd) - tmp_path.unlink(missing_ok = True) - raise - try: - os.chmod(path, 0o600) - except OSError: - pass - - -def _connect_auth_db() -> sqlite3.Connection: - auth_dir = STUDIO_HOME / "auth" - auth_dir.mkdir(parents = True, exist_ok = True) - conn = sqlite3.connect(auth_dir / "auth.db") - conn.execute( - """ - CREATE TABLE IF NOT EXISTS auth_user ( - id INTEGER PRIMARY KEY, - username TEXT UNIQUE NOT NULL, - password_salt TEXT NOT NULL, - password_hash TEXT NOT NULL, - jwt_secret TEXT NOT NULL, - must_change_password INTEGER NOT NULL DEFAULT 0 - ); - """ - ) - conn.execute( - """ - CREATE TABLE IF NOT EXISTS refresh_tokens ( - id INTEGER PRIMARY KEY, - token_hash TEXT NOT NULL, - username TEXT NOT NULL, - expires_at TEXT NOT NULL, - is_desktop INTEGER NOT NULL DEFAULT 0 - ); - """ - ) - conn.execute( - """ - CREATE TABLE IF NOT EXISTS api_keys ( - id INTEGER PRIMARY KEY AUTOINCREMENT, - username TEXT NOT NULL, - key_prefix TEXT NOT NULL, - key_hash TEXT NOT NULL UNIQUE, - name TEXT NOT NULL DEFAULT '', - created_at TEXT NOT NULL, - last_used_at TEXT, - expires_at TEXT, - is_active INTEGER NOT NULL DEFAULT 1 - ); - """ - ) - conn.execute( - """ - CREATE TABLE IF NOT EXISTS app_secrets ( - key TEXT PRIMARY KEY, - value TEXT NOT NULL - ); - """ - ) - auth_columns = {row[1] for row in conn.execute("PRAGMA table_info(auth_user)")} - if "must_change_password" not in auth_columns: - conn.execute( - "ALTER TABLE auth_user ADD COLUMN must_change_password INTEGER NOT NULL DEFAULT 0" - ) - refresh_columns = { - row[1] for row in conn.execute("PRAGMA table_info(refresh_tokens)") - } - if "is_desktop" not in refresh_columns: - conn.execute( - "ALTER TABLE refresh_tokens ADD COLUMN is_desktop INTEGER NOT NULL DEFAULT 0" - ) - conn.commit() - return conn - - -def _pbkdf2_hex(value: str, salt: bytes) -> str: - return hashlib.pbkdf2_hmac( - "sha256", - value.encode("utf-8"), - salt, - PBKDF2_ITERATIONS, - ).hex() - - -def _hash_password(password: str) -> tuple[str, str]: - salt = secrets.token_hex(16) - pwd_hash = _pbkdf2_hex(password, salt.encode("utf-8")) - return salt, pwd_hash - - -def _get_or_create_api_key_pbkdf2_salt(conn: sqlite3.Connection) -> bytes: - row = conn.execute( - "SELECT value FROM app_secrets WHERE key = ?", - (API_KEY_PBKDF2_SALT_KEY,), - ).fetchone() - if row is None: - salt_hex = secrets.token_hex(32) - conn.execute( - "INSERT OR IGNORE INTO app_secrets (key, value) VALUES (?, ?)", - (API_KEY_PBKDF2_SALT_KEY, salt_hex), - ) - row = conn.execute( - "SELECT value FROM app_secrets WHERE key = ?", - (API_KEY_PBKDF2_SALT_KEY,), - ).fetchone() - return bytes.fromhex(row[0]) - - -def _ensure_cli_default_admin(conn: sqlite3.Connection) -> None: - row = conn.execute( - "SELECT 1 FROM auth_user WHERE username = ?", - (DEFAULT_ADMIN_USERNAME,), - ).fetchone() - if row is not None: - return - - bootstrap_password = secrets.token_urlsafe(32) - password_salt, password_hash = _hash_password(bootstrap_password) - conn.execute( - """ - INSERT INTO auth_user ( - username, - password_salt, - password_hash, - jwt_secret, - must_change_password - ) - VALUES (?, ?, ?, ?, ?) - """, - ( - DEFAULT_ADMIN_USERNAME, - password_salt, - password_hash, - secrets.token_urlsafe(64), - 1, - ), - ) - _write_auth_secret( - STUDIO_HOME / "auth" / BOOTSTRAP_PASSWORD_FILE, - bootstrap_password, - ) - - -def _create_desktop_secret_in_cli() -> str: - raw_secret = DESKTOP_SECRET_PREFIX + secrets.token_urlsafe(48) - now = datetime.now(timezone.utc).isoformat() - conn = _connect_auth_db() - try: - _ensure_cli_default_admin(conn) - secret_hash = _pbkdf2_hex(raw_secret, _get_or_create_api_key_pbkdf2_salt(conn)) - conn.execute( - "INSERT OR REPLACE INTO app_secrets (key, value) VALUES (?, ?)", - (DESKTOP_SECRET_HASH_KEY, secret_hash), - ) - conn.execute( - "INSERT OR REPLACE INTO app_secrets (key, value) VALUES (?, ?)", - (DESKTOP_SECRET_CREATED_AT_KEY, now), - ) - conn.commit() - return raw_secret - finally: - conn.close() - - def _load_model_via_http( port: int, api_key: str, @@ -506,11 +283,6 @@ def studio_default( host: str = typer.Option("127.0.0.1", "--host", "-H"), frontend: Optional[Path] = typer.Option(None, "--frontend", "-f"), silent: bool = typer.Option(False, "--silent", "-q"), - api_only: bool = typer.Option( - False, - "--api-only", - help = "Run API server only, no frontend serving (for Tauri desktop app)", - ), ): """Launch the Unsloth Studio server.""" # Runs before any subcommand; covers run/setup/update/etc in one place. @@ -524,49 +296,49 @@ def studio_default( studio_venv_dir = STUDIO_HOME / "unsloth_studio" in_studio_venv = sys.prefix.startswith(str(studio_venv_dir)) - if not in_studio_venv: - studio_python = _studio_venv_python() - run_py = _find_run_py() - if studio_python and run_py: - if not silent: - typer.echo("Launching Unsloth Studio... Please wait...") - args = [ - str(studio_python), - str(run_py), - "--host", - host, - "--port", - str(port), - ] - if frontend: - args.extend(["--frontend", str(frontend)]) - if silent: - args.append("--silent") - if api_only: - args.append("--api-only") - # On Windows, os.execvp() spawns a child but the parent lingers, - # so Ctrl+C only kills the parent leaving the child orphaned. - # Use subprocess.run() on Windows so the parent waits for the child. - if sys.platform == "win32": - import subprocess as _sp + if not in_studio_venv: + studio_python = _studio_venv_python() + run_py = _find_run_py() + if studio_python and run_py: + if not silent: + typer.echo("Launching Unsloth Studio... Please wait...") + args = [ + str(studio_python), + str(run_py), + "--host", + host, + "--port", + str(port), + ] + if frontend: + args.extend(["--frontend", str(frontend)]) + if silent: + args.append("--silent") + # On Windows, os.execvp() spawns a child but the parent lingers, + # so Ctrl+C only kills the parent leaving the child orphaned. + # Use subprocess.run() on Windows so the parent waits for the child. + if sys.platform == "win32": + import subprocess as _sp - proc = _sp.Popen(args, **_windows_hidden_subprocess_kwargs()) - try: - rc = proc.wait() - except KeyboardInterrupt: - # Child has its own signal handler — let it finish - rc = proc.wait() - if rc != 0: - typer.echo( - f"\nError: Studio server exited unexpectedly (code {rc}).", - err = True, - ) - typer.echo( - "Check the error above. If a package is missing, " - "re-run: unsloth studio setup", - err = True, - ) - raise typer.Exit(rc) + proc = _sp.Popen(args) + try: + rc = proc.wait() + except KeyboardInterrupt: + # Child has its own signal handler — let it finish + rc = proc.wait() + if rc != 0: + typer.echo( + f"\nError: Studio server exited unexpectedly (code {rc}).", + err = True, + ) + typer.echo( + "Check the error above. If a package is missing, " + "re-run: unsloth studio setup", + err = True, + ) + raise typer.Exit(rc) + else: + os.execvp(str(studio_python), args) else: typer.echo("Studio not set up. Run install.sh first.") raise typer.Exit(1) @@ -579,7 +351,7 @@ def studio_default( display_host = _resolve_external_ip() if host == "0.0.0.0" else host typer.echo(f"Starting Unsloth Studio on http://{display_host}:{port}") - run_kwargs = dict(host = host, port = port, silent = silent, api_only = api_only) + run_kwargs = dict(host = host, port = port, silent = silent) if frontend is not None: run_kwargs["frontend_path"] = frontend run_server(**run_kwargs) @@ -1013,6 +785,7 @@ def _run_setup_script(*, verbose: bool = False) -> None: env = {**os.environ, "UNSLOTH_VERBOSE": "1"} if verbose else None if platform.system() == "Windows": +<<<<<<< HEAD powershell_args = ["powershell.exe"] if _should_hide_windows_subprocesses(): powershell_args.extend( @@ -1049,13 +822,18 @@ def _run_setup_script(*, verbose: bool = False) -> None: # explicitly inheritable, which works alongside # CREATE_NO_WINDOW. Empty update.log on the windows-latest # CI was the smoking gun (run 25533694490 and 25534292239). +======= +>>>>>>> e2252b36 (fix failed to start on docker) result = subprocess.run( - powershell_args, + ["powershell", "-ExecutionPolicy", "Bypass", "-File", str(script)], env = env, +<<<<<<< HEAD stdin = _stream_for_subprocess(sys.stdin), stdout = _stream_for_subprocess(sys.stdout), stderr = _stream_for_subprocess(sys.stderr), **_windows_hidden_subprocess_kwargs(), +======= +>>>>>>> e2252b36 (fix failed to start on docker) ) else: result = subprocess.run(["bash", str(script)], env = env) @@ -1111,6 +889,7 @@ def update( # ── unsloth studio reset-password ──────────────────────────────────── +<<<<<<< HEAD @studio_app.command("desktop-capabilities", hidden = True) def desktop_capabilities( json_output: bool = typer.Option( @@ -1151,6 +930,8 @@ def provision_desktop_auth(): typer.echo("Desktop auth ready.") +======= +>>>>>>> e2252b36 (fix failed to start on docker) @studio_app.command("reset-password") def reset_password(): """Reset the Studio admin password. @@ -1161,18 +942,13 @@ def reset_password(): """ auth_dir = STUDIO_HOME / "auth" db_file = auth_dir / "auth.db" - stale_files = [ - auth_dir / BOOTSTRAP_PASSWORD_FILE, - auth_dir / DESKTOP_SECRET_FILE, - ] - had_db = db_file.exists() + pw_file = auth_dir / ".bootstrap_password" - db_file.unlink(missing_ok = True) - for path in stale_files: - path.unlink(missing_ok = True) - - if not had_db: + if not db_file.exists(): typer.echo("No auth database found -- nothing to reset.") raise typer.Exit(0) + db_file.unlink(missing_ok = True) + pw_file.unlink(missing_ok = True) + typer.echo("Auth database deleted. Restart Unsloth Studio to get a new password.") From e637a14fb4dc6bdd36b9fde39899a121cb0b16ad Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Fri, 8 May 2026 00:40:06 +0400 Subject: [PATCH 26/26] studio: add --no-deps to base.txt install in Docker paths MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The 'Updating base packages' pip_install calls in the elif local_repo and final else branches resolved base.txt's transitive deps without --no-deps. On Docker builds with local-version torch (e.g. torch==2.11.0+cu130), uv treats the +cu130 segment as non-canonical PEP 440 and decides the installed torch doesn't satisfy a bare 'torch' constraint, then pulls PyPI's torch==2.10.0 (CPU-only) and uninstalls the cu130 build. This breaks every CUDA-dependent package downstream. This branch is named feature/docker-studio-* — torch and every other CUDA-adjacent package is pre-installed and pinned by the Dockerfile. There's no need to resolve transitive deps; --upgrade-package already targets only unsloth + unsloth-zoo as intended. --- studio/install_python_stack.py | 14 ++++++++++++++ 1 file changed, 14 insertions(+) diff --git a/studio/install_python_stack.py b/studio/install_python_stack.py index fd3115e0ad..20c249234a 100644 --- a/studio/install_python_stack.py +++ b/studio/install_python_stack.py @@ -1096,10 +1096,17 @@ def install_python_stack() -> int: # Local dev install: update deps from base.txt, then overlay the # local checkout as an editable install (--no-deps so torch is # never re-resolved). + # --no-deps on the base.txt install too: this branch is for Docker + # builds where torch + every CUDA-adjacent package is already pinned + # and installed; resolving base.txt's deps risks pulling PyPI's + # CPU-only torch over our local-version cu* build (the +cu130 etc. + # local segment is non-canonical PEP 440 and uv may treat it as + # not-satisfying an unconstrained `torch` requirement). _progress("base packages") pip_install( "Updating base packages", "--no-cache-dir", + "--no-deps", "--upgrade-package", "unsloth", "--upgrade-package", @@ -1136,10 +1143,17 @@ def install_python_stack() -> int: # Update path: upgrade only unsloth + unsloth-zoo while preserving # existing torch/CUDA installations. Torch is pre-installed by # install.sh / setup.ps1; --upgrade-package targets only base pkgs. + # --no-deps: this branch is for Docker builds where torch + every + # CUDA-adjacent package is already pinned and installed by the + # Dockerfile; resolving base.txt's transitive deps risks pulling + # PyPI's CPU-only torch over our local-version cu* build (the + # +cu130 etc. local segment is non-canonical PEP 440 and uv may + # treat it as not-satisfying an unconstrained `torch` requirement). _progress("base packages") pip_install( "Updating base packages", "--no-cache-dir", + "--no-deps", "--upgrade-package", "unsloth", "--upgrade-package",