From bf0007c502b87836b9e58db5b4aca9f7b277094f Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Mon, 16 Mar 2026 13:29:51 +0000 Subject: [PATCH 1/6] feat: add UNSLOTH_STUDIO_HOME env var to override studio root Allow users to set UNSLOTH_STUDIO_HOME to relocate all studio data (venvs, assets, outputs, exports, auth, cache, tensorboard runs). Defaults to ~/.unsloth/studio when unset (no behaviour change). --- studio/backend/utils/paths/storage_roots.py | 5 ++++- studio/setup.ps1 | 9 +++++---- studio/setup.sh | 1 + unsloth_cli/commands/studio.py | 18 ++++++++++++------ 4 files changed, 22 insertions(+), 11 deletions(-) diff --git a/studio/backend/utils/paths/storage_roots.py b/studio/backend/utils/paths/storage_roots.py index 81763c3095..288d9c04a9 100644 --- a/studio/backend/utils/paths/storage_roots.py +++ b/studio/backend/utils/paths/storage_roots.py @@ -9,12 +9,15 @@ import tempfile def studio_root() -> Path: + custom = os.environ.get("UNSLOTH_STUDIO_HOME") + if custom: + return Path(custom).expanduser().resolve() return Path.home() / ".unsloth" / "studio" def cache_root() -> Path: """Central cache directory for all studio downloads (models, datasets, etc.).""" - return Path.home() / ".unsloth" / "studio" / "cache" + return studio_root() / "cache" def assets_root() -> Path: diff --git a/studio/setup.ps1 b/studio/setup.ps1 index 7091920ab9..4a791ef2b1 100644 --- a/studio/setup.ps1 +++ b/studio/setup.ps1 @@ -966,9 +966,10 @@ if (-not $PythonCmd) { Write-Host "[OK] Using $PythonCmd ($(& $PythonCmd --version 2>&1))" -ForegroundColor Green -# Always create a .venv for isolation -- even for pip installs. -# Created in the repo root (parent of studio/). -$VenvDir = Join-Path $env:USERPROFILE ".unsloth\studio\.venv" +# ── Studio home (configurable via UNSLOTH_STUDIO_HOME) ── +$StudioHome = if ($env:UNSLOTH_STUDIO_HOME) { $env:UNSLOTH_STUDIO_HOME } else { Join-Path $env:USERPROFILE ".unsloth\studio" } + +$VenvDir = Join-Path $StudioHome ".venv" if (-not (Test-Path $VenvDir)) { Write-Host " Creating virtual environment at $VenvDir..." -ForegroundColor Cyan & $PythonCmd -m venv $VenvDir @@ -1090,7 +1091,7 @@ $ErrorActionPreference = $prevEAP # The training subprocess just prepends .venv_t5/ to sys.path -- instant switch. Write-Host "" Write-Host " Pre-installing transformers 5.x for newer model support..." -ForegroundColor Cyan -$VenvT5Dir = Join-Path $env:USERPROFILE ".unsloth\studio\.venv_t5" +$VenvT5Dir = Join-Path $StudioHome ".venv_t5" if (Test-Path $VenvT5Dir) { Remove-Item -Recurse -Force $VenvT5Dir } New-Item -ItemType Directory -Path $VenvT5Dir -Force | Out-Null $prevEAP_t5 = $ErrorActionPreference diff --git a/studio/setup.sh b/studio/setup.sh index 3e56aebf38..a039fc0bfc 100755 --- a/studio/setup.sh +++ b/studio/setup.sh @@ -251,6 +251,7 @@ install_python_stack() { python "$SCRIPT_DIR/install_python_stack.py" } +<<<<<<< HEAD # Create venv under ~/.unsloth/studio/ (shared location, not in repo). # All platforms (including Colab) use the same isolated venv so that # studio dependencies are never installed into the system Python. diff --git a/unsloth_cli/commands/studio.py b/unsloth_cli/commands/studio.py index 8a2267a79e..a39f050d5f 100644 --- a/unsloth_cli/commands/studio.py +++ b/unsloth_cli/commands/studio.py @@ -12,7 +12,13 @@ import typer studio_app = typer.Typer(help = "Unsloth Studio commands.") -STUDIO_HOME = Path.home() / ".unsloth" / "studio" + +def _studio_home() -> Path: + """Studio root, overridable via UNSLOTH_STUDIO_HOME.""" + custom = os.environ.get("UNSLOTH_STUDIO_HOME") + if custom: + return Path(custom).expanduser().resolve() + return Path.home() / ".unsloth" / "studio" # __file__ is unsloth_cli/commands/studio.py -- two parents up is the package root # (either site-packages or the repo root for editable installs). @@ -22,9 +28,9 @@ _PACKAGE_ROOT = Path(__file__).resolve().parent.parent.parent def _studio_venv_python() -> Optional[Path]: """Return the studio venv Python binary, or None if not set up.""" if platform.system() == "Windows": - p = STUDIO_HOME / ".venv" / "Scripts" / "python.exe" + p = _studio_home() / ".venv" / "Scripts" / "python.exe" else: - p = STUDIO_HOME / ".venv" / "bin" / "python" + p = _studio_home() / ".venv" / "bin" / "python" return p if p.is_file() else None @@ -44,7 +50,7 @@ def _find_run_py() -> Optional[Path]: "lib/python*/site-packages/studio/backend/run.py", "Lib/site-packages/studio/backend/run.py", ): - for match in (STUDIO_HOME / ".venv").glob(pattern): + for match in (_studio_home() / ".venv").glob(pattern): return match return None @@ -64,7 +70,7 @@ def _find_setup_script() -> Optional[Path]: f"lib/python*/site-packages/studio/{name}", f"Lib/site-packages/studio/{name}", ): - for match in (STUDIO_HOME / ".venv").glob(pattern): + for match in (_studio_home() / ".venv").glob(pattern): return match return None @@ -85,7 +91,7 @@ def studio_default( return # Always use the studio venv if it exists and we're not already in it - studio_venv_dir = STUDIO_HOME / ".venv" + studio_venv_dir = _studio_home() / ".venv" in_studio_venv = sys.prefix.startswith(str(studio_venv_dir)) if not in_studio_venv: From 4e69c0e415ba03a043db1fee5be8cc982239a77a Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Tue, 17 Mar 2026 18:38:55 +0000 Subject: [PATCH 2/6] fix: use venv_t5_root() so .venv_t5 respects UNSLOTH_STUDIO_HOME --- studio/backend/core/export/worker.py | 27 +++++++++++++++++--- studio/backend/core/inference/worker.py | 27 +++++++++++++++++--- studio/backend/core/training/worker.py | 27 +++++++++++++++++--- studio/backend/utils/models/model_config.py | 3 ++- studio/backend/utils/paths/__init__.py | 2 ++ studio/backend/utils/paths/storage_roots.py | 5 ++++ studio/backend/utils/transformers_version.py | 3 ++- studio/setup.sh | 1 - 8 files changed, 83 insertions(+), 12 deletions(-) diff --git a/studio/backend/core/export/worker.py b/studio/backend/core/export/worker.py index 6af6ff1193..ff0e944e17 100644 --- a/studio/backend/core/export/worker.py +++ b/studio/backend/core/export/worker.py @@ -49,9 +49,30 @@ def _activate_transformers_version(model_name: str) -> None: resolved = _resolve_base_model(model_name) if needs_transformers_5(resolved): - if not _ensure_venv_t5_exists(): - raise RuntimeError( - f"Cannot activate transformers 5.x: .venv_t5 missing at {_VENV_T5_DIR}" + from utils.paths.storage_roots import venv_t5_root + venv_t5 = str(venv_t5_root()) + if os.path.isdir(venv_t5): + sys.path.insert(0, venv_t5) + logger.info("Activated transformers 5.x from %s", venv_t5) + else: + # Fallback: pip install at runtime (slower, ~10-15s) + logger.warning(".venv_t5 not found at %s — installing at runtime", venv_t5) + import subprocess as sp + + os.makedirs(venv_t5, exist_ok = True) + r1 = sp.run( + [ + sys.executable, + "-m", + "pip", + "install", + "--target", + venv_t5, + "--no-deps", + "transformers==5.3.0", + ], + stdout = sp.PIPE, + stderr = sp.STDOUT, ) if _VENV_T5_DIR not in sys.path: sys.path.insert(0, _VENV_T5_DIR) diff --git a/studio/backend/core/inference/worker.py b/studio/backend/core/inference/worker.py index 2eb46f3217..91e197c896 100644 --- a/studio/backend/core/inference/worker.py +++ b/studio/backend/core/inference/worker.py @@ -51,9 +51,30 @@ def _activate_transformers_version(model_name: str) -> None: resolved = _resolve_base_model(model_name) if needs_transformers_5(resolved): - if not _ensure_venv_t5_exists(): - raise RuntimeError( - f"Cannot activate transformers 5.x: .venv_t5 missing at {_VENV_T5_DIR}" + from utils.paths.storage_roots import venv_t5_root + venv_t5 = str(venv_t5_root()) + if os.path.isdir(venv_t5): + sys.path.insert(0, venv_t5) + logger.info("Activated transformers 5.x from %s", venv_t5) + else: + # Fallback: pip install at runtime (slower, ~10-15s) + logger.warning(".venv_t5 not found at %s — installing at runtime", venv_t5) + import subprocess as sp + + os.makedirs(venv_t5, exist_ok = True) + r1 = sp.run( + [ + sys.executable, + "-m", + "pip", + "install", + "--target", + venv_t5, + "--no-deps", + "transformers==5.3.0", + ], + stdout = sp.PIPE, + stderr = sp.STDOUT, ) if _VENV_T5_DIR not in sys.path: sys.path.insert(0, _VENV_T5_DIR) diff --git a/studio/backend/core/training/worker.py b/studio/backend/core/training/worker.py index ccd805b7ac..6a0cda830a 100644 --- a/studio/backend/core/training/worker.py +++ b/studio/backend/core/training/worker.py @@ -45,9 +45,30 @@ def _activate_transformers_version(model_name: str) -> None: resolved = _resolve_base_model(model_name) if needs_transformers_5(resolved): - if not _ensure_venv_t5_exists(): - raise RuntimeError( - f"Cannot activate transformers 5.x: .venv_t5 missing at {_VENV_T5_DIR}" + from utils.paths.storage_roots import venv_t5_root + venv_t5 = str(venv_t5_root()) + if os.path.isdir(venv_t5): + sys.path.insert(0, venv_t5) + logger.info("Activated transformers 5.x from %s", venv_t5) + else: + # Fallback: pip install at runtime (slower, ~10-15s) + logger.warning(".venv_t5 not found at %s — installing at runtime", venv_t5) + import subprocess as sp + + os.makedirs(venv_t5, exist_ok = True) + r1 = sp.run( + [ + sys.executable, + "-m", + "pip", + "install", + "--target", + venv_t5, + "--no-deps", + "transformers==5.3.0", + ], + stdout = sp.PIPE, + stderr = sp.STDOUT, ) if _VENV_T5_DIR not in sys.path: sys.path.insert(0, _VENV_T5_DIR) diff --git a/studio/backend/utils/models/model_config.py b/studio/backend/utils/models/model_config.py index 13f1b5febf..2356600b3b 100644 --- a/studio/backend/utils/models/model_config.py +++ b/studio/backend/utils/models/model_config.py @@ -427,7 +427,8 @@ _VLM_MODEL_TYPES = { } # Pre-computed .venv_t5 path and backend dir for subprocess version switching. -_VENV_T5_DIR = str(Path.home() / ".unsloth" / "studio" / ".venv_t5") +from utils.paths.storage_roots import venv_t5_root +_VENV_T5_DIR = str(venv_t5_root()) _BACKEND_DIR = str(Path(__file__).resolve().parent.parent.parent) # Inline script executed in a subprocess with transformers 5.x activated. diff --git a/studio/backend/utils/paths/__init__.py b/studio/backend/utils/paths/__init__.py index 507fb1106b..e9cc9da562 100644 --- a/studio/backend/utils/paths/__init__.py +++ b/studio/backend/utils/paths/__init__.py @@ -8,6 +8,7 @@ Path utilities for model and dataset handling from .path_utils import normalize_path, is_local_path, is_model_cached, get_cache_path from .storage_roots import ( studio_root, + venv_t5_root, assets_root, datasets_root, dataset_uploads_root, @@ -36,6 +37,7 @@ __all__ = [ "is_model_cached", "get_cache_path", "studio_root", + "venv_t5_root", "assets_root", "datasets_root", "dataset_uploads_root", diff --git a/studio/backend/utils/paths/storage_roots.py b/studio/backend/utils/paths/storage_roots.py index 288d9c04a9..caed95639d 100644 --- a/studio/backend/utils/paths/storage_roots.py +++ b/studio/backend/utils/paths/storage_roots.py @@ -15,6 +15,11 @@ def studio_root() -> Path: return Path.home() / ".unsloth" / "studio" +def venv_t5_root() -> Path: + """Pre-installed transformers 5.x directory, respects UNSLOTH_STUDIO_HOME.""" + return studio_root() / ".venv_t5" + + def cache_root() -> Path: """Central cache directory for all studio downloads (models, datasets, etc.).""" return studio_root() / "cache" diff --git a/studio/backend/utils/transformers_version.py b/studio/backend/utils/transformers_version.py index 60b43500c0..fa11a1cfb7 100644 --- a/studio/backend/utils/transformers_version.py +++ b/studio/backend/utils/transformers_version.py @@ -62,7 +62,8 @@ TRANSFORMERS_5_VERSION = "5.3.0" TRANSFORMERS_DEFAULT_VERSION = "4.57.6" # Pre-installed directory for transformers 5.x — created by setup.sh / setup.ps1 -_VENV_T5_DIR = str(Path.home() / ".unsloth" / "studio" / ".venv_t5") +from utils.paths.storage_roots import venv_t5_root +_VENV_T5_DIR = str(venv_t5_root()) def _resolve_base_model(model_name: str) -> str: diff --git a/studio/setup.sh b/studio/setup.sh index a039fc0bfc..3e56aebf38 100755 --- a/studio/setup.sh +++ b/studio/setup.sh @@ -251,7 +251,6 @@ install_python_stack() { python "$SCRIPT_DIR/install_python_stack.py" } -<<<<<<< HEAD # Create venv under ~/.unsloth/studio/ (shared location, not in repo). # All platforms (including Colab) use the same isolated venv so that # studio dependencies are never installed into the system Python. From 6e2f64b3b68cd954aced4520082ad0d46d7f496f Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Tue, 17 Mar 2026 19:20:14 +0000 Subject: [PATCH 3/6] fix: persist studio home path across server restarts --- studio/backend/utils/paths/storage_roots.py | 6 ++++++ studio/setup.ps1 | 4 ++++ unsloth_cli/commands/studio.py | 7 ++++++- 3 files changed, 16 insertions(+), 1 deletion(-) diff --git a/studio/backend/utils/paths/storage_roots.py b/studio/backend/utils/paths/storage_roots.py index caed95639d..dab68a6852 100644 --- a/studio/backend/utils/paths/storage_roots.py +++ b/studio/backend/utils/paths/storage_roots.py @@ -9,9 +9,15 @@ import tempfile def studio_root() -> Path: + """Studio root: env var > config file > default.""" custom = os.environ.get("UNSLOTH_STUDIO_HOME") if custom: return Path(custom).expanduser().resolve() + conf = Path.home() / ".unsloth" / "studio_home" + if conf.is_file(): + saved = conf.read_text().strip() + if saved: + return Path(saved).expanduser().resolve() return Path.home() / ".unsloth" / "studio" diff --git a/studio/setup.ps1 b/studio/setup.ps1 index 4a791ef2b1..0aa7c0335f 100644 --- a/studio/setup.ps1 +++ b/studio/setup.ps1 @@ -968,6 +968,10 @@ Write-Host "[OK] Using $PythonCmd ($(& $PythonCmd --version 2>&1))" -ForegroundC # ── Studio home (configurable via UNSLOTH_STUDIO_HOME) ── $StudioHome = if ($env:UNSLOTH_STUDIO_HOME) { $env:UNSLOTH_STUDIO_HOME } else { Join-Path $env:USERPROFILE ".unsloth\studio" } +# Persist for future `unsloth studio` runs (survives shell restarts) +$UnslothDir = Join-Path $env:USERPROFILE ".unsloth" +if (-not (Test-Path $UnslothDir)) { New-Item -ItemType Directory -Path $UnslothDir -Force | Out-Null } +Set-Content -Path (Join-Path $UnslothDir "studio_home") -Value $StudioHome -NoNewline $VenvDir = Join-Path $StudioHome ".venv" if (-not (Test-Path $VenvDir)) { diff --git a/unsloth_cli/commands/studio.py b/unsloth_cli/commands/studio.py index a39f050d5f..933fad1c53 100644 --- a/unsloth_cli/commands/studio.py +++ b/unsloth_cli/commands/studio.py @@ -14,10 +14,15 @@ studio_app = typer.Typer(help = "Unsloth Studio commands.") def _studio_home() -> Path: - """Studio root, overridable via UNSLOTH_STUDIO_HOME.""" + """Studio root: env var > config file > default.""" custom = os.environ.get("UNSLOTH_STUDIO_HOME") if custom: return Path(custom).expanduser().resolve() + conf = Path.home() / ".unsloth" / "studio_home" + if conf.is_file(): + saved = conf.read_text().strip() + if saved: + return Path(saved).expanduser().resolve() return Path.home() / ".unsloth" / "studio" # __file__ is unsloth_cli/commands/studio.py -- two parents up is the package root From c2b0a4962712a46707fc2dc6323ab2b8df87b2bf Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Tue, 17 Mar 2026 20:00:18 +0000 Subject: [PATCH 4/6] chore: comment out llama.cpp build in setup.sh --- studio/setup.sh | 265 ++++++++++++++++++++++++------------------------ 1 file changed, 131 insertions(+), 134 deletions(-) diff --git a/studio/setup.sh b/studio/setup.sh index 3e56aebf38..cf9c89163a 100755 --- a/studio/setup.sh +++ b/studio/setup.sh @@ -372,140 +372,137 @@ if grep -qi microsoft /proc/version 2>/dev/null; then fi fi -# ── 8. Build llama.cpp binaries for GGUF inference + export ── -# Builds at ~/.unsloth/llama.cpp — a single shared location under the user's -# home directory. This is used by both the inference server and the GGUF -# export pipeline (unsloth-zoo). -# - llama-server: for GGUF model inference -# - llama-quantize: for GGUF export quantization (symlinked to root for check_llama_cpp()) -UNSLOTH_HOME="$HOME/.unsloth" -mkdir -p "$UNSLOTH_HOME" -LLAMA_CPP_DIR="$UNSLOTH_HOME/llama.cpp" -LLAMA_SERVER_BIN="$LLAMA_CPP_DIR/build/bin/llama-server" -if [ "${_SKIP_GGUF_BUILD:-}" = true ]; then - echo "" - echo "Skipping llama-server build (missing dependencies)" - echo " Install the missing packages and re-run setup to enable GGUF inference." -else -rm -rf "$LLAMA_CPP_DIR" -{ - # Check prerequisites - if ! command -v cmake &>/dev/null; then - echo "" - echo "⚠️ cmake not found — skipping llama-server build (GGUF inference won't be available)" - echo " Install cmake and re-run setup.sh to enable GGUF inference." - elif ! command -v git &>/dev/null; then - echo "" - echo "⚠️ git not found — skipping llama-server build (GGUF inference won't be available)" - else - echo "" - echo "Building llama-server for GGUF inference..." - - BUILD_OK=true - run_quiet "clone llama.cpp" git clone --depth 1 https://github.com/ggml-org/llama.cpp.git "$LLAMA_CPP_DIR" || BUILD_OK=false - - if [ "$BUILD_OK" = true ]; then - # Skip tests/examples we don't need (faster build) - CMAKE_ARGS="-DLLAMA_BUILD_TESTS=OFF -DLLAMA_BUILD_EXAMPLES=OFF -DLLAMA_BUILD_SERVER=ON -DGGML_NATIVE=ON" - - # Use ccache if available (dramatically faster rebuilds) - if command -v ccache &>/dev/null; then - CMAKE_ARGS="$CMAKE_ARGS -DCMAKE_C_COMPILER_LAUNCHER=ccache -DCMAKE_CXX_COMPILER_LAUNCHER=ccache -DCMAKE_CUDA_COMPILER_LAUNCHER=ccache" - echo " Using ccache for faster compilation" - fi - - # Detect CUDA: check nvcc on PATH, then common install locations - NVCC_PATH="" - if command -v nvcc &>/dev/null; then - NVCC_PATH="$(command -v nvcc)" - elif [ -x /usr/local/cuda/bin/nvcc ]; then - NVCC_PATH="/usr/local/cuda/bin/nvcc" - export PATH="/usr/local/cuda/bin:$PATH" - elif ls /usr/local/cuda-*/bin/nvcc &>/dev/null 2>&1; then - # Pick the newest cuda-XX.X directory - NVCC_PATH="$(ls -d /usr/local/cuda-*/bin/nvcc 2>/dev/null | sort -V | tail -1)" - export PATH="$(dirname "$NVCC_PATH"):$PATH" - fi - - if [ -n "$NVCC_PATH" ]; then - echo " Building with CUDA support (nvcc: $NVCC_PATH)..." - CMAKE_ARGS="$CMAKE_ARGS -DGGML_CUDA=ON" - - # Detect GPU compute capability and limit CUDA architectures - # Without this, cmake builds for ALL default archs (very slow) - CUDA_ARCHS="" - if command -v nvidia-smi &>/dev/null; then - # Read all GPUs, deduplicate (handles mixed-GPU hosts) - _raw_caps=$(nvidia-smi --query-gpu=compute_cap --format=csv,noheader 2>/dev/null || true) - while IFS= read -r _cap; do - _cap=$(echo "$_cap" | tr -d '[:space:]') - if [[ "$_cap" =~ ^([0-9]+)\.([0-9]+)$ ]]; then - _arch="${BASH_REMATCH[1]}${BASH_REMATCH[2]}" - # Append if not already present - case ";$CUDA_ARCHS;" in - *";$_arch;"*) ;; - *) CUDA_ARCHS="${CUDA_ARCHS:+$CUDA_ARCHS;}$_arch" ;; - esac - fi - done <<< "$_raw_caps" - fi - - if [ -n "$CUDA_ARCHS" ]; then - echo " GPU compute capabilities: ${CUDA_ARCHS//;/, } -- limiting build to detected archs" - CMAKE_ARGS="$CMAKE_ARGS -DCMAKE_CUDA_ARCHITECTURES=${CUDA_ARCHS}" - else - echo " Could not detect GPU arch -- building for all default CUDA architectures (slower)" - fi - - # Multi-threaded nvcc compilation (uses all CPU cores per .cu file) - CMAKE_ARGS="$CMAKE_ARGS -DCMAKE_CUDA_FLAGS=--threads=0" - elif [ -d /usr/local/cuda ] || nvidia-smi &>/dev/null; then - echo " CUDA driver detected but nvcc not found — building CPU-only" - echo " To enable GPU: install cuda-toolkit or add nvcc to PATH" - else - echo " Building CPU-only (no CUDA detected)..." - fi - - NCPU=$(nproc 2>/dev/null || sysctl -n hw.ncpu 2>/dev/null || echo 4) - - # Use Ninja if available (faster parallel builds than Make) - CMAKE_GENERATOR_ARGS="" - if command -v ninja &>/dev/null; then - CMAKE_GENERATOR_ARGS="-G Ninja" - fi - - run_quiet "cmake llama.cpp" cmake $CMAKE_GENERATOR_ARGS -S "$LLAMA_CPP_DIR" -B "$LLAMA_CPP_DIR/build" $CMAKE_ARGS || BUILD_OK=false - fi - - if [ "$BUILD_OK" = true ]; then - run_quiet "build llama-server" cmake --build "$LLAMA_CPP_DIR/build" --config Release --target llama-server -j"$NCPU" || BUILD_OK=false - fi - - # Also build llama-quantize (needed by unsloth-zoo's GGUF export pipeline) - if [ "$BUILD_OK" = true ]; then - run_quiet "build llama-quantize" cmake --build "$LLAMA_CPP_DIR/build" --config Release --target llama-quantize -j"$NCPU" || true - # Symlink to llama.cpp root — check_llama_cpp() looks for the binary there - QUANTIZE_BIN="$LLAMA_CPP_DIR/build/bin/llama-quantize" - if [ -f "$QUANTIZE_BIN" ]; then - ln -sf build/bin/llama-quantize "$LLAMA_CPP_DIR/llama-quantize" - fi - fi - - if [ "$BUILD_OK" = true ]; then - if [ -f "$LLAMA_SERVER_BIN" ]; then - echo "✅ llama-server built at $LLAMA_SERVER_BIN" - else - echo "⚠️ llama-server binary not found after build — GGUF inference won't be available" - fi - if [ -f "$LLAMA_CPP_DIR/llama-quantize" ]; then - echo "✅ llama-quantize available for GGUF export" - fi - else - echo "⚠️ llama-server build failed — GGUF inference won't be available, but everything else works" - fi - fi -} -fi # end _SKIP_GGUF_BUILD check +# Disabled: llama.cpp build is commented out for now. +# UNCOMMENT the block below to re-enable. +# +# # Builds at ~/.unsloth/llama.cpp — a single shared location under the user's +# # home directory. This is used by both the inference server and the GGUF +# # export pipeline (unsloth-zoo). +# # - llama-server: for GGUF model inference +# # - llama-quantize: for GGUF export quantization (symlinked to root for check_llama_cpp()) +# UNSLOTH_HOME="$HOME/.unsloth" +# mkdir -p "$UNSLOTH_HOME" +# LLAMA_CPP_DIR="$UNSLOTH_HOME/llama.cpp" +# LLAMA_SERVER_BIN="$LLAMA_CPP_DIR/build/bin/llama-server" +# rm -rf "$LLAMA_CPP_DIR" +# { +# # Check prerequisites +# if ! command -v cmake &>/dev/null; then +# echo "" +# echo "⚠️ cmake not found — skipping llama-server build (GGUF inference won't be available)" +# echo " Install cmake and re-run setup.sh to enable GGUF inference." +# elif ! command -v git &>/dev/null; then +# echo "" +# echo "⚠️ git not found — skipping llama-server build (GGUF inference won't be available)" +# else +# echo "" +# echo "Building llama-server for GGUF inference..." +# +# BUILD_OK=true +# run_quiet "clone llama.cpp" git clone --depth 1 https://github.com/ggml-org/llama.cpp.git "$LLAMA_CPP_DIR" || BUILD_OK=false +# +# if [ "$BUILD_OK" = true ]; then +# # Skip tests/examples we don't need (faster build) +# CMAKE_ARGS="-DLLAMA_BUILD_TESTS=OFF -DLLAMA_BUILD_EXAMPLES=OFF -DLLAMA_BUILD_SERVER=ON -DGGML_NATIVE=ON" +# +# # Use ccache if available (dramatically faster rebuilds) +# if command -v ccache &>/dev/null; then +# CMAKE_ARGS="$CMAKE_ARGS -DCMAKE_C_COMPILER_LAUNCHER=ccache -DCMAKE_CXX_COMPILER_LAUNCHER=ccache -DCMAKE_CUDA_COMPILER_LAUNCHER=ccache" +# echo " Using ccache for faster compilation" +# fi +# +# # Detect CUDA: check nvcc on PATH, then common install locations +# NVCC_PATH="" +# if command -v nvcc &>/dev/null; then +# NVCC_PATH="$(command -v nvcc)" +# elif [ -x /usr/local/cuda/bin/nvcc ]; then +# NVCC_PATH="/usr/local/cuda/bin/nvcc" +# export PATH="/usr/local/cuda/bin:$PATH" +# elif ls /usr/local/cuda-*/bin/nvcc &>/dev/null 2>&1; then +# # Pick the newest cuda-XX.X directory +# NVCC_PATH="$(ls -d /usr/local/cuda-*/bin/nvcc 2>/dev/null | sort -V | tail -1)" +# export PATH="$(dirname "$NVCC_PATH"):$PATH" +# fi +# +# if [ -n "$NVCC_PATH" ]; then +# echo " Building with CUDA support (nvcc: $NVCC_PATH)..." +# CMAKE_ARGS="$CMAKE_ARGS -DGGML_CUDA=ON" +# +# # Detect GPU compute capability and limit CUDA architectures +# # Without this, cmake builds for ALL default archs (very slow) +# CUDA_ARCHS="" +# if command -v nvidia-smi &>/dev/null; then +# # Read all GPUs, deduplicate (handles mixed-GPU hosts) +# _raw_caps=$(nvidia-smi --query-gpu=compute_cap --format=csv,noheader 2>/dev/null || true) +# while IFS= read -r _cap; do +# _cap=$(echo "$_cap" | tr -d '[:space:]') +# if [[ "$_cap" =~ ^([0-9]+)\.([0-9]+)$ ]]; then +# _arch="${BASH_REMATCH[1]}${BASH_REMATCH[2]}" +# # Append if not already present +# case ";$CUDA_ARCHS;" in +# *";$_arch;"*) ;; +# *) CUDA_ARCHS="${CUDA_ARCHS:+$CUDA_ARCHS;}$_arch" ;; +# esac +# fi +# done <<< "$_raw_caps" +# fi +# +# if [ -n "$CUDA_ARCHS" ]; then +# echo " GPU compute capabilities: ${CUDA_ARCHS//;/, } -- limiting build to detected archs" +# CMAKE_ARGS="$CMAKE_ARGS -DCMAKE_CUDA_ARCHITECTURES=${CUDA_ARCHS}" +# else +# echo " Could not detect GPU arch -- building for all default CUDA architectures (slower)" +# fi +# +# # Multi-threaded nvcc compilation (uses all CPU cores per .cu file) +# CMAKE_ARGS="$CMAKE_ARGS -DCMAKE_CUDA_FLAGS=--threads=0" +# elif [ -d /usr/local/cuda ] || nvidia-smi &>/dev/null; then +# echo " CUDA driver detected but nvcc not found — building CPU-only" +# echo " To enable GPU: install cuda-toolkit or add nvcc to PATH" +# else +# echo " Building CPU-only (no CUDA detected)..." +# fi +# +# NCPU=$(nproc 2>/dev/null || sysctl -n hw.ncpu 2>/dev/null || echo 4) +# +# # Use Ninja if available (faster parallel builds than Make) +# CMAKE_GENERATOR_ARGS="" +# if command -v ninja &>/dev/null; then +# CMAKE_GENERATOR_ARGS="-G Ninja" +# fi +# +# run_quiet "cmake llama.cpp" cmake $CMAKE_GENERATOR_ARGS -S "$LLAMA_CPP_DIR" -B "$LLAMA_CPP_DIR/build" $CMAKE_ARGS || BUILD_OK=false +# fi +# +# if [ "$BUILD_OK" = true ]; then +# run_quiet "build llama-server" cmake --build "$LLAMA_CPP_DIR/build" --config Release --target llama-server -j"$NCPU" || BUILD_OK=false +# fi +# +# # Also build llama-quantize (needed by unsloth-zoo's GGUF export pipeline) +# if [ "$BUILD_OK" = true ]; then +# run_quiet "build llama-quantize" cmake --build "$LLAMA_CPP_DIR/build" --config Release --target llama-quantize -j"$NCPU" || true +# # Symlink to llama.cpp root — check_llama_cpp() looks for the binary there +# QUANTIZE_BIN="$LLAMA_CPP_DIR/build/bin/llama-quantize" +# if [ -f "$QUANTIZE_BIN" ]; then +# ln -sf build/bin/llama-quantize "$LLAMA_CPP_DIR/llama-quantize" +# fi +# fi +# +# if [ "$BUILD_OK" = true ]; then +# if [ -f "$LLAMA_SERVER_BIN" ]; then +# echo "✅ llama-server built at $LLAMA_SERVER_BIN" +# else +# echo "⚠️ llama-server binary not found after build — GGUF inference won't be available" +# fi +# if [ -f "$LLAMA_CPP_DIR/llama-quantize" ]; then +# echo "✅ llama-quantize available for GGUF export" +# fi +# else +# echo "⚠️ llama-server build failed — GGUF inference won't be available, but everything else works" +# fi +# fi +# } +>>>>>>> 0f9e2ccd (chore: comment out llama.cpp build in setup.sh) echo "" if [ "$IS_COLAB" = true ]; then From 286f954da66395565bc96bed2defebb40d8a86a0 Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Thu, 19 Mar 2026 13:36:13 +0000 Subject: [PATCH 5/6] fix setup.sh marker removal --- studio/setup.sh | 1 - 1 file changed, 1 deletion(-) diff --git a/studio/setup.sh b/studio/setup.sh index cf9c89163a..d64aff0583 100755 --- a/studio/setup.sh +++ b/studio/setup.sh @@ -502,7 +502,6 @@ fi # fi # fi # } ->>>>>>> 0f9e2ccd (chore: comment out llama.cpp build in setup.sh) echo "" if [ "$IS_COLAB" = true ]; then From aa7c0b6d6cf00ebe68f97b0d73988323fa6ab92a Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Thu, 19 Mar 2026 14:18:00 +0000 Subject: [PATCH 6/6] recover working install files --- studio/install_python_stack.py | 12 +- studio/setup.sh | 208 +++++++++------------------------ 2 files changed, 58 insertions(+), 162 deletions(-) diff --git a/studio/install_python_stack.py b/studio/install_python_stack.py index a141c64425..20011a298a 100644 --- a/studio/install_python_stack.py +++ b/studio/install_python_stack.py @@ -140,15 +140,15 @@ def _bootstrap_uv() -> bool: global UV_NEEDS_SYSTEM if not shutil.which("uv"): return False - # Probe: try a dry-run install targeting the current Python explicitly. - # Without --python, uv can ignore the activated venv on some platforms. + # Probe: try a dry-run install without --system. + # If uv can't find a venv it exits with code 2. probe = subprocess.run( - ["uv", "pip", "install", "--dry-run", "--python", sys.executable, "pip"], + ["uv", "pip", "install", "--dry-run", "pip"], stdout = subprocess.PIPE, stderr = subprocess.STDOUT, ) if probe.returncode != 0: - # Retry with --system (some envs need it when uv can't find a venv) + # Retry with --system to confirm it works probe_sys = subprocess.run( ["uv", "pip", "install", "--dry-run", "--system", "pip"], stdout = subprocess.PIPE, @@ -204,10 +204,6 @@ def _build_uv_cmd(args: tuple[str, ...]) -> list[str]: cmd = ["uv", "pip", "install"] if UV_NEEDS_SYSTEM: cmd.append("--system") - # Always pass --python so uv targets the correct environment. - # Without this, uv can ignore an activated venv and install into - # the system Python (observed on Colab and similar environments). - cmd.extend(["--python", sys.executable]) cmd.extend(_translate_pip_args_for_uv(args)) cmd.append("--torch-backend=auto") return cmd diff --git a/studio/setup.sh b/studio/setup.sh index d64aff0583..ae47f7e362 100755 --- a/studio/setup.sh +++ b/studio/setup.sh @@ -41,26 +41,13 @@ if [[ "$keynames" == *$'\nCOLAB_'* ]]; then fi # ── Detect whether frontend needs building ── -# Skip if dist/ exists AND no tracked input is newer than dist/. -# Checks top-level config/entry files and src/, public/ recursively. -# This handles: PyPI installs (dist/ bundled), repeat runs (no changes), -# and upgrades/pulls (source newer than dist/ triggers rebuild). -_NEED_FRONTEND_BUILD=true -if [ -d "$SCRIPT_DIR/frontend/dist" ]; then - # Check all top-level files (package.json, bun.lock, vite.config.ts, index.html, etc.) - _changed=$(find "$SCRIPT_DIR/frontend" -maxdepth 1 -type f \ - -newer "$SCRIPT_DIR/frontend/dist" -print -quit 2>/dev/null) - # Check src/ and public/ recursively (|| true guards against set -e when dirs are missing) - if [ -z "$_changed" ]; then - _changed=$(find "$SCRIPT_DIR/frontend/src" "$SCRIPT_DIR/frontend/public" \ - -type f -newer "$SCRIPT_DIR/frontend/dist" -print -quit 2>/dev/null) || true - fi - if [ -z "$_changed" ]; then - _NEED_FRONTEND_BUILD=false - fi -fi -if [ "$_NEED_FRONTEND_BUILD" = false ]; then - echo "✅ Frontend already built and up to date -- skipping Node/npm check." +# Only skip when BOTH conditions are true: +# 1. We're inside site-packages (PyPI / pip install, not editable) +# 2. dist/ already exists (pre-built in the wheel) +# Otherwise always (re)build — handles upgrades, editable installs, and +# pip-from-source where dist/ was never built. +if [[ "$SCRIPT_DIR" == */site-packages/* ]] && [ -d "$SCRIPT_DIR/frontend/dist" ]; then + echo "✅ Frontend pre-built (PyPI) — skipping Node/npm check." else NEED_NODE=true if command -v node &>/dev/null && command -v npm &>/dev/null; then @@ -159,27 +146,12 @@ run_quiet "npm run build" npm run build _restore_gitignores trap - EXIT - -# Validate CSS output -- catch truncated Tailwind builds -_MAX_CSS=$(find "$SCRIPT_DIR/frontend/dist/assets" -name '*.css' -exec wc -c {} + 2>/dev/null | sort -n | tail -1 | awk '{print $1}') -if [ -z "$_MAX_CSS" ]; then - echo "⚠️ WARNING: No CSS files were emitted. The frontend build may have failed." -elif [ "$_MAX_CSS" -lt 100000 ]; then - echo "⚠️ WARNING: Largest CSS file is only $((_MAX_CSS / 1024))KB (expected >100KB)." - echo " Tailwind may not have scanned all source files. Check for .gitignore interference." -fi - +cd "$SCRIPT_DIR/backend/core/data_recipe/oxc-validator" +run_quiet "npm install (oxc validator runtime)" npm install cd "$SCRIPT_DIR" echo "✅ Frontend built to frontend/dist" -fi # end frontend build check - -# ── oxc-validator runtime (needs npm -- skip if not available) ── -if [ -d "$SCRIPT_DIR/backend/core/data_recipe/oxc-validator" ] && command -v npm &>/dev/null; then - cd "$SCRIPT_DIR/backend/core/data_recipe/oxc-validator" - run_quiet "npm install (oxc validator runtime)" npm install - cd "$SCRIPT_DIR" -fi +fi # end frontend dist check # ── 6. Python venv + deps ── @@ -251,127 +223,58 @@ install_python_stack() { python "$SCRIPT_DIR/install_python_stack.py" } -# Create venv under ~/.unsloth/studio/ (shared location, not in repo). -# All platforms (including Colab) use the same isolated venv so that -# studio dependencies are never installed into the system Python. -STUDIO_HOME="$HOME/.unsloth/studio" -VENV_DIR="$STUDIO_HOME/.venv" -VENV_T5_DIR="$STUDIO_HOME/.venv_t5" -mkdir -p "$STUDIO_HOME" - -# Clean up legacy in-repo venvs if they exist -[ -d "$REPO_ROOT/.venv" ] && rm -rf "$REPO_ROOT/.venv" -[ -d "$REPO_ROOT/.venv_overlay" ] && rm -rf "$REPO_ROOT/.venv_overlay" -[ -d "$REPO_ROOT/.venv_t5" ] && rm -rf "$REPO_ROOT/.venv_t5" - -rm -rf "$VENV_DIR" -rm -rf "$VENV_T5_DIR" -# Try creating venv with pip; fall back to --without-pip + bootstrap -# (some environments like Colab have broken ensurepip) -if ! "$BEST_PY" -m venv "$VENV_DIR" 2>/dev/null; then - "$BEST_PY" -m venv --without-pip "$VENV_DIR" - source "$VENV_DIR/bin/activate" - curl -sS https://bootstrap.pypa.io/get-pip.py | python > /dev/null +if [ "$IS_COLAB" = true ]; then + # Colab: install packages directly without venv + install_python_stack else + # Local: create venv under studio home (shared location, not in repo) + # Configurable via UNSLOTH_STUDIO_HOME; defaults to ~/.unsloth/studio + STUDIO_HOME="${UNSLOTH_STUDIO_HOME:-$HOME/.unsloth/studio}" + echo " Studio home: $STUDIO_HOME" + # Persist for future `unsloth studio` runs (survives shell restarts) + mkdir -p "$HOME/.unsloth" + echo "$STUDIO_HOME" > "$HOME/.unsloth/studio_home" + VENV_DIR="$STUDIO_HOME/.venv" + VENV_T5_DIR="$STUDIO_HOME/.venv_t5" + mkdir -p "$STUDIO_HOME" + + # Clean up legacy in-repo venvs if they exist + [ -d "$REPO_ROOT/.venv" ] && rm -rf "$REPO_ROOT/.venv" + [ -d "$REPO_ROOT/.venv_overlay" ] && rm -rf "$REPO_ROOT/.venv_overlay" + [ -d "$REPO_ROOT/.venv_t5" ] && rm -rf "$REPO_ROOT/.venv_t5" + + rm -rf "$VENV_DIR" + rm -rf "$VENV_T5_DIR" + "$BEST_PY" -m venv "$VENV_DIR" source "$VENV_DIR/bin/activate" -fi + cd "$SCRIPT_DIR" + install_python_stack -# ── Ensure uv is available (much faster than pip) ── -USE_UV=false -if command -v uv &>/dev/null; then - USE_UV=true -elif curl -LsSf https://astral.sh/uv/install.sh | sh > /dev/null 2>&1; then - export PATH="$HOME/.local/bin:$PATH" - command -v uv &>/dev/null && USE_UV=true -fi - -# Helper: install a package, preferring uv with pip fallback -fast_install() { - if [ "$USE_UV" = true ]; then - uv pip install --python "$(command -v python)" "$@" && return 0 - fi - python -m pip install "$@" -} - -cd "$SCRIPT_DIR" -install_python_stack - -# ── 6b. Pre-install transformers 5.x into .venv_t5/ ── -# Models like GLM-4.7-Flash need transformers>=5.3.0. Instead of pip-installing -# at runtime (slow, ~10-15s), we pre-install into a separate directory. -# The training subprocess just prepends .venv_t5/ to sys.path -- instant switch. -echo "" -echo " Pre-installing transformers 5.x for newer model support..." -mkdir -p "$VENV_T5_DIR" -run_quiet "install transformers 5.x" fast_install --target "$VENV_T5_DIR" --no-deps "transformers==5.3.0" -run_quiet "install huggingface_hub for t5" fast_install --target "$VENV_T5_DIR" --no-deps "huggingface_hub==1.7.1" -run_quiet "install hf_xet for t5" fast_install --target "$VENV_T5_DIR" --no-deps "hf_xet==1.4.2" -# tiktoken is needed by Qwen-family tokenizers. Install with deps since -# regex/requests may be missing on Windows. -run_quiet "install tiktoken for t5" fast_install --target "$VENV_T5_DIR" "tiktoken" -echo "✅ Transformers 5.x pre-installed to $VENV_T5_DIR/" - -# ── 7. WSL: pre-install GGUF build dependencies ── -# On WSL, sudo requires a password and can't be entered during GGUF export -# (runs in a non-interactive subprocess). Install build deps here instead. -if grep -qi microsoft /proc/version 2>/dev/null; then + # ── 6b. Pre-install transformers 5.x into .venv_t5/ ── + # Models like GLM-4.7-Flash need transformers>=5.3.0. Instead of pip-installing + # at runtime (slow, ~10-15s), we pre-install into a separate directory. + # The training subprocess just prepends .venv_t5/ to sys.path — instant switch. echo "" - echo "⚠️ WSL detected -- installing build dependencies for GGUF export..." - _GGUF_DEPS="pciutils build-essential cmake curl git libcurl4-openssl-dev" + echo " Pre-installing transformers 5.x for newer model support..." + mkdir -p "$VENV_T5_DIR" + run_quiet "pip install transformers 5.x" pip install --target "$VENV_T5_DIR" --no-deps "transformers==5.3.0" + run_quiet "pip install huggingface_hub for t5" pip install --target "$VENV_T5_DIR" --no-deps "huggingface_hub==1.3.0" + echo "✅ Transformers 5.x pre-installed to $VENV_T5_DIR/" - # Try without sudo first (works when already root) - apt-get update -y >/dev/null 2>&1 || true - apt-get install -y $_GGUF_DEPS >/dev/null 2>&1 || true - - # Check which packages are still missing - _STILL_MISSING="" - for _pkg in $_GGUF_DEPS; do - case "$_pkg" in - build-essential) command -v gcc >/dev/null 2>&1 || _STILL_MISSING="$_STILL_MISSING $_pkg" ;; - pciutils) command -v lspci >/dev/null 2>&1 || _STILL_MISSING="$_STILL_MISSING $_pkg" ;; - libcurl4-openssl-dev) dpkg -s "$_pkg" >/dev/null 2>&1 || _STILL_MISSING="$_STILL_MISSING $_pkg" ;; - *) command -v "$_pkg" >/dev/null 2>&1 || _STILL_MISSING="$_STILL_MISSING $_pkg" ;; - esac - done - _STILL_MISSING=$(echo "$_STILL_MISSING" | sed 's/^ *//') - - if [ -z "$_STILL_MISSING" ]; then + # ── 7. WSL: pre-install GGUF build dependencies ── + # On WSL, sudo requires a password and can't be entered during GGUF export + # (runs in a non-interactive subprocess). Install build deps here instead. + if grep -qi microsoft /proc/version 2>/dev/null; then + echo "" + echo "⚠️ WSL detected — installing build dependencies for GGUF export..." + echo " You may be prompted for your password." + sudo apt-get update -y + sudo apt-get install -y build-essential cmake curl git libcurl4-openssl-dev echo "✅ GGUF build dependencies installed" - elif command -v sudo >/dev/null 2>&1; then - echo "" - echo " !!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!" - echo " WARNING: We require sudo elevated permissions to install:" - echo " $_STILL_MISSING" - echo " If you accept, we'll run sudo now, and it'll prompt your password." - echo " !!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!" - echo "" - printf " Accept? [Y/n] " - if [ -r /dev/tty ]; then - read -r REPLY