From 46b1a78e52e557db0a54a5971f1fbc7e7e1a789c Mon Sep 17 00:00:00 2001 From: samit Date: Fri, 27 Feb 2026 06:00:28 -0800 Subject: [PATCH 01/66] deleted duplicate definitions --- studio/backend/core/inference/inference.py | 131 --------------------- 1 file changed, 131 deletions(-) diff --git a/studio/backend/core/inference/inference.py b/studio/backend/core/inference/inference.py index 1147c281b7..329f5d944b 100644 --- a/studio/backend/core/inference/inference.py +++ b/studio/backend/core/inference/inference.py @@ -268,47 +268,6 @@ class InferenceBackend: logger.error(traceback.format_exc()) return False, None - def load_adapter(self, base_model_name: str, adapter_path: str, adapter_name: str = None) -> bool: - """ - Load a LoRA adapter onto the base model if it's not already registered. - This method is idempotent. - """ - if base_model_name not in self.models: - logger.error(f"Base model {base_model_name} not loaded") - return False - - model = self.models[base_model_name].get("model") - if model is None: - logger.error(f"Model object for {base_model_name} is None.") - return False - - if adapter_name is None: - adapter_name = adapter_path.split("/")[-1].replace(".", "_") - - # If we've loaded this adapter before, we don't need to do anything. - if adapter_name in self.models[base_model_name].get("loaded_adapters", {}): - logger.info(f"Adapter '{adapter_name}' is already registered. Skipping.") - return True - - try: - logger.info(f"Loading new adapter '{adapter_name}' from '{adapter_path}' onto {base_model_name}") - - # Unsloth modifies the model in-place and returns None. Do NOT re-assign. - model.load_adapter(adapter_path, adapter_name=adapter_name) - - # Update our internal registry so we don't load it again. - self.models[base_model_name]["loaded_adapters"][adapter_name] = adapter_path - - total_adapters = len(getattr(model, 'peft_config', {})) - logger.info(f"Adapter '{adapter_name}' loaded successfully. (Total adapters on model: {total_adapters})") - return True - except Exception as e: - logger.error(f"Failed to load adapter '{adapter_name}': {e}") - import traceback - logger.error(traceback.format_exc()) - return False - pass - def enable_adapter(self, base_model_name: str, adapter_name: str) -> bool: """Enable specific adapter (for generation)""" if base_model_name not in self.models: @@ -341,55 +300,6 @@ class InferenceBackend: logger.error(f"Failed to disable adapters: {e}") return False - # In backend/inference.py - - def load_for_eval(self, lora_path: str, max_seq_length: int = 2048, - dtype = None, load_in_4bit: bool = True, - hf_token: Optional[str] = None) -> Tuple[bool, Optional[str], Optional[str]]: - """ - Prepare for eval: ensure base model and the specified adapter are loaded. - """ - try: - from utils.models import ModelConfig - lora_config = ModelConfig.from_lora_path(lora_path, hf_token) - if not lora_config: - return False, None, None - - base_model_name = lora_config.base_model - - # 1. Load the base model if it's not already in memory (this logic is correct) - if base_model_name not in self.models or not self.models[base_model_name].get("model"): - logger.info(f"Base model '{base_model_name}' not loaded, loading now.") - base_config = ModelConfig.from_ui_selection(base_model_name, None, is_lora=False) - if not self.load_model(base_config, max_seq_length, dtype, load_in_4bit, hf_token): - return False, None, None - else: - logger.info(f"Base model '{base_model_name}' is already in memory.") - - self.active_model_name = base_model_name - - # 2. Delegate to our now-idempotent load_adapter function. - # It will handle all cases: first adapter, or subsequent adapters. - adapter_name = lora_path.split("/")[-1].replace(".", "_") - adapter_success = self.load_adapter( - base_model_name=base_model_name, - adapter_path=lora_path, - adapter_name=adapter_name - ) - - if not adapter_success: - return False, base_model_name, None - - return True, base_model_name, adapter_name - - except Exception as e: - logger.error(f"Error during load_for_eval: {e}") - import traceback - logger.error(traceback.format_exc()) - return False, None, None - pass - - def load_for_eval(self, lora_path: str, max_seq_length: int = 2048, dtype = None, load_in_4bit: bool = True, hf_token: Optional[str] = None) -> Tuple[bool, Optional[str], Optional[str]]: @@ -1272,47 +1182,6 @@ class InferenceBackend: """Get name of currently loading model""" return next(iter(self.loading_models)) if self.loading_models else None - def load_model_simple(self, - model_path: str, - hf_token: Optional[str] = None, - max_seq_length: int = 2048, - load_in_4bit: bool = True) -> bool: - """ - Simple model loading wrapper for chat interface. - Accepts model path as string and handles ModelConfig creation internally. - - Args: - model_path: Model name or path (e.g., "unsloth/llama-3-8b") - hf_token: HuggingFace token for gated models - max_seq_length: Maximum sequence length - load_in_4bit: Whether to use 4-bit quantization - - Returns: - bool: True if successful, False otherwise - """ - try: - # Create config from string path - config = ModelConfig.from_ui_selection( - model_path, - lora_path=None, # No LoRA for chat - is_lora=False - ) - - # Call existing load_model with config - return self.load_model( - config=config, - max_seq_length=max_seq_length, - dtype=None, # Auto-detect - load_in_4bit=load_in_4bit, - hf_token=hf_token - ) - - except Exception as e: - logger.error(f"Error in load_model_simple: {e}") - return False - - - def load_model_simple(self, model_path: str, hf_token: Optional[str] = None, From 2714789381bd379e7dba66537d352a321c1ff6ca Mon Sep 17 00:00:00 2001 From: samit Date: Sat, 28 Feb 2026 01:17:43 -0800 Subject: [PATCH 02/66] added auth to dataset endpopints --- studio/backend/routes/datasets.py | 8 ++++++-- 1 file changed, 6 insertions(+), 2 deletions(-) diff --git a/studio/backend/routes/datasets.py b/studio/backend/routes/datasets.py index cb1ea75e33..a53e223015 100644 --- a/studio/backend/routes/datasets.py +++ b/studio/backend/routes/datasets.py @@ -5,7 +5,7 @@ import base64 import io import sys from pathlib import Path -from fastapi import APIRouter, HTTPException +from fastapi import APIRouter, Depends, HTTPException import logging # Add backend directory to path @@ -15,6 +15,7 @@ if str(backend_path) not in sys.path: # Import dataset utilities from utils.datasets import check_dataset_format +from auth.authentication import get_current_subject router = APIRouter() logger = logging.getLogger(__name__) @@ -84,7 +85,10 @@ DATA_EXTS = ( @router.post("/check-format", response_model=CheckFormatResponse) -def check_format(request: CheckFormatRequest): +def check_format( + request: CheckFormatRequest, + current_subject: str = Depends(get_current_subject), +): """ Check if a dataset requires manual column mapping. From 39a2fef4d9866aa9ce28c13650657b0fb76a1a75 Mon Sep 17 00:00:00 2001 From: samit Date: Sat, 28 Feb 2026 02:10:12 -0800 Subject: [PATCH 03/66] updated fetch to auth fetch in the frontend --- studio/frontend/src/features/training/api/datasets-api.ts | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/studio/frontend/src/features/training/api/datasets-api.ts b/studio/frontend/src/features/training/api/datasets-api.ts index 7bba75ca38..b2348abed8 100644 --- a/studio/frontend/src/features/training/api/datasets-api.ts +++ b/studio/frontend/src/features/training/api/datasets-api.ts @@ -1,3 +1,4 @@ +import { authFetch } from "@/features/auth"; import type { CheckFormatResponse } from "../types/datasets"; type CheckDatasetFormatArgs = { @@ -15,7 +16,7 @@ export async function checkDatasetFormat({ split, isVlm, }: CheckDatasetFormatArgs): Promise { - const res = await fetch("/api/datasets/check-format", { + const res = await authFetch("/api/datasets/check-format", { method: "POST", headers: { "Content-Type": "application/json" }, body: JSON.stringify({ From 26996bc609aa3f84dcfd62cee362dcd93500b4a8 Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Fri, 27 Feb 2026 19:45:32 +0000 Subject: [PATCH 04/66] Extract shared install_python_stack.py for cross-platform setup --- install_python_stack.py | 182 +++++++++++ setup.ps1 | 660 ++++++++++++++++++++++++++++++++++++++++ setup.sh | 25 +- 3 files changed, 843 insertions(+), 24 deletions(-) create mode 100644 install_python_stack.py create mode 100644 setup.ps1 diff --git a/install_python_stack.py b/install_python_stack.py new file mode 100644 index 0000000000..e31e96232e --- /dev/null +++ b/install_python_stack.py @@ -0,0 +1,182 @@ +#!/usr/bin/env python3 +"""Cross-platform Python dependency installer for Unsloth Studio. + +Called by both setup.sh (Linux / WSL) and setup.ps1 (Windows) after the +virtual environment is already activated. Expects `pip` and `python` on +PATH to point at the venv. +""" + +from __future__ import annotations + +import os +import subprocess +import sys +import urllib.request +from pathlib import Path + +# ── Paths ────────────────────────────────────────────────────────────── +SCRIPT_DIR = Path(__file__).resolve().parent +REQ_ROOT = SCRIPT_DIR / "studio" / "backend" / "requirements" +SINGLE_ENV = REQ_ROOT / "single-env" +CONSTRAINTS = SINGLE_ENV / "constraints.txt" + +# ── Helpers ──────────────────────────────────────────────────────────── + +def _green(msg: str) -> str: + return f"\033[92m{msg}\033[0m" + +def _cyan(msg: str) -> str: + return f"\033[96m{msg}\033[0m" + +def _red(msg: str) -> str: + return f"\033[91m{msg}\033[0m" + + +def run(label: str, cmd: list[str], *, quiet: bool = True) -> None: + """Run a command; on failure print output and exit.""" + print(_cyan(f" {label}...")) + result = subprocess.run( + cmd, + stdout=subprocess.PIPE if quiet else None, + stderr=subprocess.STDOUT if quiet else None, + ) + if result.returncode != 0: + print(_red(f"❌ {label} failed (exit code {result.returncode}):")) + if result.stdout: + print(result.stdout.decode(errors="replace")) + sys.exit(result.returncode) + + +def pip_install( + label: str, + *args: str, + req: Path | None = None, + constrain: bool = True, +) -> None: + """Build and run a pip install command.""" + cmd = [sys.executable, "-m", "pip", "install"] + cmd.extend(args) + if constrain and CONSTRAINTS.is_file(): + cmd.extend(["-c", str(CONSTRAINTS)]) + if req is not None: + cmd.extend(["-r", str(req)]) + run(label, cmd) + + +def download_file(url: str, dest: Path) -> None: + """Download a file using urllib (no curl dependency).""" + urllib.request.urlretrieve(url, dest) + + +def patch_package_file(package_name: str, relative_path: str, url: str) -> None: + """Download a file from url and overwrite a file inside an installed package.""" + result = subprocess.run( + [sys.executable, "-m", "pip", "show", package_name], + capture_output=True, text=True, + ) + if result.returncode != 0: + print(_red(f" ⚠️ Could not find package {package_name}, skipping patch")) + return + + location = None + for line in result.stdout.splitlines(): + if line.lower().startswith("location:"): + location = line.split(":", 1)[1].strip() + break + + if not location: + print(_red(f" ⚠️ Could not determine location of {package_name}")) + return + + dest = Path(location) / relative_path + print(_cyan(f" Patching {dest.name} in {package_name}...")) + download_file(url, dest) + + +# ── Main install sequence ───────────────────────────────────────────── + +def install_python_stack() -> int: + print(_cyan("── Installing Python stack ──")) + + # 1. Upgrade pip + run("Upgrading pip", [sys.executable, "-m", "pip", "install", "--upgrade", "pip"]) + + # 2. Core packages: unsloth-zoo + unsloth + pip_install( + "Installing unsloth-zoo + unsloth", + "--no-cache-dir", + req=REQ_ROOT / "base.txt", + ) + + # 3. Extra dependencies + pip_install( + "Installing additional unsloth dependencies", + "--no-cache-dir", + req=REQ_ROOT / "extras.txt", + ) + + # 4. Overrides (torchao, transformers) — force-reinstall + pip_install( + "Installing torchao + transformers overrides", + "--force-reinstall", "--no-cache-dir", + req=REQ_ROOT / "overrides.txt", + ) + + # 5. Triton kernels (no-deps, from source) + pip_install( + "Installing triton kernels", + "--no-deps", "--no-cache-dir", + req=REQ_ROOT / "triton-kernels.txt", + constrain=False, + ) + + # 6. Patch: override llama_cpp.py with fix from unsloth-zoo main branch + patch_package_file( + "unsloth-zoo", + os.path.join("unsloth_zoo", "llama_cpp.py"), + "https://raw.githubusercontent.com/unslothai/unsloth-zoo/refs/heads/main/unsloth_zoo/llama_cpp.py", + ) + + # 7. Patch: override vision.py with fix from unsloth PR #4091 + patch_package_file( + "unsloth", + os.path.join("unsloth", "models", "vision.py"), + "https://raw.githubusercontent.com/unslothai/unsloth/80e0108a684c882965a02a8ed851e3473c1145ab/unsloth/models/vision.py", + ) + + # 8. Studio dependencies + pip_install( + "Installing studio dependencies", + "--no-cache-dir", + req=REQ_ROOT / "studio.txt", + ) + + # 9. Data-designer dependencies + pip_install( + "Installing data-designer dependencies", + "--no-cache-dir", + req=SINGLE_ENV / "data-designer-deps.txt", + ) + + # 10. Data-designer packages (no-deps to avoid conflicts) + pip_install( + "Installing data-designer", + "--no-cache-dir", "--no-deps", + req=SINGLE_ENV / "data-designer.txt", + ) + + # 11. Patch metadata for single-env compatibility + run( + "Patching single-env metadata", + [sys.executable, str(SINGLE_ENV / "patch_metadata.py")], + ) + + # 12. Final check + run("Running pip check", [sys.executable, "-m", "pip", "check"], quiet=False) + + print(_green("✅ Python dependencies installed")) + return 0 + + +if __name__ == "__main__": + sys.exit(install_python_stack()) diff --git a/setup.ps1 b/setup.ps1 new file mode 100644 index 0000000000..357c9b0d98 --- /dev/null +++ b/setup.ps1 @@ -0,0 +1,660 @@ +#Requires -Version 5.1 +<# +.SYNOPSIS + Full environment setup for Unsloth Studio on Windows (bundled version). +.DESCRIPTION + Always installs Node.js if needed. When running from pip install: + skips frontend build (already bundled). When running from git repo: + full setup including frontend build. + Requires an NVIDIA GPU -- CPU-only machines are not supported. +.NOTES + Usage: powershell -ExecutionPolicy Bypass -File setup.ps1 +#> + +$ErrorActionPreference = "Stop" +$ScriptDir = Split-Path -Parent $MyInvocation.MyCommand.Path +$PackageDir = Split-Path -Parent $ScriptDir + +# Detect if running from pip install (no frontend/ dir two levels up) +$FrontendDir = Join-Path $ScriptDir "..\..\frontend" +$IsPipInstall = -not (Test-Path $FrontendDir) + +# ───────────────────────────────────────────── +# Helper functions +# ───────────────────────────────────────────── + +# Reload ALL environment variables from registry. +# Picks up changes made by installers (winget, msi, etc.) including +# Path, CUDA_PATH, CUDA_PATH_V*, and any other vars they set. +function Refresh-Environment { + foreach ($level in @('Machine', 'User')) { + $vars = [System.Environment]::GetEnvironmentVariables($level) + foreach ($key in $vars.Keys) { + if ($key -eq 'Path') { continue } + Set-Item -Path "Env:$key" -Value $vars[$key] -ErrorAction SilentlyContinue + } + } + $machinePath = [System.Environment]::GetEnvironmentVariable('Path', 'Machine') + $userPath = [System.Environment]::GetEnvironmentVariable('Path', 'User') + $env:Path = "$machinePath;$userPath" +} + +# Find nvcc on PATH, CUDA_PATH, or standard toolkit dirs. +# Returns the path to nvcc.exe, or $null if not found. +function Find-Nvcc { + # 1. Check nvcc on PATH + $cmd = Get-Command nvcc -ErrorAction SilentlyContinue + if ($cmd) { return $cmd.Source } + + # 2. Check CUDA_PATH env var + $cudaRoot = [Environment]::GetEnvironmentVariable('CUDA_PATH', 'Process') + if (-not $cudaRoot) { $cudaRoot = [Environment]::GetEnvironmentVariable('CUDA_PATH', 'Machine') } + if (-not $cudaRoot) { $cudaRoot = [Environment]::GetEnvironmentVariable('CUDA_PATH', 'User') } + if ($cudaRoot -and (Test-Path (Join-Path $cudaRoot 'bin\nvcc.exe'))) { + return (Join-Path $cudaRoot 'bin\nvcc.exe') + } + + # 3. Scan standard toolkit directory + $toolkitBase = 'C:\Program Files\NVIDIA GPU Computing Toolkit\CUDA' + if (Test-Path $toolkitBase) { + $latest = Get-ChildItem -Directory $toolkitBase | Sort-Object Name | Select-Object -Last 1 + if ($latest -and (Test-Path (Join-Path $latest.FullName 'bin\nvcc.exe'))) { + return (Join-Path $latest.FullName 'bin\nvcc.exe') + } + } + + return $null +} + +# Detect CUDA Compute Capability via nvidia-smi. +# Returns e.g. "80" for A100 (8.0), "89" for RTX 4090 (8.9), etc. +# Returns $null if detection fails. +function Get-CudaComputeCapability { + $nvSmi = Get-Command nvidia-smi -ErrorAction SilentlyContinue + if (-not $nvSmi) { return $null } + + try { + $raw = & nvidia-smi --query-gpu=compute_cap --format=csv,noheader 2>$null + if ($LASTEXITCODE -ne 0 -or -not $raw) { return $null } + + # nvidia-smi may return multiple GPUs; take the first one + $cap = ($raw -split "`n")[0].Trim() + if ($cap -match '^(\d+)\.(\d+)$') { + $major = $Matches[1] + $minor = $Matches[2] + return "$major$minor" + } + } catch { } + + return $null +} + +# Detect driver's max CUDA version from nvidia-smi and return the highest +# compatible PyTorch CUDA index tag (e.g. "cu128"). +# PyTorch on Windows ships CPU-only by default from PyPI; CUDA wheels live at +# https://download.pytorch.org/whl/. The tag must not exceed the driver's +# capability: e.g. driver "CUDA Version: 12.9" → cu128 (not cu130). +function Get-PytorchCudaTag { + $nvSmi = Get-Command nvidia-smi -ErrorAction SilentlyContinue + if (-not $nvSmi) { return "cu124" } + + try { + # 2>&1 | Out-String merges stderr into stdout then converts to a single + # string. Plain 2>$null doesn't fully suppress stderr in PS 5.1 — + # ErrorRecord objects leak into $output and break the -match. + $output = & nvidia-smi 2>&1 | Out-String + if ($output -match 'CUDA Version:\s+(\d+)\.(\d+)') { + $major = [int]$Matches[1] + $minor = [int]$Matches[2] + # PyTorch 2.10 offers: cu124, cu126, cu128, cu130 + if ($major -ge 13) { return "cu130" } + if ($major -eq 12 -and $minor -ge 8) { return "cu128" } + if ($major -eq 12 -and $minor -ge 6) { return "cu126" } + return "cu124" + } + } catch { } + + return "cu124" +} + +# Find Visual Studio Build Tools for cmake -G flag. +# Strategy: (1) vswhere, (2) scan filesystem (handles broken vswhere registration). +# Returns @{ Generator = "Visual Studio 17 2022"; InstallPath = "C:\..."; Source = "..." } or $null. +function Find-VsBuildTools { + $map = @{ '2022' = '17'; '2019' = '16'; '2017' = '15' } + + # --- Try vswhere first (works when VS is properly registered) --- + $vsw = "${env:ProgramFiles(x86)}\Microsoft Visual Studio\Installer\vswhere.exe" + if (Test-Path $vsw) { + $info = & $vsw -latest -requires Microsoft.VisualStudio.Component.VC.Tools.x86.x64 -property catalog_productLineVersion 2>$null + $path = & $vsw -latest -requires Microsoft.VisualStudio.Component.VC.Tools.x86.x64 -property installationPath 2>$null + if ($info -and $path) { + $y = $info.Trim() + $n = $map[$y] + if ($n) { + return @{ Generator = "Visual Studio $n $y"; InstallPath = $path.Trim(); Source = 'vswhere' } + } + } + } + + # --- Scan filesystem (handles broken vswhere registration after winget cycles) --- + $roots = @($env:ProgramFiles, ${env:ProgramFiles(x86)}) + $editions = @('BuildTools', 'Community', 'Professional', 'Enterprise') + $years = @('2022', '2019', '2017') + + foreach ($y in $years) { + foreach ($r in $roots) { + foreach ($ed in $editions) { + $candidate = Join-Path $r "Microsoft Visual Studio\$y\$ed" + if (Test-Path $candidate) { + $vcDir = Join-Path $candidate "VC\Tools\MSVC" + if (Test-Path $vcDir) { + $cl = Get-ChildItem -Path $vcDir -Filter "cl.exe" -Recurse -ErrorAction SilentlyContinue | Select-Object -First 1 + if ($cl) { + $n = $map[$y] + if ($n) { + return @{ Generator = "Visual Studio $n $y"; InstallPath = $candidate; Source = "filesystem ($ed)"; ClExe = $cl.FullName } + } + } + } + } + } + } + } + + return $null +} + +# ───────────────────────────────────────────── +# Banner +# ───────────────────────────────────────────── +Write-Host "+==============================================+" -ForegroundColor Green +Write-Host "| Unsloth Studio Setup (Windows) |" -ForegroundColor Green +Write-Host "+==============================================+" -ForegroundColor Green + +# ========================================================================== +# PHASE 1: System-level prerequisites (winget installs, env vars) +# All heavy system tool installs happen here BEFORE touching Python. +# ========================================================================== + +# ============================================ +# 1a. GPU requirement check +# ============================================ +$HasNvidiaSmi = $null -ne (Get-Command nvidia-smi -ErrorAction SilentlyContinue) +if (-not $HasNvidiaSmi) { + Write-Host "" + Write-Host "[ERROR] Unsloth Studio requires an NVIDIA GPU." -ForegroundColor Red + Write-Host " CPU-only machines are not supported." -ForegroundColor Red + Write-Host "" + Write-Host " If you have an NVIDIA GPU, ensure the driver is installed:" -ForegroundColor Yellow + Write-Host " https://www.nvidia.com/Download/index.aspx" -ForegroundColor Yellow + exit 1 +} +Write-Host "[OK] NVIDIA GPU detected" -ForegroundColor Green + +# ============================================ +# 1b. Git (required by pip for git+https:// deps and by npm) +# ============================================ +$HasGit = $null -ne (Get-Command git -ErrorAction SilentlyContinue) +if (-not $HasGit) { + Write-Host "Git not found -- installing via winget..." -ForegroundColor Yellow + $HasWinget = $null -ne (Get-Command winget -ErrorAction SilentlyContinue) + if ($HasWinget) { + try { + winget install Git.Git --source winget --accept-package-agreements --accept-source-agreements 2>&1 | Out-Null + Refresh-Environment + $HasGit = $null -ne (Get-Command git -ErrorAction SilentlyContinue) + } catch { } + } + if (-not $HasGit) { + Write-Host "[ERROR] Git is required but could not be installed automatically." -ForegroundColor Red + Write-Host " Install Git from https://git-scm.com/download/win and re-run." -ForegroundColor Red + exit 1 + } + Write-Host "[OK] Git installed: $(git --version)" -ForegroundColor Green +} else { + Write-Host "[OK] Git found: $(git --version)" -ForegroundColor Green +} + +# ============================================ +# 1c. CMake (required for llama.cpp build) +# ============================================ +$HasCmake = $null -ne (Get-Command cmake -ErrorAction SilentlyContinue) +if (-not $HasCmake) { + Write-Host "CMake not found -- installing via winget..." -ForegroundColor Yellow + $HasWinget = $null -ne (Get-Command winget -ErrorAction SilentlyContinue) + if ($HasWinget) { + try { + winget install Kitware.CMake --source winget --accept-package-agreements --accept-source-agreements 2>&1 | Out-Null + Refresh-Environment + $HasCmake = $null -ne (Get-Command cmake -ErrorAction SilentlyContinue) + } catch { } + } + if ($HasCmake) { + Write-Host "[OK] CMake installed" -ForegroundColor Green + } else { + Write-Host "[ERROR] CMake is required but could not be installed." -ForegroundColor Red + Write-Host " Install CMake from https://cmake.org/download/ and re-run." -ForegroundColor Red + exit 1 + } +} else { + Write-Host "[OK] CMake found: $(cmake --version | Select-Object -First 1)" -ForegroundColor Green +} + +# ============================================ +# 1d. Visual Studio Build Tools (C++ compiler for llama.cpp) +# ============================================ +$CmakeGenerator = $null +$VsInstallPath = $null +$vsResult = Find-VsBuildTools + +if (-not $vsResult) { + Write-Host "Visual Studio Build Tools not found -- installing via winget..." -ForegroundColor Yellow + Write-Host " (This is a one-time install, may take several minutes)" -ForegroundColor Gray + $HasWinget = $null -ne (Get-Command winget -ErrorAction SilentlyContinue) + if ($HasWinget) { + $prevEAPTemp = $ErrorActionPreference + $ErrorActionPreference = "Continue" + winget install Microsoft.VisualStudio.2022.BuildTools --source winget --accept-package-agreements --accept-source-agreements --override "--add Microsoft.VisualStudio.Workload.VCTools --includeRecommended --passive --wait" + $ErrorActionPreference = $prevEAPTemp + # Re-scan after install (don't trust vswhere catalog) + $vsResult = Find-VsBuildTools + } +} + +if ($vsResult) { + $CmakeGenerator = $vsResult.Generator + $VsInstallPath = $vsResult.InstallPath + Write-Host "[OK] $CmakeGenerator detected via $($vsResult.Source)" -ForegroundColor Green + if ($vsResult.ClExe) { Write-Host " cl.exe: $($vsResult.ClExe)" -ForegroundColor Gray } +} else { + Write-Host "[ERROR] Visual Studio Build Tools could not be found or installed." -ForegroundColor Red + Write-Host " Manual install:" -ForegroundColor Red + Write-Host ' 1. winget install Microsoft.VisualStudio.2022.BuildTools --source winget' -ForegroundColor Yellow + Write-Host ' 2. Open Visual Studio Installer -> Modify -> check "Desktop development with C++"' -ForegroundColor Yellow + exit 1 +} + +# ============================================ +# 1e. CUDA Toolkit (nvcc for llama.cpp build + env vars) +# ============================================ +$NvccPath = Find-Nvcc + +if (-not $NvccPath) { + Write-Host "CUDA driver detected but toolkit (nvcc) not found -- installing via winget..." -ForegroundColor Yellow + $HasWinget = $null -ne (Get-Command winget -ErrorAction SilentlyContinue) + if ($HasWinget) { + Write-Host " Installing CUDA Toolkit via winget..." -ForegroundColor Cyan + winget install --id=Nvidia.CUDA -e --source winget --accept-package-agreements --accept-source-agreements + Refresh-Environment + $NvccPath = Find-Nvcc + if ($NvccPath) { + Write-Host " [OK] CUDA Toolkit installed (nvcc: $NvccPath)" -ForegroundColor Green + } + } +} + +if (-not $NvccPath) { + Write-Host "[ERROR] CUDA Toolkit (nvcc) is required but could not be found or installed." -ForegroundColor Red + Write-Host " Install CUDA Toolkit from https://developer.nvidia.com/cuda-downloads" -ForegroundColor Yellow + exit 1 +} + +# -- Set CUDA env vars so cmake AND MSBuild can find the toolkit -- +$CudaToolkitRoot = Split-Path (Split-Path $NvccPath -Parent) -Parent +# CUDA_PATH: used by cmake's find_package(CUDAToolkit) +[Environment]::SetEnvironmentVariable('CUDA_PATH', $CudaToolkitRoot, 'Process') +# CudaToolkitDir: the MSBuild property that CUDA .targets checks directly +# Trailing backslash required -- the .targets file appends subpaths to it +[Environment]::SetEnvironmentVariable('CudaToolkitDir', "$CudaToolkitRoot\", 'Process') +# Persist CUDA_PATH to User registry if not already set +$existingSys = [Environment]::GetEnvironmentVariable('CUDA_PATH', 'Machine') +$existingUsr = [Environment]::GetEnvironmentVariable('CUDA_PATH', 'User') +if (-not $existingSys -and -not $existingUsr) { + [Environment]::SetEnvironmentVariable('CUDA_PATH', $CudaToolkitRoot, 'User') + Write-Host " Persisted CUDA_PATH to user environment" -ForegroundColor Gray +} +# Ensure nvcc's bin dir is on PATH for this process +$nvccBinDir = Split-Path $NvccPath -Parent +if ($env:PATH -notlike "*$nvccBinDir*") { + [Environment]::SetEnvironmentVariable('PATH', "$nvccBinDir;$env:PATH", 'Process') +} +# Persist nvcc bin dir to User PATH so it works in new terminals +$userPath = [Environment]::GetEnvironmentVariable('Path', 'User') +if (-not $userPath -or $userPath -notlike "*$nvccBinDir*") { + if ($userPath) { + [Environment]::SetEnvironmentVariable('Path', "$nvccBinDir;$userPath", 'User') + } else { + [Environment]::SetEnvironmentVariable('Path', "$nvccBinDir", 'User') + } + Write-Host " Persisted CUDA bin dir to user PATH" -ForegroundColor Gray +} + +Write-Host "[OK] CUDA Toolkit: $NvccPath" -ForegroundColor Green +Write-Host " CUDA_PATH = $CudaToolkitRoot" -ForegroundColor Gray +Write-Host " CudaToolkitDir = $CudaToolkitRoot\" -ForegroundColor Gray + +# Detect compute capability (used later for llama.cpp cmake) +$CudaArch = Get-CudaComputeCapability +if ($CudaArch) { + Write-Host " Compute Capability = $($CudaArch.Insert($CudaArch.Length-1, '.')) (sm_$CudaArch)" -ForegroundColor Gray +} else { + Write-Host " [WARN] Could not detect compute capability -- cmake will use defaults" -ForegroundColor Yellow +} + +# ============================================ +# 1f. Node.js / npm (always -- needed regardless of install method) +# ============================================ +$NeedNode = $true +try { + $NodeVersion = (node -v 2>$null) + $NpmVersion = (npm -v 2>$null) + if ($NodeVersion -and $NpmVersion) { + $NodeMajor = [int]($NodeVersion -replace 'v','').Split('.')[0] + $NpmMajor = [int]$NpmVersion.Split('.')[0] + + if ($NodeMajor -ge 20 -and $NpmMajor -ge 11) { + Write-Host "[OK] Node $NodeVersion and npm $NpmVersion already meet requirements." -ForegroundColor Green + $NeedNode = $false + } else { + Write-Host "[WARN] Node $NodeVersion / npm $NpmVersion too old." -ForegroundColor Yellow + } + } +} catch { + Write-Host "[WARN] Node/npm not found." -ForegroundColor Yellow +} + +if ($NeedNode) { + Write-Host "Installing Node.js via winget..." -ForegroundColor Cyan + try { + winget install OpenJS.NodeJS.LTS --source winget --accept-package-agreements --accept-source-agreements + Refresh-Environment + } catch { + Write-Host "[ERROR] Could not install Node.js automatically." -ForegroundColor Red + Write-Host "Please install Node.js >= 20 from https://nodejs.org/" -ForegroundColor Red + exit 1 + } +} + +Write-Host "[OK] Node $(node -v) | npm $(npm -v)" -ForegroundColor Green + +Write-Host "" +Write-Host "--- System prerequisites ready ---" -ForegroundColor Green +Write-Host "" + +# ========================================================================== +# PHASE 2: Frontend build (skip if pip-installed -- already bundled) +# ========================================================================== +if ($IsPipInstall) { + Write-Host "[OK] Running from pip install - frontend already bundled, skipping build" -ForegroundColor Green +} else { + $RepoRoot = (Resolve-Path (Join-Path $ScriptDir "..\..")).Path + + Write-Host "" + Write-Host "Building frontend..." -ForegroundColor Cyan + Push-Location (Join-Path $RepoRoot "frontend") + npm install 2>&1 | Out-Null + npm run build 2>&1 | Out-Null + Pop-Location + + $PackageBuildDir = Join-Path $PackageDir "studio\frontend\build" + if (Test-Path $PackageBuildDir) { Remove-Item -Recurse -Force $PackageBuildDir } + Copy-Item -Recurse (Join-Path $RepoRoot "frontend\build") $PackageBuildDir + + Write-Host "[OK] Frontend built" -ForegroundColor Green +} + +# ========================================================================== +# PHASE 3: Python environment + dependencies +# ========================================================================== +Write-Host "" +Write-Host "Setting up Python environment..." -ForegroundColor Cyan + +# Find Python +$PythonCmd = $null +foreach ($candidate in @("python3.12", "python3.11", "python3.10", "python3.9", "python3", "python")) { + try { + $ver = & $candidate --version 2>&1 + if ($ver -match 'Python 3\.(\d+)') { + $minor = [int]$Matches[1] + if ($minor -le 12) { + $PythonCmd = $candidate + break + } + } + } catch { } +} + +if (-not $PythonCmd) { + Write-Host "[ERROR] No Python <= 3.12 found." -ForegroundColor Red + exit 1 +} + +Write-Host "[OK] Using $PythonCmd ($(& $PythonCmd --version 2>&1))" -ForegroundColor Green + +# Always create a .venv for isolation -- even for pip installs. +# Created in the current working directory (where user ran the command). +$VenvDir = Join-Path (Get-Location) ".venv" +if (-not (Test-Path $VenvDir)) { + Write-Host " Creating virtual environment at $VenvDir..." -ForegroundColor Cyan + & $PythonCmd -m venv $VenvDir +} else { + Write-Host " Reusing existing virtual environment at $VenvDir" -ForegroundColor Green +} + +# pip and python write to stderr even on success (progress bars, warnings). +# With $ErrorActionPreference = "Stop" (set at top of script), PS 5.1 +# converts stderr lines into terminating ErrorRecords, breaking output. +# Lower to "Continue" for the pip/python section. +$prevEAP = $ErrorActionPreference +$ErrorActionPreference = "Continue" + +$ActivateScript = Join-Path $VenvDir "Scripts\Activate.ps1" +. $ActivateScript +pip install --upgrade pip 2>&1 | Out-Null + +# if (-not $IsPipInstall) { +# # Running from repo: copy requirements and do editable install +# $RepoRoot = (Resolve-Path (Join-Path $ScriptDir "..\..")).Path +# $ReqsSrc = Join-Path $RepoRoot "backend\requirements" +# $ReqsDst = Join-Path $PackageDir "requirements" +# if (-not (Test-Path $ReqsDst)) { New-Item -ItemType Directory -Path $ReqsDst | Out-Null } +# Copy-Item (Join-Path $ReqsSrc "*.txt") $ReqsDst -Force + +# Write-Host " Installing CLI entry point..." -ForegroundColor Cyan +# pip install -e $RepoRoot 2>&1 | Out-Null +# } else { +# # Running from pip install: the package is in system Python but not in +# # the fresh .venv. Install it so run_install() can find its modules +# # and bundled requirements files. +# Write-Host " Installing package into venv..." -ForegroundColor Cyan +# pip install unsloth-roland-test 2>&1 | Out-Null +# } + +# Pre-install PyTorch with CUDA support. +# On Windows, the default PyPI torch wheel is CPU-only. +# We need PyTorch's CUDA index to get GPU-enabled wheels. +# PyTorch bundles its own CUDA runtime, so this works regardless +# of whether the CUDA Toolkit is installed yet. +# The CUDA tag is chosen based on the driver's max supported CUDA version. +$CuTag = Get-PytorchCudaTag +Write-Host " Installing PyTorch with CUDA support ($CuTag)..." -ForegroundColor Cyan +pip install torch torchvision torchaudio --index-url "https://download.pytorch.org/whl/$CuTag" + +# Ordered heavy dependency installation — shared cross-platform script +Write-Host " Running ordered dependency installation..." -ForegroundColor Cyan +python "$PSScriptRoot\install_python_stack.py" +# Restore ErrorActionPreference after pip/python work +$ErrorActionPreference = $prevEAP + +# ========================================================================== +# PHASE 4: Build llama.cpp with CUDA for GGUF inference + export +# ========================================================================== +# Builds at ~/.unsloth/llama.cpp/ (persistent across pip upgrades). +# We build: +# - llama-server: for GGUF model inference +# - llama-quantize: for GGUF export quantization +# Prerequisites (git, cmake, VS Build Tools, CUDA Toolkit) already installed in Phase 1. +$LlamaCppDir = Join-Path $env:USERPROFILE ".unsloth\llama.cpp" +$BuildDir = Join-Path $LlamaCppDir "build" +$LlamaServerBin = Join-Path $BuildDir "bin\Release\llama-server.exe" + +if (Test-Path $LlamaServerBin) { + Write-Host "" + Write-Host "[OK] llama-server already exists at $LlamaServerBin" -ForegroundColor Green +} else { + Write-Host "" + Write-Host "Building llama.cpp with CUDA support..." -ForegroundColor Cyan + Write-Host " This typically takes 5-10 minutes on first build." -ForegroundColor Gray + Write-Host "" + + # Start total build timer + $totalSw = [System.Diagnostics.Stopwatch]::StartNew() + + # Native commands (git, cmake) write to stderr even on success. + # With $ErrorActionPreference = "Stop" (set at top of script), PS 5.1 + # converts stderr lines into terminating ErrorRecords, breaking output. + # Lower to "Continue" for the build section. + $prevEAP = $ErrorActionPreference + $ErrorActionPreference = "Continue" + + $BuildOk = $true + $FailedStep = "" + + # -- Step A: Clone or pull llama.cpp -- + $UnslothDir = Join-Path $env:USERPROFILE ".unsloth" + if (-not (Test-Path $UnslothDir)) { New-Item -ItemType Directory -Path $UnslothDir -Force | Out-Null } + + if (Test-Path (Join-Path $LlamaCppDir ".git")) { + Write-Host " llama.cpp repo already cloned, pulling latest..." -ForegroundColor Gray + git -C $LlamaCppDir pull + if ($LASTEXITCODE -ne 0) { + Write-Host " [WARN] git pull failed -- using existing source" -ForegroundColor Yellow + } + } else { + Write-Host " Cloning llama.cpp..." -ForegroundColor Gray + if (Test-Path $LlamaCppDir) { Remove-Item -Recurse -Force $LlamaCppDir } + git clone --depth 1 https://github.com/ggml-org/llama.cpp.git $LlamaCppDir + if ($LASTEXITCODE -ne 0) { + $BuildOk = $false + $FailedStep = "git clone" + } + } + + # -- Step B: cmake configure (CUDA + Unsloth flags) -- + if ($BuildOk) { + Write-Host "" + Write-Host "--- cmake configure ---" -ForegroundColor Cyan + + $CmakeArgs = @( + '-S', $LlamaCppDir, + '-B', $BuildDir, + '-G', $CmakeGenerator, + '-Wno-dev' + ) + # Tell cmake exactly where VS is (bypasses registry lookup) + if ($VsInstallPath) { + $CmakeArgs += "-DCMAKE_GENERATOR_INSTANCE=$VsInstallPath" + } + # Common flags + $CmakeArgs += '-DBUILD_SHARED_LIBS=OFF' + $CmakeArgs += '-DLLAMA_CURL=OFF' + $CmakeArgs += '-DCMAKE_POLICY_DEFAULT_CMP0194=NEW' + $CmakeArgs += '-DCMAKE_EXE_LINKER_FLAGS=/NODEFAULTLIB:LIBCMT' + # CUDA flags (Unsloth-aligned) + $CmakeArgs += '-DGGML_CUDA=ON' + $CmakeArgs += "-DCUDAToolkit_ROOT=$CudaToolkitRoot" + $CmakeArgs += "-DCMAKE_CUDA_COMPILER=$NvccPath" + $CmakeArgs += '-DGGML_CUDA_FA_ALL_QUANTS=ON' + $CmakeArgs += '-DGGML_CUDA_F16=OFF' + $CmakeArgs += '-DGGML_CUDA_GRAPHS=OFF' + $CmakeArgs += '-DGGML_CUDA_FORCE_CUBLAS=OFF' + $CmakeArgs += '-DGGML_CUDA_PEER_MAX_BATCH_SIZE=8192' + if ($CudaArch) { + $CmakeArgs += "-DCMAKE_CUDA_ARCHITECTURES=$CudaArch" + } + + Write-Host " cmake args:" -ForegroundColor Gray + foreach ($arg in $CmakeArgs) { + Write-Host " $arg" -ForegroundColor Gray + } + Write-Host "" + + cmake @CmakeArgs + if ($LASTEXITCODE -ne 0) { + $BuildOk = $false + $FailedStep = "cmake configure" + } + } + + # -- Step C: Build llama-server -- + $NumCpu = [Environment]::ProcessorCount + if ($NumCpu -lt 1) { $NumCpu = 4 } + + if ($BuildOk) { + Write-Host "" + Write-Host "--- cmake build (llama-server) ---" -ForegroundColor Cyan + Write-Host " Parallel jobs: $NumCpu" -ForegroundColor Gray + Write-Host "" + + cmake --build $BuildDir --config Release --target llama-server -j $NumCpu + if ($LASTEXITCODE -ne 0) { + $BuildOk = $false + $FailedStep = "cmake build (llama-server)" + } + } + + # -- Step D: Build llama-quantize (optional, best-effort) -- + if ($BuildOk) { + Write-Host "" + Write-Host "--- cmake build (llama-quantize) ---" -ForegroundColor Cyan + cmake --build $BuildDir --config Release --target llama-quantize -j $NumCpu + if ($LASTEXITCODE -ne 0) { + Write-Host " [WARN] llama-quantize build failed (GGUF export may be unavailable)" -ForegroundColor Yellow + } + } + + # Restore ErrorActionPreference + $ErrorActionPreference = $prevEAP + + # Stop timer + $totalSw.Stop() + $totalMin = [math]::Floor($totalSw.Elapsed.TotalMinutes) + $totalSec = [math]::Round($totalSw.Elapsed.TotalSeconds % 60, 1) + + # -- Summary -- + Write-Host "" + if ($BuildOk -and (Test-Path $LlamaServerBin)) { + Write-Host "[OK] llama-server built at $LlamaServerBin" -ForegroundColor Green + $QuantizeBin = Join-Path $BuildDir "bin\Release\llama-quantize.exe" + if (Test-Path $QuantizeBin) { + Write-Host "[OK] llama-quantize available for GGUF export" -ForegroundColor Green + } + Write-Host " Build time: ${totalMin}m ${totalSec}s" -ForegroundColor Cyan + } else { + # Check alternate paths (some cmake generators don't use Release subdir) + $altBin = Join-Path $BuildDir "bin\llama-server.exe" + if ($BuildOk -and (Test-Path $altBin)) { + Write-Host "[OK] llama-server built at $altBin" -ForegroundColor Green + Write-Host " Build time: ${totalMin}m ${totalSec}s" -ForegroundColor Cyan + } else { + Write-Host "[FAILED] llama.cpp build failed at step: $FailedStep (${totalMin}m ${totalSec}s)" -ForegroundColor Red + Write-Host " To retry: delete $LlamaCppDir and re-run setup." -ForegroundColor Yellow + exit 1 + } + } +} + +# ============================================ +# Done +# ============================================ +Write-Host "" +Write-Host "+==============================================+" -ForegroundColor Green +Write-Host "| Setup Complete! |" -ForegroundColor Green +Write-Host "| |" -ForegroundColor Green +Write-Host "| Activate venv: |" -ForegroundColor Green +Write-Host "| cmd: .venv\Scripts\activate.bat |" -ForegroundColor Green +Write-Host "| PS: .\.venv\Scripts\Activate.ps1 |" -ForegroundColor Green +Write-Host "| |" -ForegroundColor Green +Write-Host "| Then run: unsloth-roland-test studio |" -ForegroundColor Green +Write-Host "+==============================================+" -ForegroundColor Green \ No newline at end of file diff --git a/setup.sh b/setup.sh index 8314ef6d75..409df0b798 100755 --- a/setup.sh +++ b/setup.sh @@ -170,30 +170,7 @@ SINGLE_ENV_DATA_DESIGNER_DEPS="$REQ_ROOT/single-env/data-designer-deps.txt" SINGLE_ENV_PATCH="$REQ_ROOT/single-env/patch_metadata.py" install_python_stack() { - run_quiet "pip upgrade" pip install --upgrade pip - echo " Installing unsloth-zoo + unsloth..." - run_quiet "pip install unsloth" pip install --no-cache-dir -c "$SINGLE_ENV_CONSTRAINTS" -r "$REQ_ROOT/base.txt" - echo " Installing additional unsloth dependencies..." - run_quiet "pip install extras" pip install --no-cache-dir -c "$SINGLE_ENV_CONSTRAINTS" -r "$REQ_ROOT/extras.txt" - run_quiet "pip install torchao+transformers" pip install --force-reinstall --no-cache-dir -c "$SINGLE_ENV_CONSTRAINTS" -r "$REQ_ROOT/overrides.txt" - run_quiet "pip install triton_kernels" pip install --no-deps --no-cache-dir -r "$REQ_ROOT/triton-kernels.txt" - # Patch: override llama_cpp.py with fix from unsloth-zoo branch - LLAMA_CPP_DST="$(pip show unsloth-zoo | grep -i '^Location:' | awk '{print $2}')/unsloth_zoo/llama_cpp.py" - curl -sSL "https://raw.githubusercontent.com/unslothai/unsloth-zoo/refs/heads/main/unsloth_zoo/llama_cpp.py" \ - -o "$LLAMA_CPP_DST" - # Patch: override vision.py with fix from unsloth PR: https://github.com/unslothai/unsloth/pull/4091 until next pypi release - VISION_DST="$(pip show unsloth | grep -i '^Location:' | awk '{print $2}')/unsloth/models/vision.py" - curl -sSL "https://raw.githubusercontent.com/unslothai/unsloth/80e0108a684c882965a02a8ed851e3473c1145ab/unsloth/models/vision.py" \ - -o "$VISION_DST" - echo " Installing studio dependencies..." - run_quiet "pip install studio" pip install --no-cache-dir -c "$SINGLE_ENV_CONSTRAINTS" -r "$REQ_ROOT/studio.txt" - echo " Installing data-designer dependencies..." - run_quiet "pip install data-designer deps" pip install --no-cache-dir -c "$SINGLE_ENV_CONSTRAINTS" -r "$SINGLE_ENV_DATA_DESIGNER_DEPS" - echo " Installing data-designer..." - run_quiet "pip install data-designer" pip install --no-cache-dir --no-deps -c "$SINGLE_ENV_CONSTRAINTS" -r "$SINGLE_ENV_DATA_DESIGNER" - run_quiet "patch single-env metadata" python "$SINGLE_ENV_PATCH" - run_quiet "pip check" pip check - echo "✅ Python dependencies installed" + python "$SCRIPT_DIR/install_python_stack.py" } if [ "$IS_COLAB" = true ]; then From 9dc7af08a0e69148c2805f506c2065a8c1438876 Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Fri, 27 Feb 2026 20:31:57 +0000 Subject: [PATCH 05/66] add setup.bat --- install_python_stack.py | 72 +++++++++++++++++++++++++++++++++++++---- setup.bat | 2 ++ setup.ps1 | 55 ++++++++++++++++++++++++++++--- 3 files changed, 117 insertions(+), 12 deletions(-) create mode 100644 setup.bat diff --git a/install_python_stack.py b/install_python_stack.py index e31e96232e..b9526af71b 100644 --- a/install_python_stack.py +++ b/install_python_stack.py @@ -11,25 +11,53 @@ from __future__ import annotations import os import subprocess import sys +import tempfile import urllib.request from pathlib import Path +IS_WINDOWS = sys.platform == "win32" + # ── Paths ────────────────────────────────────────────────────────────── SCRIPT_DIR = Path(__file__).resolve().parent REQ_ROOT = SCRIPT_DIR / "studio" / "backend" / "requirements" SINGLE_ENV = REQ_ROOT / "single-env" CONSTRAINTS = SINGLE_ENV / "constraints.txt" -# ── Helpers ──────────────────────────────────────────────────────────── +# ── Color support ────────────────────────────────────────────────────── + +def _enable_colors() -> bool: + """Try to enable ANSI color support. Returns True if available.""" + if not hasattr(sys.stdout, "fileno"): + return False + try: + if not os.isatty(sys.stdout.fileno()): + return False + except Exception: + return False + if IS_WINDOWS: + try: + import ctypes + kernel32 = ctypes.windll.kernel32 + # Enable ENABLE_VIRTUAL_TERMINAL_PROCESSING (0x0004) on stdout + handle = kernel32.GetStdHandle(-11) # STD_OUTPUT_HANDLE + mode = ctypes.c_ulong() + kernel32.GetConsoleMode(handle, ctypes.byref(mode)) + kernel32.SetConsoleMode(handle, mode.value | 0x0004) + return True + except Exception: + return False + return True # Unix terminals support ANSI by default + +_HAS_COLOR = _enable_colors() def _green(msg: str) -> str: - return f"\033[92m{msg}\033[0m" + return f"\033[92m{msg}\033[0m" if _HAS_COLOR else msg def _cyan(msg: str) -> str: - return f"\033[96m{msg}\033[0m" + return f"\033[96m{msg}\033[0m" if _HAS_COLOR else msg def _red(msg: str) -> str: - return f"\033[91m{msg}\033[0m" + return f"\033[91m{msg}\033[0m" if _HAS_COLOR else msg def run(label: str, cmd: list[str], *, quiet: bool = True) -> None: @@ -47,6 +75,25 @@ def run(label: str, cmd: list[str], *, quiet: bool = True) -> None: sys.exit(result.returncode) +# Packages to skip on Windows (require special build steps) +WINDOWS_SKIP_PACKAGES = {"open_spiel"} + + +def _filter_requirements(req: Path, skip: set[str]) -> Path: + """Return a temp copy of a requirements file with certain packages removed.""" + lines = req.read_text(encoding="utf-8").splitlines(keepends=True) + filtered = [ + line for line in lines + if not any(line.strip().lower().startswith(pkg) for pkg in skip) + ] + tmp = tempfile.NamedTemporaryFile( + mode="w", suffix=".txt", delete=False, encoding="utf-8", + ) + tmp.writelines(filtered) + tmp.close() + return Path(tmp.name) + + def pip_install( label: str, *args: str, @@ -58,9 +105,20 @@ def pip_install( cmd.extend(args) if constrain and CONSTRAINTS.is_file(): cmd.extend(["-c", str(CONSTRAINTS)]) - if req is not None: - cmd.extend(["-r", str(req)]) - run(label, cmd) + actual_req = req + if req is not None and IS_WINDOWS and WINDOWS_SKIP_PACKAGES: + actual_req = _filter_requirements(req, WINDOWS_SKIP_PACKAGES) + if actual_req is not None: + cmd.extend(["-r", str(actual_req)]) + try: + run(label, cmd) + finally: + # Clean up temp file if we created one + if actual_req is not None and actual_req != req: + actual_req.unlink(missing_ok=True) + if req is not None and actual_req != req: + skipped = WINDOWS_SKIP_PACKAGES + print(_cyan(f" (Skipped on Windows: {', '.join(skipped)})")) def download_file(url: str, dest: Path) -> None: diff --git a/setup.bat b/setup.bat new file mode 100644 index 0000000000..ef16abd263 --- /dev/null +++ b/setup.bat @@ -0,0 +1,2 @@ +@echo off +powershell -ExecutionPolicy Bypass -File "%~dp0setup.ps1" %* diff --git a/setup.ps1 b/setup.ps1 index 357c9b0d98..9be3f2efb0 100644 --- a/setup.ps1 +++ b/setup.ps1 @@ -479,7 +479,7 @@ pip install --upgrade pip 2>&1 | Out-Null # The CUDA tag is chosen based on the driver's max supported CUDA version. $CuTag = Get-PytorchCudaTag Write-Host " Installing PyTorch with CUDA support ($CuTag)..." -ForegroundColor Cyan -pip install torch torchvision torchaudio --index-url "https://download.pytorch.org/whl/$CuTag" +pip install torch torchvision torchaudio --index-url "https://download.pytorch.org/whl/$CuTag" 2>&1 | Out-Null # Ordered heavy dependency installation — shared cross-platform script Write-Host " Running ordered dependency installation..." -ForegroundColor Cyan @@ -645,6 +645,45 @@ if (Test-Path $LlamaServerBin) { } } +# ============================================ +# Add shell aliases (PowerShell profile + cmd batch files) +# ============================================ +Write-Host "" +$RepoDir = $PSScriptRoot +$VenvPython = Join-Path $RepoDir ".venv\Scripts\python.exe" +$CliScript = Join-Path $RepoDir "cli.py" +$FrontendDist = Join-Path $RepoDir "studio\frontend\dist" +$AliasAdded = $false + +# --- PowerShell profile: add functions --- +$ProfileDir = Split-Path $PROFILE -Parent +if (-not (Test-Path $ProfileDir)) { New-Item -ItemType Directory -Path $ProfileDir -Force | Out-Null } +if (-not (Test-Path $PROFILE)) { New-Item -ItemType File -Path $PROFILE -Force | Out-Null } + +if (-not (Select-String -Path $PROFILE -Pattern "unsloth-studio" -Quiet -ErrorAction SilentlyContinue)) { + $block = @" + +# Unsloth Studio launcher +function unsloth-studio { & "$VenvPython" "$CliScript" studio -f "$FrontendDist" @args } +function unsloth-ui { & "$VenvPython" "$CliScript" studio -f "$FrontendDist" @args } +"@ + Add-Content -Path $PROFILE -Value $block + Write-Host "[OK] Aliases 'unsloth-studio' and 'unsloth-ui' added to $PROFILE" -ForegroundColor Green + $AliasAdded = $true +} else { + Write-Host "[OK] Aliases 'unsloth-studio' and 'unsloth-ui' already exist in $PROFILE" -ForegroundColor Green +} + +# --- cmd.exe: create batch files on PATH so they work from regular terminal --- +$BatDir = Join-Path $RepoDir ".venv\Scripts" +foreach ($name in @("unsloth-studio", "unsloth-ui")) { + $batPath = Join-Path $BatDir "$name.bat" + if (-not (Test-Path $batPath)) { + Set-Content -Path $batPath -Value "@echo off`r`n`"$VenvPython`" `"$CliScript`" studio -f `"$FrontendDist`" %*" + } +} +Write-Host "[OK] Batch launchers created in $BatDir (works from cmd.exe when venv is on PATH)" -ForegroundColor Green + # ============================================ # Done # ============================================ @@ -652,9 +691,15 @@ Write-Host "" Write-Host "+==============================================+" -ForegroundColor Green Write-Host "| Setup Complete! |" -ForegroundColor Green Write-Host "| |" -ForegroundColor Green -Write-Host "| Activate venv: |" -ForegroundColor Green -Write-Host "| cmd: .venv\Scripts\activate.bat |" -ForegroundColor Green -Write-Host "| PS: .\.venv\Scripts\Activate.ps1 |" -ForegroundColor Green +if ($AliasAdded) { + Write-Host "| PowerShell: run '. `$PROFILE' |" -ForegroundColor Green + Write-Host "| or open a new terminal, then: |" -ForegroundColor Green +} else { + Write-Host "| Launch with: |" -ForegroundColor Green +} Write-Host "| |" -ForegroundColor Green -Write-Host "| Then run: unsloth-roland-test studio |" -ForegroundColor Green +Write-Host "| unsloth-studio -H 0.0.0.0 -p 8000 |" -ForegroundColor Green +Write-Host "| |" -ForegroundColor Green +Write-Host "| cmd.exe: .venv\Scripts\activate.bat |" -ForegroundColor Green +Write-Host "| unsloth-studio -H 0.0.0.0 -p 8000 |" -ForegroundColor Green Write-Host "+==============================================+" -ForegroundColor Green \ No newline at end of file From ccfc00944f5166dc73d35e89d30260fdd9c5559e Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Fri, 27 Feb 2026 20:55:39 +0000 Subject: [PATCH 06/66] Fix Windows frontend build, add setup.bat, ANSI colors, aliases --- install_python_stack.py | 4 +--- setup.ps1 | 15 ++++----------- 2 files changed, 5 insertions(+), 14 deletions(-) diff --git a/install_python_stack.py b/install_python_stack.py index b9526af71b..7d3ab66b20 100644 --- a/install_python_stack.py +++ b/install_python_stack.py @@ -116,9 +116,7 @@ def pip_install( # Clean up temp file if we created one if actual_req is not None and actual_req != req: actual_req.unlink(missing_ok=True) - if req is not None and actual_req != req: - skipped = WINDOWS_SKIP_PACKAGES - print(_cyan(f" (Skipped on Windows: {', '.join(skipped)})")) + def download_file(url: str, dest: Path) -> None: diff --git a/setup.ps1 b/setup.ps1 index 9be3f2efb0..c82d03615c 100644 --- a/setup.ps1 +++ b/setup.ps1 @@ -15,8 +15,8 @@ $ErrorActionPreference = "Stop" $ScriptDir = Split-Path -Parent $MyInvocation.MyCommand.Path $PackageDir = Split-Path -Parent $ScriptDir -# Detect if running from pip install (no frontend/ dir two levels up) -$FrontendDir = Join-Path $ScriptDir "..\..\frontend" +# Detect if running from pip install (no studio/frontend/ dir in repo) +$FrontendDir = Join-Path $ScriptDir "studio\frontend" $IsPipInstall = -not (Test-Path $FrontendDir) # ───────────────────────────────────────────── @@ -388,20 +388,13 @@ Write-Host "" if ($IsPipInstall) { Write-Host "[OK] Running from pip install - frontend already bundled, skipping build" -ForegroundColor Green } else { - $RepoRoot = (Resolve-Path (Join-Path $ScriptDir "..\..")).Path - Write-Host "" Write-Host "Building frontend..." -ForegroundColor Cyan - Push-Location (Join-Path $RepoRoot "frontend") + Push-Location $FrontendDir npm install 2>&1 | Out-Null npm run build 2>&1 | Out-Null Pop-Location - - $PackageBuildDir = Join-Path $PackageDir "studio\frontend\build" - if (Test-Path $PackageBuildDir) { Remove-Item -Recurse -Force $PackageBuildDir } - Copy-Item -Recurse (Join-Path $RepoRoot "frontend\build") $PackageBuildDir - - Write-Host "[OK] Frontend built" -ForegroundColor Green + Write-Host "[OK] Frontend built to studio/frontend/dist" -ForegroundColor Green } # ========================================================================== From 721088047c5a8167569c14aa6f39bb0964be9e48 Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Fri, 27 Feb 2026 20:59:10 +0000 Subject: [PATCH 07/66] Fix npm stderr crash on Windows ErrorActionPreference --- setup.ps1 | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/setup.ps1 b/setup.ps1 index c82d03615c..239238b97d 100644 --- a/setup.ps1 +++ b/setup.ps1 @@ -390,10 +390,15 @@ if ($IsPipInstall) { } else { Write-Host "" Write-Host "Building frontend..." -ForegroundColor Cyan + # npm writes warnings to stderr; lower ErrorActionPreference so PS doesn't + # treat them as terminating errors (same pattern as the pip section below). + $prevEAP_npm = $ErrorActionPreference + $ErrorActionPreference = "Continue" Push-Location $FrontendDir npm install 2>&1 | Out-Null npm run build 2>&1 | Out-Null Pop-Location + $ErrorActionPreference = $prevEAP_npm Write-Host "[OK] Frontend built to studio/frontend/dist" -ForegroundColor Green } From e6e97ad4c1f2b603813f69841fe6ce99d68b76b8 Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Fri, 27 Feb 2026 21:15:30 +0000 Subject: [PATCH 08/66] Enforce Node LTS (v20-v22), add npm error checking, clean node_modules --- setup.ps1 | 25 ++++++++++++++++++++++--- 1 file changed, 22 insertions(+), 3 deletions(-) diff --git a/setup.ps1 b/setup.ps1 index 239238b97d..6827b0730d 100644 --- a/setup.ps1 +++ b/setup.ps1 @@ -345,6 +345,8 @@ if ($CudaArch) { # ============================================ # 1f. Node.js / npm (always -- needed regardless of install method) # ============================================ +# setup.sh installs Node LTS (v22) via nvm. We enforce the same range here: +# Node >= 20 AND <= 22 (LTS only), npm >= 11. $NeedNode = $true try { $NodeVersion = (node -v 2>$null) @@ -353,9 +355,11 @@ try { $NodeMajor = [int]($NodeVersion -replace 'v','').Split('.')[0] $NpmMajor = [int]$NpmVersion.Split('.')[0] - if ($NodeMajor -ge 20 -and $NpmMajor -ge 11) { + if ($NodeMajor -ge 20 -and $NodeMajor -le 22 -and $NpmMajor -ge 11) { Write-Host "[OK] Node $NodeVersion and npm $NpmVersion already meet requirements." -ForegroundColor Green $NeedNode = $false + } elseif ($NodeMajor -gt 22) { + Write-Host "[WARN] Node $NodeVersion is too new (non-LTS). Installing Node LTS..." -ForegroundColor Yellow } else { Write-Host "[WARN] Node $NodeVersion / npm $NpmVersion too old." -ForegroundColor Yellow } @@ -365,13 +369,13 @@ try { } if ($NeedNode) { - Write-Host "Installing Node.js via winget..." -ForegroundColor Cyan + Write-Host "Installing Node.js LTS via winget..." -ForegroundColor Cyan try { winget install OpenJS.NodeJS.LTS --source winget --accept-package-agreements --accept-source-agreements Refresh-Environment } catch { Write-Host "[ERROR] Could not install Node.js automatically." -ForegroundColor Red - Write-Host "Please install Node.js >= 20 from https://nodejs.org/" -ForegroundColor Red + Write-Host "Please install Node.js LTS (v22) from https://nodejs.org/" -ForegroundColor Red exit 1 } } @@ -395,8 +399,23 @@ if ($IsPipInstall) { $prevEAP_npm = $ErrorActionPreference $ErrorActionPreference = "Continue" Push-Location $FrontendDir + # Remove stale node_modules to avoid version conflicts + if (Test-Path "node_modules") { Remove-Item -Recurse -Force "node_modules" } npm install 2>&1 | Out-Null + if ($LASTEXITCODE -ne 0) { + Pop-Location + $ErrorActionPreference = $prevEAP_npm + Write-Host "[ERROR] npm install failed (exit code $LASTEXITCODE)" -ForegroundColor Red + Write-Host " Try running 'npm install' manually in studio/frontend/ to see errors" -ForegroundColor Yellow + exit 1 + } npm run build 2>&1 | Out-Null + if ($LASTEXITCODE -ne 0) { + Pop-Location + $ErrorActionPreference = $prevEAP_npm + Write-Host "[ERROR] npm run build failed (exit code $LASTEXITCODE)" -ForegroundColor Red + exit 1 + } Pop-Location $ErrorActionPreference = $prevEAP_npm Write-Host "[OK] Frontend built to studio/frontend/dist" -ForegroundColor Green From c7fed1ba83bb4766b19d21a2c9e0bec4ff22b30f Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Fri, 27 Feb 2026 21:22:38 +0000 Subject: [PATCH 09/66] Fix npm Invalid Version: delete package-lock.json, relax Node constraint --- setup.ps1 | 11 +++++------ 1 file changed, 5 insertions(+), 6 deletions(-) diff --git a/setup.ps1 b/setup.ps1 index 6827b0730d..cbf636983f 100644 --- a/setup.ps1 +++ b/setup.ps1 @@ -346,7 +346,7 @@ if ($CudaArch) { # 1f. Node.js / npm (always -- needed regardless of install method) # ============================================ # setup.sh installs Node LTS (v22) via nvm. We enforce the same range here: -# Node >= 20 AND <= 22 (LTS only), npm >= 11. +# Node >= 20, npm >= 11. $NeedNode = $true try { $NodeVersion = (node -v 2>$null) @@ -355,11 +355,9 @@ try { $NodeMajor = [int]($NodeVersion -replace 'v','').Split('.')[0] $NpmMajor = [int]$NpmVersion.Split('.')[0] - if ($NodeMajor -ge 20 -and $NodeMajor -le 22 -and $NpmMajor -ge 11) { + if ($NodeMajor -ge 20 -and $NpmMajor -ge 11) { Write-Host "[OK] Node $NodeVersion and npm $NpmVersion already meet requirements." -ForegroundColor Green $NeedNode = $false - } elseif ($NodeMajor -gt 22) { - Write-Host "[WARN] Node $NodeVersion is too new (non-LTS). Installing Node LTS..." -ForegroundColor Yellow } else { Write-Host "[WARN] Node $NodeVersion / npm $NpmVersion too old." -ForegroundColor Yellow } @@ -375,7 +373,7 @@ if ($NeedNode) { Refresh-Environment } catch { Write-Host "[ERROR] Could not install Node.js automatically." -ForegroundColor Red - Write-Host "Please install Node.js LTS (v22) from https://nodejs.org/" -ForegroundColor Red + Write-Host "Please install Node.js >= 20 from https://nodejs.org/" -ForegroundColor Red exit 1 } } @@ -399,8 +397,9 @@ if ($IsPipInstall) { $prevEAP_npm = $ErrorActionPreference $ErrorActionPreference = "Continue" Push-Location $FrontendDir - # Remove stale node_modules to avoid version conflicts + # Remove stale node_modules and package-lock.json to avoid version conflicts if (Test-Path "node_modules") { Remove-Item -Recurse -Force "node_modules" } + if (Test-Path "package-lock.json") { Remove-Item -Force "package-lock.json" } npm install 2>&1 | Out-Null if ($LASTEXITCODE -ne 0) { Pop-Location From 602ae716c38461fb7d8ae773df2d10dd563c401f Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Fri, 27 Feb 2026 21:50:54 +0000 Subject: [PATCH 10/66] Set short TORCHINDUCTOR_CACHE_DIR to fix Windows MAX_PATH crash --- setup.ps1 | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/setup.ps1 b/setup.ps1 index cbf636983f..10e7900d44 100644 --- a/setup.ps1 +++ b/setup.ps1 @@ -493,6 +493,15 @@ pip install --upgrade pip 2>&1 | Out-Null # PyTorch bundles its own CUDA runtime, so this works regardless # of whether the CUDA Toolkit is installed yet. # The CUDA tag is chosen based on the driver's max supported CUDA version. + +# Windows MAX_PATH (260 chars) causes Triton kernel compilation to fail because +# the auto-generated filenames are extremely long. Use a short cache directory. +$TorchCacheDir = "C:\tc" +if (-not (Test-Path $TorchCacheDir)) { New-Item -ItemType Directory -Path $TorchCacheDir -Force | Out-Null } +$env:TORCHINDUCTOR_CACHE_DIR = $TorchCacheDir +[Environment]::SetEnvironmentVariable('TORCHINDUCTOR_CACHE_DIR', $TorchCacheDir, 'User') +Write-Host "[OK] TORCHINDUCTOR_CACHE_DIR set to $TorchCacheDir (avoids MAX_PATH issues)" -ForegroundColor Green + $CuTag = Get-PytorchCudaTag Write-Host " Installing PyTorch with CUDA support ($CuTag)..." -ForegroundColor Cyan pip install torch torchvision torchaudio --index-url "https://download.pytorch.org/whl/$CuTag" 2>&1 | Out-Null From c297d7aa848e7fd0f99006971b5358b0b83c4b8c Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Fri, 27 Feb 2026 21:58:28 +0000 Subject: [PATCH 11/66] Force num_proc=1 on Windows to avoid slow spawn overhead --- studio/backend/utils/hardware/hardware.py | 11 +++++++++++ 1 file changed, 11 insertions(+) diff --git a/studio/backend/utils/hardware/hardware.py b/studio/backend/utils/hardware/hardware.py index b885e130d5..fd43e620bb 100644 --- a/studio/backend/utils/hardware/hardware.py +++ b/studio/backend/utils/hardware/hardware.py @@ -423,6 +423,11 @@ def safe_num_proc(desired: Optional[int] = None) -> int: """ Return a safe ``num_proc`` for ``dataset.map()`` calls. + On Windows, always returns 1 because Python uses ``spawn`` instead of + ``fork`` for multiprocessing — the overhead of re-importing torch, + transformers, unsloth etc. per worker is typically slower than + single-process for normal dataset sizes. + On multi-GPU machines the NVIDIA driver spawns extra background threads, making ``os.fork()`` prone to deadlocks when many workers are created. This helper caps ``num_proc`` to 4 on such machines. @@ -438,6 +443,12 @@ def safe_num_proc(desired: Optional[int] = None) -> int: A safe integer ≥ 1. """ import os + import sys + + # Windows uses 'spawn' for multiprocessing — the overhead of re-importing + # torch/transformers/unsloth per worker is typically slower than single-process. + if sys.platform == "win32": + return 1 if desired is None or not isinstance(desired, int): desired = max(1, os.cpu_count() // 3) From af90c9c3d27ae1227f5915e12c9fd0e74b31793e Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Sat, 28 Feb 2026 03:07:58 +0000 Subject: [PATCH 12/66] Fix llama-server binary lookup for Windows (.exe, Release dir, ~/.unsloth) --- studio/backend/core/inference/llama_cpp.py | 29 ++++++++++++++++------ 1 file changed, 22 insertions(+), 7 deletions(-) diff --git a/studio/backend/core/inference/llama_cpp.py b/studio/backend/core/inference/llama_cpp.py index b7b87e9cfb..cf4494df30 100644 --- a/studio/backend/core/inference/llama_cpp.py +++ b/studio/backend/core/inference/llama_cpp.py @@ -78,10 +78,14 @@ class LlamaCppBackend: Search order: 1. LLAMA_SERVER_PATH environment variable 2. ./llama.cpp/build/bin/llama-server (built by setup.sh in-tree) - 3. llama-server on PATH (system install) - 4. ./bin/llama-server (legacy: extracted binary) + 3. ~/.unsloth/llama.cpp/build/bin/Release/llama-server (built by setup.ps1 on Windows) + 4. llama-server on PATH (system install) + 5. ./bin/llama-server (legacy: extracted binary) """ import os + import sys + + binary_name = "llama-server.exe" if sys.platform == "win32" else "llama-server" # 1. Env var env_path = os.environ.get("LLAMA_SERVER_PATH") @@ -91,18 +95,29 @@ class LlamaCppBackend: # Project root: llama_cpp.py → inference/ → core/ → backend/ → studio/ → root project_root = Path(__file__).resolve().parents[4] - # 2. In-tree llama.cpp build (setup.sh builds here) - build_path = project_root / "llama.cpp" / "build" / "bin" / "llama-server" + # 2. In-tree llama.cpp build (setup.sh builds here on Linux) + build_path = project_root / "llama.cpp" / "build" / "bin" / binary_name if build_path.is_file(): return str(build_path) - # 3. System PATH + # 3. Windows MSVC build (setup.ps1 builds here — Release config) + if sys.platform == "win32": + # In-tree + win_path = project_root / "llama.cpp" / "build" / "bin" / "Release" / binary_name + if win_path.is_file(): + return str(win_path) + # ~/.unsloth (setup.ps1 default location) + home_path = Path.home() / ".unsloth" / "llama.cpp" / "build" / "bin" / "Release" / binary_name + if home_path.is_file(): + return str(home_path) + + # 4. System PATH system_path = shutil.which("llama-server") if system_path: return system_path - # 4. Legacy: extracted to bin/ - bin_path = project_root / "bin" / "llama-server" + # 5. Legacy: extracted to bin/ + bin_path = project_root / "bin" / binary_name if bin_path.is_file(): return str(bin_path) From 4ee5ada6096f0c13bf7d092eb40659dccea49b66 Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Sat, 28 Feb 2026 06:42:43 +0000 Subject: [PATCH 13/66] Add llama-cpp Windows test script, fix binary lookup paths --- test_llama_cpp.ps1 | 233 +++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 233 insertions(+) create mode 100644 test_llama_cpp.ps1 diff --git a/test_llama_cpp.ps1 b/test_llama_cpp.ps1 new file mode 100644 index 0000000000..5eb3b7db32 --- /dev/null +++ b/test_llama_cpp.ps1 @@ -0,0 +1,233 @@ +<# +.SYNOPSIS + Test script for llama.cpp compilation and binary validation on Windows. + Verifies that llama-server was built with CUDA support and can start. + +.USAGE + .\test_llama_cpp.ps1 + .\test_llama_cpp.ps1 -BinaryPath "C:\path\to\llama-server.exe" +#> +param( + [string]$BinaryPath = "" +) + +$ErrorActionPreference = "Continue" + +Write-Host "" +Write-Host "============================================" -ForegroundColor Cyan +Write-Host " llama.cpp Windows Build Test" -ForegroundColor Cyan +Write-Host "============================================" -ForegroundColor Cyan +Write-Host "" + +# ── Step 1: Locate the binary ────────────────────────────────────────── +Write-Host "1. Locating llama-server binary..." -ForegroundColor Yellow + +$SearchPaths = @() + +if ($BinaryPath) { + $SearchPaths += $BinaryPath +} + +# Add all known locations +$RepoRoot = $PSScriptRoot +$SearchPaths += Join-Path $RepoRoot "llama.cpp\build\bin\Release\llama-server.exe" +$SearchPaths += Join-Path $RepoRoot "llama.cpp\build\bin\llama-server.exe" +$SearchPaths += Join-Path $env:USERPROFILE ".unsloth\llama.cpp\build\bin\Release\llama-server.exe" + +# Check LLAMA_SERVER_PATH env var +$envPath = $env:LLAMA_SERVER_PATH +if ($envPath) { + $SearchPaths = @($envPath) + $SearchPaths +} + +# Also check system PATH +$systemPath = (Get-Command llama-server -ErrorAction SilentlyContinue) +if ($systemPath) { + $SearchPaths += $systemPath.Source +} + +$FoundBinary = $null +foreach ($p in $SearchPaths) { + if (Test-Path $p) { + $FoundBinary = $p + break + } +} + +if (-not $FoundBinary) { + Write-Host " [FAIL] llama-server.exe not found!" -ForegroundColor Red + Write-Host "" + Write-Host " Searched locations:" -ForegroundColor Gray + foreach ($p in $SearchPaths) { + Write-Host " - $p" -ForegroundColor Gray + } + Write-Host "" + Write-Host " To fix: Run setup.bat to build llama.cpp, or set:" -ForegroundColor Yellow + Write-Host ' $env:LLAMA_SERVER_PATH = "C:\path\to\llama-server.exe"' -ForegroundColor Yellow + exit 1 +} + +Write-Host " [OK] Found: $FoundBinary" -ForegroundColor Green + +# ── Step 2: Check file info ──────────────────────────────────────────── +Write-Host "" +Write-Host "2. Binary info..." -ForegroundColor Yellow + +$fileInfo = Get-Item $FoundBinary +$sizeMB = [math]::Round($fileInfo.Length / 1MB, 1) +Write-Host " Size: $sizeMB MB" -ForegroundColor Gray +Write-Host " Modified: $($fileInfo.LastWriteTime)" -ForegroundColor Gray + +# ── Step 3: Check for CUDA symbols ──────────────────────────────────── +Write-Host "" +Write-Host "3. Checking for CUDA support..." -ForegroundColor Yellow + +# Run with --help or -v and capture output to check for CUDA indicators +$helpOutput = & $FoundBinary --version 2>&1 | Out-String +if (-not $helpOutput) { + $helpOutput = "" +} + +# Check binary dependencies for CUDA DLLs using dumpbin if available +$dumpbin = (Get-Command dumpbin -ErrorAction SilentlyContinue) +$hasCudaDlls = $false + +if ($dumpbin) { + $deps = & dumpbin /dependents $FoundBinary 2>&1 | Out-String + if ($deps -match "cudart|cublas|cublasLt|nvcuda") { + $hasCudaDlls = $true + Write-Host " [OK] CUDA DLLs found in dependencies (dumpbin)" -ForegroundColor Green + # Extract CUDA DLL names + $cudaDlls = ($deps -split "`n") | Where-Object { $_ -match "cuda|cublas|nvcuda" } | ForEach-Object { $_.Trim() } + foreach ($dll in $cudaDlls) { + if ($dll) { Write-Host " - $dll" -ForegroundColor Gray } + } + } else { + Write-Host " [WARN] No CUDA DLLs found in dependencies!" -ForegroundColor Red + Write-Host " This binary was likely compiled WITHOUT -DGGML_CUDA=ON" -ForegroundColor Red + } +} else { + # Fallback: check file size (CUDA builds are typically > 50MB) + if ($sizeMB -gt 40) { + Write-Host " [LIKELY OK] Binary is $sizeMB MB (CUDA builds are typically > 50MB)" -ForegroundColor Green + } else { + Write-Host " [WARN] Binary is only $sizeMB MB (CPU-only builds are typically < 30MB)" -ForegroundColor Yellow + Write-Host " dumpbin not available for detailed check. Install VS Build Tools." -ForegroundColor Gray + } +} + +# ── Step 4: Quick startup test ───────────────────────────────────────── +Write-Host "" +Write-Host "4. Running startup test (will start and immediately stop)..." -ForegroundColor Yellow + +# Start llama-server on a random port with no model — just check it initializes +$testPort = Get-Random -Minimum 49152 -Maximum 65535 +$proc = $null + +try { + $proc = Start-Process -FilePath $FoundBinary ` + -ArgumentList "--port", $testPort, "--host", "127.0.0.1" ` + -PassThru -NoNewWindow -RedirectStandardError "$env:TEMP\llama_test_stderr.txt" ` + -RedirectStandardOutput "$env:TEMP\llama_test_stdout.txt" + + # Give it 3 seconds to start + Start-Sleep -Seconds 3 + + # Check if it crashed + if ($proc.HasExited) { + $exitCode = $proc.ExitCode + $stderr = "" + if (Test-Path "$env:TEMP\llama_test_stderr.txt") { + $stderr = Get-Content "$env:TEMP\llama_test_stderr.txt" -Raw + } + $stdout = "" + if (Test-Path "$env:TEMP\llama_test_stdout.txt") { + $stdout = Get-Content "$env:TEMP\llama_test_stdout.txt" -Raw + } + + $allOutput = "$stdout`n$stderr" + + if ($allOutput -match "failed to initialize CUDA") { + Write-Host " [FAIL] CUDA initialization failed!" -ForegroundColor Red + Write-Host " The binary was compiled without CUDA support or CUDA drivers are missing." -ForegroundColor Red + Write-Host "" + Write-Host " Rebuild with: cmake -DGGML_CUDA=ON ..." -ForegroundColor Yellow + } elseif ($allOutput -match "HTTPS is not supported") { + # This is expected when LLAMA_CURL=OFF — not a real failure + Write-Host " [OK] Binary started (HTTPS warning is expected — we use local files)" -ForegroundColor Green + } else { + Write-Host " [WARN] Process exited with code $exitCode" -ForegroundColor Yellow + } + + if ($allOutput.Trim()) { + Write-Host "" + Write-Host " --- Output ---" -ForegroundColor Gray + $allOutput.Trim().Split("`n") | ForEach-Object { Write-Host " $_" -ForegroundColor Gray } + } + } else { + Write-Host " [OK] llama-server started successfully on port $testPort" -ForegroundColor Green + + # Check CUDA detection from startup output + Start-Sleep -Seconds 1 + $stderr = "" + if (Test-Path "$env:TEMP\llama_test_stderr.txt") { + $stderr = Get-Content "$env:TEMP\llama_test_stderr.txt" -Raw + } + + if ($stderr -match "CUDA") { + if ($stderr -match "failed to initialize CUDA") { + Write-Host " [FAIL] CUDA init failed at runtime!" -ForegroundColor Red + } else { + Write-Host " [OK] CUDA detected at runtime" -ForegroundColor Green + } + } + + # Kill it + Stop-Process -Id $proc.Id -Force -ErrorAction SilentlyContinue + Write-Host " Stopped test server." -ForegroundColor Gray + } +} catch { + Write-Host " [ERROR] Could not start llama-server: $_" -ForegroundColor Red +} finally { + if ($proc -and -not $proc.HasExited) { + Stop-Process -Id $proc.Id -Force -ErrorAction SilentlyContinue + } + Remove-Item "$env:TEMP\llama_test_stderr.txt" -ErrorAction SilentlyContinue + Remove-Item "$env:TEMP\llama_test_stdout.txt" -ErrorAction SilentlyContinue +} + +# ── Step 5: Check llama-quantize ─────────────────────────────────────── +Write-Host "" +Write-Host "5. Checking llama-quantize..." -ForegroundColor Yellow + +$quantizePath = Join-Path (Split-Path $FoundBinary) "llama-quantize.exe" +if (Test-Path $quantizePath) { + $qSize = [math]::Round((Get-Item $quantizePath).Length / 1MB, 1) + Write-Host " [OK] Found: $quantizePath ($qSize MB)" -ForegroundColor Green +} else { + Write-Host " [WARN] llama-quantize.exe not found alongside llama-server" -ForegroundColor Yellow + Write-Host " GGUF export/quantization won't work without it" -ForegroundColor Yellow +} + +# ── Summary ──────────────────────────────────────────────────────────── +Write-Host "" +Write-Host "============================================" -ForegroundColor Cyan +Write-Host " Summary" -ForegroundColor Cyan +Write-Host "============================================" -ForegroundColor Cyan +Write-Host " Binary: $FoundBinary" -ForegroundColor Gray +Write-Host " Size: $sizeMB MB" -ForegroundColor Gray +if ($hasCudaDlls) { + Write-Host " CUDA: YES (confirmed via DLL deps)" -ForegroundColor Green +} elseif ($sizeMB -gt 40) { + Write-Host " CUDA: LIKELY (large binary size)" -ForegroundColor Yellow +} else { + Write-Host " CUDA: NO (rebuild with -DGGML_CUDA=ON)" -ForegroundColor Red +} +Write-Host "" + +if (-not $hasCudaDlls -and $sizeMB -le 40) { + Write-Host "To rebuild with CUDA:" -ForegroundColor Yellow + Write-Host ' 1. Delete the build dir: Remove-Item -Recurse -Force "$env:USERPROFILE\.unsloth\llama.cpp\build"' -ForegroundColor Gray + Write-Host ' 2. Re-run: .\setup.bat' -ForegroundColor Gray + Write-Host "" +} From 18134ee80802320724f38cc05b10cedb925126c9 Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Sat, 28 Feb 2026 07:07:46 +0000 Subject: [PATCH 14/66] Fix non-ASCII chars in test script for Windows PS 5.1 --- test_llama_cpp.ps1 | 18 +++++++++--------- 1 file changed, 9 insertions(+), 9 deletions(-) diff --git a/test_llama_cpp.ps1 b/test_llama_cpp.ps1 index 5eb3b7db32..6e16057640 100644 --- a/test_llama_cpp.ps1 +++ b/test_llama_cpp.ps1 @@ -19,7 +19,7 @@ Write-Host " llama.cpp Windows Build Test" -ForegroundColor Cyan Write-Host "============================================" -ForegroundColor Cyan Write-Host "" -# ── Step 1: Locate the binary ────────────────────────────────────────── +# -- Step 1: Locate the binary ------------------------------------------ Write-Host "1. Locating llama-server binary..." -ForegroundColor Yellow $SearchPaths = @() @@ -69,7 +69,7 @@ if (-not $FoundBinary) { Write-Host " [OK] Found: $FoundBinary" -ForegroundColor Green -# ── Step 2: Check file info ──────────────────────────────────────────── +# -- Step 2: Check file info -------------------------------------------- Write-Host "" Write-Host "2. Binary info..." -ForegroundColor Yellow @@ -78,7 +78,7 @@ $sizeMB = [math]::Round($fileInfo.Length / 1MB, 1) Write-Host " Size: $sizeMB MB" -ForegroundColor Gray Write-Host " Modified: $($fileInfo.LastWriteTime)" -ForegroundColor Gray -# ── Step 3: Check for CUDA symbols ──────────────────────────────────── +# -- Step 3: Check for CUDA symbols ------------------------------------ Write-Host "" Write-Host "3. Checking for CUDA support..." -ForegroundColor Yellow @@ -116,11 +116,11 @@ if ($dumpbin) { } } -# ── Step 4: Quick startup test ───────────────────────────────────────── +# -- Step 4: Quick startup test ----------------------------------------- Write-Host "" Write-Host "4. Running startup test (will start and immediately stop)..." -ForegroundColor Yellow -# Start llama-server on a random port with no model — just check it initializes +# Start llama-server on a random port with no model -- just check it initializes $testPort = Get-Random -Minimum 49152 -Maximum 65535 $proc = $null @@ -153,8 +153,8 @@ try { Write-Host "" Write-Host " Rebuild with: cmake -DGGML_CUDA=ON ..." -ForegroundColor Yellow } elseif ($allOutput -match "HTTPS is not supported") { - # This is expected when LLAMA_CURL=OFF — not a real failure - Write-Host " [OK] Binary started (HTTPS warning is expected — we use local files)" -ForegroundColor Green + # This is expected when LLAMA_CURL=OFF -- not a real failure + Write-Host " [OK] Binary started (HTTPS warning is expected -- we use local files)" -ForegroundColor Green } else { Write-Host " [WARN] Process exited with code $exitCode" -ForegroundColor Yellow } @@ -196,7 +196,7 @@ try { Remove-Item "$env:TEMP\llama_test_stdout.txt" -ErrorAction SilentlyContinue } -# ── Step 5: Check llama-quantize ─────────────────────────────────────── +# -- Step 5: Check llama-quantize --------------------------------------- Write-Host "" Write-Host "5. Checking llama-quantize..." -ForegroundColor Yellow @@ -209,7 +209,7 @@ if (Test-Path $quantizePath) { Write-Host " GGUF export/quantization won't work without it" -ForegroundColor Yellow } -# ── Summary ──────────────────────────────────────────────────────────── +# -- Summary ------------------------------------------------------------ Write-Host "" Write-Host "============================================" -ForegroundColor Cyan Write-Host " Summary" -ForegroundColor Cyan From 7377c92a4c47715fc9e3a6870d4b7720e04707c5 Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Sat, 28 Feb 2026 07:42:46 +0000 Subject: [PATCH 15/66] Auto-detect driver CUDA version, install compatible toolkit instead of latest --- setup.ps1 | 47 ++++++++++++++++++++++++++++++++++++++++++++--- 1 file changed, 44 insertions(+), 3 deletions(-) diff --git a/setup.ps1 b/setup.ps1 index 10e7900d44..c213dcbf50 100644 --- a/setup.ps1 +++ b/setup.ps1 @@ -278,14 +278,55 @@ if ($vsResult) { # ============================================ # 1e. CUDA Toolkit (nvcc for llama.cpp build + env vars) # ============================================ +# IMPORTANT: The CUDA Toolkit version must be <= the max CUDA version the +# NVIDIA driver supports. nvidia-smi reports this as "CUDA Version: X.Y". +# If we install a toolkit newer than the driver supports, llama-server will +# fail at runtime with "ggml_cuda_init: failed to initialize CUDA: (null)". + +# -- Detect max CUDA version the driver supports -- +$DriverMaxCuda = $null +try { + $smiOut = nvidia-smi 2>&1 | Out-String + if ($smiOut -match "CUDA Version:\s+([\d]+)\.([\d]+)") { + $DriverMaxCuda = "$($Matches[1]).$($Matches[2])" + Write-Host " Driver supports up to CUDA $DriverMaxCuda" -ForegroundColor Gray + } +} catch {} + $NvccPath = Find-Nvcc +# -- If toolkit is already installed, verify it's compatible with driver -- +if ($NvccPath -and $DriverMaxCuda) { + $NvccOut = & $NvccPath --version 2>&1 | Out-String + if ($NvccOut -match "release\s+([\d]+)\.([\d]+)") { + $ToolkitVersion = "$($Matches[1]).$($Matches[2])" + $tkMajor = [int]$Matches[1]; $tkMinor = [int]$Matches[2] + $drMajor = [int]$DriverMaxCuda.Split('.')[0]; $drMinor = [int]$DriverMaxCuda.Split('.')[1] + if (($tkMajor -gt $drMajor) -or ($tkMajor -eq $drMajor -and $tkMinor -gt $drMinor)) { + Write-Host "[WARN] Installed CUDA Toolkit $ToolkitVersion is NEWER than driver supports ($DriverMaxCuda)." -ForegroundColor Yellow + Write-Host " This will cause 'failed to initialize CUDA' at runtime." -ForegroundColor Yellow + Write-Host " Installing compatible CUDA Toolkit $DriverMaxCuda..." -ForegroundColor Cyan + # Force reinstall of a compatible version + $NvccPath = $null + } else { + Write-Host " [OK] CUDA Toolkit $ToolkitVersion is compatible with driver (max $DriverMaxCuda)" -ForegroundColor Green + } + } +} + if (-not $NvccPath) { - Write-Host "CUDA driver detected but toolkit (nvcc) not found -- installing via winget..." -ForegroundColor Yellow + Write-Host "CUDA driver detected but compatible toolkit (nvcc) not found -- installing via winget..." -ForegroundColor Yellow $HasWinget = $null -ne (Get-Command winget -ErrorAction SilentlyContinue) if ($HasWinget) { - Write-Host " Installing CUDA Toolkit via winget..." -ForegroundColor Cyan - winget install --id=Nvidia.CUDA -e --source winget --accept-package-agreements --accept-source-agreements + # Install the version matching the driver's max supported CUDA + $WingetVersion = if ($DriverMaxCuda) { $DriverMaxCuda } else { $null } + if ($WingetVersion) { + Write-Host " Installing CUDA Toolkit $WingetVersion via winget..." -ForegroundColor Cyan + winget install --id=Nvidia.CUDA --version=$WingetVersion -e --source winget --accept-package-agreements --accept-source-agreements + } else { + Write-Host " Installing CUDA Toolkit (latest) via winget..." -ForegroundColor Cyan + winget install --id=Nvidia.CUDA -e --source winget --accept-package-agreements --accept-source-agreements + } Refresh-Environment $NvccPath = Find-Nvcc if ($NvccPath) { From afa134445205e101b9b73b296f447ade20ea43e0 Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Sat, 28 Feb 2026 07:45:40 +0000 Subject: [PATCH 16/66] Build llama.cpp in-tree, auto-detect driver CUDA version for compatible toolkit --- setup.ps1 | 7 +++---- studio/backend/core/inference/llama_cpp.py | 6 +++--- test_llama_cpp.ps1 | 1 + 3 files changed, 7 insertions(+), 7 deletions(-) diff --git a/setup.ps1 b/setup.ps1 index c213dcbf50..1269467368 100644 --- a/setup.ps1 +++ b/setup.ps1 @@ -556,12 +556,13 @@ $ErrorActionPreference = $prevEAP # ========================================================================== # PHASE 4: Build llama.cpp with CUDA for GGUF inference + export # ========================================================================== -# Builds at ~/.unsloth/llama.cpp/ (persistent across pip upgrades). +# Builds in-tree at $REPO/llama.cpp/ (same as setup.sh on Linux). +# This directory is already in .gitignore. # We build: # - llama-server: for GGUF model inference # - llama-quantize: for GGUF export quantization # Prerequisites (git, cmake, VS Build Tools, CUDA Toolkit) already installed in Phase 1. -$LlamaCppDir = Join-Path $env:USERPROFILE ".unsloth\llama.cpp" +$LlamaCppDir = Join-Path $PSScriptRoot "llama.cpp" $BuildDir = Join-Path $LlamaCppDir "build" $LlamaServerBin = Join-Path $BuildDir "bin\Release\llama-server.exe" @@ -588,8 +589,6 @@ if (Test-Path $LlamaServerBin) { $FailedStep = "" # -- Step A: Clone or pull llama.cpp -- - $UnslothDir = Join-Path $env:USERPROFILE ".unsloth" - if (-not (Test-Path $UnslothDir)) { New-Item -ItemType Directory -Path $UnslothDir -Force | Out-Null } if (Test-Path (Join-Path $LlamaCppDir ".git")) { Write-Host " llama.cpp repo already cloned, pulling latest..." -ForegroundColor Gray diff --git a/studio/backend/core/inference/llama_cpp.py b/studio/backend/core/inference/llama_cpp.py index cf4494df30..7ae9bd142a 100644 --- a/studio/backend/core/inference/llama_cpp.py +++ b/studio/backend/core/inference/llama_cpp.py @@ -100,13 +100,13 @@ class LlamaCppBackend: if build_path.is_file(): return str(build_path) - # 3. Windows MSVC build (setup.ps1 builds here — Release config) + # 3. Windows MSVC build (Release config, in-tree — matches setup.ps1) if sys.platform == "win32": - # In-tree + # In-tree (primary — setup.ps1 now builds here) win_path = project_root / "llama.cpp" / "build" / "bin" / "Release" / binary_name if win_path.is_file(): return str(win_path) - # ~/.unsloth (setup.ps1 default location) + # Legacy: ~/.unsloth (older setup.ps1 versions built here) home_path = Path.home() / ".unsloth" / "llama.cpp" / "build" / "bin" / "Release" / binary_name if home_path.is_file(): return str(home_path) diff --git a/test_llama_cpp.ps1 b/test_llama_cpp.ps1 index 6e16057640..b98177eac6 100644 --- a/test_llama_cpp.ps1 +++ b/test_llama_cpp.ps1 @@ -32,6 +32,7 @@ if ($BinaryPath) { $RepoRoot = $PSScriptRoot $SearchPaths += Join-Path $RepoRoot "llama.cpp\build\bin\Release\llama-server.exe" $SearchPaths += Join-Path $RepoRoot "llama.cpp\build\bin\llama-server.exe" +# Legacy: older setup.ps1 built under ~/.unsloth $SearchPaths += Join-Path $env:USERPROFILE ".unsloth\llama.cpp\build\bin\Release\llama-server.exe" # Check LLAMA_SERVER_PATH env var From f80cef3abee56f7595f41e12ddd89697064fca90 Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Sat, 28 Feb 2026 07:50:44 +0000 Subject: [PATCH 17/66] Fix: scan side-by-side CUDA installs, pick compatible toolkit version --- setup.ps1 | 75 +++++++++++++++++++++++++++++++++++++++---------------- 1 file changed, 54 insertions(+), 21 deletions(-) diff --git a/setup.ps1 b/setup.ps1 index 1269467368..3fec5f22ed 100644 --- a/setup.ps1 +++ b/setup.ps1 @@ -42,6 +42,41 @@ function Refresh-Environment { # Find nvcc on PATH, CUDA_PATH, or standard toolkit dirs. # Returns the path to nvcc.exe, or $null if not found. function Find-Nvcc { + param([string]$MaxVersion = "") + + # If MaxVersion is set, we need to find a toolkit <= that version. + # CUDA toolkits install side-by-side under C:\Program Files\NVIDIA GPU Computing Toolkit\CUDA\vX.Y\ + + $toolkitBase = 'C:\Program Files\NVIDIA GPU Computing Toolkit\CUDA' + + if ($MaxVersion -and (Test-Path $toolkitBase)) { + $drMajor = [int]$MaxVersion.Split('.')[0] + $drMinor = [int]$MaxVersion.Split('.')[1] + + # Get all installed CUDA dirs, sorted descending (highest first) + $cudaDirs = Get-ChildItem -Directory $toolkitBase | Where-Object { + $_.Name -match '^v(\d+)\.(\d+)' + } | Sort-Object { [version]($_.Name -replace '^v','') } -Descending + + foreach ($dir in $cudaDirs) { + if ($dir.Name -match '^v(\d+)\.(\d+)') { + $tkMajor = [int]$Matches[1]; $tkMinor = [int]$Matches[2] + $compatible = ($tkMajor -lt $drMajor) -or ($tkMajor -eq $drMajor -and $tkMinor -le $drMinor) + if ($compatible) { + $nvcc = Join-Path $dir.FullName 'bin\nvcc.exe' + if (Test-Path $nvcc) { + return $nvcc + } + } + } + } + + # No compatible side-by-side version found + return $null + } + + # Fallback: no version constraint — pick latest or whatever is available + # 1. Check nvcc on PATH $cmd = Get-Command nvcc -ErrorAction SilentlyContinue if ($cmd) { return $cmd.Source } @@ -55,7 +90,6 @@ function Find-Nvcc { } # 3. Scan standard toolkit directory - $toolkitBase = 'C:\Program Files\NVIDIA GPU Computing Toolkit\CUDA' if (Test-Path $toolkitBase) { $latest = Get-ChildItem -Directory $toolkitBase | Sort-Object Name | Select-Object -Last 1 if ($latest -and (Test-Path (Join-Path $latest.FullName 'bin\nvcc.exe'))) { @@ -293,25 +327,16 @@ try { } } catch {} -$NvccPath = Find-Nvcc - -# -- If toolkit is already installed, verify it's compatible with driver -- -if ($NvccPath -and $DriverMaxCuda) { - $NvccOut = & $NvccPath --version 2>&1 | Out-String - if ($NvccOut -match "release\s+([\d]+)\.([\d]+)") { - $ToolkitVersion = "$($Matches[1]).$($Matches[2])" - $tkMajor = [int]$Matches[1]; $tkMinor = [int]$Matches[2] - $drMajor = [int]$DriverMaxCuda.Split('.')[0]; $drMinor = [int]$DriverMaxCuda.Split('.')[1] - if (($tkMajor -gt $drMajor) -or ($tkMajor -eq $drMajor -and $tkMinor -gt $drMinor)) { - Write-Host "[WARN] Installed CUDA Toolkit $ToolkitVersion is NEWER than driver supports ($DriverMaxCuda)." -ForegroundColor Yellow - Write-Host " This will cause 'failed to initialize CUDA' at runtime." -ForegroundColor Yellow - Write-Host " Installing compatible CUDA Toolkit $DriverMaxCuda..." -ForegroundColor Cyan - # Force reinstall of a compatible version - $NvccPath = $null - } else { - Write-Host " [OK] CUDA Toolkit $ToolkitVersion is compatible with driver (max $DriverMaxCuda)" -ForegroundColor Green - } +# -- Find a toolkit that's compatible with the driver -- +if ($DriverMaxCuda) { + $NvccPath = Find-Nvcc -MaxVersion $DriverMaxCuda + if ($NvccPath) { + Write-Host " [OK] Found compatible CUDA Toolkit (nvcc: $NvccPath)" -ForegroundColor Green + } else { + Write-Host " No CUDA Toolkit <= $DriverMaxCuda found. Will install..." -ForegroundColor Yellow } +} else { + $NvccPath = Find-Nvcc } if (-not $NvccPath) { @@ -328,7 +353,11 @@ if (-not $NvccPath) { winget install --id=Nvidia.CUDA -e --source winget --accept-package-agreements --accept-source-agreements } Refresh-Environment - $NvccPath = Find-Nvcc + if ($DriverMaxCuda) { + $NvccPath = Find-Nvcc -MaxVersion $DriverMaxCuda + } else { + $NvccPath = Find-Nvcc + } if ($NvccPath) { Write-Host " [OK] CUDA Toolkit installed (nvcc: $NvccPath)" -ForegroundColor Green } @@ -337,7 +366,11 @@ if (-not $NvccPath) { if (-not $NvccPath) { Write-Host "[ERROR] CUDA Toolkit (nvcc) is required but could not be found or installed." -ForegroundColor Red - Write-Host " Install CUDA Toolkit from https://developer.nvidia.com/cuda-downloads" -ForegroundColor Yellow + if ($DriverMaxCuda) { + Write-Host " Install CUDA Toolkit $DriverMaxCuda from https://developer.nvidia.com/cuda-toolkit-archive" -ForegroundColor Yellow + } else { + Write-Host " Install CUDA Toolkit from https://developer.nvidia.com/cuda-downloads" -ForegroundColor Yellow + } exit 1 } From d965a51b703476d9a2dd1fe212fa6f99fbf205aa Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Sat, 28 Feb 2026 07:51:31 +0000 Subject: [PATCH 18/66] Always persist compatible CUDA_PATH to User registry (overwrite stale values) --- setup.ps1 | 11 ++++------- 1 file changed, 4 insertions(+), 7 deletions(-) diff --git a/setup.ps1 b/setup.ps1 index 3fec5f22ed..4c826d4bec 100644 --- a/setup.ps1 +++ b/setup.ps1 @@ -381,13 +381,10 @@ $CudaToolkitRoot = Split-Path (Split-Path $NvccPath -Parent) -Parent # CudaToolkitDir: the MSBuild property that CUDA .targets checks directly # Trailing backslash required -- the .targets file appends subpaths to it [Environment]::SetEnvironmentVariable('CudaToolkitDir', "$CudaToolkitRoot\", 'Process') -# Persist CUDA_PATH to User registry if not already set -$existingSys = [Environment]::GetEnvironmentVariable('CUDA_PATH', 'Machine') -$existingUsr = [Environment]::GetEnvironmentVariable('CUDA_PATH', 'User') -if (-not $existingSys -and -not $existingUsr) { - [Environment]::SetEnvironmentVariable('CUDA_PATH', $CudaToolkitRoot, 'User') - Write-Host " Persisted CUDA_PATH to user environment" -ForegroundColor Gray -} +# Always persist CUDA_PATH to User registry so the compatible toolkit is used +# in future sessions (overwrites any existing value pointing to a newer, incompatible version) +[Environment]::SetEnvironmentVariable('CUDA_PATH', $CudaToolkitRoot, 'User') +Write-Host " Persisted CUDA_PATH=$CudaToolkitRoot to user environment" -ForegroundColor Gray # Ensure nvcc's bin dir is on PATH for this process $nvccBinDir = Split-Path $NvccPath -Parent if ($env:PATH -notlike "*$nvccBinDir*") { From c182b8c4380c4189c581dcde6f2727f15400f825 Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Sat, 28 Feb 2026 07:57:46 +0000 Subject: [PATCH 19/66] Fallback: try descending CUDA versions if exact driver-max install fails --- setup.ps1 | 38 +++++++++++++++++++++++++------------- 1 file changed, 25 insertions(+), 13 deletions(-) diff --git a/setup.ps1 b/setup.ps1 index 4c826d4bec..d4202dc97c 100644 --- a/setup.ps1 +++ b/setup.ps1 @@ -343,23 +343,35 @@ if (-not $NvccPath) { Write-Host "CUDA driver detected but compatible toolkit (nvcc) not found -- installing via winget..." -ForegroundColor Yellow $HasWinget = $null -ne (Get-Command winget -ErrorAction SilentlyContinue) if ($HasWinget) { - # Install the version matching the driver's max supported CUDA - $WingetVersion = if ($DriverMaxCuda) { $DriverMaxCuda } else { $null } - if ($WingetVersion) { - Write-Host " Installing CUDA Toolkit $WingetVersion via winget..." -ForegroundColor Cyan - winget install --id=Nvidia.CUDA --version=$WingetVersion -e --source winget --accept-package-agreements --accept-source-agreements + if ($DriverMaxCuda) { + # Try descending compatible versions: 12.9, 12.8, 12.6, 12.5, 12.4, 12.3 + $drMajor = [int]$DriverMaxCuda.Split('.')[0] + $drMinor = [int]$DriverMaxCuda.Split('.')[1] + $versionsToTry = @() + for ($m = $drMinor; $m -ge 0; $m--) { + $versionsToTry += "$drMajor.$m" + } + foreach ($ver in $versionsToTry) { + Write-Host " Trying CUDA Toolkit $ver via winget..." -ForegroundColor Cyan + $prevEAPCuda = $ErrorActionPreference + $ErrorActionPreference = "Continue" + winget install --id=Nvidia.CUDA --version=$ver -e --source winget --accept-package-agreements --accept-source-agreements 2>&1 | Out-Null + $ErrorActionPreference = $prevEAPCuda + Refresh-Environment + $NvccPath = Find-Nvcc -MaxVersion $DriverMaxCuda + if ($NvccPath) { + Write-Host " [OK] CUDA Toolkit installed (nvcc: $NvccPath)" -ForegroundColor Green + break + } + } } else { Write-Host " Installing CUDA Toolkit (latest) via winget..." -ForegroundColor Cyan winget install --id=Nvidia.CUDA -e --source winget --accept-package-agreements --accept-source-agreements - } - Refresh-Environment - if ($DriverMaxCuda) { - $NvccPath = Find-Nvcc -MaxVersion $DriverMaxCuda - } else { + Refresh-Environment $NvccPath = Find-Nvcc - } - if ($NvccPath) { - Write-Host " [OK] CUDA Toolkit installed (nvcc: $NvccPath)" -ForegroundColor Green + if ($NvccPath) { + Write-Host " [OK] CUDA Toolkit installed (nvcc: $NvccPath)" -ForegroundColor Green + } } } } From e956fb410684de091f3ee4daaabd7fed06856c15 Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Sat, 28 Feb 2026 08:05:00 +0000 Subject: [PATCH 20/66] Warn user to uninstall incompatible CUDA toolkit instead of failed side-by-side --- setup.ps1 | 41 +++++++++++++++++++++++++++++++++-------- 1 file changed, 33 insertions(+), 8 deletions(-) diff --git a/setup.ps1 b/setup.ps1 index d4202dc97c..49e7ebba0d 100644 --- a/setup.ps1 +++ b/setup.ps1 @@ -328,30 +328,55 @@ try { } catch {} # -- Find a toolkit that's compatible with the driver -- +$IncompatibleToolkit = $null if ($DriverMaxCuda) { $NvccPath = Find-Nvcc -MaxVersion $DriverMaxCuda if ($NvccPath) { Write-Host " [OK] Found compatible CUDA Toolkit (nvcc: $NvccPath)" -ForegroundColor Green } else { - Write-Host " No CUDA Toolkit <= $DriverMaxCuda found. Will install..." -ForegroundColor Yellow + # Check if there's an incompatible (too new) toolkit installed + $AnyNvcc = Find-Nvcc + if ($AnyNvcc) { + $NvccOut = & $AnyNvcc --version 2>&1 | Out-String + if ($NvccOut -match "release\s+([\d]+\.[\d]+)") { + $IncompatibleToolkit = $Matches[1] + } + } } } else { $NvccPath = Find-Nvcc } +# -- If incompatible toolkit is blocking, tell user to uninstall it -- +if (-not $NvccPath -and $IncompatibleToolkit) { + Write-Host "" -ForegroundColor Red + Write-Host "========================================================================" -ForegroundColor Red + Write-Host "[ERROR] CUDA Toolkit $IncompatibleToolkit is installed but INCOMPATIBLE" -ForegroundColor Red + Write-Host " with your NVIDIA driver (which supports up to CUDA $DriverMaxCuda)." -ForegroundColor Red + Write-Host "" -ForegroundColor Red + Write-Host " This will cause 'failed to initialize CUDA' errors at runtime." -ForegroundColor Red + Write-Host "" -ForegroundColor Red + Write-Host " To fix:" -ForegroundColor Yellow + Write-Host " 1. Open Control Panel -> Programs -> Uninstall a program" -ForegroundColor Yellow + Write-Host " 2. Uninstall 'NVIDIA CUDA Toolkit $IncompatibleToolkit'" -ForegroundColor Yellow + Write-Host " 3. Re-run setup.bat (it will install CUDA $DriverMaxCuda automatically)" -ForegroundColor Yellow + Write-Host "" -ForegroundColor Yellow + Write-Host " Alternatively, update your NVIDIA driver to one that supports CUDA $IncompatibleToolkit." -ForegroundColor Gray + Write-Host "========================================================================" -ForegroundColor Red + exit 1 +} + +# -- No toolkit at all: install via winget -- if (-not $NvccPath) { - Write-Host "CUDA driver detected but compatible toolkit (nvcc) not found -- installing via winget..." -ForegroundColor Yellow + Write-Host "CUDA toolkit (nvcc) not found -- installing via winget..." -ForegroundColor Yellow $HasWinget = $null -ne (Get-Command winget -ErrorAction SilentlyContinue) if ($HasWinget) { if ($DriverMaxCuda) { - # Try descending compatible versions: 12.9, 12.8, 12.6, 12.5, 12.4, 12.3 + # Try descending compatible versions $drMajor = [int]$DriverMaxCuda.Split('.')[0] $drMinor = [int]$DriverMaxCuda.Split('.')[1] - $versionsToTry = @() for ($m = $drMinor; $m -ge 0; $m--) { - $versionsToTry += "$drMajor.$m" - } - foreach ($ver in $versionsToTry) { + $ver = "$drMajor.$m" Write-Host " Trying CUDA Toolkit $ver via winget..." -ForegroundColor Cyan $prevEAPCuda = $ErrorActionPreference $ErrorActionPreference = "Continue" @@ -360,7 +385,7 @@ if (-not $NvccPath) { Refresh-Environment $NvccPath = Find-Nvcc -MaxVersion $DriverMaxCuda if ($NvccPath) { - Write-Host " [OK] CUDA Toolkit installed (nvcc: $NvccPath)" -ForegroundColor Green + Write-Host " [OK] CUDA Toolkit $ver installed (nvcc: $NvccPath)" -ForegroundColor Green break } } From 0430d22cc285fd8eeeaf0602ac3482274d67a52d Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Sat, 28 Feb 2026 08:41:54 +0000 Subject: [PATCH 21/66] Auto-add CUDA DLLs to PATH when launching llama-server on Windows --- setup.ps1 | 28 ++++++++++------------ studio/backend/core/inference/llama_cpp.py | 26 ++++++++++++++++---- 2 files changed, 35 insertions(+), 19 deletions(-) diff --git a/setup.ps1 b/setup.ps1 index 49e7ebba0d..fae0ce7529 100644 --- a/setup.ps1 +++ b/setup.ps1 @@ -820,18 +820,16 @@ Write-Host "[OK] Batch launchers created in $BatDir (works from cmd.exe when ven # Done # ============================================ Write-Host "" -Write-Host "+==============================================+" -ForegroundColor Green -Write-Host "| Setup Complete! |" -ForegroundColor Green -Write-Host "| |" -ForegroundColor Green -if ($AliasAdded) { - Write-Host "| PowerShell: run '. `$PROFILE' |" -ForegroundColor Green - Write-Host "| or open a new terminal, then: |" -ForegroundColor Green -} else { - Write-Host "| Launch with: |" -ForegroundColor Green -} -Write-Host "| |" -ForegroundColor Green -Write-Host "| unsloth-studio -H 0.0.0.0 -p 8000 |" -ForegroundColor Green -Write-Host "| |" -ForegroundColor Green -Write-Host "| cmd.exe: .venv\Scripts\activate.bat |" -ForegroundColor Green -Write-Host "| unsloth-studio -H 0.0.0.0 -p 8000 |" -ForegroundColor Green -Write-Host "+==============================================+" -ForegroundColor Green \ No newline at end of file +Write-Host "+===============================================+" -ForegroundColor Green +Write-Host "| Setup Complete! |" -ForegroundColor Green +Write-Host "| |" -ForegroundColor Green +Write-Host "| IMPORTANT: Open a NEW terminal, then: |" -ForegroundColor Yellow +Write-Host "| |" -ForegroundColor Green +Write-Host "| cmd.exe: |" -ForegroundColor Green +Write-Host "| .venv\Scripts\activate.bat |" -ForegroundColor Green +Write-Host "| unsloth-studio -H 0.0.0.0 -p 8000 |" -ForegroundColor Green +Write-Host "| |" -ForegroundColor Green +Write-Host "| PowerShell: |" -ForegroundColor Green +Write-Host "| .venv\Scripts\Activate.ps1 |" -ForegroundColor Green +Write-Host "| unsloth-studio -H 0.0.0.0 -p 8000 |" -ForegroundColor Green +Write-Host "+===============================================+" -ForegroundColor Green \ No newline at end of file diff --git a/studio/backend/core/inference/llama_cpp.py b/studio/backend/core/inference/llama_cpp.py index 7ae9bd142a..023edb959b 100644 --- a/studio/backend/core/inference/llama_cpp.py +++ b/studio/backend/core/inference/llama_cpp.py @@ -235,13 +235,31 @@ class LlamaCppBackend: logger.info(f"Starting llama-server: {' '.join(cmd)}") - # Set LD_LIBRARY_PATH so llama-server can find its shared libs - # (libmtmd.so, libllama.so, etc.) which live next to the binary + # Set library paths so llama-server can find its shared libs and CUDA DLLs import os + import sys env = os.environ.copy() binary_dir = str(Path(binary).parent) - existing_ld = env.get("LD_LIBRARY_PATH", "") - env["LD_LIBRARY_PATH"] = f"{binary_dir}:{existing_ld}" if existing_ld else binary_dir + + if sys.platform == "win32": + # On Windows, CUDA DLLs (cublas64_12.dll, cudart64_12.dll, etc.) + # must be on PATH. Add CUDA_PATH\bin if available. + path_dirs = [binary_dir] + cuda_path = os.environ.get("CUDA_PATH", "") + if cuda_path: + cuda_bin = os.path.join(cuda_path, "bin") + if os.path.isdir(cuda_bin): + path_dirs.append(cuda_bin) + # Some CUDA installs put DLLs in bin\x64 + cuda_bin_x64 = os.path.join(cuda_path, "bin", "x64") + if os.path.isdir(cuda_bin_x64): + path_dirs.append(cuda_bin_x64) + existing_path = env.get("PATH", "") + env["PATH"] = ";".join(path_dirs) + ";" + existing_path + else: + # Linux: set LD_LIBRARY_PATH for shared libs next to the binary + existing_ld = env.get("LD_LIBRARY_PATH", "") + env["LD_LIBRARY_PATH"] = f"{binary_dir}:{existing_ld}" if existing_ld else binary_dir self._stdout_lines = [] self._process = subprocess.Popen( From 9b6a928a78181752fb50e765438b57c1839562b0 Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Sat, 28 Feb 2026 08:43:35 +0000 Subject: [PATCH 22/66] Simplify completion banner: no venv activation needed --- setup.ps1 | 7 +------ 1 file changed, 1 insertion(+), 6 deletions(-) diff --git a/setup.ps1 b/setup.ps1 index fae0ce7529..07dff2090f 100644 --- a/setup.ps1 +++ b/setup.ps1 @@ -823,13 +823,8 @@ Write-Host "" Write-Host "+===============================================+" -ForegroundColor Green Write-Host "| Setup Complete! |" -ForegroundColor Green Write-Host "| |" -ForegroundColor Green -Write-Host "| IMPORTANT: Open a NEW terminal, then: |" -ForegroundColor Yellow +Write-Host "| IMPORTANT: Open a NEW terminal, then run: |" -ForegroundColor Yellow Write-Host "| |" -ForegroundColor Green -Write-Host "| cmd.exe: |" -ForegroundColor Green -Write-Host "| .venv\Scripts\activate.bat |" -ForegroundColor Green Write-Host "| unsloth-studio -H 0.0.0.0 -p 8000 |" -ForegroundColor Green Write-Host "| |" -ForegroundColor Green -Write-Host "| PowerShell: |" -ForegroundColor Green -Write-Host "| .venv\Scripts\Activate.ps1 |" -ForegroundColor Green -Write-Host "| unsloth-studio -H 0.0.0.0 -p 8000 |" -ForegroundColor Green Write-Host "+===============================================+" -ForegroundColor Green \ No newline at end of file From 2a5e03945f97482b7c43b1c35cb96b970857290f Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Sat, 28 Feb 2026 08:46:09 +0000 Subject: [PATCH 23/66] Add .venv/Scripts to User PATH so unsloth-studio works without activation --- setup.ps1 | 14 ++++++++++++-- 1 file changed, 12 insertions(+), 2 deletions(-) diff --git a/setup.ps1 b/setup.ps1 index 07dff2090f..265b13eb35 100644 --- a/setup.ps1 +++ b/setup.ps1 @@ -806,7 +806,7 @@ function unsloth-ui { & "$VenvPython" "$CliScript" studio -f "$FrontendDist" Write-Host "[OK] Aliases 'unsloth-studio' and 'unsloth-ui' already exist in $PROFILE" -ForegroundColor Green } -# --- cmd.exe: create batch files on PATH so they work from regular terminal --- +# --- cmd.exe: create batch files and ensure they're on PATH --- $BatDir = Join-Path $RepoDir ".venv\Scripts" foreach ($name in @("unsloth-studio", "unsloth-ui")) { $batPath = Join-Path $BatDir "$name.bat" @@ -814,7 +814,17 @@ foreach ($name in @("unsloth-studio", "unsloth-ui")) { Set-Content -Path $batPath -Value "@echo off`r`n`"$VenvPython`" `"$CliScript`" studio -f `"$FrontendDist`" %*" } } -Write-Host "[OK] Batch launchers created in $BatDir (works from cmd.exe when venv is on PATH)" -ForegroundColor Green +# Persist .venv\Scripts to User PATH so commands work in new cmd.exe terminals without activation +$userPath = [Environment]::GetEnvironmentVariable('Path', 'User') +if (-not $userPath -or $userPath -notlike "*$BatDir*") { + if ($userPath) { + [Environment]::SetEnvironmentVariable('Path', "$BatDir;$userPath", 'User') + } else { + [Environment]::SetEnvironmentVariable('Path', "$BatDir", 'User') + } + Write-Host " Persisted $BatDir to User PATH" -ForegroundColor Gray +} +Write-Host "[OK] Batch launchers created (works from any new cmd.exe or PowerShell)" -ForegroundColor Green # ============================================ # Done From 70d1567fe38f4a6fb2ae76a3c0b0eddeb0039d3a Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Sat, 28 Feb 2026 09:01:00 +0000 Subject: [PATCH 24/66] Download GGUF via huggingface_hub instead of llama-server -hf (fixes HTTPS not supported on Windows) --- studio/backend/core/inference/llama_cpp.py | 55 +++++++++++++++++++--- 1 file changed, 48 insertions(+), 7 deletions(-) diff --git a/studio/backend/core/inference/llama_cpp.py b/studio/backend/core/inference/llama_cpp.py index 023edb959b..e0e2eebcf2 100644 --- a/studio/backend/core/inference/llama_cpp.py +++ b/studio/backend/core/inference/llama_cpp.py @@ -199,16 +199,58 @@ class LlamaCppBackend: # Build command based on mode if hf_repo: - hf_spec = f"{hf_repo}:{hf_variant}" if hf_variant else hf_repo + # Download the GGUF file ourselves using huggingface_hub + # (llama-server's -hf flag requires HTTPS/curl which may not + # be available, e.g. Windows builds with -DLLAMA_CURL=OFF) + try: + from huggingface_hub import hf_hub_download + except ImportError: + raise RuntimeError( + "huggingface_hub is required for HF model loading. " + "Install it with: pip install huggingface_hub" + ) + + # Determine the filename from the variant (e.g., "Q4_K_M" -> find matching file) + gguf_filename = None + if hf_variant: + # Try common naming patterns + try: + from huggingface_hub import list_repo_files + files = list_repo_files(hf_repo, token=hf_token) + variant_lower = hf_variant.lower() + for f in files: + if f.endswith(".gguf") and variant_lower in f.lower(): + gguf_filename = f + break + except Exception as e: + logger.warning(f"Could not list repo files: {e}") + + if not gguf_filename: + # Fallback: construct common filename pattern + # e.g., "unsloth/gemma-3-4b-it-GGUF" + "Q4_K_M" -> try model name + repo_name = hf_repo.split("/")[-1].replace("-GGUF", "") + gguf_filename = f"{repo_name}-{hf_variant}.gguf" + + logger.info(f"Downloading GGUF: {hf_repo}/{gguf_filename}") + try: + local_path = hf_hub_download( + repo_id=hf_repo, + filename=gguf_filename, + token=hf_token, + ) + except Exception as e: + raise RuntimeError( + f"Failed to download GGUF file '{gguf_filename}' from {hf_repo}: {e}" + ) + + logger.info(f"GGUF downloaded to: {local_path}") cmd = [ binary, - "-hf", hf_spec, + "-m", local_path, "--port", str(self._port), "-c", str(n_ctx), "-ngl", str(n_gpu_layers), ] - if hf_token: - cmd.extend(["--hf-token", hf_token]) elif gguf_path: if not Path(gguf_path).is_file(): raise FileNotFoundError(f"GGUF file not found: {gguf_path}") @@ -282,9 +324,8 @@ class LlamaCppBackend: self._is_vision = is_vision self._model_identifier = model_identifier - # HF mode: llama-server downloads before becoming healthy — need longer timeout - timeout = 600.0 if hf_repo else 120.0 - if not self._wait_for_health(timeout=timeout): + # Wait for llama-server to become healthy + if not self._wait_for_health(timeout=120.0): self._kill_process() raise RuntimeError( "llama-server failed to start. " From 2a9aba50176e0f81e55041c1206ae3f641e6bc84 Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Sat, 28 Feb 2026 09:14:22 +0000 Subject: [PATCH 25/66] Add vcpkg/curl[ssl] for HTTPS support in llama-server, enable LLAMA_CURL=ON --- .gitignore | 3 +++ setup.ps1 | 42 ++++++++++++++++++++++++++++++++++++++++-- 2 files changed, 43 insertions(+), 2 deletions(-) diff --git a/.gitignore b/.gitignore index 044775e846..633e3041c6 100755 --- a/.gitignore +++ b/.gitignore @@ -28,6 +28,9 @@ unsloth_training_checkpoints/ # llama.cpp build (built by setup.sh, shared with unsloth-zoo export) llama.cpp/ +# vcpkg (installed by setup.ps1 for curl/SSL) +vcpkg/ + # Built binaries (llama-server etc.) bin/ diff --git a/setup.ps1 b/setup.ps1 index 265b13eb35..336d20a777 100644 --- a/setup.ps1 +++ b/setup.ps1 @@ -620,13 +620,45 @@ python "$PSScriptRoot\install_python_stack.py" # Restore ErrorActionPreference after pip/python work $ErrorActionPreference = $prevEAP +# ========================================================================== +# PHASE 3.5: Install vcpkg + curl (for HTTPS support in llama-server) +# ========================================================================== +# llama-server needs curl + OpenSSL to download models from HuggingFace via -hf. +# We use vcpkg to install curl with SSL support. +$VcpkgDir = Join-Path $PSScriptRoot "vcpkg" +$VcpkgExe = Join-Path $VcpkgDir "vcpkg.exe" + +if (-not (Test-Path $VcpkgExe)) { + Write-Host "" + Write-Host "Installing vcpkg (for curl/SSL)..." -ForegroundColor Cyan + if (Test-Path $VcpkgDir) { Remove-Item -Recurse -Force $VcpkgDir } + git clone --depth 1 https://github.com/microsoft/vcpkg.git $VcpkgDir + if ($LASTEXITCODE -eq 0) { + & (Join-Path $VcpkgDir "bootstrap-vcpkg.bat") -disableMetrics + } +} + +if (Test-Path $VcpkgExe) { + # Install curl with SSL support (this also installs OpenSSL as a dependency) + $CurlInstalled = & $VcpkgExe list 2>&1 | Select-String "curl" + if (-not $CurlInstalled) { + Write-Host " Installing curl[ssl] via vcpkg (includes OpenSSL)..." -ForegroundColor Cyan + & $VcpkgExe install "curl[ssl]:x64-windows" + } + $VcpkgToolchainFile = Join-Path $VcpkgDir "scripts\buildsystems\vcpkg.cmake" + Write-Host "[OK] vcpkg ready (toolchain: $VcpkgToolchainFile)" -ForegroundColor Green +} else { + Write-Host "[WARN] vcpkg not available -- llama-server will be built without HTTPS support" -ForegroundColor Yellow + $VcpkgToolchainFile = $null +} + # ========================================================================== # PHASE 4: Build llama.cpp with CUDA for GGUF inference + export # ========================================================================== # Builds in-tree at $REPO/llama.cpp/ (same as setup.sh on Linux). # This directory is already in .gitignore. # We build: -# - llama-server: for GGUF model inference +# - llama-server: for GGUF model inference (with HTTPS if vcpkg/curl available) # - llama-quantize: for GGUF export quantization # Prerequisites (git, cmake, VS Build Tools, CUDA Toolkit) already installed in Phase 1. $LlamaCppDir = Join-Path $PSScriptRoot "llama.cpp" @@ -690,8 +722,14 @@ if (Test-Path $LlamaServerBin) { } # Common flags $CmakeArgs += '-DBUILD_SHARED_LIBS=OFF' - $CmakeArgs += '-DLLAMA_CURL=OFF' $CmakeArgs += '-DCMAKE_POLICY_DEFAULT_CMP0194=NEW' + # HTTPS support via vcpkg curl + OpenSSL + if ($VcpkgToolchainFile -and (Test-Path $VcpkgToolchainFile)) { + $CmakeArgs += "-DCMAKE_TOOLCHAIN_FILE=$VcpkgToolchainFile" + $CmakeArgs += '-DLLAMA_CURL=ON' + } else { + $CmakeArgs += '-DLLAMA_CURL=OFF' + } $CmakeArgs += '-DCMAKE_EXE_LINKER_FLAGS=/NODEFAULTLIB:LIBCMT' # CUDA flags (Unsloth-aligned) $CmakeArgs += '-DGGML_CUDA=ON' From 6d606cd18c57a53adee7da358bc03e597c9129d0 Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Sat, 28 Feb 2026 09:16:19 +0000 Subject: [PATCH 26/66] Simplify: use winget OpenSSL.Dev instead of vcpkg for HTTPS support --- .gitignore | 3 --- setup.ps1 | 67 ++++++++++++++++++++++++++++++++---------------------- 2 files changed, 40 insertions(+), 30 deletions(-) diff --git a/.gitignore b/.gitignore index 633e3041c6..044775e846 100755 --- a/.gitignore +++ b/.gitignore @@ -28,9 +28,6 @@ unsloth_training_checkpoints/ # llama.cpp build (built by setup.sh, shared with unsloth-zoo export) llama.cpp/ -# vcpkg (installed by setup.ps1 for curl/SSL) -vcpkg/ - # Built binaries (llama-server etc.) bin/ diff --git a/setup.ps1 b/setup.ps1 index 336d20a777..7b5413a519 100644 --- a/setup.ps1 +++ b/setup.ps1 @@ -621,35 +621,48 @@ python "$PSScriptRoot\install_python_stack.py" $ErrorActionPreference = $prevEAP # ========================================================================== -# PHASE 3.5: Install vcpkg + curl (for HTTPS support in llama-server) +# PHASE 3.5: Install OpenSSL dev (for HTTPS support in llama-server) # ========================================================================== -# llama-server needs curl + OpenSSL to download models from HuggingFace via -hf. -# We use vcpkg to install curl with SSL support. -$VcpkgDir = Join-Path $PSScriptRoot "vcpkg" -$VcpkgExe = Join-Path $VcpkgDir "vcpkg.exe" +# llama-server needs OpenSSL to download models from HuggingFace via -hf. +# ShiningLight.OpenSSL.Dev includes headers + libs that cmake can find. +$OpenSslAvailable = $false -if (-not (Test-Path $VcpkgExe)) { - Write-Host "" - Write-Host "Installing vcpkg (for curl/SSL)..." -ForegroundColor Cyan - if (Test-Path $VcpkgDir) { Remove-Item -Recurse -Force $VcpkgDir } - git clone --depth 1 https://github.com/microsoft/vcpkg.git $VcpkgDir - if ($LASTEXITCODE -eq 0) { - & (Join-Path $VcpkgDir "bootstrap-vcpkg.bat") -disableMetrics +# Check if OpenSSL dev is already installed (look for include dir) +$OpenSslRoots = @( + 'C:\Program Files\OpenSSL-Win64', + 'C:\Program Files\OpenSSL', + 'C:\OpenSSL-Win64' +) +$OpenSslRoot = $null +foreach ($root in $OpenSslRoots) { + if (Test-Path (Join-Path $root 'include\openssl\ssl.h')) { + $OpenSslRoot = $root + break } } -if (Test-Path $VcpkgExe) { - # Install curl with SSL support (this also installs OpenSSL as a dependency) - $CurlInstalled = & $VcpkgExe list 2>&1 | Select-String "curl" - if (-not $CurlInstalled) { - Write-Host " Installing curl[ssl] via vcpkg (includes OpenSSL)..." -ForegroundColor Cyan - & $VcpkgExe install "curl[ssl]:x64-windows" - } - $VcpkgToolchainFile = Join-Path $VcpkgDir "scripts\buildsystems\vcpkg.cmake" - Write-Host "[OK] vcpkg ready (toolchain: $VcpkgToolchainFile)" -ForegroundColor Green +if ($OpenSslRoot) { + $OpenSslAvailable = $true + Write-Host "[OK] OpenSSL dev found at $OpenSslRoot" -ForegroundColor Green } else { - Write-Host "[WARN] vcpkg not available -- llama-server will be built without HTTPS support" -ForegroundColor Yellow - $VcpkgToolchainFile = $null + Write-Host "" + Write-Host "Installing OpenSSL dev (for HTTPS in llama-server)..." -ForegroundColor Cyan + $HasWinget = $null -ne (Get-Command winget -ErrorAction SilentlyContinue) + if ($HasWinget) { + winget install -e --id ShiningLight.OpenSSL.Dev --accept-package-agreements --accept-source-agreements + # Re-check after install + foreach ($root in $OpenSslRoots) { + if (Test-Path (Join-Path $root 'include\openssl\ssl.h')) { + $OpenSslRoot = $root + $OpenSslAvailable = $true + Write-Host "[OK] OpenSSL dev installed at $OpenSslRoot" -ForegroundColor Green + break + } + } + } + if (-not $OpenSslAvailable) { + Write-Host "[WARN] OpenSSL dev not available -- llama-server will be built without HTTPS" -ForegroundColor Yellow + } } # ========================================================================== @@ -723,10 +736,10 @@ if (Test-Path $LlamaServerBin) { # Common flags $CmakeArgs += '-DBUILD_SHARED_LIBS=OFF' $CmakeArgs += '-DCMAKE_POLICY_DEFAULT_CMP0194=NEW' - # HTTPS support via vcpkg curl + OpenSSL - if ($VcpkgToolchainFile -and (Test-Path $VcpkgToolchainFile)) { - $CmakeArgs += "-DCMAKE_TOOLCHAIN_FILE=$VcpkgToolchainFile" - $CmakeArgs += '-DLLAMA_CURL=ON' + # HTTPS support via OpenSSL + if ($OpenSslAvailable -and $OpenSslRoot) { + $CmakeArgs += "-DOPENSSL_ROOT_DIR=$OpenSslRoot" + $CmakeArgs += '-DLLAMA_OPENSSL=ON' } else { $CmakeArgs += '-DLLAMA_CURL=OFF' } From 77d1378d04499ee8ad05fe6191ff01b1f2a298aa Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Sat, 28 Feb 2026 11:28:15 +0000 Subject: [PATCH 27/66] Remove unused CMP0194 cmake policy (eliminates cmake warning) --- setup.ps1 | 1 - 1 file changed, 1 deletion(-) diff --git a/setup.ps1 b/setup.ps1 index 7b5413a519..f892ca52ee 100644 --- a/setup.ps1 +++ b/setup.ps1 @@ -735,7 +735,6 @@ if (Test-Path $LlamaServerBin) { } # Common flags $CmakeArgs += '-DBUILD_SHARED_LIBS=OFF' - $CmakeArgs += '-DCMAKE_POLICY_DEFAULT_CMP0194=NEW' # HTTPS support via OpenSSL if ($OpenSslAvailable -and $OpenSslRoot) { $CmakeArgs += "-DOPENSSL_ROOT_DIR=$OpenSslRoot" From 148e93cb835d8b340b4462e73a958893b606303e Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Sun, 1 Mar 2026 01:51:04 +0000 Subject: [PATCH 28/66] Auto-enable Windows Long Paths via UAC elevation during setup --- setup.ps1 | 34 ++++++++++++++++++++++++++++++++++ 1 file changed, 34 insertions(+) diff --git a/setup.ps1 b/setup.ps1 index f892ca52ee..0d0ce5dea2 100644 --- a/setup.ps1 +++ b/setup.ps1 @@ -226,6 +226,40 @@ if (-not $HasNvidiaSmi) { } Write-Host "[OK] NVIDIA GPU detected" -ForegroundColor Green +# ============================================ +# 1a.5. Windows Long Paths (required for deep node_modules / Python paths) +# ============================================ +$LongPathsEnabled = $false +try { + $regVal = Get-ItemProperty -Path "HKLM:\SYSTEM\CurrentControlSet\Control\FileSystem" -Name "LongPathsEnabled" -ErrorAction SilentlyContinue + if ($regVal -and $regVal.LongPathsEnabled -eq 1) { + $LongPathsEnabled = $true + } +} catch {} + +if ($LongPathsEnabled) { + Write-Host "[OK] Windows Long Paths enabled" -ForegroundColor Green +} else { + Write-Host "Windows Long Paths not enabled (required for Triton compilation and deep dependency paths)." -ForegroundColor Yellow + Write-Host " Requesting admin access to fix..." -ForegroundColor Yellow + try { + # Spawn an elevated process to set the registry key (triggers UAC prompt) + $proc = Start-Process -FilePath "reg.exe" ` + -ArgumentList 'add "HKLM\SYSTEM\CurrentControlSet\Control\FileSystem" /v LongPathsEnabled /t REG_DWORD /d 1 /f' ` + -Verb RunAs -Wait -PassThru -ErrorAction Stop + if ($proc.ExitCode -eq 0) { + $LongPathsEnabled = $true + Write-Host "[OK] Windows Long Paths enabled (via UAC)" -ForegroundColor Green + } else { + Write-Host "[WARN] Failed to enable Long Paths (exit code: $($proc.ExitCode))" -ForegroundColor Yellow + } + } catch { + Write-Host "[WARN] Could not enable Long Paths (UAC was declined or not available)" -ForegroundColor Yellow + Write-Host " Run this manually in an Admin terminal:" -ForegroundColor Yellow + Write-Host ' reg add "HKLM\SYSTEM\CurrentControlSet\Control\FileSystem" /v LongPathsEnabled /t REG_DWORD /d 1 /f' -ForegroundColor Cyan + } +} + # ============================================ # 1b. Git (required by pip for git+https:// deps and by npm) # ============================================ From 33be2ac0930a543b0dc5e59a77bfc08d5665c1c8 Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Sun, 1 Mar 2026 04:33:04 +0000 Subject: [PATCH 29/66] Add Python 3.12 prerequisite check with auto-install via winget --- setup.ps1 | 37 +++++++++++++++++++++++++++++++++++++ 1 file changed, 37 insertions(+) diff --git a/setup.ps1 b/setup.ps1 index 0d0ce5dea2..05c0f1f0d9 100644 --- a/setup.ps1 +++ b/setup.ps1 @@ -522,6 +522,43 @@ if ($NeedNode) { Write-Host "[OK] Node $(node -v) | npm $(npm -v)" -ForegroundColor Green +# ============================================ +# 1g. Python (>= 3.10, prefer 3.12) +# ============================================ +$HasPython = $null -ne (Get-Command python -ErrorAction SilentlyContinue) +$NeedPython = $true + +if ($HasPython) { + $PyVer = python --version 2>&1 + if ($PyVer -match "(\d+)\.(\d+)") { + $PyMajor = [int]$Matches[1]; $PyMinor = [int]$Matches[2] + if ($PyMajor -eq 3 -and $PyMinor -ge 10) { + Write-Host "[OK] Python $PyVer" -ForegroundColor Green + $NeedPython = $false + } else { + Write-Host "[WARN] Python $PyVer is too old (need >= 3.10)" -ForegroundColor Yellow + } + } +} else { + Write-Host "[WARN] Python not found." -ForegroundColor Yellow +} + +if ($NeedPython) { + Write-Host "Installing Python 3.12 via winget..." -ForegroundColor Cyan + $HasWinget = $null -ne (Get-Command winget -ErrorAction SilentlyContinue) + if ($HasWinget) { + winget install -e --id Python.Python.3.12 --source winget --accept-package-agreements --accept-source-agreements + Refresh-Environment + } + $HasPython = $null -ne (Get-Command python -ErrorAction SilentlyContinue) + if (-not $HasPython) { + Write-Host "[ERROR] Python could not be installed automatically." -ForegroundColor Red + Write-Host " Install Python 3.12 from https://python.org/downloads/" -ForegroundColor Yellow + exit 1 + } + Write-Host "[OK] Python $(python --version)" -ForegroundColor Green +} + Write-Host "" Write-Host "--- System prerequisites ready ---" -ForegroundColor Green Write-Host "" From 204123a212956a7c42a94ff7ba9ebb098167beed Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Sun, 1 Mar 2026 04:55:59 +0000 Subject: [PATCH 30/66] Tighten Python bounds to >= 3.11, < 3.14 (matching setup.sh), only auto-install if missing --- setup.ps1 | 20 ++++++++++---------- 1 file changed, 10 insertions(+), 10 deletions(-) diff --git a/setup.ps1 b/setup.ps1 index 05c0f1f0d9..fd5447b124 100644 --- a/setup.ps1 +++ b/setup.ps1 @@ -523,28 +523,27 @@ if ($NeedNode) { Write-Host "[OK] Node $(node -v) | npm $(npm -v)" -ForegroundColor Green # ============================================ -# 1g. Python (>= 3.10, prefer 3.12) +# 1g. Python (>= 3.11 and < 3.14, matching setup.sh) # ============================================ $HasPython = $null -ne (Get-Command python -ErrorAction SilentlyContinue) -$NeedPython = $true +$PythonOk = $false if ($HasPython) { $PyVer = python --version 2>&1 if ($PyVer -match "(\d+)\.(\d+)") { $PyMajor = [int]$Matches[1]; $PyMinor = [int]$Matches[2] - if ($PyMajor -eq 3 -and $PyMinor -ge 10) { + if ($PyMajor -eq 3 -and $PyMinor -ge 11 -and $PyMinor -lt 14) { Write-Host "[OK] Python $PyVer" -ForegroundColor Green - $NeedPython = $false + $PythonOk = $true } else { - Write-Host "[WARN] Python $PyVer is too old (need >= 3.10)" -ForegroundColor Yellow + Write-Host "[ERROR] Python $PyVer is outside supported range (need >= 3.11 and < 3.14)." -ForegroundColor Red + Write-Host " Install Python 3.12 from https://python.org/downloads/" -ForegroundColor Yellow + exit 1 } } } else { - Write-Host "[WARN] Python not found." -ForegroundColor Yellow -} - -if ($NeedPython) { - Write-Host "Installing Python 3.12 via winget..." -ForegroundColor Cyan + # No Python at all -- install 3.12 + Write-Host "Python not found -- installing Python 3.12 via winget..." -ForegroundColor Yellow $HasWinget = $null -ne (Get-Command winget -ErrorAction SilentlyContinue) if ($HasWinget) { winget install -e --id Python.Python.3.12 --source winget --accept-package-agreements --accept-source-agreements @@ -557,6 +556,7 @@ if ($NeedPython) { exit 1 } Write-Host "[OK] Python $(python --version)" -ForegroundColor Green + $PythonOk = $true } Write-Host "" From e7619a12919bd430b71a19a745abd9f9d6a06f37 Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Mon, 2 Mar 2026 04:04:41 +0000 Subject: [PATCH 31/66] Move llama.cpp clone/build from in-tree to ~/.unsloth/llama.cpp - setup.sh: builds at ~/.unsloth/llama.cpp instead of ./llama.cpp - setup.ps1: builds at %USERPROFILE%/.unsloth/llama.cpp - inference llama_cpp.py: searches ~/.unsloth/ first, in-tree as legacy - export.py: updated comments (unsloth-zoo handles path natively) --- setup.ps1 | 11 ++++--- setup.sh | 9 ++++-- studio/backend/core/export/export.py | 5 ++- studio/backend/core/inference/llama_cpp.py | 36 ++++++++++++---------- 4 files changed, 35 insertions(+), 26 deletions(-) diff --git a/setup.ps1 b/setup.ps1 index fd5447b124..8a2ec56033 100644 --- a/setup.ps1 +++ b/setup.ps1 @@ -739,13 +739,16 @@ if ($OpenSslRoot) { # ========================================================================== # PHASE 4: Build llama.cpp with CUDA for GGUF inference + export # ========================================================================== -# Builds in-tree at $REPO/llama.cpp/ (same as setup.sh on Linux). -# This directory is already in .gitignore. +# Builds at ~/.unsloth/llama.cpp — a single shared location under the user's +# home directory. This is used by both the inference server and the GGUF +# export pipeline (unsloth-zoo). # We build: -# - llama-server: for GGUF model inference (with HTTPS if vcpkg/curl available) +# - llama-server: for GGUF model inference (with HTTPS if OpenSSL available) # - llama-quantize: for GGUF export quantization # Prerequisites (git, cmake, VS Build Tools, CUDA Toolkit) already installed in Phase 1. -$LlamaCppDir = Join-Path $PSScriptRoot "llama.cpp" +$UnslothHome = Join-Path $env:USERPROFILE ".unsloth" +if (-not (Test-Path $UnslothHome)) { New-Item -ItemType Directory -Force $UnslothHome | Out-Null } +$LlamaCppDir = Join-Path $UnslothHome "llama.cpp" $BuildDir = Join-Path $LlamaCppDir "build" $LlamaServerBin = Join-Path $BuildDir "bin\Release\llama-server.exe" diff --git a/setup.sh b/setup.sh index 409df0b798..07ea131125 100755 --- a/setup.sh +++ b/setup.sh @@ -197,11 +197,14 @@ else fi # ── 8. Build llama.cpp binaries for GGUF inference + export ── -# Builds in-tree at $REPO/llama.cpp/. This directory is shared with -# unsloth-zoo's GGUF export pipeline. We build: +# Builds at ~/.unsloth/llama.cpp — a single shared location under the user's +# home directory. This is used by both the inference server and the GGUF +# export pipeline (unsloth-zoo). # - llama-server: for GGUF model inference # - llama-quantize: for GGUF export quantization (symlinked to root for check_llama_cpp()) -LLAMA_CPP_DIR="$SCRIPT_DIR/llama.cpp" +UNSLOTH_HOME="$HOME/.unsloth" +mkdir -p "$UNSLOTH_HOME" +LLAMA_CPP_DIR="$UNSLOTH_HOME/llama.cpp" LLAMA_SERVER_BIN="$LLAMA_CPP_DIR/build/bin/llama-server" rm -rf "$LLAMA_CPP_DIR" { diff --git a/studio/backend/core/export/export.py b/studio/backend/core/export/export.py index bc4e267f75..3d3db2d560 100644 --- a/studio/backend/core/export/export.py +++ b/studio/backend/core/export/export.py @@ -418,9 +418,8 @@ class ExportBackend: pre_existing_ggufs = set(glob.glob(os.path.join(cwd, "*.gguf"))) # Pass absolute path — no os.chdir needed. - # unsloth saves intermediate HF model files into model_save_path, - # while check_llama_cpp("llama.cpp") resolves against cwd (repo root) - # where setup.sh already built llama.cpp with quantizer. + # unsloth saves intermediate HF model files into model_save_path. + # unsloth-zoo's check_llama_cpp() uses ~/.unsloth/llama.cpp by default. model_save_path = os.path.join(abs_save_dir, "model") self.current_model.save_pretrained_gguf( model_save_path, diff --git a/studio/backend/core/inference/llama_cpp.py b/studio/backend/core/inference/llama_cpp.py index e0e2eebcf2..09f8a2fcc9 100644 --- a/studio/backend/core/inference/llama_cpp.py +++ b/studio/backend/core/inference/llama_cpp.py @@ -77,10 +77,11 @@ class LlamaCppBackend: Search order: 1. LLAMA_SERVER_PATH environment variable - 2. ./llama.cpp/build/bin/llama-server (built by setup.sh in-tree) - 3. ~/.unsloth/llama.cpp/build/bin/Release/llama-server (built by setup.ps1 on Windows) - 4. llama-server on PATH (system install) - 5. ./bin/llama-server (legacy: extracted binary) + 2. ~/.unsloth/llama.cpp/build/bin/llama-server (Linux, built by setup.sh) + 3. ~/.unsloth/llama.cpp/build/bin/Release/llama-server.exe (Windows, built by setup.ps1) + 4. ./llama.cpp/build/bin/llama-server (legacy: in-tree build) + 5. llama-server on PATH (system install) + 6. ./bin/llama-server (legacy: extracted binary) """ import os import sys @@ -92,31 +93,34 @@ class LlamaCppBackend: if env_path and Path(env_path).is_file(): return env_path - # Project root: llama_cpp.py → inference/ → core/ → backend/ → studio/ → root - project_root = Path(__file__).resolve().parents[4] + # 2. ~/.unsloth/llama.cpp (primary — setup.sh / setup.ps1 build here) + unsloth_home = Path.home() / ".unsloth" / "llama.cpp" + home_linux = unsloth_home / "build" / "bin" / binary_name + if home_linux.is_file(): + return str(home_linux) - # 2. In-tree llama.cpp build (setup.sh builds here on Linux) + # 3. Windows MSVC build has Release subdir + if sys.platform == "win32": + home_win = unsloth_home / "build" / "bin" / "Release" / binary_name + if home_win.is_file(): + return str(home_win) + + # 4. Legacy: in-tree build (older setup.sh / setup.ps1 versions) + project_root = Path(__file__).resolve().parents[4] build_path = project_root / "llama.cpp" / "build" / "bin" / binary_name if build_path.is_file(): return str(build_path) - - # 3. Windows MSVC build (Release config, in-tree — matches setup.ps1) if sys.platform == "win32": - # In-tree (primary — setup.ps1 now builds here) win_path = project_root / "llama.cpp" / "build" / "bin" / "Release" / binary_name if win_path.is_file(): return str(win_path) - # Legacy: ~/.unsloth (older setup.ps1 versions built here) - home_path = Path.home() / ".unsloth" / "llama.cpp" / "build" / "bin" / "Release" / binary_name - if home_path.is_file(): - return str(home_path) - # 4. System PATH + # 5. System PATH system_path = shutil.which("llama-server") if system_path: return system_path - # 5. Legacy: extracted to bin/ + # 6. Legacy: extracted to bin/ bin_path = project_root / "bin" / binary_name if bin_path.is_file(): return str(bin_path) From 7f339b5c97229b9e08f2cd04e7e3870031934a61 Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Mon, 2 Mar 2026 10:45:09 +0000 Subject: [PATCH 32/66] Patch unsloth-zoo llama_cpp.py and unsloth save.py from windows-support branch --- install_python_stack.py | 13 ++++++++++--- 1 file changed, 10 insertions(+), 3 deletions(-) diff --git a/install_python_stack.py b/install_python_stack.py index 7d3ab66b20..8f2b5fba8c 100644 --- a/install_python_stack.py +++ b/install_python_stack.py @@ -186,20 +186,27 @@ def install_python_stack() -> int: constrain=False, ) - # 6. Patch: override llama_cpp.py with fix from unsloth-zoo main branch + # 6. Patch: override llama_cpp.py with fix from unsloth-zoo feature/llama-cpp-windows-support branch patch_package_file( "unsloth-zoo", os.path.join("unsloth_zoo", "llama_cpp.py"), - "https://raw.githubusercontent.com/unslothai/unsloth-zoo/refs/heads/main/unsloth_zoo/llama_cpp.py", + "https://raw.githubusercontent.com/unslothai/unsloth-zoo/refs/heads/feature/llama-cpp-windows-support/unsloth_zoo/llama_cpp.py", ) - # 7. Patch: override vision.py with fix from unsloth PR #4091 + # 7a. Patch: override vision.py with fix from unsloth PR #4091 patch_package_file( "unsloth", os.path.join("unsloth", "models", "vision.py"), "https://raw.githubusercontent.com/unslothai/unsloth/80e0108a684c882965a02a8ed851e3473c1145ab/unsloth/models/vision.py", ) + # 7b. Patch : override save.py with fix from feature/llama-cpp-windows-support + patch_package_file( + "unsloth", + os.path.join("unsloth", "save.py"), + "https://raw.githubusercontent.com/unslothai/unsloth/refs/heads/feature/llama-cpp-windows-support/unsloth/save.py", + ) + # 8. Studio dependencies pip_install( "Installing studio dependencies", From fe87826d1f5984335b690d16c1a67ce65f2666a6 Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Tue, 3 Mar 2026 09:34:35 +0000 Subject: [PATCH 33/66] fix: make pip check non-fatal and install jedi for Colab compatibility --- setup.sh | 8 +++++++- 1 file changed, 7 insertions(+), 1 deletion(-) diff --git a/setup.sh b/setup.sh index 8314ef6d75..9cc6f527bd 100755 --- a/setup.sh +++ b/setup.sh @@ -191,8 +191,14 @@ install_python_stack() { run_quiet "pip install data-designer deps" pip install --no-cache-dir -c "$SINGLE_ENV_CONSTRAINTS" -r "$SINGLE_ENV_DATA_DESIGNER_DEPS" echo " Installing data-designer..." run_quiet "pip install data-designer" pip install --no-cache-dir --no-deps -c "$SINGLE_ENV_CONSTRAINTS" -r "$SINGLE_ENV_DATA_DESIGNER" + # Colab's bundled IPython 7.34 requires jedi but doesn't ship it + run_quiet "pip install jedi" pip install --no-cache-dir jedi run_quiet "patch single-env metadata" python "$SINGLE_ENV_PATCH" - run_quiet "pip check" pip check + # pip check can flag minor transitive-dependency version mismatches that + # don't actually break anything. Warn instead of aborting. + if ! pip check > /dev/null 2>&1; then + echo "⚠️ pip check reports dependency conflicts (safe to ignore)" + fi echo "✅ Python dependencies installed" } From 5f98d232d032c6e596e5bf286b2d200d8d704560 Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Tue, 3 Mar 2026 17:03:01 +0000 Subject: [PATCH 34/66] fix: align llama-server binary discovery with upstream unsloth-zoo paths --- studio/backend/core/inference/llama_cpp.py | 53 +++++++++++++++++----- 1 file changed, 42 insertions(+), 11 deletions(-) diff --git a/studio/backend/core/inference/llama_cpp.py b/studio/backend/core/inference/llama_cpp.py index 09f8a2fcc9..f47b9132db 100644 --- a/studio/backend/core/inference/llama_cpp.py +++ b/studio/backend/core/inference/llama_cpp.py @@ -76,25 +76,51 @@ class LlamaCppBackend: Locate the llama-server binary. Search order: - 1. LLAMA_SERVER_PATH environment variable - 2. ~/.unsloth/llama.cpp/build/bin/llama-server (Linux, built by setup.sh) - 3. ~/.unsloth/llama.cpp/build/bin/Release/llama-server.exe (Windows, built by setup.ps1) - 4. ./llama.cpp/build/bin/llama-server (legacy: in-tree build) - 5. llama-server on PATH (system install) - 6. ./bin/llama-server (legacy: extracted binary) + 1. LLAMA_SERVER_PATH environment variable (direct path to binary) + 1b. UNSLOTH_LLAMA_CPP_PATH env var (custom llama.cpp install dir) + 2. ~/.unsloth/llama.cpp/llama-server (make build, root dir) + 3. ~/.unsloth/llama.cpp/build/bin/llama-server (cmake build, Linux) + 4. ~/.unsloth/llama.cpp/build/bin/Release/llama-server.exe (cmake build, Windows) + 5. ./llama.cpp/llama-server (legacy: make build, root dir) + 6. ./llama.cpp/build/bin/llama-server (legacy: cmake in-tree build) + 7. llama-server on PATH (system install) + 8. ./bin/llama-server (legacy: extracted binary) """ import os import sys binary_name = "llama-server.exe" if sys.platform == "win32" else "llama-server" - # 1. Env var + # 1. Env var — direct path to binary env_path = os.environ.get("LLAMA_SERVER_PATH") if env_path and Path(env_path).is_file(): return env_path - # 2. ~/.unsloth/llama.cpp (primary — setup.sh / setup.ps1 build here) + # 1b. UNSLOTH_LLAMA_CPP_PATH — custom llama.cpp install directory + custom_llama_cpp = os.environ.get("UNSLOTH_LLAMA_CPP_PATH") + if custom_llama_cpp: + custom_dir = Path(custom_llama_cpp) + # Root dir (make builds) + root_bin = custom_dir / binary_name + if root_bin.is_file(): + return str(root_bin) + # build/bin/ (cmake builds on Linux) + cmake_bin = custom_dir / "build" / "bin" / binary_name + if cmake_bin.is_file(): + return str(cmake_bin) + # build/bin/Release/ (cmake builds on Windows) + if sys.platform == "win32": + win_bin = custom_dir / "build" / "bin" / "Release" / binary_name + if win_bin.is_file(): + return str(win_bin) + + # 2–4. ~/.unsloth/llama.cpp (primary — setup.sh / setup.ps1 build here) unsloth_home = Path.home() / ".unsloth" / "llama.cpp" + # Root dir (make builds copy binaries here) + home_root = unsloth_home / binary_name + if home_root.is_file(): + return str(home_root) + # build/bin/ (cmake builds on Linux) home_linux = unsloth_home / "build" / "bin" / binary_name if home_linux.is_file(): return str(home_linux) @@ -105,8 +131,13 @@ class LlamaCppBackend: if home_win.is_file(): return str(home_win) - # 4. Legacy: in-tree build (older setup.sh / setup.ps1 versions) + # 5–6. Legacy: in-tree build (older setup.sh / setup.ps1 versions) project_root = Path(__file__).resolve().parents[4] + # Root dir (make builds) + root_path = project_root / "llama.cpp" / binary_name + if root_path.is_file(): + return str(root_path) + # build/bin/ (cmake builds) build_path = project_root / "llama.cpp" / "build" / "bin" / binary_name if build_path.is_file(): return str(build_path) @@ -115,12 +146,12 @@ class LlamaCppBackend: if win_path.is_file(): return str(win_path) - # 5. System PATH + # 7. System PATH system_path = shutil.which("llama-server") if system_path: return system_path - # 6. Legacy: extracted to bin/ + # 8. Legacy: extracted to bin/ bin_path = project_root / "bin" / binary_name if bin_path.is_file(): return str(bin_path) From 8c333e2b811778e0c7593f3308e4432548f7eb45 Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Tue, 3 Mar 2026 17:31:59 +0000 Subject: [PATCH 35/66] chore: add cross-platform Python installer with updated unsloth patch URLs --- install_python_stack.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/install_python_stack.py b/install_python_stack.py index 8f2b5fba8c..c9b73034b4 100644 --- a/install_python_stack.py +++ b/install_python_stack.py @@ -190,7 +190,7 @@ def install_python_stack() -> int: patch_package_file( "unsloth-zoo", os.path.join("unsloth_zoo", "llama_cpp.py"), - "https://raw.githubusercontent.com/unslothai/unsloth-zoo/refs/heads/feature/llama-cpp-windows-support/unsloth_zoo/llama_cpp.py", + "https://raw.githubusercontent.com/unslothai/unsloth-zoo/refs/heads/main/unsloth_zoo/llama_cpp.py", ) # 7a. Patch: override vision.py with fix from unsloth PR #4091 @@ -204,7 +204,7 @@ def install_python_stack() -> int: patch_package_file( "unsloth", os.path.join("unsloth", "save.py"), - "https://raw.githubusercontent.com/unslothai/unsloth/refs/heads/feature/llama-cpp-windows-support/unsloth/save.py", + "https://raw.githubusercontent.com/unslothai/unsloth/refs/heads/main/unsloth/save.py", ) # 8. Studio dependencies From 981b618c1c431fa4b42f377716d66bc70b5bba40 Mon Sep 17 00:00:00 2001 From: imagineer99 Date: Tue, 3 Mar 2026 18:37:46 +0000 Subject: [PATCH 36/66] fix: prevent select scroll-lock margin from shifting layout --- studio/frontend/src/components/ui/select.tsx | 36 ++++++++++---------- studio/frontend/src/index.css | 3 ++ 2 files changed, 21 insertions(+), 18 deletions(-) diff --git a/studio/frontend/src/components/ui/select.tsx b/studio/frontend/src/components/ui/select.tsx index 52fe4e5a75..4c500af08e 100644 --- a/studio/frontend/src/components/ui/select.tsx +++ b/studio/frontend/src/components/ui/select.tsx @@ -4,8 +4,8 @@ import { Select as SelectPrimitive } from "radix-ui"; import type * as React from "react"; import { createContext, useContext, useState } from "react"; -import { cn } from "@/lib/utils"; -import { useDialogPortalContainer } from "@/components/ui/dialog"; +import { cn } from "@/lib/utils"; +import { useDialogPortalContainer } from "@/components/ui/dialog"; import { ArrowDown01Icon, ArrowUp01Icon, @@ -92,22 +92,22 @@ function SelectTrigger({ ); } -function SelectContent({ - className, - children, - position = "item-aligned", - align = "center", - container, - ...props -}: React.ComponentProps & { - container?: HTMLElement | null; -}) { - const dialogContainer = useDialogPortalContainer(); - return ( - - & { + container?: HTMLElement | null; +}) { + const dialogContainer = useDialogPortalContainer(); + return ( + + Date: Tue, 3 Mar 2026 18:42:35 +0000 Subject: [PATCH 37/66] Updated README --- README.md | 118 ++++++++++++++++++++++++++++++++++++++++++++++++------ 1 file changed, 106 insertions(+), 12 deletions(-) diff --git a/README.md b/README.md index 3d2810fae5..9656f1c634 100644 --- a/README.md +++ b/README.md @@ -30,28 +30,119 @@ ## Quick Start -### One-command setup +### Prerequisites + +| Requirement | Linux / WSL | Windows | +|---|---|---| +| **GPU** | NVIDIA GPU with working driver | NVIDIA GPU with working driver | +| **Python** | 3.11 – 3.13 | 3.11 – 3.13 | +| **Git** | Pre-installed on most distros | Auto-installed by setup script (via `winget`) | +| **CMake** | Pre-installed or `sudo apt install cmake` | Auto-installed by setup script (via `winget`) | +| **C++ compiler** | `build-essential` (auto-detected) | Visual Studio Build Tools 2022 (auto-installed by setup script) | +| **CUDA Toolkit** | Optional — setup auto-detects `nvcc` | Auto-installed by setup script (version matched to driver) | + +> [!NOTE] +> On **WSL**, the setup script will also run `sudo apt-get install build-essential cmake curl git libcurl4-openssl-dev` so that GGUF export works in non-interactive subprocesses. You may be prompted for your password during setup. + +--- + +### Linux / Windows WSL ```bash +# 1. Clone the repo +git clone https://github.com/unslothai/unsloth-studio.git +cd unsloth-studio + +# 2. Run setup (installs Node, builds frontend, creates .venv, builds llama.cpp) bash setup.sh + +# 3. Open a new terminal (or source your shell rc), then launch: +unsloth-studio -H 0.0.0.0 -p 8000 ``` -This script will: -1. Install **Node.js ≥ 20** via nvm (if needed) -2. Build the frontend to `studio/frontend/dist` -3. Create a Python virtual environment and install all dependencies (including `unsloth`) -4. Register a convenient `unsloth-ui` shell alias +
+What does setup.sh do? -### Launch the studio +1. Installs **Node.js ≥ 20** via nvm (if needed) +2. Runs `npm install && npm run build` for the React frontend +3. Detects the best **Python 3.11 – 3.13** on your system and creates a `.venv` +4. Installs all Python dependencies (unsloth, PyTorch with CUDA, triton kernels, etc.) +5. On **WSL**: pre-installs build dependencies via `apt-get` +6. Clones and builds **llama.cpp** at `~/.unsloth/llama.cpp` (GPU-accelerated if CUDA is found) +7. Registers `unsloth-studio` and `unsloth-ui` shell aliases in your shell rc (bash, zsh, fish, or ksh) + +
+ +--- + +### Windows (Native) + +> [!IMPORTANT] +> Requires an **NVIDIA GPU** — CPU-only machines are not supported on Windows. + +```powershell +# 1. Clone the repo +git clone https://github.com/unslothai/unsloth-studio.git +cd unsloth-studio + +# 2. Run setup (Right-click → "Run with PowerShell", or from a terminal): +.\setup.bat +# Or directly: +powershell -ExecutionPolicy Bypass -File setup.ps1 +``` + +After setup completes, **open a new terminal** and run: + +```powershell +# PowerShell +unsloth-studio -H 0.0.0.0 -p 8000 + +# Or cmd.exe +unsloth-studio -H 0.0.0.0 -p 8000 +``` + +
+What does setup.ps1 do? + +1. Enables **Windows Long Paths** (required for deep dependency trees — prompts for UAC) +2. Auto-installs missing system tools via `winget`: **Git**, **CMake**, **Visual Studio Build Tools 2022**, **CUDA Toolkit** (version-matched to your driver), **Node.js LTS**, **Python 3.12**, **OpenSSL dev** +3. Builds the React frontend (`npm install && npm run build`) +4. Creates a `.venv` and installs all Python dependencies (including CUDA-enabled PyTorch from the official index) +5. Sets `TORCHINDUCTOR_CACHE_DIR=C:\tc` to avoid Windows MAX_PATH issues with Triton +6. Clones and builds **llama.cpp** at `%USERPROFILE%\.unsloth\llama.cpp` with CUDA + Visual Studio +7. Registers `unsloth-studio` and `unsloth-ui` commands in both PowerShell profile and `cmd.exe` (via batch files on PATH) + +
+ +--- + +### Google Colab + +The setup script auto-detects Colab and installs everything into the existing system Python (no venv): + +```python +!bash setup.sh +``` + +--- + +### Launching the Studio + +After setup on any platform, the command is the same: ```bash -# After setup, open a new terminal (or source ~/.bashrc), then inside your working directory: -unsloth-ui -H 0.0.0.0 -p 8000 +unsloth-studio -H 0.0.0.0 -p 8000 ``` -On **first launch**, a one-time setup token is printed to the console. Use it in the browser to create your admin account. +| Flag | Description | +|---|---| +| `-H` / `--host` | Bind address (`0.0.0.0` for all interfaces, `127.0.0.1` for local only) | +| `-p` / `--port` | Port number (default: `8000`) | -As this repo is in continuous development, please make sure to run the setup.sh file everytime you pull new changes from the repo. +On **first launch**, a one-time setup token is printed to the console. Open the URL shown in your browser and use this token to create your admin account. + +> [!TIP] +> This repo is in active development. After pulling new changes, **always re-run the setup script** (`bash setup.sh` or `.\setup.bat`) to pick up dependency and build updates. ## API Reference @@ -105,7 +196,10 @@ new-ui-prototype/ │ ├── export.py │ ├── ui.py │ └── studio.py -├── setup.sh # One-command bootstrap script +├── setup.sh # Bootstrap script (Linux / WSL / Colab) +├── setup.ps1 # Bootstrap script (Windows native) +├── setup.bat # Wrapper to launch setup.ps1 via double-click +├── install_python_stack.py # Cross-platform Python dependency installer └── studio/ ├── backend/ │ ├── main.py # FastAPI app & middleware From 7bc5bc26f6a92eb51e7ae97f30049055bca39eaf Mon Sep 17 00:00:00 2001 From: imagineer99 Date: Tue, 3 Mar 2026 20:15:23 +0000 Subject: [PATCH 38/66] fix: sanitize dataset script errors and persist training start error --- .../hf-dataset-subset-split-selectors.tsx | 2 +- .../training/hooks/use-training-actions.ts | 22 ++++++++-- .../training/stores/training-runtime-store.ts | 1 - .../src/hooks/use-hf-dataset-splits.ts | 44 ++++++++++++++++++- 4 files changed, 63 insertions(+), 6 deletions(-) diff --git a/studio/frontend/src/features/training/components/hf-dataset-subset-split-selectors.tsx b/studio/frontend/src/features/training/components/hf-dataset-subset-split-selectors.tsx index d21fd55ff4..148341e4de 100644 --- a/studio/frontend/src/features/training/components/hf-dataset-subset-split-selectors.tsx +++ b/studio/frontend/src/features/training/components/hf-dataset-subset-split-selectors.tsx @@ -106,7 +106,7 @@ export function HfDatasetSubsetSplitSelectors({ : "rounded-lg border border-amber-200 bg-amber-50 px-3.5 py-2.5 text-xs text-amber-700 dark:border-amber-800 dark:bg-amber-950 dark:text-amber-400" } > - Could not fetch dataset splits: {error} + {error} )} diff --git a/studio/frontend/src/features/training/hooks/use-training-actions.ts b/studio/frontend/src/features/training/hooks/use-training-actions.ts index bfe035cf4a..1f3db35ce5 100644 --- a/studio/frontend/src/features/training/hooks/use-training-actions.ts +++ b/studio/frontend/src/features/training/hooks/use-training-actions.ts @@ -16,6 +16,19 @@ const ROLE_REMAP: Record> = { sharegpt: { user: "human", assistant: "gpt", system: "system" }, }; +function normalizeTrainingStartError(message: string): string { + const normalized = message.toLowerCase(); + const isLegacyDatasetScriptError = + normalized.includes("failed to check dataset format") && + normalized.includes("dataset scripts are no longer supported"); + + if (isLegacyDatasetScriptError) { + return "This Hub dataset relies on a legacy custom script and isn’t supported in this training flow."; + } + + return message; +} + export function useTrainingActions() { const isStarting = useTrainingRuntimeStore((state) => state.isStarting); const startError = useTrainingRuntimeStore((state) => state.startError); @@ -79,7 +92,9 @@ export function useTrainingActions() { const response = await startTraining(payload); if (response.status === "error") { - runtimeStore.setStartError(response.error || response.message); + const rawMessage = response.error || response.message; + const safeMessage = normalizeTrainingStartError(rawMessage); + runtimeStore.setStartError(safeMessage); runtimeStore.setStarting(false); return false; } @@ -88,9 +103,10 @@ export function useTrainingActions() { await syncTrainingRuntimeFromBackend(); return true; } catch (error) { - const message = + const rawMessage = error instanceof Error ? error.message : "Failed to start training"; - runtimeStore.setStartError(message); + const safeMessage = normalizeTrainingStartError(rawMessage); + runtimeStore.setStartError(safeMessage); runtimeStore.setStarting(false); return false; } diff --git a/studio/frontend/src/features/training/stores/training-runtime-store.ts b/studio/frontend/src/features/training/stores/training-runtime-store.ts index 7d5c86549b..41b0fcca07 100644 --- a/studio/frontend/src/features/training/stores/training-runtime-store.ts +++ b/studio/frontend/src/features/training/stores/training-runtime-store.ts @@ -187,7 +187,6 @@ export const useTrainingRuntimeStore = create()((set) => ( evalEnabled: payload.eval_enabled ?? state.evalEnabled, message: payload.message, error: payload.error, - startError: null, currentStep: typeof detailStep === "number" ? Math.max(detailStep, 0) : state.currentStep, totalSteps: diff --git a/studio/frontend/src/hooks/use-hf-dataset-splits.ts b/studio/frontend/src/hooks/use-hf-dataset-splits.ts index 7b8e4906ec..bc769b5b4b 100644 --- a/studio/frontend/src/hooks/use-hf-dataset-splits.ts +++ b/studio/frontend/src/hooks/use-hf-dataset-splits.ts @@ -35,6 +35,37 @@ export interface HfDatasetSplitsResult { const HF_SPLITS_API = "https://datasets-server.huggingface.co/splits"; +function normalizeDatasetSplitsError(message: string): string { + const normalized = message.toLowerCase(); + + // datasets-server returns technical script/runtime details for legacy datasets. + if ( + normalized.includes("dataset scripts are no longer supported") || + normalized.includes("runs arbitrary python code") || + normalized.includes(".py") + ) { + return "We can’t load subset/split options for this Hub dataset because it relies on a legacy custom script."; + } + + if ( + normalized.includes("unauthorized") || + normalized.includes("forbidden") || + normalized.includes("access token") || + normalized.includes("private") || + normalized.includes("gated") || + normalized.includes("401") || + normalized.includes("403") + ) { + return "Unable to load dataset splits. This dataset may be private or gated. Add a Hugging Face token with access and try again."; + } + + if (normalized.includes("not found") || normalized.includes("404")) { + return "Dataset not found. Check the dataset name and try again."; + } + + return "Unable to load dataset split options for this dataset."; +} + // --------------------------------------------------------------------------- // Hook // --------------------------------------------------------------------------- @@ -101,7 +132,18 @@ export function useHfDatasetSplits( }) .catch((err) => { if (!controller.signal.aborted) { - setError(err.message || "Failed to fetch dataset splits"); + const rawErrorMessage = + err instanceof Error + ? err.message + : typeof err === "string" + ? err + : "Failed to fetch dataset splits"; + console.warn("[useHfDatasetSplits] Failed to fetch dataset splits", { + datasetName, + message: rawErrorMessage, + error: err, + }); + setError(normalizeDatasetSplitsError(rawErrorMessage)); setEntries([]); } }) From 84ad61bfa24f38e3ee24c942e2e11c7f7302b12f Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Wed, 4 Mar 2026 06:13:47 +0000 Subject: [PATCH 39/66] Remove overly broad .py check from dataset error normalization --- studio/frontend/src/hooks/use-hf-dataset-splits.ts | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/studio/frontend/src/hooks/use-hf-dataset-splits.ts b/studio/frontend/src/hooks/use-hf-dataset-splits.ts index bc769b5b4b..cda2fcbc34 100644 --- a/studio/frontend/src/hooks/use-hf-dataset-splits.ts +++ b/studio/frontend/src/hooks/use-hf-dataset-splits.ts @@ -41,8 +41,7 @@ function normalizeDatasetSplitsError(message: string): string { // datasets-server returns technical script/runtime details for legacy datasets. if ( normalized.includes("dataset scripts are no longer supported") || - normalized.includes("runs arbitrary python code") || - normalized.includes(".py") + normalized.includes("runs arbitrary python code") ) { return "We can’t load subset/split options for this Hub dataset because it relies on a legacy custom script."; } From 3c4bf80cc26567babbf5029610c982a0a898a388 Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Wed, 4 Mar 2026 06:42:37 +0000 Subject: [PATCH 40/66] fix: cast URL image columns to HF Image() type in VLM conversion --- studio/backend/utils/datasets/format_conversion.py | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/studio/backend/utils/datasets/format_conversion.py b/studio/backend/utils/datasets/format_conversion.py index 6436e7a82a..4e6a6d9919 100644 --- a/studio/backend/utils/datasets/format_conversion.py +++ b/studio/backend/utils/datasets/format_conversion.py @@ -254,8 +254,15 @@ def convert_to_vlm_format( list: List of dicts with 'messages' field """ from PIL import Image + from datasets import Image as datasets_Image from .vlm_processing import generate_smart_vlm_instruction + # Cast string image columns (URLs or local paths) to HF Image() type + # so HuggingFace handles downloading, decoding, and caching transparently. + sample_value = next(iter(dataset))[image_column] + if isinstance(sample_value, str): + dataset = dataset.cast_column(image_column, datasets_Image()) + # Generate smart instruction if not provided if instruction is None: instruction_info = generate_smart_vlm_instruction( From c6d82a6fadfcd2d4ca4b10d4f68d8f1767b6a799 Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Wed, 4 Mar 2026 06:42:48 +0000 Subject: [PATCH 41/66] fix: abort training pipeline on dataset conversion failure --- studio/backend/core/training/trainer.py | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/studio/backend/core/training/trainer.py b/studio/backend/core/training/trainer.py index e4e7a474be..c3376c10a2 100644 --- a/studio/backend/core/training/trainer.py +++ b/studio/backend/core/training/trainer.py @@ -468,6 +468,14 @@ class UnslothTrainer: print("Stopped during dataset formatting\n") return None + # Abort if dataset formatting/conversion failed + if not dataset_info.get("success", True): + errors = dataset_info.get("errors", []) + error_msg = "; ".join(errors) if errors else "Dataset formatting failed" + logger.error(f"Dataset conversion failed: {error_msg}") + self._update_progress(error=error_msg) + return None + self._update_progress(status_message=f"Dataset formatted and ready for training") print(f"Dataset formatted successfully\n") From 060e2e1abfdb1647aa07eb390307dba2668883db Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Wed, 4 Mar 2026 07:39:35 +0000 Subject: [PATCH 42/66] test: add URL image loading comparison script --- studio/tests/test_url_image_loading.py | 132 +++++++++++++++++++++++++ 1 file changed, 132 insertions(+) create mode 100644 studio/tests/test_url_image_loading.py diff --git a/studio/tests/test_url_image_loading.py b/studio/tests/test_url_image_loading.py new file mode 100644 index 0000000000..4899bab8e3 --- /dev/null +++ b/studio/tests/test_url_image_loading.py @@ -0,0 +1,132 @@ +""" +Reproduce: VLM URL image loading with HF datasets. +Tests cast_column(Image()) vs manual download approaches. +Dataset: google-research-datasets/conceptual_captions (subset: labeled) +""" +from datasets import load_dataset, Image as datasets_Image, Dataset +from PIL import Image as PILImage +from io import BytesIO +from itertools import islice +import time + +DATASET = "google-research-datasets/conceptual_captions" +SUBSET = "labeled" +SPLIT = "train" +N_SAMPLES = 20 # small slice for testing + +print("=" * 60) +print("Loading dataset (streaming, first N samples)...") +print("=" * 60) +ds = load_dataset(DATASET, name=SUBSET, split=SPLIT, streaming=True) +rows = list(islice(ds, N_SAMPLES)) +dataset = Dataset.from_list(rows) + +print(f"Loaded {len(dataset)} samples") +print(f"Columns: {dataset.column_names}") +print(f"First image_url: {dataset[0]['image_url'][:100]}...") +print() + +# ─── Test 1: cast_column(Image()) — what we tried ─── +print("=" * 60) +print("TEST 1: cast_column(Image()) approach") +print("=" * 60) +try: + ds_cast = dataset.cast_column("image_url", datasets_Image()) + success, fail = 0, 0 + t0 = time.time() + for i, sample in enumerate(ds_cast): + try: + img = sample["image_url"] + if img is not None: + print(f" [{i}] OK — {img.size} {img.mode}") + success += 1 + else: + print(f" [{i}] None returned") + fail += 1 + except Exception as e: + print(f" [{i}] FAILED: {type(e).__name__}: {str(e)[:80]}") + fail += 1 + elapsed = time.time() - t0 + print(f"\nResult: {success} ok, {fail} failed, {elapsed:.1f}s") +except Exception as e: + print(f"CRASHED during iteration: {type(e).__name__}: {str(e)[:120]}") +print() + +# ─── Test 2: Manual download with requests.Session ─── +print("=" * 60) +print("TEST 2: requests.Session() approach") +print("=" * 60) +try: + import requests + session = requests.Session() + success, fail = 0, 0 + t0 = time.time() + for i, sample in enumerate(dataset): + url = sample["image_url"] + try: + resp = session.get(url, timeout=10) + resp.raise_for_status() + img = PILImage.open(BytesIO(resp.content)).convert("RGB") + print(f" [{i}] OK — {img.size} {img.mode}") + success += 1 + except Exception as e: + print(f" [{i}] FAILED: {type(e).__name__}: {str(e)[:80]}") + fail += 1 + elapsed = time.time() - t0 + print(f"\nResult: {success} ok, {fail} failed, {elapsed:.1f}s") +except Exception as e: + print(f"CRASHED: {type(e).__name__}: {str(e)[:120]}") +print() + +# ─── Test 3: urllib (stdlib) ─── +print("=" * 60) +print("TEST 3: urllib approach (stdlib)") +print("=" * 60) +try: + from urllib.request import urlopen, Request + success, fail = 0, 0 + t0 = time.time() + for i, sample in enumerate(dataset): + url = sample["image_url"] + try: + req = Request(url, headers={"User-Agent": "Mozilla/5.0"}) + with urlopen(req, timeout=10) as resp: + img = PILImage.open(BytesIO(resp.read())).convert("RGB") + print(f" [{i}] OK — {img.size} {img.mode}") + success += 1 + except Exception as e: + print(f" [{i}] FAILED: {type(e).__name__}: {str(e)[:80]}") + fail += 1 + elapsed = time.time() - t0 + print(f"\nResult: {success} ok, {fail} failed, {elapsed:.1f}s") +except Exception as e: + print(f"CRASHED: {type(e).__name__}: {str(e)[:120]}") +print() + +# ─── Test 4: fsspec directly with expand=True ─── +print("=" * 60) +print("TEST 4: fsspec.open() with expand=True") +print("=" * 60) +try: + import fsspec + success, fail = 0, 0 + t0 = time.time() + for i, sample in enumerate(dataset): + url = sample["image_url"] + try: + with fsspec.open(url, "rb", expand=True) as f: + img = PILImage.open(BytesIO(f.read())).convert("RGB") + print(f" [{i}] OK — {img.size} {img.mode}") + success += 1 + except Exception as e: + print(f" [{i}] FAILED: {type(e).__name__}: {str(e)[:80]}") + fail += 1 + elapsed = time.time() - t0 + print(f"\nResult: {success} ok, {fail} failed, {elapsed:.1f}s") +except Exception as e: + print(f"CRASHED: {type(e).__name__}: {str(e)[:120]}") + +print() +print("=" * 60) +print("DONE — compare success rates and timing above") +print("=" * 60) From 9cbd3d44a7079fa837ee69bd6e8045133e32fcc9 Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Wed, 4 Mar 2026 07:50:55 +0000 Subject: [PATCH 43/66] fix: use fsspec for URL image downloads with per-sample error handling --- .../utils/datasets/format_conversion.py | 50 +++++++++++++------ 1 file changed, 36 insertions(+), 14 deletions(-) diff --git a/studio/backend/utils/datasets/format_conversion.py b/studio/backend/utils/datasets/format_conversion.py index 4e6a6d9919..37bd3c103c 100644 --- a/studio/backend/utils/datasets/format_conversion.py +++ b/studio/backend/utils/datasets/format_conversion.py @@ -254,15 +254,8 @@ def convert_to_vlm_format( list: List of dicts with 'messages' field """ from PIL import Image - from datasets import Image as datasets_Image from .vlm_processing import generate_smart_vlm_instruction - # Cast string image columns (URLs or local paths) to HF Image() type - # so HuggingFace handles downloading, decoding, and caching transparently. - sample_value = next(iter(dataset))[image_column] - if isinstance(sample_value, str): - dataset = dataset.cast_column(image_column, datasets_Image()) - # Generate smart instruction if not provided if instruction is None: instruction_info = generate_smart_vlm_instruction( @@ -288,12 +281,17 @@ def convert_to_vlm_format( def _convert_single_sample(sample): """Convert a single sample to VLM format.""" - # Get image (might be PIL Image or path) + # Get image (might be PIL Image, local path, or URL) image_data = sample[image_column] - # Handle image paths if isinstance(image_data, str): - image_data = Image.open(image_data).convert("RGB") + if image_data.startswith(("http://", "https://")): + import fsspec + from io import BytesIO + with fsspec.open(image_data, "rb", expand=True) as f: + image_data = Image.open(BytesIO(f.read())).convert("RGB") + else: + image_data = Image.open(image_data).convert("RGB") # Get text text_data = sample[text_column] @@ -324,11 +322,35 @@ def convert_to_vlm_format( # Return dict with messages return {"messages": messages} - # Use list comprehension and return the LIST directly - print(f"🔄 Converting {len(dataset)} samples to VLM format...") - converted_list = [_convert_single_sample(sample) for sample in dataset] + # Convert samples, skipping any with broken/unreachable images + total = len(dataset) + print(f"🔄 Converting {total} samples to VLM format...") + converted_list = [] + failed_count = 0 + for sample in dataset: + try: + converted_list.append(_convert_single_sample(sample)) + except Exception as e: + failed_count += 1 - print(f"✅ Converted {len(converted_list)} samples") + if failed_count > 0: + fail_rate = failed_count / total + print(f"⚠️ Skipped {failed_count}/{total} ({fail_rate:.0%}) samples with broken/unreachable images") + + if fail_rate >= 0.3: + raise ValueError( + f"{fail_rate:.0%} of images failed to download ({failed_count}/{total}). " + "This dataset has too many broken or unreachable image URLs to be usable for training. " + "Consider using a dataset with embedded images instead." + ) + + if len(converted_list) == 0: + raise ValueError( + f"All {total} samples failed during VLM conversion — no usable images found. " + "This dataset may contain only image URLs that are no longer accessible." + ) + + print(f"✅ Converted {len(converted_list)}/{total} samples") # Return list, NOT Dataset return converted_list From 7804a4db2e308eedf6a50f631670c6d879f77a72 Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Wed, 4 Mar 2026 08:05:40 +0000 Subject: [PATCH 44/66] fix: add early probe to fail fast on datasets with too many broken image URLs --- .../utils/datasets/format_conversion.py | 32 +++++++++++++------ 1 file changed, 23 insertions(+), 9 deletions(-) diff --git a/studio/backend/utils/datasets/format_conversion.py b/studio/backend/utils/datasets/format_conversion.py index 37bd3c103c..f9da8b1a1e 100644 --- a/studio/backend/utils/datasets/format_conversion.py +++ b/studio/backend/utils/datasets/format_conversion.py @@ -322,28 +322,42 @@ def convert_to_vlm_format( # Return dict with messages return {"messages": messages} - # Convert samples, skipping any with broken/unreachable images + # Convert samples, skipping any with broken/unreachable images. + # For URL-based datasets, check the first PROBE_SIZE samples early to + # fail fast if too many images are broken, before downloading millions. + PROBE_SIZE = 5000 + MAX_FAIL_RATE = 0.3 + total = len(dataset) + has_urls = isinstance(next(iter(dataset))[image_column], str) + probe_needed = has_urls and total > PROBE_SIZE + print(f"🔄 Converting {total} samples to VLM format...") converted_list = [] failed_count = 0 - for sample in dataset: + + for i, sample in enumerate(dataset): try: converted_list.append(_convert_single_sample(sample)) except Exception as e: failed_count += 1 + # Early exit check after probing the first batch + if probe_needed and (i + 1) == PROBE_SIZE: + fail_rate = failed_count / PROBE_SIZE + if fail_rate >= MAX_FAIL_RATE: + raise ValueError( + f"{fail_rate:.0%} of the first {PROBE_SIZE} images failed to download " + f"({failed_count}/{PROBE_SIZE}). " + "This dataset has too many broken or unreachable image URLs. " + "Consider using a dataset with embedded images instead." + ) + print(f"✅ Probe passed: {failed_count}/{PROBE_SIZE} ({fail_rate:.0%}) failures in first batch, continuing...") + if failed_count > 0: fail_rate = failed_count / total print(f"⚠️ Skipped {failed_count}/{total} ({fail_rate:.0%}) samples with broken/unreachable images") - if fail_rate >= 0.3: - raise ValueError( - f"{fail_rate:.0%} of images failed to download ({failed_count}/{total}). " - "This dataset has too many broken or unreachable image URLs to be usable for training. " - "Consider using a dataset with embedded images instead." - ) - if len(converted_list) == 0: raise ValueError( f"All {total} samples failed during VLM conversion — no usable images found. " From cc11f066b156c4da1bd30c8206f0f8ce096d51a5 Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Wed, 4 Mar 2026 13:30:27 +0000 Subject: [PATCH 45/66] feat: add tqdm progress bar to VLM conversion and download benchmark test --- .../utils/datasets/format_conversion.py | 9 +++- studio/tests/test_url_download_benchmark.py | 54 +++++++++++++++++++ 2 files changed, 62 insertions(+), 1 deletion(-) create mode 100644 studio/tests/test_url_download_benchmark.py diff --git a/studio/backend/utils/datasets/format_conversion.py b/studio/backend/utils/datasets/format_conversion.py index f9da8b1a1e..e784e8e645 100644 --- a/studio/backend/utils/datasets/format_conversion.py +++ b/studio/backend/utils/datasets/format_conversion.py @@ -332,20 +332,26 @@ def convert_to_vlm_format( has_urls = isinstance(next(iter(dataset))[image_column], str) probe_needed = has_urls and total > PROBE_SIZE + from tqdm import tqdm + print(f"🔄 Converting {total} samples to VLM format...") converted_list = [] failed_count = 0 - for i, sample in enumerate(dataset): + pbar = tqdm(dataset, total=total, desc="Converting VLM samples", unit="sample") + for i, sample in enumerate(pbar): try: converted_list.append(_convert_single_sample(sample)) except Exception as e: failed_count += 1 + pbar.set_postfix(ok=len(converted_list), failed=failed_count, refresh=False) + # Early exit check after probing the first batch if probe_needed and (i + 1) == PROBE_SIZE: fail_rate = failed_count / PROBE_SIZE if fail_rate >= MAX_FAIL_RATE: + pbar.close() raise ValueError( f"{fail_rate:.0%} of the first {PROBE_SIZE} images failed to download " f"({failed_count}/{PROBE_SIZE}). " @@ -353,6 +359,7 @@ def convert_to_vlm_format( "Consider using a dataset with embedded images instead." ) print(f"✅ Probe passed: {failed_count}/{PROBE_SIZE} ({fail_rate:.0%}) failures in first batch, continuing...") + pbar.close() if failed_count > 0: fail_rate = failed_count / total diff --git a/studio/tests/test_url_download_benchmark.py b/studio/tests/test_url_download_benchmark.py new file mode 100644 index 0000000000..e29a8e05b6 --- /dev/null +++ b/studio/tests/test_url_download_benchmark.py @@ -0,0 +1,54 @@ +""" +Benchmark: fsspec URL image download throughput at different dataset sizes. +Dataset: google-research-datasets/conceptual_captions (subset: labeled) + +Tests sizes: 100, 200, 300, 500, 1000, 1500, 2000 +Reports: time, success/fail rate, throughput (images/sec) +""" +from datasets import load_dataset, Dataset +from PIL import Image as PILImage +from io import BytesIO +from itertools import islice +import fsspec +import time + +DATASET = "google-research-datasets/conceptual_captions" +SUBSET = "labeled" +SPLIT = "train" +SIZES = [100, 200, 300, 500, 1000, 1500, 2000] + +# Load the max we need in one go +max_size = max(SIZES) +print(f"Loading {max_size} samples from {DATASET} (streaming)...") +ds = load_dataset(DATASET, name=SUBSET, split=SPLIT, streaming=True) +rows = list(islice(ds, max_size)) +full_dataset = Dataset.from_list(rows) +print(f"Loaded {len(full_dataset)} samples") +print(f"Columns: {full_dataset.column_names}") +print() + +print(f"{'Size':>6} | {'Time':>8} | {'OK':>6} | {'Fail':>6} | {'Fail%':>6} | {'img/s':>7}") +print("-" * 55) + +for size in SIZES: + dataset = full_dataset.select(range(size)) + success, fail = 0, 0 + t0 = time.time() + + for sample in dataset: + url = sample["image_url"] + try: + with fsspec.open(url, "rb", expand=True) as f: + img = PILImage.open(BytesIO(f.read())).convert("RGB") + success += 1 + except Exception: + fail += 1 + + elapsed = time.time() - t0 + fail_pct = (fail / size) * 100 + throughput = success / elapsed if elapsed > 0 else 0 + + print(f"{size:>6} | {elapsed:>7.1f}s | {success:>6} | {fail:>6} | {fail_pct:>5.1f}% | {throughput:>6.1f}/s") + +print() +print("Done.") From 760f4a360981f2b32ebb0cdc12565bd76644ae1d Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Wed, 4 Mar 2026 14:30:11 +0000 Subject: [PATCH 46/66] test: add parallel download benchmark with ThreadPoolExecutor --- studio/tests/test_url_parallel_benchmark.py | 79 +++++++++++++++++++++ 1 file changed, 79 insertions(+) create mode 100644 studio/tests/test_url_parallel_benchmark.py diff --git a/studio/tests/test_url_parallel_benchmark.py b/studio/tests/test_url_parallel_benchmark.py new file mode 100644 index 0000000000..a0d160136c --- /dev/null +++ b/studio/tests/test_url_parallel_benchmark.py @@ -0,0 +1,79 @@ +""" +Benchmark: parallel fsspec URL image downloads with ThreadPoolExecutor. +Tests different worker counts to find optimal parallelism. +Dataset: google-research-datasets/conceptual_captions (subset: labeled) +""" +from datasets import load_dataset, Dataset +from PIL import Image as PILImage +from io import BytesIO +from itertools import islice +from concurrent.futures import ThreadPoolExecutor, as_completed +import fsspec +import time +import os + +DATASET = "google-research-datasets/conceptual_captions" +SUBSET = "labeled" +SPLIT = "train" +N_SAMPLES = 500 + +# safe_num_proc formula from studio/backend/utils/hardware/hardware.py +cpu_count = os.cpu_count() +safe_workers = max(1, cpu_count // 3) +print(f"CPU count: {cpu_count}, safe_num_proc: {safe_workers}") + +WORKER_COUNTS = [1, 4, 8, 16, 32, safe_workers] +# Deduplicate and sort +WORKER_COUNTS = sorted(set(WORKER_COUNTS)) + +print(f"Loading {N_SAMPLES} samples from {DATASET} (streaming)...") +ds = load_dataset(DATASET, name=SUBSET, split=SPLIT, streaming=True) +rows = list(islice(ds, N_SAMPLES)) +dataset = Dataset.from_list(rows) +urls = [row["image_url"] for row in dataset] +print(f"Loaded {len(urls)} URLs") +print() + + +def download_single(url): + """Download a single image URL using fsspec. Returns PIL image or raises.""" + with fsspec.open(url, "rb", expand=True) as f: + img = PILImage.open(BytesIO(f.read())).convert("RGB") + return img + + +print(f"{'Workers':>8} | {'Time':>8} | {'OK':>6} | {'Fail':>6} | {'Fail%':>6} | {'img/s':>7} | {'Speedup':>8}") +print("-" * 70) + +baseline_throughput = None + +for n_workers in WORKER_COUNTS: + success, fail = 0, 0 + t0 = time.time() + + with ThreadPoolExecutor(max_workers=n_workers) as pool: + futures = {pool.submit(download_single, url): url for url in urls} + for future in as_completed(futures): + try: + img = future.result(timeout=30) + success += 1 + except Exception: + fail += 1 + + elapsed = time.time() - t0 + fail_pct = (fail / N_SAMPLES) * 100 + throughput = success / elapsed if elapsed > 0 else 0 + + if baseline_throughput is None: + baseline_throughput = throughput + speedup = throughput / baseline_throughput if baseline_throughput > 0 else 0 + + label = f"{n_workers}" + if n_workers == safe_workers: + label += "*" # mark the safe_num_proc value + + print(f"{label:>8} | {elapsed:>7.1f}s | {success:>6} | {fail:>6} | {fail_pct:>5.1f}% | {throughput:>6.1f}/s | {speedup:>7.1f}x") + +print() +print("* = safe_num_proc value") +print("Done.") From 02b17ec6d9fb1b8a4c9c4b186a77dd7f4e84fc50 Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Wed, 4 Mar 2026 21:48:40 +0000 Subject: [PATCH 47/66] feat: add index range dataset slicing to studio training page Add Start/End index inputs under Advanced in the dataset card, allowing users to slice a dataset by row range before training. Wired end-to-end: frontend store, API payload, backend Pydantic model, and trainer dataset loading (inclusive on both ends). --- studio/backend/core/training/trainer.py | 16 +- studio/backend/core/training/training.py | 6 +- studio/backend/models/training.py | 2 + studio/backend/routes/training.py | 2 + .../studio/sections/dataset-section.tsx | 141 ++++++++++++------ .../src/features/training/api/mappers.ts | 11 ++ .../training/stores/training-config-store.ts | 12 +- .../src/features/training/types/api.ts | 2 + .../src/features/training/types/config.ts | 4 + 9 files changed, 148 insertions(+), 48 deletions(-) diff --git a/studio/backend/core/training/trainer.py b/studio/backend/core/training/trainer.py index c3376c10a2..f5b2f18245 100644 --- a/studio/backend/core/training/trainer.py +++ b/studio/backend/core/training/trainer.py @@ -353,7 +353,9 @@ class UnslothTrainer: subset: str = None, train_split: str = "train", eval_split: str = None, - eval_steps: float = 0.00) -> Optional[tuple]: + eval_steps: float = 0.00, + dataset_slice_start: int = None, + dataset_slice_end: int = None) -> Optional[tuple]: """ Load and prepare dataset for training. @@ -445,6 +447,18 @@ class UnslothTrainer: if dataset is None: raise ValueError("No dataset provided") + # Apply index range slicing if requested (inclusive on both ends) + if dataset_slice_start is not None or dataset_slice_end is not None: + total_rows = len(dataset) + start = dataset_slice_start if dataset_slice_start is not None else 0 + end = dataset_slice_end if dataset_slice_end is not None else total_rows - 1 + # Clamp to valid range + start = max(0, min(start, total_rows - 1)) + end = max(start, min(end, total_rows - 1)) + dataset = dataset.select(range(start, end + 1)) + print(f"Sliced dataset to rows [{start}, {end}]: {len(dataset)} of {total_rows} rows\n") + self._update_progress(status_message=f"Sliced dataset to {len(dataset)} rows (indices {start}-{end})") + # Check if stopped before applying template if self.should_stop: print("Stopped before applying chat template\n") diff --git a/studio/backend/core/training/training.py b/studio/backend/core/training/training.py index 9123d36b39..153f4335e3 100644 --- a/studio/backend/core/training/training.py +++ b/studio/backend/core/training/training.py @@ -116,7 +116,9 @@ class TrainingBackend: train_split: str = "train", eval_split: str = None, eval_steps: float = 0.00, - is_dataset_multimodal: bool = False) -> bool: + is_dataset_multimodal: bool = False, + dataset_slice_start: int = None, + dataset_slice_end: int = None) -> bool: """ Start training. @@ -224,6 +226,8 @@ class TrainingBackend: train_split=train_split, eval_split=eval_split, eval_steps=eval_steps, + dataset_slice_start=dataset_slice_start, + dataset_slice_end=dataset_slice_end, ) # Unpack: load_and_format_dataset returns (dataset, eval_dataset) diff --git a/studio/backend/models/training.py b/studio/backend/models/training.py index 54de974100..b6b30989bd 100644 --- a/studio/backend/models/training.py +++ b/studio/backend/models/training.py @@ -22,6 +22,8 @@ class TrainingStartRequest(BaseModel): train_split: Optional[str] = Field("train", description="Training split name") eval_split: Optional[str] = Field(None, description="Eval split name. None = auto-detect") eval_steps: float = Field(0.00, description="Fraction of total steps between evals (0-1)") + dataset_slice_start: Optional[int] = Field(None, description="Inclusive start row index for dataset slicing") + dataset_slice_end: Optional[int] = Field(None, description="Inclusive end row index for dataset slicing") @model_validator(mode="before") @classmethod diff --git a/studio/backend/routes/training.py b/studio/backend/routes/training.py index f8de2f639f..497daaedd3 100644 --- a/studio/backend/routes/training.py +++ b/studio/backend/routes/training.py @@ -149,6 +149,8 @@ async def start_training( "train_split": request.train_split, "eval_split": request.eval_split, "eval_steps": request.eval_steps, + "dataset_slice_start": request.dataset_slice_start, + "dataset_slice_end": request.dataset_slice_end, "custom_format_mapping": request.custom_format_mapping, "num_epochs": request.num_epochs, "learning_rate": request.learning_rate, diff --git a/studio/frontend/src/features/studio/sections/dataset-section.tsx b/studio/frontend/src/features/studio/sections/dataset-section.tsx index 57e50bced5..00ee520120 100644 --- a/studio/frontend/src/features/studio/sections/dataset-section.tsx +++ b/studio/frontend/src/features/studio/sections/dataset-section.tsx @@ -13,6 +13,7 @@ import { ComboboxItem, ComboboxList, } from "@/components/ui/combobox"; +import { Input } from "@/components/ui/input"; import { InputGroupAddon } from "@/components/ui/input-group"; import { Select, @@ -75,6 +76,10 @@ export function DatasetSection() { setDatasetEvalSplit, hfToken, modelType, + datasetSliceStart, + setDatasetSliceStart, + datasetSliceEnd, + setDatasetSliceEnd, } = useTrainingConfigStore( useShallow((s) => ({ dataset: s.dataset, @@ -89,6 +94,10 @@ export function DatasetSection() { setDatasetEvalSplit: s.setDatasetEvalSplit, hfToken: s.hfToken, modelType: s.modelType, + datasetSliceStart: s.datasetSliceStart, + setDatasetSliceStart: s.setDatasetSliceStart, + datasetSliceEnd: s.datasetSliceEnd, + setDatasetSliceEnd: s.setDatasetSliceEnd, })), ); @@ -293,51 +302,93 @@ export function DatasetSection() { Advanced -
- - Target Format - - - - - - Format of your training data. Auto-detect works for most - datasets.{" "} - - Read more - - - - - +
+
+ + Target Format + + + + + + Format of your training data. Auto-detect works for most + datasets.{" "} + + Read more + + + + + +
+
+ + Index Range + + + + + + Slice the dataset by row index. Both start and end are + inclusive. Leave empty to use all rows. + + + +
+ + setDatasetSliceStart(e.target.value || null) + } + /> + + setDatasetSliceEnd(e.target.value || null) + } + /> +
+
diff --git a/studio/frontend/src/features/training/api/mappers.ts b/studio/frontend/src/features/training/api/mappers.ts index 1adfbd8b6d..cd0d1f14e1 100644 --- a/studio/frontend/src/features/training/api/mappers.ts +++ b/studio/frontend/src/features/training/api/mappers.ts @@ -4,6 +4,15 @@ import type { TrainingStartRequest } from "../types/api"; const BACKEND_LORA_TYPE = "LoRA/QLoRA"; const BACKEND_FULL_TYPE = "Full Finetuning"; +function parseSliceValue(value: string | null): number | null { + if (value == null) return null; + const trimmed = value.trim(); + if (!trimmed) return null; + const num = Number(trimmed); + if (!Number.isFinite(num) || !Number.isInteger(num)) return null; + return num; +} + export function toBackendTrainingType(trainingMethod: string): string { return trainingMethod === "full" ? BACKEND_FULL_TYPE : BACKEND_LORA_TYPE; } @@ -27,6 +36,8 @@ export function buildTrainingStartPayload( subset: hfDataset ? config.datasetSubset : null, train_split: hfDataset ? config.datasetSplit : null, eval_split: hfDataset ? config.datasetEvalSplit : null, + dataset_slice_start: parseSliceValue(config.datasetSliceStart), + dataset_slice_end: parseSliceValue(config.datasetSliceEnd), local_datasets: [], format_type: config.datasetFormat, custom_format_mapping: customFormatMapping, diff --git a/studio/frontend/src/features/training/stores/training-config-store.ts b/studio/frontend/src/features/training/stores/training-config-store.ts index b2d1858716..93c742ba98 100644 --- a/studio/frontend/src/features/training/stores/training-config-store.ts +++ b/studio/frontend/src/features/training/stores/training-config-store.ts @@ -28,6 +28,8 @@ const initialState: TrainingConfigState = { datasetSplit: null, datasetEvalSplit: null, datasetManualMapping: emptyManualMapping(), + datasetSliceStart: null, + datasetSliceEnd: null, uploadedFile: null, isCheckingVision: false, isVisionModel: false, @@ -255,6 +257,8 @@ export const useTrainingConfigStore = create()( datasetSplit: null, datasetEvalSplit: null, datasetManualMapping: emptyManualMapping(), + datasetSliceStart: null, + datasetSliceEnd: null, isDatasetMultimodal: null, isCheckingDataset: false, }); @@ -311,6 +315,8 @@ export const useTrainingConfigStore = create()( }, setDatasetManualMapping: (datasetManualMapping) => set({ datasetManualMapping }), + setDatasetSliceStart: (datasetSliceStart) => set({ datasetSliceStart }), + setDatasetSliceEnd: (datasetSliceEnd) => set({ datasetSliceEnd }), setUploadedFile: (uploadedFile) => set({ uploadedFile }), setEpochs: (epochs) => set({ epochs }), setContextLength: (contextLength) => set({ contextLength }), @@ -368,7 +374,7 @@ export const useTrainingConfigStore = create()( }, { name: "unsloth_training_config_v1", - version: 6, + version: 7, migrate: (persisted, version) => { const s = persisted as Record; if (version < 2 && s.datasetSubset == null && s.datasetConfig != null) { @@ -387,6 +393,10 @@ export const useTrainingConfigStore = create()( if (version < 6 && s.datasetEvalSplit == null) { s.datasetEvalSplit = null; } + if (version < 7) { + s.datasetSliceStart ??= null; + s.datasetSliceEnd ??= null; + } return s as unknown as TrainingConfigStore; }, partialize: partializePersistedState, diff --git a/studio/frontend/src/features/training/types/api.ts b/studio/frontend/src/features/training/types/api.ts index 22f02e4331..e2fcc04ad4 100644 --- a/studio/frontend/src/features/training/types/api.ts +++ b/studio/frontend/src/features/training/types/api.ts @@ -8,6 +8,8 @@ export interface TrainingStartRequest { subset: string | null; train_split: string | null; eval_split: string | null; + dataset_slice_start: number | null; + dataset_slice_end: number | null; local_datasets: string[]; format_type: string; custom_format_mapping?: Record | null; diff --git a/studio/frontend/src/features/training/types/config.ts b/studio/frontend/src/features/training/types/config.ts index 6c2feec172..0e48d18861 100644 --- a/studio/frontend/src/features/training/types/config.ts +++ b/studio/frontend/src/features/training/types/config.ts @@ -26,6 +26,8 @@ export interface TrainingConfigState { datasetSplit: string | null; datasetEvalSplit: string | null; datasetManualMapping: DatasetManualMapping; + datasetSliceStart: string | null; + datasetSliceEnd: string | null; uploadedFile: string | null; epochs: number; contextLength: number; @@ -84,6 +86,8 @@ export interface TrainingConfigActions { setDatasetSplit: (split: string | null) => void; setDatasetEvalSplit: (split: string | null) => void; setDatasetManualMapping: (mapping: DatasetManualMapping) => void; + setDatasetSliceStart: (value: string | null) => void; + setDatasetSliceEnd: (value: string | null) => void; setUploadedFile: (file: string | null) => void; setEpochs: (epochs: number) => void; setContextLength: (length: number) => void; From 8d9f195ec0d262ebbaadad489ed003a16862276f Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Wed, 4 Mar 2026 22:35:17 +0000 Subject: [PATCH 48/66] refactor: move index range fields next to eval split in 3-col grid Place Slice Start and Slice End inputs alongside the Eval Split selector in a single row (grid-cols-3) so the dataset card stays compact. Remove the duplicate controls from the Advanced section. --- .../studio/sections/dataset-section.tsx | 137 +++++++----------- .../hf-dataset-subset-split-selectors.tsx | 102 +++++++++++-- 2 files changed, 141 insertions(+), 98 deletions(-) diff --git a/studio/frontend/src/features/studio/sections/dataset-section.tsx b/studio/frontend/src/features/studio/sections/dataset-section.tsx index 00ee520120..338a264579 100644 --- a/studio/frontend/src/features/studio/sections/dataset-section.tsx +++ b/studio/frontend/src/features/studio/sections/dataset-section.tsx @@ -13,7 +13,6 @@ import { ComboboxItem, ComboboxList, } from "@/components/ui/combobox"; -import { Input } from "@/components/ui/input"; import { InputGroupAddon } from "@/components/ui/input-group"; import { Select, @@ -291,6 +290,10 @@ export function DatasetSection() { setDatasetSplit={setDatasetSplit} datasetEvalSplit={datasetEvalSplit} setDatasetEvalSplit={setDatasetEvalSplit} + datasetSliceStart={datasetSliceStart} + setDatasetSliceStart={setDatasetSliceStart} + datasetSliceEnd={datasetSliceEnd} + setDatasetSliceEnd={setDatasetSliceEnd} /> @@ -302,93 +305,51 @@ export function DatasetSection() { Advanced -
-
- - Target Format - - - - - - Format of your training data. Auto-detect works for most - datasets.{" "} - - Read more - - - - - -
-
- - Index Range - - - - - - Slice the dataset by row index. Both start and end are - inclusive. Leave empty to use all rows. - - - -
- - setDatasetSliceStart(e.target.value || null) - } - /> - - setDatasetSliceEnd(e.target.value || null) - } - /> -
-
+
+ + Target Format + + + + + + Format of your training data. Auto-detect works for most + datasets.{" "} + + Read more + + + + +
diff --git a/studio/frontend/src/features/training/components/hf-dataset-subset-split-selectors.tsx b/studio/frontend/src/features/training/components/hf-dataset-subset-split-selectors.tsx index d21fd55ff4..5a21ec6d83 100644 --- a/studio/frontend/src/features/training/components/hf-dataset-subset-split-selectors.tsx +++ b/studio/frontend/src/features/training/components/hf-dataset-subset-split-selectors.tsx @@ -5,6 +5,7 @@ import { SelectTrigger, SelectValue, } from "@/components/ui/select"; +import { Input } from "@/components/ui/input"; import { Spinner } from "@/components/ui/spinner"; import { Tooltip, @@ -31,6 +32,10 @@ type Props = { setDatasetSplit: (v: string | null) => void; datasetEvalSplit: string | null; setDatasetEvalSplit: (v: string | null) => void; + datasetSliceStart?: string | null; + setDatasetSliceStart?: (v: string | null) => void; + datasetSliceEnd?: string | null; + setDatasetSliceEnd?: (v: string | null) => void; }; export function HfDatasetSubsetSplitSelectors({ @@ -44,6 +49,10 @@ export function HfDatasetSubsetSplitSelectors({ setDatasetSplit, datasetEvalSplit, setDatasetEvalSplit, + datasetSliceStart, + setDatasetSliceStart, + datasetSliceEnd, + setDatasetSliceEnd, }: Props) { const { subsets: hfSubsets, @@ -155,16 +164,89 @@ export function HfDatasetSubsetSplitSelectors({ /> )} - + {variant === "studio" && setDatasetSliceStart && setDatasetSliceEnd ? ( +
+ +
+ + Slice Start + + + + + + Inclusive start row index. Leave empty to start from the beginning. + + + + + setDatasetSliceStart(e.target.value || null) + } + /> +
+
+ + Slice End + + + + + + Inclusive end row index. Leave empty to use all remaining rows. + + + + + setDatasetSliceEnd(e.target.value || null) + } + /> +
+
+ ) : ( + + )} )} From ed702b48500f284fca3321b53fb64bba2bbd3ee3 Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Wed, 4 Mar 2026 23:07:36 +0000 Subject: [PATCH 49/66] refactor: move train split slice controls back to Advanced section Place Train Split Start / End inputs inside the Advanced collapsible with descriptive tooltips clarifying they slice the training split. Revert the selectors component to its original eval-split-only layout. --- .../studio/sections/dataset-section.tsx | 164 ++++++++++++------ .../hf-dataset-subset-split-selectors.tsx | 102 ++--------- 2 files changed, 125 insertions(+), 141 deletions(-) diff --git a/studio/frontend/src/features/studio/sections/dataset-section.tsx b/studio/frontend/src/features/studio/sections/dataset-section.tsx index 338a264579..bb65f704ac 100644 --- a/studio/frontend/src/features/studio/sections/dataset-section.tsx +++ b/studio/frontend/src/features/studio/sections/dataset-section.tsx @@ -13,6 +13,7 @@ import { ComboboxItem, ComboboxList, } from "@/components/ui/combobox"; +import { Input } from "@/components/ui/input"; import { InputGroupAddon } from "@/components/ui/input-group"; import { Select, @@ -290,10 +291,6 @@ export function DatasetSection() { setDatasetSplit={setDatasetSplit} datasetEvalSplit={datasetEvalSplit} setDatasetEvalSplit={setDatasetEvalSplit} - datasetSliceStart={datasetSliceStart} - setDatasetSliceStart={setDatasetSliceStart} - datasetSliceEnd={datasetSliceEnd} - setDatasetSliceEnd={setDatasetSliceEnd} /> @@ -305,51 +302,120 @@ export function DatasetSection() { Advanced -
- - Target Format - - - - - - Format of your training data. Auto-detect works for most - datasets.{" "} - - Read more - - - - - +
+
+ + Target Format + + + + + + Format of your training data. Auto-detect works for most + datasets.{" "} + + Read more + + + + + +
+
+
+ + Train Split Start + + + + + + Only train on a subset of your training split by + specifying a start row index (inclusive, 0-based). + Useful for resuming from a checkpoint or debugging + with a smaller slice. Leave empty to start from the + first row. + + + + + setDatasetSliceStart(e.target.value || null) + } + /> +
+
+ + Train Split End + + + + + + Last row index to include from the training split + (inclusive, 0-based). For example, set Start to 0 and + End to 99 to train on the first 100 rows. Leave empty + to use all remaining rows. + + + + + setDatasetSliceEnd(e.target.value || null) + } + /> +
+
diff --git a/studio/frontend/src/features/training/components/hf-dataset-subset-split-selectors.tsx b/studio/frontend/src/features/training/components/hf-dataset-subset-split-selectors.tsx index 5a21ec6d83..d21fd55ff4 100644 --- a/studio/frontend/src/features/training/components/hf-dataset-subset-split-selectors.tsx +++ b/studio/frontend/src/features/training/components/hf-dataset-subset-split-selectors.tsx @@ -5,7 +5,6 @@ import { SelectTrigger, SelectValue, } from "@/components/ui/select"; -import { Input } from "@/components/ui/input"; import { Spinner } from "@/components/ui/spinner"; import { Tooltip, @@ -32,10 +31,6 @@ type Props = { setDatasetSplit: (v: string | null) => void; datasetEvalSplit: string | null; setDatasetEvalSplit: (v: string | null) => void; - datasetSliceStart?: string | null; - setDatasetSliceStart?: (v: string | null) => void; - datasetSliceEnd?: string | null; - setDatasetSliceEnd?: (v: string | null) => void; }; export function HfDatasetSubsetSplitSelectors({ @@ -49,10 +44,6 @@ export function HfDatasetSubsetSplitSelectors({ setDatasetSplit, datasetEvalSplit, setDatasetEvalSplit, - datasetSliceStart, - setDatasetSliceStart, - datasetSliceEnd, - setDatasetSliceEnd, }: Props) { const { subsets: hfSubsets, @@ -164,89 +155,16 @@ export function HfDatasetSubsetSplitSelectors({ /> )} - {variant === "studio" && setDatasetSliceStart && setDatasetSliceEnd ? ( -
- -
- - Slice Start - - - - - - Inclusive start row index. Leave empty to start from the beginning. - - - - - setDatasetSliceStart(e.target.value || null) - } - /> -
-
- - Slice End - - - - - - Inclusive end row index. Leave empty to use all remaining rows. - - - - - setDatasetSliceEnd(e.target.value || null) - } - /> -
-
- ) : ( - - )} + )} From 43feb6c2e2f2d179f2cabaeec331fa852c37b08d Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Wed, 4 Mar 2026 23:15:49 +0000 Subject: [PATCH 50/66] fix: remove unnecessary tooltip copy from train split start --- .../frontend/src/features/studio/sections/dataset-section.tsx | 4 +--- 1 file changed, 1 insertion(+), 3 deletions(-) diff --git a/studio/frontend/src/features/studio/sections/dataset-section.tsx b/studio/frontend/src/features/studio/sections/dataset-section.tsx index bb65f704ac..28254adcc7 100644 --- a/studio/frontend/src/features/studio/sections/dataset-section.tsx +++ b/studio/frontend/src/features/studio/sections/dataset-section.tsx @@ -368,9 +368,7 @@ export function DatasetSection() { Only train on a subset of your training split by specifying a start row index (inclusive, 0-based). - Useful for resuming from a checkpoint or debugging - with a smaller slice. Leave empty to start from the - first row. + Leave empty to start from the first row. From 9333f99dd3f101f92acac98682358c1718f3f71b Mon Sep 17 00:00:00 2001 From: Roland Tannous <115670425+rolandtannous@users.noreply.github.com> Date: Thu, 5 Mar 2026 03:21:07 +0400 Subject: [PATCH 51/66] Revert "Add index range dataset slicing to Studio training page" --- studio/backend/core/training/trainer.py | 24 +-- studio/backend/core/training/training.py | 6 +- studio/backend/models/training.py | 2 - studio/backend/routes/training.py | 2 - .../utils/datasets/format_conversion.py | 64 +------ .../studio/sections/dataset-section.tsx | 166 +++++------------- .../src/features/training/api/mappers.ts | 11 -- .../training/stores/training-config-store.ts | 12 +- .../src/features/training/types/api.ts | 2 - .../src/features/training/types/config.ts | 4 - studio/tests/test_url_download_benchmark.py | 54 ------ studio/tests/test_url_image_loading.py | 132 -------------- studio/tests/test_url_parallel_benchmark.py | 79 --------- 13 files changed, 55 insertions(+), 503 deletions(-) delete mode 100644 studio/tests/test_url_download_benchmark.py delete mode 100644 studio/tests/test_url_image_loading.py delete mode 100644 studio/tests/test_url_parallel_benchmark.py diff --git a/studio/backend/core/training/trainer.py b/studio/backend/core/training/trainer.py index f5b2f18245..e4e7a474be 100644 --- a/studio/backend/core/training/trainer.py +++ b/studio/backend/core/training/trainer.py @@ -353,9 +353,7 @@ class UnslothTrainer: subset: str = None, train_split: str = "train", eval_split: str = None, - eval_steps: float = 0.00, - dataset_slice_start: int = None, - dataset_slice_end: int = None) -> Optional[tuple]: + eval_steps: float = 0.00) -> Optional[tuple]: """ Load and prepare dataset for training. @@ -447,18 +445,6 @@ class UnslothTrainer: if dataset is None: raise ValueError("No dataset provided") - # Apply index range slicing if requested (inclusive on both ends) - if dataset_slice_start is not None or dataset_slice_end is not None: - total_rows = len(dataset) - start = dataset_slice_start if dataset_slice_start is not None else 0 - end = dataset_slice_end if dataset_slice_end is not None else total_rows - 1 - # Clamp to valid range - start = max(0, min(start, total_rows - 1)) - end = max(start, min(end, total_rows - 1)) - dataset = dataset.select(range(start, end + 1)) - print(f"Sliced dataset to rows [{start}, {end}]: {len(dataset)} of {total_rows} rows\n") - self._update_progress(status_message=f"Sliced dataset to {len(dataset)} rows (indices {start}-{end})") - # Check if stopped before applying template if self.should_stop: print("Stopped before applying chat template\n") @@ -482,14 +468,6 @@ class UnslothTrainer: print("Stopped during dataset formatting\n") return None - # Abort if dataset formatting/conversion failed - if not dataset_info.get("success", True): - errors = dataset_info.get("errors", []) - error_msg = "; ".join(errors) if errors else "Dataset formatting failed" - logger.error(f"Dataset conversion failed: {error_msg}") - self._update_progress(error=error_msg) - return None - self._update_progress(status_message=f"Dataset formatted and ready for training") print(f"Dataset formatted successfully\n") diff --git a/studio/backend/core/training/training.py b/studio/backend/core/training/training.py index 153f4335e3..9123d36b39 100644 --- a/studio/backend/core/training/training.py +++ b/studio/backend/core/training/training.py @@ -116,9 +116,7 @@ class TrainingBackend: train_split: str = "train", eval_split: str = None, eval_steps: float = 0.00, - is_dataset_multimodal: bool = False, - dataset_slice_start: int = None, - dataset_slice_end: int = None) -> bool: + is_dataset_multimodal: bool = False) -> bool: """ Start training. @@ -226,8 +224,6 @@ class TrainingBackend: train_split=train_split, eval_split=eval_split, eval_steps=eval_steps, - dataset_slice_start=dataset_slice_start, - dataset_slice_end=dataset_slice_end, ) # Unpack: load_and_format_dataset returns (dataset, eval_dataset) diff --git a/studio/backend/models/training.py b/studio/backend/models/training.py index b6b30989bd..54de974100 100644 --- a/studio/backend/models/training.py +++ b/studio/backend/models/training.py @@ -22,8 +22,6 @@ class TrainingStartRequest(BaseModel): train_split: Optional[str] = Field("train", description="Training split name") eval_split: Optional[str] = Field(None, description="Eval split name. None = auto-detect") eval_steps: float = Field(0.00, description="Fraction of total steps between evals (0-1)") - dataset_slice_start: Optional[int] = Field(None, description="Inclusive start row index for dataset slicing") - dataset_slice_end: Optional[int] = Field(None, description="Inclusive end row index for dataset slicing") @model_validator(mode="before") @classmethod diff --git a/studio/backend/routes/training.py b/studio/backend/routes/training.py index 497daaedd3..f8de2f639f 100644 --- a/studio/backend/routes/training.py +++ b/studio/backend/routes/training.py @@ -149,8 +149,6 @@ async def start_training( "train_split": request.train_split, "eval_split": request.eval_split, "eval_steps": request.eval_steps, - "dataset_slice_start": request.dataset_slice_start, - "dataset_slice_end": request.dataset_slice_end, "custom_format_mapping": request.custom_format_mapping, "num_epochs": request.num_epochs, "learning_rate": request.learning_rate, diff --git a/studio/backend/utils/datasets/format_conversion.py b/studio/backend/utils/datasets/format_conversion.py index e784e8e645..6436e7a82a 100644 --- a/studio/backend/utils/datasets/format_conversion.py +++ b/studio/backend/utils/datasets/format_conversion.py @@ -281,17 +281,12 @@ def convert_to_vlm_format( def _convert_single_sample(sample): """Convert a single sample to VLM format.""" - # Get image (might be PIL Image, local path, or URL) + # Get image (might be PIL Image or path) image_data = sample[image_column] + # Handle image paths if isinstance(image_data, str): - if image_data.startswith(("http://", "https://")): - import fsspec - from io import BytesIO - with fsspec.open(image_data, "rb", expand=True) as f: - image_data = Image.open(BytesIO(f.read())).convert("RGB") - else: - image_data = Image.open(image_data).convert("RGB") + image_data = Image.open(image_data).convert("RGB") # Get text text_data = sample[text_column] @@ -322,56 +317,11 @@ def convert_to_vlm_format( # Return dict with messages return {"messages": messages} - # Convert samples, skipping any with broken/unreachable images. - # For URL-based datasets, check the first PROBE_SIZE samples early to - # fail fast if too many images are broken, before downloading millions. - PROBE_SIZE = 5000 - MAX_FAIL_RATE = 0.3 + # Use list comprehension and return the LIST directly + print(f"🔄 Converting {len(dataset)} samples to VLM format...") + converted_list = [_convert_single_sample(sample) for sample in dataset] - total = len(dataset) - has_urls = isinstance(next(iter(dataset))[image_column], str) - probe_needed = has_urls and total > PROBE_SIZE - - from tqdm import tqdm - - print(f"🔄 Converting {total} samples to VLM format...") - converted_list = [] - failed_count = 0 - - pbar = tqdm(dataset, total=total, desc="Converting VLM samples", unit="sample") - for i, sample in enumerate(pbar): - try: - converted_list.append(_convert_single_sample(sample)) - except Exception as e: - failed_count += 1 - - pbar.set_postfix(ok=len(converted_list), failed=failed_count, refresh=False) - - # Early exit check after probing the first batch - if probe_needed and (i + 1) == PROBE_SIZE: - fail_rate = failed_count / PROBE_SIZE - if fail_rate >= MAX_FAIL_RATE: - pbar.close() - raise ValueError( - f"{fail_rate:.0%} of the first {PROBE_SIZE} images failed to download " - f"({failed_count}/{PROBE_SIZE}). " - "This dataset has too many broken or unreachable image URLs. " - "Consider using a dataset with embedded images instead." - ) - print(f"✅ Probe passed: {failed_count}/{PROBE_SIZE} ({fail_rate:.0%}) failures in first batch, continuing...") - pbar.close() - - if failed_count > 0: - fail_rate = failed_count / total - print(f"⚠️ Skipped {failed_count}/{total} ({fail_rate:.0%}) samples with broken/unreachable images") - - if len(converted_list) == 0: - raise ValueError( - f"All {total} samples failed during VLM conversion — no usable images found. " - "This dataset may contain only image URLs that are no longer accessible." - ) - - print(f"✅ Converted {len(converted_list)}/{total} samples") + print(f"✅ Converted {len(converted_list)} samples") # Return list, NOT Dataset return converted_list diff --git a/studio/frontend/src/features/studio/sections/dataset-section.tsx b/studio/frontend/src/features/studio/sections/dataset-section.tsx index 28254adcc7..57e50bced5 100644 --- a/studio/frontend/src/features/studio/sections/dataset-section.tsx +++ b/studio/frontend/src/features/studio/sections/dataset-section.tsx @@ -13,7 +13,6 @@ import { ComboboxItem, ComboboxList, } from "@/components/ui/combobox"; -import { Input } from "@/components/ui/input"; import { InputGroupAddon } from "@/components/ui/input-group"; import { Select, @@ -76,10 +75,6 @@ export function DatasetSection() { setDatasetEvalSplit, hfToken, modelType, - datasetSliceStart, - setDatasetSliceStart, - datasetSliceEnd, - setDatasetSliceEnd, } = useTrainingConfigStore( useShallow((s) => ({ dataset: s.dataset, @@ -94,10 +89,6 @@ export function DatasetSection() { setDatasetEvalSplit: s.setDatasetEvalSplit, hfToken: s.hfToken, modelType: s.modelType, - datasetSliceStart: s.datasetSliceStart, - setDatasetSliceStart: s.setDatasetSliceStart, - datasetSliceEnd: s.datasetSliceEnd, - setDatasetSliceEnd: s.setDatasetSliceEnd, })), ); @@ -302,118 +293,51 @@ export function DatasetSection() { Advanced -
-
- - Target Format - - - - - - Format of your training data. Auto-detect works for most - datasets.{" "} - - Read more - - - - - -
-
-
- - Train Split Start - - - - - - Only train on a subset of your training split by - specifying a start row index (inclusive, 0-based). - Leave empty to start from the first row. - - - - - setDatasetSliceStart(e.target.value || null) - } - /> -
-
- - Train Split End - - - - - - Last row index to include from the training split - (inclusive, 0-based). For example, set Start to 0 and - End to 99 to train on the first 100 rows. Leave empty - to use all remaining rows. - - - - - setDatasetSliceEnd(e.target.value || null) - } - /> -
-
+
+ + Target Format + + + + + + Format of your training data. Auto-detect works for most + datasets.{" "} + + Read more + + + + +
diff --git a/studio/frontend/src/features/training/api/mappers.ts b/studio/frontend/src/features/training/api/mappers.ts index cd0d1f14e1..1adfbd8b6d 100644 --- a/studio/frontend/src/features/training/api/mappers.ts +++ b/studio/frontend/src/features/training/api/mappers.ts @@ -4,15 +4,6 @@ import type { TrainingStartRequest } from "../types/api"; const BACKEND_LORA_TYPE = "LoRA/QLoRA"; const BACKEND_FULL_TYPE = "Full Finetuning"; -function parseSliceValue(value: string | null): number | null { - if (value == null) return null; - const trimmed = value.trim(); - if (!trimmed) return null; - const num = Number(trimmed); - if (!Number.isFinite(num) || !Number.isInteger(num)) return null; - return num; -} - export function toBackendTrainingType(trainingMethod: string): string { return trainingMethod === "full" ? BACKEND_FULL_TYPE : BACKEND_LORA_TYPE; } @@ -36,8 +27,6 @@ export function buildTrainingStartPayload( subset: hfDataset ? config.datasetSubset : null, train_split: hfDataset ? config.datasetSplit : null, eval_split: hfDataset ? config.datasetEvalSplit : null, - dataset_slice_start: parseSliceValue(config.datasetSliceStart), - dataset_slice_end: parseSliceValue(config.datasetSliceEnd), local_datasets: [], format_type: config.datasetFormat, custom_format_mapping: customFormatMapping, diff --git a/studio/frontend/src/features/training/stores/training-config-store.ts b/studio/frontend/src/features/training/stores/training-config-store.ts index 93c742ba98..b2d1858716 100644 --- a/studio/frontend/src/features/training/stores/training-config-store.ts +++ b/studio/frontend/src/features/training/stores/training-config-store.ts @@ -28,8 +28,6 @@ const initialState: TrainingConfigState = { datasetSplit: null, datasetEvalSplit: null, datasetManualMapping: emptyManualMapping(), - datasetSliceStart: null, - datasetSliceEnd: null, uploadedFile: null, isCheckingVision: false, isVisionModel: false, @@ -257,8 +255,6 @@ export const useTrainingConfigStore = create()( datasetSplit: null, datasetEvalSplit: null, datasetManualMapping: emptyManualMapping(), - datasetSliceStart: null, - datasetSliceEnd: null, isDatasetMultimodal: null, isCheckingDataset: false, }); @@ -315,8 +311,6 @@ export const useTrainingConfigStore = create()( }, setDatasetManualMapping: (datasetManualMapping) => set({ datasetManualMapping }), - setDatasetSliceStart: (datasetSliceStart) => set({ datasetSliceStart }), - setDatasetSliceEnd: (datasetSliceEnd) => set({ datasetSliceEnd }), setUploadedFile: (uploadedFile) => set({ uploadedFile }), setEpochs: (epochs) => set({ epochs }), setContextLength: (contextLength) => set({ contextLength }), @@ -374,7 +368,7 @@ export const useTrainingConfigStore = create()( }, { name: "unsloth_training_config_v1", - version: 7, + version: 6, migrate: (persisted, version) => { const s = persisted as Record; if (version < 2 && s.datasetSubset == null && s.datasetConfig != null) { @@ -393,10 +387,6 @@ export const useTrainingConfigStore = create()( if (version < 6 && s.datasetEvalSplit == null) { s.datasetEvalSplit = null; } - if (version < 7) { - s.datasetSliceStart ??= null; - s.datasetSliceEnd ??= null; - } return s as unknown as TrainingConfigStore; }, partialize: partializePersistedState, diff --git a/studio/frontend/src/features/training/types/api.ts b/studio/frontend/src/features/training/types/api.ts index e2fcc04ad4..22f02e4331 100644 --- a/studio/frontend/src/features/training/types/api.ts +++ b/studio/frontend/src/features/training/types/api.ts @@ -8,8 +8,6 @@ export interface TrainingStartRequest { subset: string | null; train_split: string | null; eval_split: string | null; - dataset_slice_start: number | null; - dataset_slice_end: number | null; local_datasets: string[]; format_type: string; custom_format_mapping?: Record | null; diff --git a/studio/frontend/src/features/training/types/config.ts b/studio/frontend/src/features/training/types/config.ts index 0e48d18861..6c2feec172 100644 --- a/studio/frontend/src/features/training/types/config.ts +++ b/studio/frontend/src/features/training/types/config.ts @@ -26,8 +26,6 @@ export interface TrainingConfigState { datasetSplit: string | null; datasetEvalSplit: string | null; datasetManualMapping: DatasetManualMapping; - datasetSliceStart: string | null; - datasetSliceEnd: string | null; uploadedFile: string | null; epochs: number; contextLength: number; @@ -86,8 +84,6 @@ export interface TrainingConfigActions { setDatasetSplit: (split: string | null) => void; setDatasetEvalSplit: (split: string | null) => void; setDatasetManualMapping: (mapping: DatasetManualMapping) => void; - setDatasetSliceStart: (value: string | null) => void; - setDatasetSliceEnd: (value: string | null) => void; setUploadedFile: (file: string | null) => void; setEpochs: (epochs: number) => void; setContextLength: (length: number) => void; diff --git a/studio/tests/test_url_download_benchmark.py b/studio/tests/test_url_download_benchmark.py deleted file mode 100644 index e29a8e05b6..0000000000 --- a/studio/tests/test_url_download_benchmark.py +++ /dev/null @@ -1,54 +0,0 @@ -""" -Benchmark: fsspec URL image download throughput at different dataset sizes. -Dataset: google-research-datasets/conceptual_captions (subset: labeled) - -Tests sizes: 100, 200, 300, 500, 1000, 1500, 2000 -Reports: time, success/fail rate, throughput (images/sec) -""" -from datasets import load_dataset, Dataset -from PIL import Image as PILImage -from io import BytesIO -from itertools import islice -import fsspec -import time - -DATASET = "google-research-datasets/conceptual_captions" -SUBSET = "labeled" -SPLIT = "train" -SIZES = [100, 200, 300, 500, 1000, 1500, 2000] - -# Load the max we need in one go -max_size = max(SIZES) -print(f"Loading {max_size} samples from {DATASET} (streaming)...") -ds = load_dataset(DATASET, name=SUBSET, split=SPLIT, streaming=True) -rows = list(islice(ds, max_size)) -full_dataset = Dataset.from_list(rows) -print(f"Loaded {len(full_dataset)} samples") -print(f"Columns: {full_dataset.column_names}") -print() - -print(f"{'Size':>6} | {'Time':>8} | {'OK':>6} | {'Fail':>6} | {'Fail%':>6} | {'img/s':>7}") -print("-" * 55) - -for size in SIZES: - dataset = full_dataset.select(range(size)) - success, fail = 0, 0 - t0 = time.time() - - for sample in dataset: - url = sample["image_url"] - try: - with fsspec.open(url, "rb", expand=True) as f: - img = PILImage.open(BytesIO(f.read())).convert("RGB") - success += 1 - except Exception: - fail += 1 - - elapsed = time.time() - t0 - fail_pct = (fail / size) * 100 - throughput = success / elapsed if elapsed > 0 else 0 - - print(f"{size:>6} | {elapsed:>7.1f}s | {success:>6} | {fail:>6} | {fail_pct:>5.1f}% | {throughput:>6.1f}/s") - -print() -print("Done.") diff --git a/studio/tests/test_url_image_loading.py b/studio/tests/test_url_image_loading.py deleted file mode 100644 index 4899bab8e3..0000000000 --- a/studio/tests/test_url_image_loading.py +++ /dev/null @@ -1,132 +0,0 @@ -""" -Reproduce: VLM URL image loading with HF datasets. -Tests cast_column(Image()) vs manual download approaches. -Dataset: google-research-datasets/conceptual_captions (subset: labeled) -""" -from datasets import load_dataset, Image as datasets_Image, Dataset -from PIL import Image as PILImage -from io import BytesIO -from itertools import islice -import time - -DATASET = "google-research-datasets/conceptual_captions" -SUBSET = "labeled" -SPLIT = "train" -N_SAMPLES = 20 # small slice for testing - -print("=" * 60) -print("Loading dataset (streaming, first N samples)...") -print("=" * 60) -ds = load_dataset(DATASET, name=SUBSET, split=SPLIT, streaming=True) -rows = list(islice(ds, N_SAMPLES)) -dataset = Dataset.from_list(rows) - -print(f"Loaded {len(dataset)} samples") -print(f"Columns: {dataset.column_names}") -print(f"First image_url: {dataset[0]['image_url'][:100]}...") -print() - -# ─── Test 1: cast_column(Image()) — what we tried ─── -print("=" * 60) -print("TEST 1: cast_column(Image()) approach") -print("=" * 60) -try: - ds_cast = dataset.cast_column("image_url", datasets_Image()) - success, fail = 0, 0 - t0 = time.time() - for i, sample in enumerate(ds_cast): - try: - img = sample["image_url"] - if img is not None: - print(f" [{i}] OK — {img.size} {img.mode}") - success += 1 - else: - print(f" [{i}] None returned") - fail += 1 - except Exception as e: - print(f" [{i}] FAILED: {type(e).__name__}: {str(e)[:80]}") - fail += 1 - elapsed = time.time() - t0 - print(f"\nResult: {success} ok, {fail} failed, {elapsed:.1f}s") -except Exception as e: - print(f"CRASHED during iteration: {type(e).__name__}: {str(e)[:120]}") -print() - -# ─── Test 2: Manual download with requests.Session ─── -print("=" * 60) -print("TEST 2: requests.Session() approach") -print("=" * 60) -try: - import requests - session = requests.Session() - success, fail = 0, 0 - t0 = time.time() - for i, sample in enumerate(dataset): - url = sample["image_url"] - try: - resp = session.get(url, timeout=10) - resp.raise_for_status() - img = PILImage.open(BytesIO(resp.content)).convert("RGB") - print(f" [{i}] OK — {img.size} {img.mode}") - success += 1 - except Exception as e: - print(f" [{i}] FAILED: {type(e).__name__}: {str(e)[:80]}") - fail += 1 - elapsed = time.time() - t0 - print(f"\nResult: {success} ok, {fail} failed, {elapsed:.1f}s") -except Exception as e: - print(f"CRASHED: {type(e).__name__}: {str(e)[:120]}") -print() - -# ─── Test 3: urllib (stdlib) ─── -print("=" * 60) -print("TEST 3: urllib approach (stdlib)") -print("=" * 60) -try: - from urllib.request import urlopen, Request - success, fail = 0, 0 - t0 = time.time() - for i, sample in enumerate(dataset): - url = sample["image_url"] - try: - req = Request(url, headers={"User-Agent": "Mozilla/5.0"}) - with urlopen(req, timeout=10) as resp: - img = PILImage.open(BytesIO(resp.read())).convert("RGB") - print(f" [{i}] OK — {img.size} {img.mode}") - success += 1 - except Exception as e: - print(f" [{i}] FAILED: {type(e).__name__}: {str(e)[:80]}") - fail += 1 - elapsed = time.time() - t0 - print(f"\nResult: {success} ok, {fail} failed, {elapsed:.1f}s") -except Exception as e: - print(f"CRASHED: {type(e).__name__}: {str(e)[:120]}") -print() - -# ─── Test 4: fsspec directly with expand=True ─── -print("=" * 60) -print("TEST 4: fsspec.open() with expand=True") -print("=" * 60) -try: - import fsspec - success, fail = 0, 0 - t0 = time.time() - for i, sample in enumerate(dataset): - url = sample["image_url"] - try: - with fsspec.open(url, "rb", expand=True) as f: - img = PILImage.open(BytesIO(f.read())).convert("RGB") - print(f" [{i}] OK — {img.size} {img.mode}") - success += 1 - except Exception as e: - print(f" [{i}] FAILED: {type(e).__name__}: {str(e)[:80]}") - fail += 1 - elapsed = time.time() - t0 - print(f"\nResult: {success} ok, {fail} failed, {elapsed:.1f}s") -except Exception as e: - print(f"CRASHED: {type(e).__name__}: {str(e)[:120]}") - -print() -print("=" * 60) -print("DONE — compare success rates and timing above") -print("=" * 60) diff --git a/studio/tests/test_url_parallel_benchmark.py b/studio/tests/test_url_parallel_benchmark.py deleted file mode 100644 index a0d160136c..0000000000 --- a/studio/tests/test_url_parallel_benchmark.py +++ /dev/null @@ -1,79 +0,0 @@ -""" -Benchmark: parallel fsspec URL image downloads with ThreadPoolExecutor. -Tests different worker counts to find optimal parallelism. -Dataset: google-research-datasets/conceptual_captions (subset: labeled) -""" -from datasets import load_dataset, Dataset -from PIL import Image as PILImage -from io import BytesIO -from itertools import islice -from concurrent.futures import ThreadPoolExecutor, as_completed -import fsspec -import time -import os - -DATASET = "google-research-datasets/conceptual_captions" -SUBSET = "labeled" -SPLIT = "train" -N_SAMPLES = 500 - -# safe_num_proc formula from studio/backend/utils/hardware/hardware.py -cpu_count = os.cpu_count() -safe_workers = max(1, cpu_count // 3) -print(f"CPU count: {cpu_count}, safe_num_proc: {safe_workers}") - -WORKER_COUNTS = [1, 4, 8, 16, 32, safe_workers] -# Deduplicate and sort -WORKER_COUNTS = sorted(set(WORKER_COUNTS)) - -print(f"Loading {N_SAMPLES} samples from {DATASET} (streaming)...") -ds = load_dataset(DATASET, name=SUBSET, split=SPLIT, streaming=True) -rows = list(islice(ds, N_SAMPLES)) -dataset = Dataset.from_list(rows) -urls = [row["image_url"] for row in dataset] -print(f"Loaded {len(urls)} URLs") -print() - - -def download_single(url): - """Download a single image URL using fsspec. Returns PIL image or raises.""" - with fsspec.open(url, "rb", expand=True) as f: - img = PILImage.open(BytesIO(f.read())).convert("RGB") - return img - - -print(f"{'Workers':>8} | {'Time':>8} | {'OK':>6} | {'Fail':>6} | {'Fail%':>6} | {'img/s':>7} | {'Speedup':>8}") -print("-" * 70) - -baseline_throughput = None - -for n_workers in WORKER_COUNTS: - success, fail = 0, 0 - t0 = time.time() - - with ThreadPoolExecutor(max_workers=n_workers) as pool: - futures = {pool.submit(download_single, url): url for url in urls} - for future in as_completed(futures): - try: - img = future.result(timeout=30) - success += 1 - except Exception: - fail += 1 - - elapsed = time.time() - t0 - fail_pct = (fail / N_SAMPLES) * 100 - throughput = success / elapsed if elapsed > 0 else 0 - - if baseline_throughput is None: - baseline_throughput = throughput - speedup = throughput / baseline_throughput if baseline_throughput > 0 else 0 - - label = f"{n_workers}" - if n_workers == safe_workers: - label += "*" # mark the safe_num_proc value - - print(f"{label:>8} | {elapsed:>7.1f}s | {success:>6} | {fail:>6} | {fail_pct:>5.1f}% | {throughput:>6.1f}/s | {speedup:>7.1f}x") - -print() -print("* = safe_num_proc value") -print("Done.") From 64889cd5fcb8d48b35160cc9afab02ea04a48b20 Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Wed, 4 Mar 2026 21:48:40 +0000 Subject: [PATCH 52/66] feat: add index range dataset slicing to studio training page Add Start/End index inputs under Advanced in the dataset card, allowing users to slice a dataset by row range before training. Wired end-to-end: frontend store, API payload, backend Pydantic model, and trainer dataset loading (inclusive on both ends). --- studio/backend/core/training/trainer.py | 16 +- studio/backend/core/training/training.py | 6 +- studio/backend/models/training.py | 2 + studio/backend/routes/training.py | 2 + .../studio/sections/dataset-section.tsx | 141 ++++++++++++------ .../src/features/training/api/mappers.ts | 11 ++ .../training/stores/training-config-store.ts | 12 +- .../src/features/training/types/api.ts | 2 + .../src/features/training/types/config.ts | 4 + 9 files changed, 148 insertions(+), 48 deletions(-) diff --git a/studio/backend/core/training/trainer.py b/studio/backend/core/training/trainer.py index e4e7a474be..5ce273f43c 100644 --- a/studio/backend/core/training/trainer.py +++ b/studio/backend/core/training/trainer.py @@ -353,7 +353,9 @@ class UnslothTrainer: subset: str = None, train_split: str = "train", eval_split: str = None, - eval_steps: float = 0.00) -> Optional[tuple]: + eval_steps: float = 0.00, + dataset_slice_start: int = None, + dataset_slice_end: int = None) -> Optional[tuple]: """ Load and prepare dataset for training. @@ -445,6 +447,18 @@ class UnslothTrainer: if dataset is None: raise ValueError("No dataset provided") + # Apply index range slicing if requested (inclusive on both ends) + if dataset_slice_start is not None or dataset_slice_end is not None: + total_rows = len(dataset) + start = dataset_slice_start if dataset_slice_start is not None else 0 + end = dataset_slice_end if dataset_slice_end is not None else total_rows - 1 + # Clamp to valid range + start = max(0, min(start, total_rows - 1)) + end = max(start, min(end, total_rows - 1)) + dataset = dataset.select(range(start, end + 1)) + print(f"Sliced dataset to rows [{start}, {end}]: {len(dataset)} of {total_rows} rows\n") + self._update_progress(status_message=f"Sliced dataset to {len(dataset)} rows (indices {start}-{end})") + # Check if stopped before applying template if self.should_stop: print("Stopped before applying chat template\n") diff --git a/studio/backend/core/training/training.py b/studio/backend/core/training/training.py index 9123d36b39..153f4335e3 100644 --- a/studio/backend/core/training/training.py +++ b/studio/backend/core/training/training.py @@ -116,7 +116,9 @@ class TrainingBackend: train_split: str = "train", eval_split: str = None, eval_steps: float = 0.00, - is_dataset_multimodal: bool = False) -> bool: + is_dataset_multimodal: bool = False, + dataset_slice_start: int = None, + dataset_slice_end: int = None) -> bool: """ Start training. @@ -224,6 +226,8 @@ class TrainingBackend: train_split=train_split, eval_split=eval_split, eval_steps=eval_steps, + dataset_slice_start=dataset_slice_start, + dataset_slice_end=dataset_slice_end, ) # Unpack: load_and_format_dataset returns (dataset, eval_dataset) diff --git a/studio/backend/models/training.py b/studio/backend/models/training.py index 54de974100..b6b30989bd 100644 --- a/studio/backend/models/training.py +++ b/studio/backend/models/training.py @@ -22,6 +22,8 @@ class TrainingStartRequest(BaseModel): train_split: Optional[str] = Field("train", description="Training split name") eval_split: Optional[str] = Field(None, description="Eval split name. None = auto-detect") eval_steps: float = Field(0.00, description="Fraction of total steps between evals (0-1)") + dataset_slice_start: Optional[int] = Field(None, description="Inclusive start row index for dataset slicing") + dataset_slice_end: Optional[int] = Field(None, description="Inclusive end row index for dataset slicing") @model_validator(mode="before") @classmethod diff --git a/studio/backend/routes/training.py b/studio/backend/routes/training.py index f8de2f639f..497daaedd3 100644 --- a/studio/backend/routes/training.py +++ b/studio/backend/routes/training.py @@ -149,6 +149,8 @@ async def start_training( "train_split": request.train_split, "eval_split": request.eval_split, "eval_steps": request.eval_steps, + "dataset_slice_start": request.dataset_slice_start, + "dataset_slice_end": request.dataset_slice_end, "custom_format_mapping": request.custom_format_mapping, "num_epochs": request.num_epochs, "learning_rate": request.learning_rate, diff --git a/studio/frontend/src/features/studio/sections/dataset-section.tsx b/studio/frontend/src/features/studio/sections/dataset-section.tsx index 57e50bced5..00ee520120 100644 --- a/studio/frontend/src/features/studio/sections/dataset-section.tsx +++ b/studio/frontend/src/features/studio/sections/dataset-section.tsx @@ -13,6 +13,7 @@ import { ComboboxItem, ComboboxList, } from "@/components/ui/combobox"; +import { Input } from "@/components/ui/input"; import { InputGroupAddon } from "@/components/ui/input-group"; import { Select, @@ -75,6 +76,10 @@ export function DatasetSection() { setDatasetEvalSplit, hfToken, modelType, + datasetSliceStart, + setDatasetSliceStart, + datasetSliceEnd, + setDatasetSliceEnd, } = useTrainingConfigStore( useShallow((s) => ({ dataset: s.dataset, @@ -89,6 +94,10 @@ export function DatasetSection() { setDatasetEvalSplit: s.setDatasetEvalSplit, hfToken: s.hfToken, modelType: s.modelType, + datasetSliceStart: s.datasetSliceStart, + setDatasetSliceStart: s.setDatasetSliceStart, + datasetSliceEnd: s.datasetSliceEnd, + setDatasetSliceEnd: s.setDatasetSliceEnd, })), ); @@ -293,51 +302,93 @@ export function DatasetSection() { Advanced -
- - Target Format - - - - - - Format of your training data. Auto-detect works for most - datasets.{" "} - - Read more - - - - - +
+
+ + Target Format + + + + + + Format of your training data. Auto-detect works for most + datasets.{" "} + + Read more + + + + + +
+
+ + Index Range + + + + + + Slice the dataset by row index. Both start and end are + inclusive. Leave empty to use all rows. + + + +
+ + setDatasetSliceStart(e.target.value || null) + } + /> + + setDatasetSliceEnd(e.target.value || null) + } + /> +
+
diff --git a/studio/frontend/src/features/training/api/mappers.ts b/studio/frontend/src/features/training/api/mappers.ts index 1adfbd8b6d..cd0d1f14e1 100644 --- a/studio/frontend/src/features/training/api/mappers.ts +++ b/studio/frontend/src/features/training/api/mappers.ts @@ -4,6 +4,15 @@ import type { TrainingStartRequest } from "../types/api"; const BACKEND_LORA_TYPE = "LoRA/QLoRA"; const BACKEND_FULL_TYPE = "Full Finetuning"; +function parseSliceValue(value: string | null): number | null { + if (value == null) return null; + const trimmed = value.trim(); + if (!trimmed) return null; + const num = Number(trimmed); + if (!Number.isFinite(num) || !Number.isInteger(num)) return null; + return num; +} + export function toBackendTrainingType(trainingMethod: string): string { return trainingMethod === "full" ? BACKEND_FULL_TYPE : BACKEND_LORA_TYPE; } @@ -27,6 +36,8 @@ export function buildTrainingStartPayload( subset: hfDataset ? config.datasetSubset : null, train_split: hfDataset ? config.datasetSplit : null, eval_split: hfDataset ? config.datasetEvalSplit : null, + dataset_slice_start: parseSliceValue(config.datasetSliceStart), + dataset_slice_end: parseSliceValue(config.datasetSliceEnd), local_datasets: [], format_type: config.datasetFormat, custom_format_mapping: customFormatMapping, diff --git a/studio/frontend/src/features/training/stores/training-config-store.ts b/studio/frontend/src/features/training/stores/training-config-store.ts index b2d1858716..93c742ba98 100644 --- a/studio/frontend/src/features/training/stores/training-config-store.ts +++ b/studio/frontend/src/features/training/stores/training-config-store.ts @@ -28,6 +28,8 @@ const initialState: TrainingConfigState = { datasetSplit: null, datasetEvalSplit: null, datasetManualMapping: emptyManualMapping(), + datasetSliceStart: null, + datasetSliceEnd: null, uploadedFile: null, isCheckingVision: false, isVisionModel: false, @@ -255,6 +257,8 @@ export const useTrainingConfigStore = create()( datasetSplit: null, datasetEvalSplit: null, datasetManualMapping: emptyManualMapping(), + datasetSliceStart: null, + datasetSliceEnd: null, isDatasetMultimodal: null, isCheckingDataset: false, }); @@ -311,6 +315,8 @@ export const useTrainingConfigStore = create()( }, setDatasetManualMapping: (datasetManualMapping) => set({ datasetManualMapping }), + setDatasetSliceStart: (datasetSliceStart) => set({ datasetSliceStart }), + setDatasetSliceEnd: (datasetSliceEnd) => set({ datasetSliceEnd }), setUploadedFile: (uploadedFile) => set({ uploadedFile }), setEpochs: (epochs) => set({ epochs }), setContextLength: (contextLength) => set({ contextLength }), @@ -368,7 +374,7 @@ export const useTrainingConfigStore = create()( }, { name: "unsloth_training_config_v1", - version: 6, + version: 7, migrate: (persisted, version) => { const s = persisted as Record; if (version < 2 && s.datasetSubset == null && s.datasetConfig != null) { @@ -387,6 +393,10 @@ export const useTrainingConfigStore = create()( if (version < 6 && s.datasetEvalSplit == null) { s.datasetEvalSplit = null; } + if (version < 7) { + s.datasetSliceStart ??= null; + s.datasetSliceEnd ??= null; + } return s as unknown as TrainingConfigStore; }, partialize: partializePersistedState, diff --git a/studio/frontend/src/features/training/types/api.ts b/studio/frontend/src/features/training/types/api.ts index 22f02e4331..e2fcc04ad4 100644 --- a/studio/frontend/src/features/training/types/api.ts +++ b/studio/frontend/src/features/training/types/api.ts @@ -8,6 +8,8 @@ export interface TrainingStartRequest { subset: string | null; train_split: string | null; eval_split: string | null; + dataset_slice_start: number | null; + dataset_slice_end: number | null; local_datasets: string[]; format_type: string; custom_format_mapping?: Record | null; diff --git a/studio/frontend/src/features/training/types/config.ts b/studio/frontend/src/features/training/types/config.ts index 6c2feec172..0e48d18861 100644 --- a/studio/frontend/src/features/training/types/config.ts +++ b/studio/frontend/src/features/training/types/config.ts @@ -26,6 +26,8 @@ export interface TrainingConfigState { datasetSplit: string | null; datasetEvalSplit: string | null; datasetManualMapping: DatasetManualMapping; + datasetSliceStart: string | null; + datasetSliceEnd: string | null; uploadedFile: string | null; epochs: number; contextLength: number; @@ -84,6 +86,8 @@ export interface TrainingConfigActions { setDatasetSplit: (split: string | null) => void; setDatasetEvalSplit: (split: string | null) => void; setDatasetManualMapping: (mapping: DatasetManualMapping) => void; + setDatasetSliceStart: (value: string | null) => void; + setDatasetSliceEnd: (value: string | null) => void; setUploadedFile: (file: string | null) => void; setEpochs: (epochs: number) => void; setContextLength: (length: number) => void; From 44905e92167af4649c1960d1aa92741aab7aca3e Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Wed, 4 Mar 2026 22:35:17 +0000 Subject: [PATCH 53/66] refactor: move index range fields next to eval split in 3-col grid Place Slice Start and Slice End inputs alongside the Eval Split selector in a single row (grid-cols-3) so the dataset card stays compact. Remove the duplicate controls from the Advanced section. --- .../studio/sections/dataset-section.tsx | 137 +++++++----------- .../hf-dataset-subset-split-selectors.tsx | 102 +++++++++++-- 2 files changed, 141 insertions(+), 98 deletions(-) diff --git a/studio/frontend/src/features/studio/sections/dataset-section.tsx b/studio/frontend/src/features/studio/sections/dataset-section.tsx index 00ee520120..338a264579 100644 --- a/studio/frontend/src/features/studio/sections/dataset-section.tsx +++ b/studio/frontend/src/features/studio/sections/dataset-section.tsx @@ -13,7 +13,6 @@ import { ComboboxItem, ComboboxList, } from "@/components/ui/combobox"; -import { Input } from "@/components/ui/input"; import { InputGroupAddon } from "@/components/ui/input-group"; import { Select, @@ -291,6 +290,10 @@ export function DatasetSection() { setDatasetSplit={setDatasetSplit} datasetEvalSplit={datasetEvalSplit} setDatasetEvalSplit={setDatasetEvalSplit} + datasetSliceStart={datasetSliceStart} + setDatasetSliceStart={setDatasetSliceStart} + datasetSliceEnd={datasetSliceEnd} + setDatasetSliceEnd={setDatasetSliceEnd} /> @@ -302,93 +305,51 @@ export function DatasetSection() { Advanced -
-
- - Target Format - - - - - - Format of your training data. Auto-detect works for most - datasets.{" "} - - Read more - - - - - -
-
- - Index Range - - - - - - Slice the dataset by row index. Both start and end are - inclusive. Leave empty to use all rows. - - - -
- - setDatasetSliceStart(e.target.value || null) - } - /> - - setDatasetSliceEnd(e.target.value || null) - } - /> -
-
+
+ + Target Format + + + + + + Format of your training data. Auto-detect works for most + datasets.{" "} + + Read more + + + + +
diff --git a/studio/frontend/src/features/training/components/hf-dataset-subset-split-selectors.tsx b/studio/frontend/src/features/training/components/hf-dataset-subset-split-selectors.tsx index 148341e4de..35bfec8cd3 100644 --- a/studio/frontend/src/features/training/components/hf-dataset-subset-split-selectors.tsx +++ b/studio/frontend/src/features/training/components/hf-dataset-subset-split-selectors.tsx @@ -5,6 +5,7 @@ import { SelectTrigger, SelectValue, } from "@/components/ui/select"; +import { Input } from "@/components/ui/input"; import { Spinner } from "@/components/ui/spinner"; import { Tooltip, @@ -31,6 +32,10 @@ type Props = { setDatasetSplit: (v: string | null) => void; datasetEvalSplit: string | null; setDatasetEvalSplit: (v: string | null) => void; + datasetSliceStart?: string | null; + setDatasetSliceStart?: (v: string | null) => void; + datasetSliceEnd?: string | null; + setDatasetSliceEnd?: (v: string | null) => void; }; export function HfDatasetSubsetSplitSelectors({ @@ -44,6 +49,10 @@ export function HfDatasetSubsetSplitSelectors({ setDatasetSplit, datasetEvalSplit, setDatasetEvalSplit, + datasetSliceStart, + setDatasetSliceStart, + datasetSliceEnd, + setDatasetSliceEnd, }: Props) { const { subsets: hfSubsets, @@ -155,16 +164,89 @@ export function HfDatasetSubsetSplitSelectors({ /> )} - + {variant === "studio" && setDatasetSliceStart && setDatasetSliceEnd ? ( +
+ +
+ + Slice Start + + + + + + Inclusive start row index. Leave empty to start from the beginning. + + + + + setDatasetSliceStart(e.target.value || null) + } + /> +
+
+ + Slice End + + + + + + Inclusive end row index. Leave empty to use all remaining rows. + + + + + setDatasetSliceEnd(e.target.value || null) + } + /> +
+
+ ) : ( + + )} )} From 263cd19c44bd3a9b2c3cb1c9369bd219fec11fa7 Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Wed, 4 Mar 2026 23:07:36 +0000 Subject: [PATCH 54/66] refactor: move train split slice controls back to Advanced section Place Train Split Start / End inputs inside the Advanced collapsible with descriptive tooltips clarifying they slice the training split. Revert the selectors component to its original eval-split-only layout. --- .../studio/sections/dataset-section.tsx | 164 ++++++++++++------ .../hf-dataset-subset-split-selectors.tsx | 102 ++--------- 2 files changed, 125 insertions(+), 141 deletions(-) diff --git a/studio/frontend/src/features/studio/sections/dataset-section.tsx b/studio/frontend/src/features/studio/sections/dataset-section.tsx index 338a264579..bb65f704ac 100644 --- a/studio/frontend/src/features/studio/sections/dataset-section.tsx +++ b/studio/frontend/src/features/studio/sections/dataset-section.tsx @@ -13,6 +13,7 @@ import { ComboboxItem, ComboboxList, } from "@/components/ui/combobox"; +import { Input } from "@/components/ui/input"; import { InputGroupAddon } from "@/components/ui/input-group"; import { Select, @@ -290,10 +291,6 @@ export function DatasetSection() { setDatasetSplit={setDatasetSplit} datasetEvalSplit={datasetEvalSplit} setDatasetEvalSplit={setDatasetEvalSplit} - datasetSliceStart={datasetSliceStart} - setDatasetSliceStart={setDatasetSliceStart} - datasetSliceEnd={datasetSliceEnd} - setDatasetSliceEnd={setDatasetSliceEnd} /> @@ -305,51 +302,120 @@ export function DatasetSection() { Advanced -
- - Target Format - - - - - - Format of your training data. Auto-detect works for most - datasets.{" "} - - Read more - - - - - +
+
+ + Target Format + + + + + + Format of your training data. Auto-detect works for most + datasets.{" "} + + Read more + + + + + +
+
+
+ + Train Split Start + + + + + + Only train on a subset of your training split by + specifying a start row index (inclusive, 0-based). + Useful for resuming from a checkpoint or debugging + with a smaller slice. Leave empty to start from the + first row. + + + + + setDatasetSliceStart(e.target.value || null) + } + /> +
+
+ + Train Split End + + + + + + Last row index to include from the training split + (inclusive, 0-based). For example, set Start to 0 and + End to 99 to train on the first 100 rows. Leave empty + to use all remaining rows. + + + + + setDatasetSliceEnd(e.target.value || null) + } + /> +
+
diff --git a/studio/frontend/src/features/training/components/hf-dataset-subset-split-selectors.tsx b/studio/frontend/src/features/training/components/hf-dataset-subset-split-selectors.tsx index 35bfec8cd3..148341e4de 100644 --- a/studio/frontend/src/features/training/components/hf-dataset-subset-split-selectors.tsx +++ b/studio/frontend/src/features/training/components/hf-dataset-subset-split-selectors.tsx @@ -5,7 +5,6 @@ import { SelectTrigger, SelectValue, } from "@/components/ui/select"; -import { Input } from "@/components/ui/input"; import { Spinner } from "@/components/ui/spinner"; import { Tooltip, @@ -32,10 +31,6 @@ type Props = { setDatasetSplit: (v: string | null) => void; datasetEvalSplit: string | null; setDatasetEvalSplit: (v: string | null) => void; - datasetSliceStart?: string | null; - setDatasetSliceStart?: (v: string | null) => void; - datasetSliceEnd?: string | null; - setDatasetSliceEnd?: (v: string | null) => void; }; export function HfDatasetSubsetSplitSelectors({ @@ -49,10 +44,6 @@ export function HfDatasetSubsetSplitSelectors({ setDatasetSplit, datasetEvalSplit, setDatasetEvalSplit, - datasetSliceStart, - setDatasetSliceStart, - datasetSliceEnd, - setDatasetSliceEnd, }: Props) { const { subsets: hfSubsets, @@ -164,89 +155,16 @@ export function HfDatasetSubsetSplitSelectors({ /> )} - {variant === "studio" && setDatasetSliceStart && setDatasetSliceEnd ? ( -
- -
- - Slice Start - - - - - - Inclusive start row index. Leave empty to start from the beginning. - - - - - setDatasetSliceStart(e.target.value || null) - } - /> -
-
- - Slice End - - - - - - Inclusive end row index. Leave empty to use all remaining rows. - - - - - setDatasetSliceEnd(e.target.value || null) - } - /> -
-
- ) : ( - - )} + )} From 945cfa9460cae7f40d04e525c9b8c2525615d8b9 Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Wed, 4 Mar 2026 23:15:49 +0000 Subject: [PATCH 55/66] fix: remove unnecessary tooltip copy from train split start --- .../frontend/src/features/studio/sections/dataset-section.tsx | 4 +--- 1 file changed, 1 insertion(+), 3 deletions(-) diff --git a/studio/frontend/src/features/studio/sections/dataset-section.tsx b/studio/frontend/src/features/studio/sections/dataset-section.tsx index bb65f704ac..28254adcc7 100644 --- a/studio/frontend/src/features/studio/sections/dataset-section.tsx +++ b/studio/frontend/src/features/studio/sections/dataset-section.tsx @@ -368,9 +368,7 @@ export function DatasetSection() { Only train on a subset of your training split by specifying a start row index (inclusive, 0-based). - Useful for resuming from a checkpoint or debugging - with a smaller slice. Leave empty to start from the - first row. + Leave empty to start from the first row. From bc244aeb2334ce7e73d5fbd7120eb1cf7853911a Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Wed, 4 Mar 2026 06:42:37 +0000 Subject: [PATCH 56/66] fix: cast URL image columns to HF Image() type in VLM conversion --- studio/backend/utils/datasets/format_conversion.py | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/studio/backend/utils/datasets/format_conversion.py b/studio/backend/utils/datasets/format_conversion.py index 6436e7a82a..4e6a6d9919 100644 --- a/studio/backend/utils/datasets/format_conversion.py +++ b/studio/backend/utils/datasets/format_conversion.py @@ -254,8 +254,15 @@ def convert_to_vlm_format( list: List of dicts with 'messages' field """ from PIL import Image + from datasets import Image as datasets_Image from .vlm_processing import generate_smart_vlm_instruction + # Cast string image columns (URLs or local paths) to HF Image() type + # so HuggingFace handles downloading, decoding, and caching transparently. + sample_value = next(iter(dataset))[image_column] + if isinstance(sample_value, str): + dataset = dataset.cast_column(image_column, datasets_Image()) + # Generate smart instruction if not provided if instruction is None: instruction_info = generate_smart_vlm_instruction( From b376c54c213d6812188ec2b3d75c98c880ab7de8 Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Wed, 4 Mar 2026 06:42:48 +0000 Subject: [PATCH 57/66] fix: abort training pipeline on dataset conversion failure --- studio/backend/core/training/trainer.py | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/studio/backend/core/training/trainer.py b/studio/backend/core/training/trainer.py index 5ce273f43c..f5b2f18245 100644 --- a/studio/backend/core/training/trainer.py +++ b/studio/backend/core/training/trainer.py @@ -482,6 +482,14 @@ class UnslothTrainer: print("Stopped during dataset formatting\n") return None + # Abort if dataset formatting/conversion failed + if not dataset_info.get("success", True): + errors = dataset_info.get("errors", []) + error_msg = "; ".join(errors) if errors else "Dataset formatting failed" + logger.error(f"Dataset conversion failed: {error_msg}") + self._update_progress(error=error_msg) + return None + self._update_progress(status_message=f"Dataset formatted and ready for training") print(f"Dataset formatted successfully\n") From e6eea64df59d286a9a27a3c9ff4b5c6aca161c19 Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Wed, 4 Mar 2026 07:39:35 +0000 Subject: [PATCH 58/66] test: add URL image loading comparison script --- studio/tests/test_url_image_loading.py | 132 +++++++++++++++++++++++++ 1 file changed, 132 insertions(+) create mode 100644 studio/tests/test_url_image_loading.py diff --git a/studio/tests/test_url_image_loading.py b/studio/tests/test_url_image_loading.py new file mode 100644 index 0000000000..4899bab8e3 --- /dev/null +++ b/studio/tests/test_url_image_loading.py @@ -0,0 +1,132 @@ +""" +Reproduce: VLM URL image loading with HF datasets. +Tests cast_column(Image()) vs manual download approaches. +Dataset: google-research-datasets/conceptual_captions (subset: labeled) +""" +from datasets import load_dataset, Image as datasets_Image, Dataset +from PIL import Image as PILImage +from io import BytesIO +from itertools import islice +import time + +DATASET = "google-research-datasets/conceptual_captions" +SUBSET = "labeled" +SPLIT = "train" +N_SAMPLES = 20 # small slice for testing + +print("=" * 60) +print("Loading dataset (streaming, first N samples)...") +print("=" * 60) +ds = load_dataset(DATASET, name=SUBSET, split=SPLIT, streaming=True) +rows = list(islice(ds, N_SAMPLES)) +dataset = Dataset.from_list(rows) + +print(f"Loaded {len(dataset)} samples") +print(f"Columns: {dataset.column_names}") +print(f"First image_url: {dataset[0]['image_url'][:100]}...") +print() + +# ─── Test 1: cast_column(Image()) — what we tried ─── +print("=" * 60) +print("TEST 1: cast_column(Image()) approach") +print("=" * 60) +try: + ds_cast = dataset.cast_column("image_url", datasets_Image()) + success, fail = 0, 0 + t0 = time.time() + for i, sample in enumerate(ds_cast): + try: + img = sample["image_url"] + if img is not None: + print(f" [{i}] OK — {img.size} {img.mode}") + success += 1 + else: + print(f" [{i}] None returned") + fail += 1 + except Exception as e: + print(f" [{i}] FAILED: {type(e).__name__}: {str(e)[:80]}") + fail += 1 + elapsed = time.time() - t0 + print(f"\nResult: {success} ok, {fail} failed, {elapsed:.1f}s") +except Exception as e: + print(f"CRASHED during iteration: {type(e).__name__}: {str(e)[:120]}") +print() + +# ─── Test 2: Manual download with requests.Session ─── +print("=" * 60) +print("TEST 2: requests.Session() approach") +print("=" * 60) +try: + import requests + session = requests.Session() + success, fail = 0, 0 + t0 = time.time() + for i, sample in enumerate(dataset): + url = sample["image_url"] + try: + resp = session.get(url, timeout=10) + resp.raise_for_status() + img = PILImage.open(BytesIO(resp.content)).convert("RGB") + print(f" [{i}] OK — {img.size} {img.mode}") + success += 1 + except Exception as e: + print(f" [{i}] FAILED: {type(e).__name__}: {str(e)[:80]}") + fail += 1 + elapsed = time.time() - t0 + print(f"\nResult: {success} ok, {fail} failed, {elapsed:.1f}s") +except Exception as e: + print(f"CRASHED: {type(e).__name__}: {str(e)[:120]}") +print() + +# ─── Test 3: urllib (stdlib) ─── +print("=" * 60) +print("TEST 3: urllib approach (stdlib)") +print("=" * 60) +try: + from urllib.request import urlopen, Request + success, fail = 0, 0 + t0 = time.time() + for i, sample in enumerate(dataset): + url = sample["image_url"] + try: + req = Request(url, headers={"User-Agent": "Mozilla/5.0"}) + with urlopen(req, timeout=10) as resp: + img = PILImage.open(BytesIO(resp.read())).convert("RGB") + print(f" [{i}] OK — {img.size} {img.mode}") + success += 1 + except Exception as e: + print(f" [{i}] FAILED: {type(e).__name__}: {str(e)[:80]}") + fail += 1 + elapsed = time.time() - t0 + print(f"\nResult: {success} ok, {fail} failed, {elapsed:.1f}s") +except Exception as e: + print(f"CRASHED: {type(e).__name__}: {str(e)[:120]}") +print() + +# ─── Test 4: fsspec directly with expand=True ─── +print("=" * 60) +print("TEST 4: fsspec.open() with expand=True") +print("=" * 60) +try: + import fsspec + success, fail = 0, 0 + t0 = time.time() + for i, sample in enumerate(dataset): + url = sample["image_url"] + try: + with fsspec.open(url, "rb", expand=True) as f: + img = PILImage.open(BytesIO(f.read())).convert("RGB") + print(f" [{i}] OK — {img.size} {img.mode}") + success += 1 + except Exception as e: + print(f" [{i}] FAILED: {type(e).__name__}: {str(e)[:80]}") + fail += 1 + elapsed = time.time() - t0 + print(f"\nResult: {success} ok, {fail} failed, {elapsed:.1f}s") +except Exception as e: + print(f"CRASHED: {type(e).__name__}: {str(e)[:120]}") + +print() +print("=" * 60) +print("DONE — compare success rates and timing above") +print("=" * 60) From 9487d17b9409b7de35359c3444e07ef21041f4f1 Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Wed, 4 Mar 2026 07:50:55 +0000 Subject: [PATCH 59/66] fix: use fsspec for URL image downloads with per-sample error handling --- .../utils/datasets/format_conversion.py | 50 +++++++++++++------ 1 file changed, 36 insertions(+), 14 deletions(-) diff --git a/studio/backend/utils/datasets/format_conversion.py b/studio/backend/utils/datasets/format_conversion.py index 4e6a6d9919..37bd3c103c 100644 --- a/studio/backend/utils/datasets/format_conversion.py +++ b/studio/backend/utils/datasets/format_conversion.py @@ -254,15 +254,8 @@ def convert_to_vlm_format( list: List of dicts with 'messages' field """ from PIL import Image - from datasets import Image as datasets_Image from .vlm_processing import generate_smart_vlm_instruction - # Cast string image columns (URLs or local paths) to HF Image() type - # so HuggingFace handles downloading, decoding, and caching transparently. - sample_value = next(iter(dataset))[image_column] - if isinstance(sample_value, str): - dataset = dataset.cast_column(image_column, datasets_Image()) - # Generate smart instruction if not provided if instruction is None: instruction_info = generate_smart_vlm_instruction( @@ -288,12 +281,17 @@ def convert_to_vlm_format( def _convert_single_sample(sample): """Convert a single sample to VLM format.""" - # Get image (might be PIL Image or path) + # Get image (might be PIL Image, local path, or URL) image_data = sample[image_column] - # Handle image paths if isinstance(image_data, str): - image_data = Image.open(image_data).convert("RGB") + if image_data.startswith(("http://", "https://")): + import fsspec + from io import BytesIO + with fsspec.open(image_data, "rb", expand=True) as f: + image_data = Image.open(BytesIO(f.read())).convert("RGB") + else: + image_data = Image.open(image_data).convert("RGB") # Get text text_data = sample[text_column] @@ -324,11 +322,35 @@ def convert_to_vlm_format( # Return dict with messages return {"messages": messages} - # Use list comprehension and return the LIST directly - print(f"🔄 Converting {len(dataset)} samples to VLM format...") - converted_list = [_convert_single_sample(sample) for sample in dataset] + # Convert samples, skipping any with broken/unreachable images + total = len(dataset) + print(f"🔄 Converting {total} samples to VLM format...") + converted_list = [] + failed_count = 0 + for sample in dataset: + try: + converted_list.append(_convert_single_sample(sample)) + except Exception as e: + failed_count += 1 - print(f"✅ Converted {len(converted_list)} samples") + if failed_count > 0: + fail_rate = failed_count / total + print(f"⚠️ Skipped {failed_count}/{total} ({fail_rate:.0%}) samples with broken/unreachable images") + + if fail_rate >= 0.3: + raise ValueError( + f"{fail_rate:.0%} of images failed to download ({failed_count}/{total}). " + "This dataset has too many broken or unreachable image URLs to be usable for training. " + "Consider using a dataset with embedded images instead." + ) + + if len(converted_list) == 0: + raise ValueError( + f"All {total} samples failed during VLM conversion — no usable images found. " + "This dataset may contain only image URLs that are no longer accessible." + ) + + print(f"✅ Converted {len(converted_list)}/{total} samples") # Return list, NOT Dataset return converted_list From 63f723cc36e1dded0240f810fe0d40dca91504f0 Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Wed, 4 Mar 2026 08:05:40 +0000 Subject: [PATCH 60/66] fix: add early probe to fail fast on datasets with too many broken image URLs --- .../utils/datasets/format_conversion.py | 32 +++++++++++++------ 1 file changed, 23 insertions(+), 9 deletions(-) diff --git a/studio/backend/utils/datasets/format_conversion.py b/studio/backend/utils/datasets/format_conversion.py index 37bd3c103c..f9da8b1a1e 100644 --- a/studio/backend/utils/datasets/format_conversion.py +++ b/studio/backend/utils/datasets/format_conversion.py @@ -322,28 +322,42 @@ def convert_to_vlm_format( # Return dict with messages return {"messages": messages} - # Convert samples, skipping any with broken/unreachable images + # Convert samples, skipping any with broken/unreachable images. + # For URL-based datasets, check the first PROBE_SIZE samples early to + # fail fast if too many images are broken, before downloading millions. + PROBE_SIZE = 5000 + MAX_FAIL_RATE = 0.3 + total = len(dataset) + has_urls = isinstance(next(iter(dataset))[image_column], str) + probe_needed = has_urls and total > PROBE_SIZE + print(f"🔄 Converting {total} samples to VLM format...") converted_list = [] failed_count = 0 - for sample in dataset: + + for i, sample in enumerate(dataset): try: converted_list.append(_convert_single_sample(sample)) except Exception as e: failed_count += 1 + # Early exit check after probing the first batch + if probe_needed and (i + 1) == PROBE_SIZE: + fail_rate = failed_count / PROBE_SIZE + if fail_rate >= MAX_FAIL_RATE: + raise ValueError( + f"{fail_rate:.0%} of the first {PROBE_SIZE} images failed to download " + f"({failed_count}/{PROBE_SIZE}). " + "This dataset has too many broken or unreachable image URLs. " + "Consider using a dataset with embedded images instead." + ) + print(f"✅ Probe passed: {failed_count}/{PROBE_SIZE} ({fail_rate:.0%}) failures in first batch, continuing...") + if failed_count > 0: fail_rate = failed_count / total print(f"⚠️ Skipped {failed_count}/{total} ({fail_rate:.0%}) samples with broken/unreachable images") - if fail_rate >= 0.3: - raise ValueError( - f"{fail_rate:.0%} of images failed to download ({failed_count}/{total}). " - "This dataset has too many broken or unreachable image URLs to be usable for training. " - "Consider using a dataset with embedded images instead." - ) - if len(converted_list) == 0: raise ValueError( f"All {total} samples failed during VLM conversion — no usable images found. " From 1f03754c95dfbc9fe11ca5244f5b611094c54769 Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Wed, 4 Mar 2026 13:30:27 +0000 Subject: [PATCH 61/66] feat: add tqdm progress bar to VLM conversion and download benchmark test --- .../utils/datasets/format_conversion.py | 9 +++- studio/tests/test_url_download_benchmark.py | 54 +++++++++++++++++++ 2 files changed, 62 insertions(+), 1 deletion(-) create mode 100644 studio/tests/test_url_download_benchmark.py diff --git a/studio/backend/utils/datasets/format_conversion.py b/studio/backend/utils/datasets/format_conversion.py index f9da8b1a1e..e784e8e645 100644 --- a/studio/backend/utils/datasets/format_conversion.py +++ b/studio/backend/utils/datasets/format_conversion.py @@ -332,20 +332,26 @@ def convert_to_vlm_format( has_urls = isinstance(next(iter(dataset))[image_column], str) probe_needed = has_urls and total > PROBE_SIZE + from tqdm import tqdm + print(f"🔄 Converting {total} samples to VLM format...") converted_list = [] failed_count = 0 - for i, sample in enumerate(dataset): + pbar = tqdm(dataset, total=total, desc="Converting VLM samples", unit="sample") + for i, sample in enumerate(pbar): try: converted_list.append(_convert_single_sample(sample)) except Exception as e: failed_count += 1 + pbar.set_postfix(ok=len(converted_list), failed=failed_count, refresh=False) + # Early exit check after probing the first batch if probe_needed and (i + 1) == PROBE_SIZE: fail_rate = failed_count / PROBE_SIZE if fail_rate >= MAX_FAIL_RATE: + pbar.close() raise ValueError( f"{fail_rate:.0%} of the first {PROBE_SIZE} images failed to download " f"({failed_count}/{PROBE_SIZE}). " @@ -353,6 +359,7 @@ def convert_to_vlm_format( "Consider using a dataset with embedded images instead." ) print(f"✅ Probe passed: {failed_count}/{PROBE_SIZE} ({fail_rate:.0%}) failures in first batch, continuing...") + pbar.close() if failed_count > 0: fail_rate = failed_count / total diff --git a/studio/tests/test_url_download_benchmark.py b/studio/tests/test_url_download_benchmark.py new file mode 100644 index 0000000000..e29a8e05b6 --- /dev/null +++ b/studio/tests/test_url_download_benchmark.py @@ -0,0 +1,54 @@ +""" +Benchmark: fsspec URL image download throughput at different dataset sizes. +Dataset: google-research-datasets/conceptual_captions (subset: labeled) + +Tests sizes: 100, 200, 300, 500, 1000, 1500, 2000 +Reports: time, success/fail rate, throughput (images/sec) +""" +from datasets import load_dataset, Dataset +from PIL import Image as PILImage +from io import BytesIO +from itertools import islice +import fsspec +import time + +DATASET = "google-research-datasets/conceptual_captions" +SUBSET = "labeled" +SPLIT = "train" +SIZES = [100, 200, 300, 500, 1000, 1500, 2000] + +# Load the max we need in one go +max_size = max(SIZES) +print(f"Loading {max_size} samples from {DATASET} (streaming)...") +ds = load_dataset(DATASET, name=SUBSET, split=SPLIT, streaming=True) +rows = list(islice(ds, max_size)) +full_dataset = Dataset.from_list(rows) +print(f"Loaded {len(full_dataset)} samples") +print(f"Columns: {full_dataset.column_names}") +print() + +print(f"{'Size':>6} | {'Time':>8} | {'OK':>6} | {'Fail':>6} | {'Fail%':>6} | {'img/s':>7}") +print("-" * 55) + +for size in SIZES: + dataset = full_dataset.select(range(size)) + success, fail = 0, 0 + t0 = time.time() + + for sample in dataset: + url = sample["image_url"] + try: + with fsspec.open(url, "rb", expand=True) as f: + img = PILImage.open(BytesIO(f.read())).convert("RGB") + success += 1 + except Exception: + fail += 1 + + elapsed = time.time() - t0 + fail_pct = (fail / size) * 100 + throughput = success / elapsed if elapsed > 0 else 0 + + print(f"{size:>6} | {elapsed:>7.1f}s | {success:>6} | {fail:>6} | {fail_pct:>5.1f}% | {throughput:>6.1f}/s") + +print() +print("Done.") From db30b4105f44b5a271f88c1e8b4bad48e47cdc10 Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Wed, 4 Mar 2026 14:30:11 +0000 Subject: [PATCH 62/66] test: add parallel download benchmark with ThreadPoolExecutor --- studio/tests/test_url_parallel_benchmark.py | 79 +++++++++++++++++++++ 1 file changed, 79 insertions(+) create mode 100644 studio/tests/test_url_parallel_benchmark.py diff --git a/studio/tests/test_url_parallel_benchmark.py b/studio/tests/test_url_parallel_benchmark.py new file mode 100644 index 0000000000..a0d160136c --- /dev/null +++ b/studio/tests/test_url_parallel_benchmark.py @@ -0,0 +1,79 @@ +""" +Benchmark: parallel fsspec URL image downloads with ThreadPoolExecutor. +Tests different worker counts to find optimal parallelism. +Dataset: google-research-datasets/conceptual_captions (subset: labeled) +""" +from datasets import load_dataset, Dataset +from PIL import Image as PILImage +from io import BytesIO +from itertools import islice +from concurrent.futures import ThreadPoolExecutor, as_completed +import fsspec +import time +import os + +DATASET = "google-research-datasets/conceptual_captions" +SUBSET = "labeled" +SPLIT = "train" +N_SAMPLES = 500 + +# safe_num_proc formula from studio/backend/utils/hardware/hardware.py +cpu_count = os.cpu_count() +safe_workers = max(1, cpu_count // 3) +print(f"CPU count: {cpu_count}, safe_num_proc: {safe_workers}") + +WORKER_COUNTS = [1, 4, 8, 16, 32, safe_workers] +# Deduplicate and sort +WORKER_COUNTS = sorted(set(WORKER_COUNTS)) + +print(f"Loading {N_SAMPLES} samples from {DATASET} (streaming)...") +ds = load_dataset(DATASET, name=SUBSET, split=SPLIT, streaming=True) +rows = list(islice(ds, N_SAMPLES)) +dataset = Dataset.from_list(rows) +urls = [row["image_url"] for row in dataset] +print(f"Loaded {len(urls)} URLs") +print() + + +def download_single(url): + """Download a single image URL using fsspec. Returns PIL image or raises.""" + with fsspec.open(url, "rb", expand=True) as f: + img = PILImage.open(BytesIO(f.read())).convert("RGB") + return img + + +print(f"{'Workers':>8} | {'Time':>8} | {'OK':>6} | {'Fail':>6} | {'Fail%':>6} | {'img/s':>7} | {'Speedup':>8}") +print("-" * 70) + +baseline_throughput = None + +for n_workers in WORKER_COUNTS: + success, fail = 0, 0 + t0 = time.time() + + with ThreadPoolExecutor(max_workers=n_workers) as pool: + futures = {pool.submit(download_single, url): url for url in urls} + for future in as_completed(futures): + try: + img = future.result(timeout=30) + success += 1 + except Exception: + fail += 1 + + elapsed = time.time() - t0 + fail_pct = (fail / N_SAMPLES) * 100 + throughput = success / elapsed if elapsed > 0 else 0 + + if baseline_throughput is None: + baseline_throughput = throughput + speedup = throughput / baseline_throughput if baseline_throughput > 0 else 0 + + label = f"{n_workers}" + if n_workers == safe_workers: + label += "*" # mark the safe_num_proc value + + print(f"{label:>8} | {elapsed:>7.1f}s | {success:>6} | {fail:>6} | {fail_pct:>5.1f}% | {throughput:>6.1f}/s | {speedup:>7.1f}x") + +print() +print("* = safe_num_proc value") +print("Done.") From e04b9d53d683a9c224eb6b7481081136edf353ea Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Wed, 4 Mar 2026 23:40:38 +0000 Subject: [PATCH 63/66] feat: parallel URL image probe with time estimate and progress reporting - Add 200-sample parallel probe using ThreadPoolExecutor + safe_num_proc to estimate download speed and failure rate before full conversion - Abort with clear error if >=30% of probe images fail to download - Show estimated download time in the training overlay modal - Parallel batch conversion for URL-based datasets (vs sequential for local) - Add warning field to /check-format response for URL-based image datasets - Display URL warning in dataset preview dialog (amber banner) - Thread progress_callback from trainer through format_and_template_dataset to convert_to_vlm_format for real-time status updates --- studio/backend/core/training/trainer.py | 1 + studio/backend/models/datasets.py | 1 + studio/backend/routes/datasets.py | 16 ++ .../backend/utils/datasets/dataset_utils.py | 3 + .../utils/datasets/format_conversion.py | 167 +++++++++++++++--- .../sections/dataset-preview-dialog.tsx | 7 + .../src/features/training/types/datasets.ts | 1 + 7 files changed, 169 insertions(+), 27 deletions(-) diff --git a/studio/backend/core/training/trainer.py b/studio/backend/core/training/trainer.py index f5b2f18245..f2e43f76f4 100644 --- a/studio/backend/core/training/trainer.py +++ b/studio/backend/core/training/trainer.py @@ -475,6 +475,7 @@ class UnslothTrainer: format_type=format_type, dataset_name=dataset_source, custom_format_mapping=custom_format_mapping, + progress_callback=self._update_progress, ) # Check if stopped during formatting diff --git a/studio/backend/models/datasets.py b/studio/backend/models/datasets.py index 81adef7577..18f6ec224b 100644 --- a/studio/backend/models/datasets.py +++ b/studio/backend/models/datasets.py @@ -34,3 +34,4 @@ class CheckFormatResponse(BaseModel): detected_text_column: Optional[str] = None preview_samples: Optional[List[Dict]] = None total_rows: Optional[int] = None + warning: Optional[str] = None diff --git a/studio/backend/routes/datasets.py b/studio/backend/routes/datasets.py index a53e223015..4a475ff8c2 100644 --- a/studio/backend/routes/datasets.py +++ b/studio/backend/routes/datasets.py @@ -211,6 +211,21 @@ def check_format( else: preview_samples = _serialize_preview_rows(preview_slice) + # Lightweight URL-based image detection for VLM datasets + warning = None + image_col = result.get("detected_image_column") + if image_col and image_col in (result.get("columns") or []): + try: + sample_val = preview_slice[0][image_col] + if isinstance(sample_val, str) and sample_val.startswith(("http://", "https://")): + warning = ( + "This dataset contains image URLs instead of embedded images. " + "Images will be downloaded during training, which may be slow for large datasets." + ) + logger.info(f"URL-based image column detected: {image_col}") + except Exception: + pass + return CheckFormatResponse( requires_manual_mapping=result["requires_manual_mapping"], detected_format=result["detected_format"], @@ -222,6 +237,7 @@ def check_format( detected_text_column=result.get("detected_text_column"), preview_samples=preview_samples, total_rows=total_rows, + warning=warning, ) except HTTPException: diff --git a/studio/backend/utils/datasets/dataset_utils.py b/studio/backend/utils/datasets/dataset_utils.py index 3a4d54f93f..9e1f54a75c 100644 --- a/studio/backend/utils/datasets/dataset_utils.py +++ b/studio/backend/utils/datasets/dataset_utils.py @@ -593,6 +593,7 @@ def format_and_template_dataset( aliases_for_assistant=["gpt", "assistant", "output",], batch_size=1000, num_proc=None, + progress_callback=None, ): """ Convenience function that combines format_dataset and apply_chat_template_to_dataset. @@ -638,6 +639,7 @@ def format_and_template_dataset( text_column=user_vlm_text_column, image_column=user_vlm_image_column, dataset_name=dataset_name, + progress_callback=progress_callback, ) warnings.append(f"Applied user VLM mapping: image='{user_vlm_image_column}', text='{user_vlm_text_column}'") @@ -734,6 +736,7 @@ def format_and_template_dataset( text_column=vlm_text_column, image_column=vlm_image_column, dataset_name=dataset_name, + progress_callback=progress_callback, ) if vlm_instruction: diff --git a/studio/backend/utils/datasets/format_conversion.py b/studio/backend/utils/datasets/format_conversion.py index e784e8e645..c5c9a4d6e7 100644 --- a/studio/backend/utils/datasets/format_conversion.py +++ b/studio/backend/utils/datasets/format_conversion.py @@ -238,24 +238,51 @@ def convert_alpaca_to_chatml(dataset, batch_size=1000, num_proc=None): return dataset.map(_convert, **dataset_map_kwargs) +def _format_eta(seconds): + """Format seconds into a human-readable ETA string.""" + if seconds < 60: + return f"{seconds:.0f}s" + elif seconds < 3600: + m, s = divmod(int(seconds), 60) + return f"{m}m {s}s" + else: + h, remainder = divmod(int(seconds), 3600) + m, _ = divmod(remainder, 60) + return f"{h}h {m}m" + + def convert_to_vlm_format( dataset, instruction=None, text_column="text", image_column="image", dataset_name=None, + progress_callback=None, ): """ Converts simple {image, text} format to VLM messages format. Returns a LIST, not a HuggingFace Dataset (to preserve PIL Images). + For URL-based image datasets, runs a 200-sample parallel probe first to + estimate download speed and failure rate, then reports time estimate or + warning through progress_callback before proceeding with the full conversion. + + Args: + progress_callback: Optional callable(status_message=str) to report + progress to the training overlay. + Returns: list: List of dicts with 'messages' field """ from PIL import Image from .vlm_processing import generate_smart_vlm_instruction + def _notify(msg): + """Send status update to the training overlay if callback is available.""" + if progress_callback: + progress_callback(status_message=msg) + # Generate smart instruction if not provided if instruction is None: instruction_info = generate_smart_vlm_instruction( @@ -322,48 +349,133 @@ def convert_to_vlm_format( # Return dict with messages return {"messages": messages} - # Convert samples, skipping any with broken/unreachable images. - # For URL-based datasets, check the first PROBE_SIZE samples early to - # fail fast if too many images are broken, before downloading millions. - PROBE_SIZE = 5000 - MAX_FAIL_RATE = 0.3 - total = len(dataset) has_urls = isinstance(next(iter(dataset))[image_column], str) - probe_needed = has_urls and total > PROBE_SIZE + # ── URL probe: 200 samples with parallel workers to estimate speed + failure rate ── + PROBE_SIZE = 200 + MAX_FAIL_RATE = 0.3 + + if has_urls and total > PROBE_SIZE: + import time + from concurrent.futures import ThreadPoolExecutor, as_completed + from utils.hardware import safe_num_proc + + num_workers = safe_num_proc() + _notify(f"Probing {PROBE_SIZE} image URLs with {num_workers} workers...") + print(f"🔍 Probing {PROBE_SIZE}/{total} image URLs with {num_workers} workers...") + + probe_samples = [dataset[i] for i in range(PROBE_SIZE)] + probe_ok = 0 + probe_fail = 0 + probe_start = time.time() + + with ThreadPoolExecutor(max_workers=num_workers) as executor: + futures = {executor.submit(_convert_single_sample, s): s for s in probe_samples} + for future in as_completed(futures): + try: + future.result() + probe_ok += 1 + except Exception: + probe_fail += 1 + + probe_elapsed = time.time() - probe_start + probe_total = probe_ok + probe_fail + fail_rate = probe_fail / probe_total if probe_total > 0 else 0 + throughput = probe_total / probe_elapsed if probe_elapsed > 0 else 0 + + if fail_rate >= MAX_FAIL_RATE: + msg = ( + f"⚠️ {fail_rate:.0%} of the first {PROBE_SIZE} images failed to download " + f"({probe_fail}/{probe_total}). " + "This dataset has too many broken or unreachable image URLs. " + "Consider using a dataset with embedded images instead." + ) + print(msg) + _notify(msg) + raise ValueError(msg) + + # Estimate total time for remaining samples + remaining = total - PROBE_SIZE + estimated_seconds = remaining / throughput if throughput > 0 else 0 + eta_str = _format_eta(estimated_seconds) + + info_msg = ( + f"Downloading {total:,} images ({num_workers} workers, ~{throughput:.1f} img/s). " + f"Estimated time: ~{eta_str}" + ) + if probe_fail > 0: + info_msg += f" | {fail_rate:.0%} broken URLs will be skipped" + + print(f"✅ Probe passed: {probe_ok}/{probe_total} ok, {probe_fail} failed ({fail_rate:.0%}), {throughput:.1f} img/s") + print(f"⏱️ Estimated time for {total:,} samples: ~{eta_str}") + _notify(info_msg) + + # ── Full conversion with progress ── from tqdm import tqdm print(f"🔄 Converting {total} samples to VLM format...") converted_list = [] failed_count = 0 - pbar = tqdm(dataset, total=total, desc="Converting VLM samples", unit="sample") - for i, sample in enumerate(pbar): - try: - converted_list.append(_convert_single_sample(sample)) - except Exception as e: - failed_count += 1 + if has_urls: + # Parallel conversion for URL-based datasets + import time + from concurrent.futures import ThreadPoolExecutor, as_completed + from utils.hardware import safe_num_proc - pbar.set_postfix(ok=len(converted_list), failed=failed_count, refresh=False) + num_workers = safe_num_proc() + batch_size = 500 + start_time = time.time() - # Early exit check after probing the first batch - if probe_needed and (i + 1) == PROBE_SIZE: - fail_rate = failed_count / PROBE_SIZE - if fail_rate >= MAX_FAIL_RATE: - pbar.close() - raise ValueError( - f"{fail_rate:.0%} of the first {PROBE_SIZE} images failed to download " - f"({failed_count}/{PROBE_SIZE}). " - "This dataset has too many broken or unreachable image URLs. " - "Consider using a dataset with embedded images instead." - ) - print(f"✅ Probe passed: {failed_count}/{PROBE_SIZE} ({fail_rate:.0%}) failures in first batch, continuing...") - pbar.close() + for batch_start in range(0, total, batch_size): + batch_end = min(batch_start + batch_size, total) + batch_samples = [dataset[i] for i in range(batch_start, batch_end)] + + with ThreadPoolExecutor(max_workers=num_workers) as executor: + futures = {executor.submit(_convert_single_sample, s): i for i, s in enumerate(batch_samples)} + batch_results = [None] * len(batch_samples) + for future in as_completed(futures): + idx = futures[future] + try: + batch_results[idx] = future.result() + except Exception: + failed_count += 1 + + converted_list.extend(r for r in batch_results if r is not None) + + # Progress update every batch + elapsed = time.time() - start_time + done = batch_end + rate = done / elapsed if elapsed > 0 else 0 + remaining_time = (total - done) / rate if rate > 0 else 0 + eta_str = _format_eta(remaining_time) + progress_msg = f"Downloading images: {done:,}/{total:,} ({done*100//total}%) | ~{eta_str} remaining | {failed_count} skipped" + print(f" [{done}/{total}] {rate:.1f} img/s, {failed_count} failed, ETA {eta_str}") + _notify(progress_msg) + else: + # Sequential conversion for local/embedded images (fast, no I/O bottleneck) + pbar = tqdm(dataset, total=total, desc="Converting VLM samples", unit="sample") + for sample in pbar: + try: + converted_list.append(_convert_single_sample(sample)) + except Exception: + failed_count += 1 + pbar.set_postfix(ok=len(converted_list), failed=failed_count, refresh=False) + pbar.close() if failed_count > 0: fail_rate = failed_count / total print(f"⚠️ Skipped {failed_count}/{total} ({fail_rate:.0%}) samples with broken/unreachable images") + # For datasets that skipped the probe (small URL datasets), check fail rate now + if has_urls and fail_rate >= MAX_FAIL_RATE: + msg = ( + f"⚠️ {fail_rate:.0%} of images failed to download ({failed_count}/{total}). " + "This dataset has too many broken or unreachable image URLs. " + "Consider using a dataset with embedded images instead." + ) + _notify(msg) + raise ValueError(msg) if len(converted_list) == 0: raise ValueError( @@ -372,6 +484,7 @@ def convert_to_vlm_format( ) print(f"✅ Converted {len(converted_list)}/{total} samples") + _notify(f"Converted {len(converted_list):,}/{total:,} images successfully") # Return list, NOT Dataset return converted_list diff --git a/studio/frontend/src/features/studio/sections/dataset-preview-dialog.tsx b/studio/frontend/src/features/studio/sections/dataset-preview-dialog.tsx index 00a738bc64..516f3612c0 100644 --- a/studio/frontend/src/features/studio/sections/dataset-preview-dialog.tsx +++ b/studio/frontend/src/features/studio/sections/dataset-preview-dialog.tsx @@ -340,6 +340,13 @@ export function DatasetPreviewDialog({ />
+ {data.warning && ( +
+ + {data.warning} +
+ )} + {mappingEnabled && ( Date: Wed, 4 Mar 2026 23:42:23 +0000 Subject: [PATCH 64/66] fix: clear dataset slice state when switching to uploaded file Prevents stale slice values from silently truncating uploaded datasets. --- .../src/features/training/stores/training-config-store.ts | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/studio/frontend/src/features/training/stores/training-config-store.ts b/studio/frontend/src/features/training/stores/training-config-store.ts index 93c742ba98..8aeb54af78 100644 --- a/studio/frontend/src/features/training/stores/training-config-store.ts +++ b/studio/frontend/src/features/training/stores/training-config-store.ts @@ -317,7 +317,8 @@ export const useTrainingConfigStore = create()( set({ datasetManualMapping }), setDatasetSliceStart: (datasetSliceStart) => set({ datasetSliceStart }), setDatasetSliceEnd: (datasetSliceEnd) => set({ datasetSliceEnd }), - setUploadedFile: (uploadedFile) => set({ uploadedFile }), + setUploadedFile: (uploadedFile) => + set({ uploadedFile, datasetSliceStart: null, datasetSliceEnd: null }), setEpochs: (epochs) => set({ epochs }), setContextLength: (contextLength) => set({ contextLength }), setLearningRate: (learningRate) => set({ learningRate }), From 2116cc5ccaf2f8df8672e2922011a70a7470308e Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Thu, 5 Mar 2026 06:06:47 +0000 Subject: [PATCH 65/66] fix: remove benchmark scripts from git tracking These are standalone benchmark scripts that were force-added despite being gitignored. They have no test functions and run network calls at module level, which breaks pytest collection in CI. --- studio/tests/test_url_download_benchmark.py | 54 -------- studio/tests/test_url_image_loading.py | 132 -------------------- studio/tests/test_url_parallel_benchmark.py | 79 ------------ 3 files changed, 265 deletions(-) delete mode 100644 studio/tests/test_url_download_benchmark.py delete mode 100644 studio/tests/test_url_image_loading.py delete mode 100644 studio/tests/test_url_parallel_benchmark.py diff --git a/studio/tests/test_url_download_benchmark.py b/studio/tests/test_url_download_benchmark.py deleted file mode 100644 index e29a8e05b6..0000000000 --- a/studio/tests/test_url_download_benchmark.py +++ /dev/null @@ -1,54 +0,0 @@ -""" -Benchmark: fsspec URL image download throughput at different dataset sizes. -Dataset: google-research-datasets/conceptual_captions (subset: labeled) - -Tests sizes: 100, 200, 300, 500, 1000, 1500, 2000 -Reports: time, success/fail rate, throughput (images/sec) -""" -from datasets import load_dataset, Dataset -from PIL import Image as PILImage -from io import BytesIO -from itertools import islice -import fsspec -import time - -DATASET = "google-research-datasets/conceptual_captions" -SUBSET = "labeled" -SPLIT = "train" -SIZES = [100, 200, 300, 500, 1000, 1500, 2000] - -# Load the max we need in one go -max_size = max(SIZES) -print(f"Loading {max_size} samples from {DATASET} (streaming)...") -ds = load_dataset(DATASET, name=SUBSET, split=SPLIT, streaming=True) -rows = list(islice(ds, max_size)) -full_dataset = Dataset.from_list(rows) -print(f"Loaded {len(full_dataset)} samples") -print(f"Columns: {full_dataset.column_names}") -print() - -print(f"{'Size':>6} | {'Time':>8} | {'OK':>6} | {'Fail':>6} | {'Fail%':>6} | {'img/s':>7}") -print("-" * 55) - -for size in SIZES: - dataset = full_dataset.select(range(size)) - success, fail = 0, 0 - t0 = time.time() - - for sample in dataset: - url = sample["image_url"] - try: - with fsspec.open(url, "rb", expand=True) as f: - img = PILImage.open(BytesIO(f.read())).convert("RGB") - success += 1 - except Exception: - fail += 1 - - elapsed = time.time() - t0 - fail_pct = (fail / size) * 100 - throughput = success / elapsed if elapsed > 0 else 0 - - print(f"{size:>6} | {elapsed:>7.1f}s | {success:>6} | {fail:>6} | {fail_pct:>5.1f}% | {throughput:>6.1f}/s") - -print() -print("Done.") diff --git a/studio/tests/test_url_image_loading.py b/studio/tests/test_url_image_loading.py deleted file mode 100644 index 4899bab8e3..0000000000 --- a/studio/tests/test_url_image_loading.py +++ /dev/null @@ -1,132 +0,0 @@ -""" -Reproduce: VLM URL image loading with HF datasets. -Tests cast_column(Image()) vs manual download approaches. -Dataset: google-research-datasets/conceptual_captions (subset: labeled) -""" -from datasets import load_dataset, Image as datasets_Image, Dataset -from PIL import Image as PILImage -from io import BytesIO -from itertools import islice -import time - -DATASET = "google-research-datasets/conceptual_captions" -SUBSET = "labeled" -SPLIT = "train" -N_SAMPLES = 20 # small slice for testing - -print("=" * 60) -print("Loading dataset (streaming, first N samples)...") -print("=" * 60) -ds = load_dataset(DATASET, name=SUBSET, split=SPLIT, streaming=True) -rows = list(islice(ds, N_SAMPLES)) -dataset = Dataset.from_list(rows) - -print(f"Loaded {len(dataset)} samples") -print(f"Columns: {dataset.column_names}") -print(f"First image_url: {dataset[0]['image_url'][:100]}...") -print() - -# ─── Test 1: cast_column(Image()) — what we tried ─── -print("=" * 60) -print("TEST 1: cast_column(Image()) approach") -print("=" * 60) -try: - ds_cast = dataset.cast_column("image_url", datasets_Image()) - success, fail = 0, 0 - t0 = time.time() - for i, sample in enumerate(ds_cast): - try: - img = sample["image_url"] - if img is not None: - print(f" [{i}] OK — {img.size} {img.mode}") - success += 1 - else: - print(f" [{i}] None returned") - fail += 1 - except Exception as e: - print(f" [{i}] FAILED: {type(e).__name__}: {str(e)[:80]}") - fail += 1 - elapsed = time.time() - t0 - print(f"\nResult: {success} ok, {fail} failed, {elapsed:.1f}s") -except Exception as e: - print(f"CRASHED during iteration: {type(e).__name__}: {str(e)[:120]}") -print() - -# ─── Test 2: Manual download with requests.Session ─── -print("=" * 60) -print("TEST 2: requests.Session() approach") -print("=" * 60) -try: - import requests - session = requests.Session() - success, fail = 0, 0 - t0 = time.time() - for i, sample in enumerate(dataset): - url = sample["image_url"] - try: - resp = session.get(url, timeout=10) - resp.raise_for_status() - img = PILImage.open(BytesIO(resp.content)).convert("RGB") - print(f" [{i}] OK — {img.size} {img.mode}") - success += 1 - except Exception as e: - print(f" [{i}] FAILED: {type(e).__name__}: {str(e)[:80]}") - fail += 1 - elapsed = time.time() - t0 - print(f"\nResult: {success} ok, {fail} failed, {elapsed:.1f}s") -except Exception as e: - print(f"CRASHED: {type(e).__name__}: {str(e)[:120]}") -print() - -# ─── Test 3: urllib (stdlib) ─── -print("=" * 60) -print("TEST 3: urllib approach (stdlib)") -print("=" * 60) -try: - from urllib.request import urlopen, Request - success, fail = 0, 0 - t0 = time.time() - for i, sample in enumerate(dataset): - url = sample["image_url"] - try: - req = Request(url, headers={"User-Agent": "Mozilla/5.0"}) - with urlopen(req, timeout=10) as resp: - img = PILImage.open(BytesIO(resp.read())).convert("RGB") - print(f" [{i}] OK — {img.size} {img.mode}") - success += 1 - except Exception as e: - print(f" [{i}] FAILED: {type(e).__name__}: {str(e)[:80]}") - fail += 1 - elapsed = time.time() - t0 - print(f"\nResult: {success} ok, {fail} failed, {elapsed:.1f}s") -except Exception as e: - print(f"CRASHED: {type(e).__name__}: {str(e)[:120]}") -print() - -# ─── Test 4: fsspec directly with expand=True ─── -print("=" * 60) -print("TEST 4: fsspec.open() with expand=True") -print("=" * 60) -try: - import fsspec - success, fail = 0, 0 - t0 = time.time() - for i, sample in enumerate(dataset): - url = sample["image_url"] - try: - with fsspec.open(url, "rb", expand=True) as f: - img = PILImage.open(BytesIO(f.read())).convert("RGB") - print(f" [{i}] OK — {img.size} {img.mode}") - success += 1 - except Exception as e: - print(f" [{i}] FAILED: {type(e).__name__}: {str(e)[:80]}") - fail += 1 - elapsed = time.time() - t0 - print(f"\nResult: {success} ok, {fail} failed, {elapsed:.1f}s") -except Exception as e: - print(f"CRASHED: {type(e).__name__}: {str(e)[:120]}") - -print() -print("=" * 60) -print("DONE — compare success rates and timing above") -print("=" * 60) diff --git a/studio/tests/test_url_parallel_benchmark.py b/studio/tests/test_url_parallel_benchmark.py deleted file mode 100644 index a0d160136c..0000000000 --- a/studio/tests/test_url_parallel_benchmark.py +++ /dev/null @@ -1,79 +0,0 @@ -""" -Benchmark: parallel fsspec URL image downloads with ThreadPoolExecutor. -Tests different worker counts to find optimal parallelism. -Dataset: google-research-datasets/conceptual_captions (subset: labeled) -""" -from datasets import load_dataset, Dataset -from PIL import Image as PILImage -from io import BytesIO -from itertools import islice -from concurrent.futures import ThreadPoolExecutor, as_completed -import fsspec -import time -import os - -DATASET = "google-research-datasets/conceptual_captions" -SUBSET = "labeled" -SPLIT = "train" -N_SAMPLES = 500 - -# safe_num_proc formula from studio/backend/utils/hardware/hardware.py -cpu_count = os.cpu_count() -safe_workers = max(1, cpu_count // 3) -print(f"CPU count: {cpu_count}, safe_num_proc: {safe_workers}") - -WORKER_COUNTS = [1, 4, 8, 16, 32, safe_workers] -# Deduplicate and sort -WORKER_COUNTS = sorted(set(WORKER_COUNTS)) - -print(f"Loading {N_SAMPLES} samples from {DATASET} (streaming)...") -ds = load_dataset(DATASET, name=SUBSET, split=SPLIT, streaming=True) -rows = list(islice(ds, N_SAMPLES)) -dataset = Dataset.from_list(rows) -urls = [row["image_url"] for row in dataset] -print(f"Loaded {len(urls)} URLs") -print() - - -def download_single(url): - """Download a single image URL using fsspec. Returns PIL image or raises.""" - with fsspec.open(url, "rb", expand=True) as f: - img = PILImage.open(BytesIO(f.read())).convert("RGB") - return img - - -print(f"{'Workers':>8} | {'Time':>8} | {'OK':>6} | {'Fail':>6} | {'Fail%':>6} | {'img/s':>7} | {'Speedup':>8}") -print("-" * 70) - -baseline_throughput = None - -for n_workers in WORKER_COUNTS: - success, fail = 0, 0 - t0 = time.time() - - with ThreadPoolExecutor(max_workers=n_workers) as pool: - futures = {pool.submit(download_single, url): url for url in urls} - for future in as_completed(futures): - try: - img = future.result(timeout=30) - success += 1 - except Exception: - fail += 1 - - elapsed = time.time() - t0 - fail_pct = (fail / N_SAMPLES) * 100 - throughput = success / elapsed if elapsed > 0 else 0 - - if baseline_throughput is None: - baseline_throughput = throughput - speedup = throughput / baseline_throughput if baseline_throughput > 0 else 0 - - label = f"{n_workers}" - if n_workers == safe_workers: - label += "*" # mark the safe_num_proc value - - print(f"{label:>8} | {elapsed:>7.1f}s | {success:>6} | {fail:>6} | {fail_pct:>5.1f}% | {throughput:>6.1f}/s | {speedup:>7.1f}x") - -print() -print("* = safe_num_proc value") -print("Done.") From a1706c894f7d59b8372f955527e0c6178905e7b7 Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Thu, 5 Mar 2026 06:10:10 +0000 Subject: [PATCH 66/66] fix: check for http(s) prefix instead of bare string type for URL detection --- studio/backend/utils/datasets/format_conversion.py | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/studio/backend/utils/datasets/format_conversion.py b/studio/backend/utils/datasets/format_conversion.py index c5c9a4d6e7..41a9617857 100644 --- a/studio/backend/utils/datasets/format_conversion.py +++ b/studio/backend/utils/datasets/format_conversion.py @@ -350,7 +350,8 @@ def convert_to_vlm_format( return {"messages": messages} total = len(dataset) - has_urls = isinstance(next(iter(dataset))[image_column], str) + first_image = next(iter(dataset))[image_column] + has_urls = isinstance(first_image, str) and first_image.startswith(("http://", "https://")) # ── URL probe: 200 samples with parallel workers to estimate speed + failure rate ── PROBE_SIZE = 200