diff --git a/README.md b/README.md index fef79e730a..9ad41546cf 100644 --- a/README.md +++ b/README.md @@ -23,7 +23,7 @@ Notebooks are beginner friendly. Read our [guide](https://docs.unsloth.ai/get-st | Unsloth supports | Free Notebooks | Performance | Memory use | |-----------|---------|--------|----------| | **Qwen3 (14B)** | [▶️ Start for free](https://colab.research.google.com/github/unslothai/notebooks/blob/main/nb/Qwen3_(14B)-Reasoning-Conversational.ipynb) | 2x faster | 70% less | -| **GRPO (R1 reasoning)** | [▶️ Start for free](https://colab.research.google.com/github/unslothai/notebooks/blob/main/nb/Llama3.1_(8B)-GRPO.ipynb) | 2x faster | 80% less | +| **GRPO (reasoning)** | [▶️ Start for free](https://colab.research.google.com/github/unslothai/notebooks/blob/main/nb/Llama3.1_(8B)-GRPO.ipynb) | 2x faster | 80% less | | **Gemma 3 (4B)** | [▶️ Start for free](https://colab.research.google.com/github/unslothai/notebooks/blob/main/nb/Gemma3_(4B).ipynb) | 1.6x faster | 60% less | | **Llama 3.2 (3B)** | [▶️ Start for free](https://colab.research.google.com/github/unslothai/notebooks/blob/main/nb/Llama3.2_(1B_and_3B)-Conversational.ipynb) | 2x faster | 70% less | | **Phi-4 (14B)** | [▶️ Start for free](https://colab.research.google.com/github/unslothai/notebooks/blob/main/nb/Phi_4-Conversational.ipynb) | 2x faster | 70% less | @@ -31,7 +31,7 @@ Notebooks are beginner friendly. Read our [guide](https://docs.unsloth.ai/get-st | **Llama 3.1 (8B)** | [▶️ Start for free](https://colab.research.google.com/github/unslothai/notebooks/blob/main/nb/Llama3.1_(8B)-Alpaca.ipynb) | 2x faster | 70% less | | **Mistral v0.3 (7B)** | [▶️ Start for free](https://colab.research.google.com/github/unslothai/notebooks/blob/main/nb/Mistral_v0.3_(7B)-Conversational.ipynb) | 2.2x faster | 75% less | | **Ollama** | [▶️ Start for free](https://colab.research.google.com/github/unslothai/notebooks/blob/main/nb/Llama3_(8B)-Ollama.ipynb) | 1.9x faster | 60% less | -| **DPO Zephyr** | [▶️ Start for free](https://colab.research.google.com/github/unslothai/notebooks/blob/main/nb/Zephyr_(7B)-DPO.ipynb) | 1.9x faster | 50% less | +| **Orpheus-TTS (3B)** | [▶️ Start for free](https://colab.research.google.com/github/unslothai/notebooks/blob/main/nb/Orpheus_(3B)-TTS.ipynb) | 1.5x faster | 50% less | - See [all our notebooks](https://docs.unsloth.ai/get-started/unsloth-notebooks) and [all our models](https://docs.unsloth.ai/get-started/all-our-models) - **Kaggle Notebooks** for [Llama 3.2](https://www.kaggle.com/danielhanchen/kaggle-llama-3-2-1b-3b-unsloth-notebook), [Llama 3.1 (8B)](https://www.kaggle.com/danielhanchen/kaggle-llama-3-1-8b-unsloth-notebook), [Phi-4 (14B)](https://www.kaggle.com/code/danielhanchen/phi-4-finetuning-unsloth-notebook), [Mistral (7B)](https://www.kaggle.com/code/danielhanchen/kaggle-mistral-7b-unsloth-notebook) diff --git a/pyproject.toml b/pyproject.toml index 6e85d76de2..7f049d6f21 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -37,11 +37,11 @@ triton = [ ] huggingface = [ - "unsloth_zoo>=2025.4.4", + "unsloth_zoo>=2025.5.2", "packaging", "tyro", - "transformers>=4.51.3,!=4.47.0", - "datasets>=2.16.0", + "transformers==4.51.3,!=4.47.0", + "datasets>=3.4.1", "sentencepiece>=0.2.0", "tqdm", "psutil", @@ -381,11 +381,11 @@ colab-ampere-torch220 = [ "flash-attn>=2.6.3", ] colab-new = [ - "unsloth_zoo>=2025.4.4", + "unsloth_zoo>=2025.5.2", "packaging", "tyro", - "transformers>=4.51.3,!=4.47.0", - "datasets>=2.16.0", + "transformers==4.51.3,!=4.47.0", + "datasets>=3.4.1", "sentencepiece>=0.2.0", "tqdm", "psutil", @@ -550,6 +550,22 @@ cu128-ampere-torch270 = [ "unsloth[flashattention]", ] +intel-gpu-torch260 = [ + "unsloth[huggingface]", + + "pytorch_triton_xpu @ https://download.pytorch.org/whl/pytorch_triton_xpu-3.2.0-cp39-cp39-linux_x86_64.whl#sha256=147607f190a7d7aa24ba454def5977fbbfec792fdae18e4ed278cfec29b69271 ; platform_system == 'Linux' and python_version == '3.9' and platform_machine == 'x86_64'", + "pytorch_triton_xpu @ https://download.pytorch.org/whl/pytorch_triton_xpu-3.2.0-cp310-cp310-linux_x86_64.whl#sha256=23aa423fa1542afc34f67eb3ba8ef20060f6d1b3a4697eaeab22b11c92b30f2b ; platform_system == 'Linux' and python_version == '3.10' and platform_machine == 'x86_64'", + "pytorch_triton_xpu @ https://download.pytorch.org/whl/pytorch_triton_xpu-3.2.0-cp311-cp311-linux_x86_64.whl#sha256=bcfa995229bbfd9ffd8d6c8d9f6428d393e876fa6e23ee3c20e3c0d73ca75ca5 ; platform_system == 'Linux' and python_version == '3.11' and platform_machine == 'x86_64'", + "pytorch_triton_xpu @ https://download.pytorch.org/whl/pytorch_triton_xpu-3.2.0-cp312-cp312-linux_x86_64.whl#sha256=bd340903d03470708df3442438acb8b7e08087ab9e61fbe349b2872bf9257ab0 ; platform_system == 'Linux' and python_version == '3.12' and platform_machine == 'x86_64'", + "pytorch_triton_xpu @ https://download.pytorch.org/whl/pytorch_triton_xpu-3.2.0-cp313-cp313-linux_x86_64.whl#sha256=814dccc8a07159e6eca74bed70091bc8fea2d9dd87b0d91845f9f38cde62f01c ; platform_system == 'Linux' and python_version == '3.13' and platform_machine == 'x86_64'", + + "torch @ https://download.pytorch.org/whl/xpu/torch-2.6.0%2Bxpu-cp39-cp39-linux_x86_64.whl#sha256=6a8adf6dc4c089406e8b3a7e58ab57a463bddf9b07130d2576e76eced43e92af ; platform_system == 'Linux' and python_version == '3.9' and platform_machine == 'x86_64'", + "torch @ https://download.pytorch.org/whl/xpu/torch-2.6.0%2Bxpu-cp310-cp310-linux_x86_64.whl#sha256=ff4561cbf07c83bbccaa0f6e9bb0e6dcf721bacd53c9c43c4eb0e7331b4792f9 ; platform_system == 'Linux' and python_version == '3.10' and platform_machine == 'x86_64'", + "torch @ https://download.pytorch.org/whl/xpu/torch-2.6.0%2Bxpu-cp311-cp311-linux_x86_64.whl#sha256=12005f66b810ddd3ab93f86c4522bcfdd412cbd27fc9d189b661ff7509bc5e8a ; platform_system == 'Linux' and python_version == '3.11' and platform_machine == 'x86_64'", + "torch @ https://download.pytorch.org/whl/xpu/torch-2.6.0%2Bxpu-cp312-cp312-linux_x86_64.whl#sha256=c4c5c67625cdacf35765c2b94e61fe166e3c3f4a14521b1212a59ad1b3eb0f2e ; platform_system == 'Linux' and python_version == '3.12' and platform_machine == 'x86_64'", + "torch @ https://download.pytorch.org/whl/xpu/torch-2.6.0%2Bxpu-cp313-cp313-linux_x86_64.whl#sha256=e6864f7a60a5ecc43d5d38f59a16e5dd132384f73dfd3a697f74944026038f7b ; platform_system == 'Linux' and python_version == '3.13' and platform_machine == 'x86_64'", +] + [project.urls] homepage = "http://www.unsloth.ai" documentation = "https://github.com/unslothai/unsloth" diff --git a/unsloth/__init__.py b/unsloth/__init__.py index 713453cc0e..cf4dbe09c0 100644 --- a/unsloth/__init__.py +++ b/unsloth/__init__.py @@ -46,12 +46,6 @@ pass # Fixes https://github.com/unslothai/unsloth/issues/1266 os.environ["PROTOCOL_BUFFERS_PYTHON_IMPLEMENTATION"] = "python" -# Reduce VRAM usage by reducing fragmentation -# And optimize pinning of memory -os.environ["PYTORCH_CUDA_ALLOC_CONF"] = \ - "expandable_segments:True,"\ - "roundup_power2_divisions:[32:256,64:128,256:64,>:32]" - # [TODO] Check why some GPUs don't work # "pinned_use_cuda_host_register:True,"\ # "pinned_num_register_threads:8" @@ -84,9 +78,25 @@ except Exception as exception: raise exception pass +def get_device_type(): + if hasattr(torch, "cuda") and torch.cuda.is_available(): + return "cuda" + elif hasattr(torch, "xpu") and torch.xpu.is_available(): + return "xpu" + raise NotImplementedError("Unsloth currently only works on NVIDIA GPUs and Intel GPUs.") +pass +DEVICE_TYPE : str = get_device_type() + +# Reduce VRAM usage by reducing fragmentation +# And optimize pinning of memory +if DEVICE_TYPE == "cuda": + os.environ["PYTORCH_CUDA_ALLOC_CONF"] = \ + "expandable_segments:True,"\ + "roundup_power2_divisions:[32:256,64:128,256:64,>:32]" + # We support Pytorch 2 # Fixes https://github.com/unslothai/unsloth/issues/38 -torch_version = torch.__version__.split(".") +torch_version = str(torch.__version__).split(".") major_torch, minor_torch = torch_version[0], torch_version[1] major_torch, minor_torch = int(major_torch), int(minor_torch) if (major_torch < 2): @@ -97,10 +107,6 @@ elif (major_torch == 2) and (minor_torch < 2): del os.environ["PYTORCH_CUDA_ALLOC_CONF"] pass -# First check if CUDA is available ie a NVIDIA GPU is seen -if not torch.cuda.is_available(): - raise NotImplementedError("Unsloth: No NVIDIA GPU found? Unsloth currently only supports GPUs!") - # Fix Xformers performance issues since 0.0.25 import importlib.util from pathlib import Path @@ -132,77 +138,89 @@ except: pass # Torch 2.4 has including_emulation -major_version, minor_version = torch.cuda.get_device_capability() -SUPPORTS_BFLOAT16 = (major_version >= 8) +if DEVICE_TYPE == "cuda": + major_version, minor_version = torch.cuda.get_device_capability() + SUPPORTS_BFLOAT16 = (major_version >= 8) -old_is_bf16_supported = torch.cuda.is_bf16_supported -if "including_emulation" in str(inspect.signature(old_is_bf16_supported)): - def is_bf16_supported(including_emulation = False): - return old_is_bf16_supported(including_emulation) - torch.cuda.is_bf16_supported = is_bf16_supported -else: - def is_bf16_supported(): return SUPPORTS_BFLOAT16 - torch.cuda.is_bf16_supported = is_bf16_supported + old_is_bf16_supported = torch.cuda.is_bf16_supported + if "including_emulation" in str(inspect.signature(old_is_bf16_supported)): + def is_bf16_supported(including_emulation = False): + return old_is_bf16_supported(including_emulation) + torch.cuda.is_bf16_supported = is_bf16_supported + else: + def is_bf16_supported(): return SUPPORTS_BFLOAT16 + torch.cuda.is_bf16_supported = is_bf16_supported + pass +elif DEVICE_TYPE == "xpu": + # torch.xpu.is_bf16_supported() does not have including_emulation + # set SUPPORTS_BFLOAT16 as torch.xpu.is_bf16_supported() + SUPPORTS_BFLOAT16 = torch.xpu.is_bf16_supported() pass + # For Gradio HF Spaces? # if "SPACE_AUTHOR_NAME" not in os.environ and "SPACE_REPO_NAME" not in os.environ: import triton -libcuda_dirs = lambda: None -if Version(triton.__version__) >= Version("3.0.0"): - try: from triton.backends.nvidia.driver import libcuda_dirs - except: pass -else: from triton.common.build import libcuda_dirs +if DEVICE_TYPE == "cuda": + libcuda_dirs = lambda: None + if Version(triton.__version__) >= Version("3.0.0"): + try: from triton.backends.nvidia.driver import libcuda_dirs + except: pass + else: from triton.common.build import libcuda_dirs -# Try loading bitsandbytes and triton -import bitsandbytes as bnb -try: - cdequantize_blockwise_fp32 = bnb.functional.lib.cdequantize_blockwise_fp32 - libcuda_dirs() -except: - warnings.warn( - "Unsloth: Running `ldconfig /usr/lib64-nvidia` to link CUDA."\ - ) - - if os.path.exists("/usr/lib64-nvidia"): - os.system("ldconfig /usr/lib64-nvidia") - elif os.path.exists("/usr/local"): - # Sometimes bitsandbytes cannot be linked properly in Runpod for example - possible_cudas = subprocess.check_output(["ls", "-al", "/usr/local"]).decode("utf-8").split("\n") - find_cuda = re.compile(r"[\s](cuda\-[\d\.]{2,})$") - possible_cudas = [find_cuda.search(x) for x in possible_cudas] - possible_cudas = [x.group(1) for x in possible_cudas if x is not None] - - # Try linking cuda folder, or everything in local - if len(possible_cudas) == 0: - os.system("ldconfig /usr/local/") - else: - find_number = re.compile(r"([\d\.]{2,})") - latest_cuda = np.argsort([float(find_number.search(x).group(1)) for x in possible_cudas])[::-1][0] - latest_cuda = possible_cudas[latest_cuda] - os.system(f"ldconfig /usr/local/{latest_cuda}") - pass - - importlib.reload(bnb) - importlib.reload(triton) + # Try loading bitsandbytes and triton + import bitsandbytes as bnb try: - libcuda_dirs = lambda: None - if Version(triton.__version__) >= Version("3.0.0"): - try: from triton.backends.nvidia.driver import libcuda_dirs - except: pass - else: from triton.common.build import libcuda_dirs cdequantize_blockwise_fp32 = bnb.functional.lib.cdequantize_blockwise_fp32 libcuda_dirs() except: warnings.warn( - "Unsloth: CUDA is not linked properly.\n"\ - "Try running `python -m bitsandbytes` then `python -m xformers.info`\n"\ - "We tried running `ldconfig /usr/lib64-nvidia` ourselves, but it didn't work.\n"\ - "You need to run in your terminal `sudo ldconfig /usr/lib64-nvidia` yourself, then import Unsloth.\n"\ - "Also try `sudo ldconfig /usr/local/cuda-xx.x` - find the latest cuda version.\n"\ - "Unsloth will still run for now, but maybe it might crash - let's hope it works!" + "Unsloth: Running `ldconfig /usr/lib64-nvidia` to link CUDA."\ ) -pass + + if os.path.exists("/usr/lib64-nvidia"): + os.system("ldconfig /usr/lib64-nvidia") + elif os.path.exists("/usr/local"): + # Sometimes bitsandbytes cannot be linked properly in Runpod for example + possible_cudas = subprocess.check_output(["ls", "-al", "/usr/local"]).decode("utf-8").split("\n") + find_cuda = re.compile(r"[\s](cuda\-[\d\.]{2,})$") + possible_cudas = [find_cuda.search(x) for x in possible_cudas] + possible_cudas = [x.group(1) for x in possible_cudas if x is not None] + + # Try linking cuda folder, or everything in local + if len(possible_cudas) == 0: + os.system("ldconfig /usr/local/") + else: + find_number = re.compile(r"([\d\.]{2,})") + latest_cuda = np.argsort([float(find_number.search(x).group(1)) for x in possible_cudas])[::-1][0] + latest_cuda = possible_cudas[latest_cuda] + os.system(f"ldconfig /usr/local/{latest_cuda}") + pass + + importlib.reload(bnb) + importlib.reload(triton) + try: + libcuda_dirs = lambda: None + if Version(triton.__version__) >= Version("3.0.0"): + try: from triton.backends.nvidia.driver import libcuda_dirs + except: pass + else: from triton.common.build import libcuda_dirs + cdequantize_blockwise_fp32 = bnb.functional.lib.cdequantize_blockwise_fp32 + libcuda_dirs() + except: + warnings.warn( + "Unsloth: CUDA is not linked properly.\n"\ + "Try running `python -m bitsandbytes` then `python -m xformers.info`\n"\ + "We tried running `ldconfig /usr/lib64-nvidia` ourselves, but it didn't work.\n"\ + "You need to run in your terminal `sudo ldconfig /usr/lib64-nvidia` yourself, then import Unsloth.\n"\ + "Also try `sudo ldconfig /usr/local/cuda-xx.x` - find the latest cuda version.\n"\ + "Unsloth will still run for now, but maybe it might crash - let's hope it works!" + ) + pass +elif DEVICE_TYPE == "xpu": + # currently intel xpu will not support bnb, will add support in the future + # TODO: check triton for intel installed properly. + pass # Check for unsloth_zoo try: diff --git a/unsloth/models/__init__.py b/unsloth/models/__init__.py index 99db55c086..fb3fc5c455 100644 --- a/unsloth/models/__init__.py +++ b/unsloth/models/__init__.py @@ -20,5 +20,5 @@ from .qwen3 import FastQwen3Model from .qwen3_moe import FastQwen3MoeModel from .granite import FastGraniteModel from .dpo import PatchDPOTrainer, PatchKTOTrainer -from ._utils import is_bfloat16_supported, __version__ -from .rl import PatchFastRL, vLLMSamplingParams +from ._utils import is_bfloat16_supported, is_vLLM_available, __version__ +from .rl import PatchFastRL, vLLMSamplingParams \ No newline at end of file diff --git a/unsloth/models/_utils.py b/unsloth/models/_utils.py index 90e0d9eb44..9ec3f8d2de 100644 --- a/unsloth/models/_utils.py +++ b/unsloth/models/_utils.py @@ -12,11 +12,12 @@ # See the License for the specific language governing permissions and # limitations under the License. -__version__ = "2025.4.8" +__version__ = "2025.5.2" __all__ = [ "SUPPORTS_BFLOAT16", "is_bfloat16_supported", + "is_vLLM_available", "prepare_model_for_kbit_training", "xformers", @@ -800,6 +801,9 @@ def is_bfloat16_supported(): return SUPPORTS_BFLOAT16 pass +def is_vLLM_available(): + return _is_package_available("vllm") +pass # Patches models to add RoPE Scaling def patch_linear_scaling( diff --git a/unsloth/models/llama.py b/unsloth/models/llama.py index 65c7334f51..207540abf5 100644 --- a/unsloth/models/llama.py +++ b/unsloth/models/llama.py @@ -1076,7 +1076,7 @@ def CausalLM_fast_forward(fast_forward_inference): if labels is not None: labels = labels.to(lm_head_device) # Output last hidden states without logits if asked - if self.training and os.environ.get("UNSLOTH_RETURN_HIDDEN_STATES", "0") == "1": + if os.environ.get("UNSLOTH_RETURN_HIDDEN_STATES", "0") == "1": if num_logits_to_keep != 0: hidden_states = hidden_states[:, -num_logits_to_keep:, :] return CausalLMOutputWithPast( @@ -1661,9 +1661,8 @@ class FastLlamaModel: ) pass if fast_inference: - import platform - if platform.system().lower() == 'windows': - print("Unsloth: vLLM does not work in Windows! Will use Unsloth inference!") + if not is_vLLM_available(): + print("Unsloth: vLLM is not installed! Will use Unsloth inference!") fast_inference = False major_version, minor_version = torch.cuda.get_device_capability() if major_version < 7: diff --git a/unsloth/models/loader.py b/unsloth/models/loader.py index f0eab18a72..a1dbc82534 100644 --- a/unsloth/models/loader.py +++ b/unsloth/models/loader.py @@ -14,6 +14,7 @@ from ._utils import ( is_bfloat16_supported, + is_vLLM_available, HAS_FLASH_ATTENTION, HAS_FLASH_ATTENTION_SOFTCAPPING, USE_MODELSCOPE, @@ -351,9 +352,8 @@ class FastLanguageModel(FastLlamaModel): pass if fast_inference: - import platform - if platform.system().lower() == 'windows': - print("Unsloth: vLLM does not work in Windows! Will use Unsloth inference!") + if not is_vLLM_available(): + print("Unsloth: vLLM is not installed! Will use Unsloth inference!") fast_inference = False pass from unsloth_zoo.vllm_utils import ( diff --git a/unsloth/models/loader_utils.py b/unsloth/models/loader_utils.py index db81b5c5c6..e2e24eae3e 100644 --- a/unsloth/models/loader_utils.py +++ b/unsloth/models/loader_utils.py @@ -21,9 +21,11 @@ SUPPORTS_FOURBIT = transformers_version >= Version("4.37") BAD_MAPPINGS = \ { - "unsloth/qwen3-32B-unsloth-bnb-4bit".lower() : "unsloth/Qwen3-32B-bnb-4bit".lower(), # 32B dynamic quant is way too big - "unsloth/qwen3-30B-A3B-unsloth-bnb-4bit".lower() : "unsloth/qwen3-30B-A3B".lower(), # HF loads MoEs too slowly - "unsloth/qwen3-30B-A3B-bnb-4bit".lower() : "unsloth/qwen3-30B-A3B".lower(), # We rather do it on the fly + "unsloth/Qwen3-32B-unsloth-bnb-4bit".lower() : "unsloth/Qwen3-32B-bnb-4bit".lower(), # 32B dynamic quant is way too big + "unsloth/Qwen3-30B-A3B-unsloth-bnb-4bit".lower() : "unsloth/Qwen3-30B-A3B".lower(), # HF loads MoEs too slowly + "unsloth/Qwen3-30B-A3B-bnb-4bit".lower() : "unsloth/Qwen3-30B-A3B".lower(), # We rather do it on the fly + "unsloth/Qwen3-30B-A3B-Base-unsloth-bnb-4bit".lower() : "unsloth/Qwen3-30B-A3B-Base".lower(), # HF loads MoEs too slowly + "unsloth/Qwen3-30B-A3B-Base-bnb-4bit".lower() : "unsloth/Qwen3-30B-A3B-Base".lower(), # We rather do it on the fly } def __get_model_name( diff --git a/unsloth/models/vision.py b/unsloth/models/vision.py index d212c224be..cadfed9430 100644 --- a/unsloth/models/vision.py +++ b/unsloth/models/vision.py @@ -112,6 +112,11 @@ def unsloth_base_fast_generate( arch = self.config.architectures[0] # Remove token_type_ids - WRONG for Gemma 3 since bidirectional attention + if hasattr(self, "generate") and hasattr(self, "forward"): + # did not combine with below since self might not have model + keys = inspect.signature(self.forward).parameters.keys() + if "token_type_ids" not in keys: + kwargs.pop("token_type_ids", None) # kwargs.pop("token_type_ids", None) # VLMs do not allow logits_to_keep