diff --git a/README.md b/README.md index 9ad41546cf..74c5e8afac 100644 --- a/README.md +++ b/README.md @@ -23,20 +23,18 @@ Notebooks are beginner friendly. Read our [guide](https://docs.unsloth.ai/get-st | Unsloth supports | Free Notebooks | Performance | Memory use | |-----------|---------|--------|----------| | **Qwen3 (14B)** | [▶️ Start for free](https://colab.research.google.com/github/unslothai/notebooks/blob/main/nb/Qwen3_(14B)-Reasoning-Conversational.ipynb) | 2x faster | 70% less | -| **GRPO (reasoning)** | [▶️ Start for free](https://colab.research.google.com/github/unslothai/notebooks/blob/main/nb/Llama3.1_(8B)-GRPO.ipynb) | 2x faster | 80% less | +| **Qwen3 (4B): GRPO** | [▶️ Start for free](https://colab.research.google.com/github/unslothai/notebooks/blob/main/nb/Qwen3_(4B)-GRPO.ipynb) | 2x faster | 80% less | | **Gemma 3 (4B)** | [▶️ Start for free](https://colab.research.google.com/github/unslothai/notebooks/blob/main/nb/Gemma3_(4B).ipynb) | 1.6x faster | 60% less | | **Llama 3.2 (3B)** | [▶️ Start for free](https://colab.research.google.com/github/unslothai/notebooks/blob/main/nb/Llama3.2_(1B_and_3B)-Conversational.ipynb) | 2x faster | 70% less | | **Phi-4 (14B)** | [▶️ Start for free](https://colab.research.google.com/github/unslothai/notebooks/blob/main/nb/Phi_4-Conversational.ipynb) | 2x faster | 70% less | | **Llama 3.2 Vision (11B)** | [▶️ Start for free](https://colab.research.google.com/github/unslothai/notebooks/blob/main/nb/Llama3.2_(11B)-Vision.ipynb) | 2x faster | 50% less | | **Llama 3.1 (8B)** | [▶️ Start for free](https://colab.research.google.com/github/unslothai/notebooks/blob/main/nb/Llama3.1_(8B)-Alpaca.ipynb) | 2x faster | 70% less | | **Mistral v0.3 (7B)** | [▶️ Start for free](https://colab.research.google.com/github/unslothai/notebooks/blob/main/nb/Mistral_v0.3_(7B)-Conversational.ipynb) | 2.2x faster | 75% less | -| **Ollama** | [▶️ Start for free](https://colab.research.google.com/github/unslothai/notebooks/blob/main/nb/Llama3_(8B)-Ollama.ipynb) | 1.9x faster | 60% less | -| **Orpheus-TTS (3B)** | [▶️ Start for free](https://colab.research.google.com/github/unslothai/notebooks/blob/main/nb/Orpheus_(3B)-TTS.ipynb) | 1.5x faster | 50% less | +| **Sesame-CSM (1B)** | [▶️ Start for free](https://colab.research.google.com/github/unslothai/notebooks/blob/main/nb/Orpheus_(3B)-TTS.ipynb) | 1.5x faster | 50% less | -- See [all our notebooks](https://docs.unsloth.ai/get-started/unsloth-notebooks) and [all our models](https://docs.unsloth.ai/get-started/all-our-models) -- **Kaggle Notebooks** for [Llama 3.2](https://www.kaggle.com/danielhanchen/kaggle-llama-3-2-1b-3b-unsloth-notebook), [Llama 3.1 (8B)](https://www.kaggle.com/danielhanchen/kaggle-llama-3-1-8b-unsloth-notebook), [Phi-4 (14B)](https://www.kaggle.com/code/danielhanchen/phi-4-finetuning-unsloth-notebook), [Mistral (7B)](https://www.kaggle.com/code/danielhanchen/kaggle-mistral-7b-unsloth-notebook) -- Don't have data? Use our [Synthetic Dataset notebook](https://colab.research.google.com/github/unslothai/notebooks/blob/main/nb/Meta_Synthetic_Data_Llama3_2_(3B).ipynb) in collaboration with Meta. -- See detailed documentation for Unsloth [here](https://docs.unsloth.ai/). +- See all our notebooks for: [Kaggle](https://github.com/unslothai/notebooks?tab=readme-ov-file#-kaggle-notebooks), [GRPO](https://docs.unsloth.ai/get-started/unsloth-notebooks#grpo-reasoning-notebooks), **[TTS](https://docs.unsloth.ai/get-started/unsloth-notebooks#text-to-speech-tts-notebooks)** & [Vision](https://docs.unsloth.ai/get-started/unsloth-notebooks#vision-multimodal-notebooks) +- See [all our models](https://docs.unsloth.ai/get-started/all-our-models) and our [Synthetic Dataset notebook](https://colab.research.google.com/github/unslothai/notebooks/blob/main/nb/Meta_Synthetic_Data_Llama3_2_%283B%29.ipynb) in collaboration with Meta +- See detailed documentation for Unsloth [here](https://docs.unsloth.ai/) ## ⚡ Quickstart @@ -47,14 +45,14 @@ pip install unsloth For Windows install instructions, see [here](https://docs.unsloth.ai/get-started/installing-+-updating/windows-installation). ## 🦥 Unsloth.ai News -- 📣 NEW! **[Qwen3](https://docs.unsloth.ai/basics/qwen3-how-to-run-and-fine-tune)** is now supported! Qwen3-30B-A3B fits on 17.5GB VRAM. +- 📣 NEW! **[Text-to-Speech (TTS)](https://docs.unsloth.ai/basics/text-to-speech-tts-fine-tuning)** is now supported, including `sesame/csm-1b` and STT `openai/whisper-large-v3`. +- 📣 NEW! **[Qwen3](https://docs.unsloth.ai/basics/qwen3-how-to-run-and-fine-tune)** is now supported. Qwen3-30B-A3B fits on 17.5GB VRAM. - 📣 NEW! Introducing **[Dynamic 2.0](https://docs.unsloth.ai/basics/unsloth-dynamic-2.0-ggufs)** quants that set new benchmarks on 5-shot MMLU & KL Divergence. -- 📣 **[Llama 4](https://unsloth.ai/blog/llama4)**, Meta's latest models including Scout & Maverick are now supported. -- 📣 NEW! [**EVERYTHING** is now supported](https://unsloth.ai/blog/gemma3#everything) incuding: FFT, ALL models (Mixtral, MOE, Cohere, Mamba) and all training algorithms (KTO, DoRA) etc. MultiGPU support coming very soon. - To enable full-finetuning, set ```full_finetuning = True``` and for 8-bit finetuning, set ```load_in_8bit = True``` +- 📣 **[Llama 4](https://unsloth.ai/blog/llama4)** by Meta, including Scout & Maverick are now supported. +- 📣 [**EVERYTHING** is now supported](https://unsloth.ai/blog/gemma3#everything) - all models (BERT, diffusion, Cohere, Mamba), FFT, etc. MultiGPU coming soon. Enable FFT with `full_finetuning = True`, 8-bit with `load_in_8bit = True`. - 📣 **Gemma 3** by Google: [Read Blog](https://unsloth.ai/blog/gemma3). We [uploaded GGUFs, 4-bit models](https://huggingface.co/collections/unsloth/gemma-3-67d12b7e8816ec6efa7e4e5b). - 📣 Introducing Long-context [Reasoning (GRPO)](https://unsloth.ai/blog/grpo) in Unsloth. Train your own reasoning model with just 5GB VRAM. Transform Llama, Phi, Mistral etc. into reasoning LLMs! -- 📣 [DeepSeek-R1](https://unsloth.ai/blog/deepseek-r1) - the most powerful open reasoning models with Llama & Qwen distillations. Run or fine-tune them now [with our guide](https://unsloth.ai/blog/deepseek-r1). All model uploads: [here](https://huggingface.co/collections/unsloth/deepseek-r1-all-versions-678e1c48f5d2fce87892ace5). +- 📣 [DeepSeek-R1](https://unsloth.ai/blog/deepseek-r1) - run or fine-tune them [with our guide](https://unsloth.ai/blog/deepseek-r1). All model uploads: [here](https://huggingface.co/collections/unsloth/deepseek-r1-all-versions-678e1c48f5d2fce87892ace5).
Click for more news @@ -64,12 +62,6 @@ For Windows install instructions, see [here](https://docs.unsloth.ai/get-started - 📣 [Llama 3.3 (70B)](https://huggingface.co/collections/unsloth/llama-33-all-versions-67535d7d994794b9d7cf5e9f), Meta's latest model is supported. - 📣 We worked with Apple to add [Cut Cross Entropy](https://arxiv.org/abs/2411.09009). Unsloth now supports 89K context for Meta's Llama 3.3 (70B) on a 80GB GPU - 13x longer than HF+FA2. For Llama 3.1 (8B), Unsloth enables 342K context, surpassing its native 128K support. - 📣 We found and helped fix a [gradient accumulation bug](https://unsloth.ai/blog/gradient)! Please update Unsloth and transformers. -- 📣 Try out [Chat interface](https://colab.research.google.com/github/unslothai/notebooks/blob/main/nb/Unsloth_Studio.ipynb)! -- 📣 NEW! Qwen-2.5 including [Coder](https://unsloth.ai/blog/qwen-coder) models are now supported with bugfixes. 14b fits in a Colab GPU! [Qwen 2.5 conversational notebook](https://colab.research.google.com/github/unslothai/notebooks/blob/main/nb/Qwen2.5_Coder_(14B)-Conversational.ipynb) -- 📣 NEW! [Mistral Small 22b notebook](https://colab.research.google.com/github/unslothai/notebooks/blob/main/nb/Mistral_Small_(22B)-Alpaca.ipynb) finetuning fits in under 16GB of VRAM! -- 📣 NEW! `pip install unsloth` now works! Head over to [pypi](https://pypi.org/project/unsloth/) to check it out! This allows non git pull installs. Use `pip install unsloth[colab-new]` for non dependency installs. -- 📣 NEW! Continued Pretraining [notebook](https://colab.research.google.com/github/unslothai/notebooks/blob/main/nb/Mistral_v0.3_(7B)-CPT.ipynb) for other languages like Korean! -- 📣 [2x faster inference](https://colab.research.google.com/github/unslothai/notebooks/blob/main/nb/Llama3.1_(8B)-Inference.ipynb) added for all our models - 📣 We cut memory usage by a [further 30%](https://unsloth.ai/blog/long-context) and now support [4x longer context windows](https://unsloth.ai/blog/long-context)!
@@ -118,7 +110,7 @@ See [here](https://github.com/unslothai/unsloth/edit/main/README.md#advanced-pip Follow the instructions to install [CUDA Toolkit](https://developer.nvidia.com/cuda-toolkit-archive). 6. **Install PyTorch:** - You will need the correct version of PyTorch that is compatibile with your CUDA drivers, so make sure to select them carefully. + You will need the correct version of PyTorch that is compatible with your CUDA drivers, so make sure to select them carefully. [Install PyTorch](https://pytorch.org/get-started/locally/). 7. **Install Unsloth:** @@ -143,7 +135,7 @@ trainer = SFTTrainer( For **advanced installation instructions** or if you see weird errors during installations: 1. Install `torch` and `triton`. Go to https://pytorch.org to install it. For example `pip install torch torchvision torchaudio triton` -2. Confirm if CUDA is installated correctly. Try `nvcc`. If that fails, you need to install `cudatoolkit` or CUDA drivers. +2. Confirm if CUDA is installed correctly. Try `nvcc`. If that fails, you need to install `cudatoolkit` or CUDA drivers. 3. Install `xformers` manually. You can try installing `vllm` and seeing if `vllm` succeeds. Check if `xformers` succeeded with `python -m xformers.info` Go to https://github.com/facebookresearch/xformers. Another option is to install `flash-attn` for Ampere GPUs. 4. Double check that your versions of Python, CUDA, CUDNN, `torch`, `triton`, and `xformers` are compatible with one another. The [PyTorch Compatibility Matrix](https://github.com/pytorch/pytorch/blob/main/RELEASE.md#release-compatibility-matrix) may be useful. 5. Finally, install `bitsandbytes` and check it with `python -m bitsandbytes` @@ -244,7 +236,7 @@ from unsloth import FastLanguageModel, FastModel import torch from trl import SFTTrainer, SFTConfig from datasets import load_dataset -max_seq_length = 2048 # Supports RoPE Scaling interally, so choose any! +max_seq_length = 2048 # Supports RoPE Scaling internally, so choose any! # Get LAION dataset url = "https://huggingface.co/datasets/laion/OIG/resolve/main/unified_chip2.jsonl" dataset = load_dataset("json", data_files = {"train" : url}, split = "train") @@ -326,6 +318,7 @@ trainer.train() ## 💡 Reinforcement Learning RL including DPO, GRPO, PPO, Reward Modelling, Online DPO all work with Unsloth. We're in 🤗Hugging Face's official docs! We're on the [GRPO docs](https://huggingface.co/learn/nlp-course/en/chapter12/6) and the [DPO docs](https://huggingface.co/docs/trl/main/en/dpo_trainer#accelerate-dpo-fine-tuning-using-unsloth)! List of RL notebooks: +- Advanced Qwen3 GRPO notebook: [Link](https://colab.research.google.com/github/unslothai/notebooks/blob/main/nb/Qwen3_(4B)-GRPO.ipynb) - ORPO notebook: [Link](https://colab.research.google.com/github/unslothai/notebooks/blob/main/nb/Llama3_(8B)-ORPO.ipynb) - DPO Zephyr notebook: [Link](https://colab.research.google.com/github/unslothai/notebooks/blob/main/nb/Zephyr_(7B)-DPO.ipynb) - KTO notebook: [Link](https://colab.research.google.com/drive/1MRgGtLWuZX4ypSfGguFgC-IblTvO2ivM?usp=sharing) @@ -439,8 +432,8 @@ You can cite the Unsloth repo as follows: ``` ### Thank You to -- Hugging Face's [TRL library](https://github.com/huggingface/trl) which serves as the basis foundation for Unsloth +- The [llama.cpp library](https://github.com/ggml-org/llama.cpp) that lets users save models with Unsloth +- The Hugging Face team and their [TRL library](https://github.com/huggingface/trl) - [Erik](https://github.com/erikwijmans) for his help adding [Apple's ML Cross Entropy](https://github.com/apple/ml-cross-entropy) in Unsloth -- [HuyNguyen-hust](https://github.com/HuyNguyen-hust) for making [RoPE Embeddings 28% faster](https://github.com/unslothai/unsloth/pull/238) -- [RandomInternetPreson](https://github.com/RandomInternetPreson) for confirming WSL support -- [152334H](https://github.com/152334H) for experimental DPO support +- [Etherl](https://github.com/Etherll) for adding support for [TTS, diffusion and BERT models](https://github.com/unslothai/notebooks/pull/34) +- And of course for every single person who has contributed or has used Unsloth! diff --git a/pyproject.toml b/pyproject.toml index 4cadd3aa67..558abd9f1a 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -37,7 +37,7 @@ triton = [ ] huggingface = [ - "unsloth_zoo>=2025.5.6", + "unsloth_zoo>=2025.5.7", "packaging", "tyro", "transformers==4.51.3,!=4.47.0", @@ -381,7 +381,7 @@ colab-ampere-torch220 = [ "flash-attn>=2.6.3", ] colab-new = [ - "unsloth_zoo>=2025.5.6", + "unsloth_zoo>=2025.5.7", "packaging", "tyro", "transformers==4.51.3,!=4.47.0", diff --git a/tests/qlora/README.md b/tests/qlora/README.md index e535c38760..c05dcf446d 100644 --- a/tests/qlora/README.md +++ b/tests/qlora/README.md @@ -39,7 +39,7 @@ For the unsloth test, the model's behavior is as expected: - after merging, the model's response contains the answer For the huggingface test, the model's behavior is as expected: -- before training, the model's response does not contains the answer +- before training, the model's response does not contain the answer - after training, the model's response contains the answer - after using peft's `merge_and_unload`, the model's response does not contain the answer - after using my custom merge function, the model's response contains the answer diff --git a/unsloth/models/_utils.py b/unsloth/models/_utils.py index 118f4f0535..63f48af659 100644 --- a/unsloth/models/_utils.py +++ b/unsloth/models/_utils.py @@ -12,7 +12,7 @@ # See the License for the specific language governing permissions and # limitations under the License. -__version__ = "2025.5.4" +__version__ = "2025.5.5" __all__ = [ "SUPPORTS_BFLOAT16", @@ -147,7 +147,7 @@ class HideLoggingMessage(logging.Filter): def filter(self, x): return not (self.text in x.getMessage()) pass -# The speedups for torchdynamo mostly come wih GPU Ampere or higher and which is not detected here. +# The speedups for torchdynamo mostly come with GPU Ampere or higher and which is not detected here. from transformers.training_args import logger as transformers_training_args_logger transformers_training_args_logger.addFilter(HideLoggingMessage("The speedups")) # torch.distributed process group is initialized, but parallel_mode != ParallelMode.DISTRIBUTED. @@ -1174,6 +1174,7 @@ def unsloth_compile_transformers( import_from_cache = False, disable = False, return_logits = False, + unsloth_force_compile = False, ): if Version(torch_version) < Version("2.4.0"): print( @@ -1184,12 +1185,12 @@ def unsloth_compile_transformers( ) return pass - if trust_remote_code: + if trust_remote_code and unsloth_force_compile == False: print( "Unsloth: We can't trace models if `trust_remote_code = True`, "\ "so turning off some optimizations!" ) - return + return model_types, False model_types = list(dict().fromkeys(model_types).keys()) if disable: return model_types, False diff --git a/unsloth/models/loader.py b/unsloth/models/loader.py index a233b26a86..9c5a7b68be 100644 --- a/unsloth/models/loader.py +++ b/unsloth/models/loader.py @@ -485,6 +485,7 @@ class FastModel(FastBaseModel): auto_model = None, whisper_language = None, whisper_task = None, + unsloth_force_compile = False, *args, **kwargs, ): if token is None: token = get_token() @@ -715,6 +716,7 @@ class FastModel(FastBaseModel): disable = False, return_logits = return_logits, trust_remote_code = trust_remote_code, + unsloth_force_compile = unsloth_force_compile, ) pass diff --git a/unsloth/models/vision.py b/unsloth/models/vision.py index 2ba7c1391b..2bff87d8d9 100644 --- a/unsloth/models/vision.py +++ b/unsloth/models/vision.py @@ -200,12 +200,14 @@ def unsloth_base_fast_generate( cache_implementation = "static" else: cache_implementation = "hybrid" + if "generation_config" in kwargs: kwargs["generation_config"].cache_implementation = cache_implementation kwargs["generation_config"].compile_config = _compile_config if cache_implementation is not None else None else: kwargs["cache_implementation"] = cache_implementation - kwargs["compile_config"] = _compile_config if cache_implementation is not None else None + if cache_implementation: + kwargs["compile_config"] = _compile_config pass try: @@ -477,10 +479,12 @@ class FastBaseModel: unsloth_base_fast_generate.__doc__ = model._old_generate.__doc__ model.generate = types.MethodType(unsloth_base_fast_generate, model) pass + model._unsloth_trust_remote_code = trust_remote_code # Post patches model = FastBaseModel.post_patch_model( model, use_gradient_checkpointing = use_gradient_checkpointing, + trust_remote_code = trust_remote_code, ) # Clear deleted GPU items for _ in range(3): @@ -514,7 +518,7 @@ class FastBaseModel: loftq_config = {}, task_type = TaskType.CAUSAL_LM, temporary_location = "_unsloth_temporary_saved_buffers", - **kwargs, + **kwargs ): if os.environ.get("UNSLOTH_ENABLE_FULL_FINETUNING", "0") == "1": print("Unsloth: Full finetuning is enabled, so .get_peft_model has no effect") @@ -561,6 +565,7 @@ class FastBaseModel: lora_dropout = lora_dropout, bias = bias, task_type = task_type, + use_rslora = use_rslora, ) model = prepare_model_for_kbit_training( model, @@ -569,10 +574,9 @@ class FastBaseModel: model = _get_peft_model(model, lora_config) # Enable gradients on modules which are trainable requires_grad_for_gradient_checkpointing(model) - - model = FastBaseModel.post_patch_model(model, use_gradient_checkpointing) + trust_remote_code = getattr(model, "_unsloth_trust_remote_code", False) + model = FastBaseModel.post_patch_model(model, use_gradient_checkpointing, trust_remote_code = trust_remote_code) model.max_seq_length = max_seq_length - # Clear deleted GPU items for _ in range(3): gc.collect() @@ -591,6 +595,7 @@ class FastBaseModel: def post_patch_model( model, use_gradient_checkpointing = True, + trust_remote_code = False, ): full_finetuning = os.environ.get("UNSLOTH_ENABLE_FULL_FINETUNING", "0") == "1" @@ -611,7 +616,7 @@ class FastBaseModel: ) from transformers.trainer import Trainer - if Trainer._inner_training_loop.__name__ != "_fast_inner_training_loop": + if Trainer._inner_training_loop.__name__ != "_fast_inner_training_loop" and trust_remote_code == False: raise RuntimeError('Unsloth: Unsuccessfully patched inner_training_loop') pass patch_saving_functions(model, vision = True) diff --git a/unsloth/save.py b/unsloth/save.py index 578f2b49a1..5652e15098 100644 --- a/unsloth/save.py +++ b/unsloth/save.py @@ -1717,7 +1717,7 @@ def push_to_ollama( tag=tag ) - print("Succesfully pushed to ollama") + print("Successfully pushed to ollama")