From 8d72323bd8d7186617584efb254c68b58b4cdd75 Mon Sep 17 00:00:00 2001 From: Dina Suehiro Jones Date: Tue, 25 Nov 2025 17:33:28 -0800 Subject: [PATCH 1/2] Fix llama tokenizer padding_side when using model.generate in inference mode (#3644) * Only restore training mode after generation, if the model started out in training mode Signed-off-by: Dina Suehiro Jones * [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci --------- Signed-off-by: Dina Suehiro Jones Co-authored-by: pre-commit-ci[bot] <66853113+pre-commit-ci[bot]@users.noreply.github.com> --- unsloth/models/llama.py | 6 +++++- unsloth/models/loader.py | 14 ++++++++++---- 2 files changed, 15 insertions(+), 5 deletions(-) diff --git a/unsloth/models/llama.py b/unsloth/models/llama.py index 9e3896244c..b1b8a8fb78 100644 --- a/unsloth/models/llama.py +++ b/unsloth/models/llama.py @@ -1988,6 +1988,9 @@ def unsloth_fast_generate( *args, **kwargs, ): + # If the model starts out in training mode, restore training mode after generation + restore_training_mode = self.training + FastLlamaModel.for_inference(self) dtype = _get_dtype(dtype_from_config(self.config)) @@ -2043,7 +2046,8 @@ def unsloth_fast_generate( # accelerate.utils.operations.send_to_device = accelerate_old_send_to_device # pass - FastLlamaModel.for_training(self) + if restore_training_mode: + FastLlamaModel.for_training(self) return output diff --git a/unsloth/models/loader.py b/unsloth/models/loader.py index 828fe17320..a8ddbfed2a 100644 --- a/unsloth/models/loader.py +++ b/unsloth/models/loader.py @@ -258,7 +258,7 @@ class FastLanguageModel(FastLlamaModel): if model_name.lower().endswith("-bf16"): load_in_4bit = False load_in_8bit = False - load_in_fp8 = False + load_in_fp8 = False load_in_16bit = True if USE_MODELSCOPE and not os.path.exists(model_name): @@ -387,7 +387,7 @@ class FastLanguageModel(FastLlamaModel): if model_name.lower().endswith("-bf16"): load_in_4bit = False load_in_8bit = False - load_in_fp8 = False + load_in_fp8 = False load_in_16bit = True model_config = AutoConfig.from_pretrained( @@ -711,10 +711,16 @@ class FastModel(FastBaseModel): ) load_in_4bit = False load_in_8bit = False - load_in_fp8 = False + load_in_fp8 = False load_in_16bit = False - if int(load_in_4bit) + int(load_in_8bit) + int(load_in_16bit) + int(load_in_fp8 != False) >= 2: + if ( + int(load_in_4bit) + + int(load_in_8bit) + + int(load_in_16bit) + + int(load_in_fp8 != False) + >= 2 + ): raise RuntimeError( "Unsloth: Can only load in 4bit or 8bit or 16bit, not a combination!\n" "Also, we by default set `load_in_4bit = True`.\n" From b86ce3399622e40051cdd0cfab91e239c9ad630d Mon Sep 17 00:00:00 2001 From: mk0walsk Date: Thu, 27 Nov 2025 04:15:27 +0200 Subject: [PATCH 2/2] Fix indefinite article usage in comments and docstrings (#3648) --- unsloth/chat_templates.py | 4 ++-- unsloth/models/falcon_h1.py | 2 +- unsloth/ollama_template_mappers.py | 2 +- unsloth/tokenizer_utils.py | 2 +- 4 files changed, 5 insertions(+), 5 deletions(-) diff --git a/unsloth/chat_templates.py b/unsloth/chat_templates.py index 823ce0ee6c..63d310af8e 100644 --- a/unsloth/chat_templates.py +++ b/unsloth/chat_templates.py @@ -2593,7 +2593,7 @@ default_system_message = \ extra_eos_tokens = None, ): """ - Creates a Ollama modelfile and a HF Jinja template from a custom + Creates an Ollama modelfile and a HF Jinja template from a custom template. You must provide 2x examples of an input & output. There is an optional system message as well. @@ -2930,7 +2930,7 @@ extra_eos_tokens = None, ): """ - Creates a Ollama modelfile and a HF Jinja template from a custom + Creates an Ollama modelfile and a HF Jinja template from a custom template. You must provide 2x examples of an input & output. There is an optional system message as well. diff --git a/unsloth/models/falcon_h1.py b/unsloth/models/falcon_h1.py index 3010d37163..c6a413ee40 100644 --- a/unsloth/models/falcon_h1.py +++ b/unsloth/models/falcon_h1.py @@ -57,7 +57,7 @@ try: FalconH1Attention, ) except ModuleNotFoundError: - # if we are on a old version of transformers technically it should fail in the try except above + # if we are on an old version of transformers technically it should fail in the try except above # but if somehow we make it here, we need to raise an error since FalconH1Attention is not available # or renamed raise ImportError( diff --git a/unsloth/ollama_template_mappers.py b/unsloth/ollama_template_mappers.py index 3ad0e334d2..ea1882e117 100644 --- a/unsloth/ollama_template_mappers.py +++ b/unsloth/ollama_template_mappers.py @@ -1520,7 +1520,7 @@ Loop over messages and look for a user-provided system message and documents {{- /* NOTE: Since Ollama collates consecutive roles, for control and documents, we - work around this by allowing the role to contain an qualifier after the + work around this by allowing the role to contain a qualifier after the role string. */ -}} diff --git a/unsloth/tokenizer_utils.py b/unsloth/tokenizer_utils.py index af6bba9de2..99651643a8 100644 --- a/unsloth/tokenizer_utils.py +++ b/unsloth/tokenizer_utils.py @@ -600,7 +600,7 @@ def load_correct_tokenizer( ### 1. Fixup tokenizer's chat_template old_chat_template = getattr(tokenizer, "chat_template", None) - # Ignore mistral type models since they don't have a add_generation_prompt + # Ignore mistral type models since they don't have an add_generation_prompt if "mistral" in str(getattr(tokenizer, "name_or_path", "")).lower(): chat_template = old_chat_template