diff --git a/unsloth/models/rl.py b/unsloth/models/rl.py index 72b911acbb..72b568790d 100644 --- a/unsloth/models/rl.py +++ b/unsloth/models/rl.py @@ -39,13 +39,15 @@ def PatchRL(FastLanguageModel): @contextmanager def unsloth_unwrap_model_for_generation(model, accelerator): # Must use for_inference to allow inference in Unsloth - with unwrap_model_for_generation(model, accelerator) as unwrapped_model: - FastLanguageModel.for_inference(unwrapped_model) - try: - yield unwrapped_model - finally: - # Finally return back training - FastLanguageModel.for_training(model) + with torch.inference_mode(): + with unwrap_model_for_generation(model, accelerator) as unwrapped_model: + FastLanguageModel.for_inference(unwrapped_model) + try: + yield unwrapped_model + finally: + # Finally return back training + FastLanguageModel.for_training(model) + pass pass pass pass