diff --git a/unsloth/models/rl_replacements.py b/unsloth/models/rl_replacements.py index abca252e7a..9b0f4e4aef 100644 --- a/unsloth/models/rl_replacements.py +++ b/unsloth/models/rl_replacements.py @@ -250,7 +250,7 @@ def grpo_trainer__get_per_token_logps(function_name, function): if function_name != "_get_per_token_logps": return function def _get_per_token_logps(self, model, input_ids, attention_mask, logits_to_keep): - if True: #os.environ.get('UNSLOTH_USE_NEW_MODEL', '0') == '0': + if True: # os.environ.get('UNSLOTH_USE_NEW_MODEL', '0') == '0': return None # Unsloth efficient GRPO # Otherwise, calculate normally: if not hasattr(self, '_autocast_dtype'):