diff --git a/unsloth/models/rl_replacements.py b/unsloth/models/rl_replacements.py index b91c808710..8e930261f6 100644 --- a/unsloth/models/rl_replacements.py +++ b/unsloth/models/rl_replacements.py @@ -202,6 +202,7 @@ grpo_compute_loss = RL_REPLACEMENTS["grpo_compute_loss"] RL_PRE_ITEMS["grpo_trainer"].append(inspect.getsource(grpo_compute_loss)) global INPUTS +INPUTS = None # Edit _get_per_token_logps to handle mixed precision def grpo_trainer_compute_loss(function_name, function):