fix cross entropy loss issue for small vocab size on amd gpu (#3503)

This commit is contained in:
wangxunx 2025-10-27 12:20:47 +08:00 committed by GitHub
commit 6fd28a4dce
2 changed files with 2 additions and 6 deletions

View file

@ -300,6 +300,7 @@ class Fast_CrossEntropyLoss(torch.autograd.Function):
if n_chunks == 1:
# For small vocabs <= 65336 like Llama, Mistral
BLOCK_SIZE, num_warps = calculate_settings(vocab_size)
if is_cdna(): num_warps = num_warps // 2
logsumexp = torch.empty(n_rows, dtype = torch.float32, device = device)
with torch_gpu_device(device):

View file

@ -298,12 +298,7 @@ def MistralForCausalLM_fast_forward(
else:
RETURN_LOGITS = os.environ.get("UNSLOTH_RETURN_LOGITS", "0") == "1"
# < 1024 Normal Unsloth uses less VRAM!
if DEVICE_TYPE == "hip":
# [TODO] AMD GPUs fail on chunked_cross_entropy loss!
# RuntimeError: Triton Error [HIP]: Code: 1, Messsage: invalid argument
RETURN_LOGITS = False
elif bsz*q_len <= 1024:
RETURN_LOGITS = True
if bsz * q_len <= 1024: RETURN_LOGITS = True
if not RETURN_LOGITS and labels is not None:
n_items = kwargs.get("num_items_in_batch", None) or kwargs.get("n_items", None)