Update llama.py

This commit is contained in:
Daniel Han 2025-02-13 19:11:38 -08:00
commit 47892d60d7

View file

@ -449,7 +449,6 @@ def LlamaAttention_fast_forward(
else:
# Grouped query attention
if SDPA_HAS_GQA:
print("##")
# Needs (batch_size, n_heads, seq_len, head_dim)
# is_casual and attention_mask must not be both set!
A = scaled_dot_product_attention(Q, K, V, attn_mask = attention_mask, is_causal = False, enable_gqa = n_groups != 1)
@ -708,8 +707,8 @@ def LlamaModel_fast_forward(
# Ignore attention_mask
if attention_mask is None:
padding_mask = None
# elif self.training:
elif attention_mask is None:
elif self.training:
# elif attention_mask is None:
attention_mask = None
padding_mask = None
else: