From 88a695da3a406825e5789fb5bd6cbe37fa35a479 Mon Sep 17 00:00:00 2001 From: Daniel Han-Chen Date: Sun, 4 Feb 2024 02:49:51 +1100 Subject: [PATCH] Update llama.py --- unsloth/models/llama.py | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/unsloth/models/llama.py b/unsloth/models/llama.py index b23f5fa4d6..fb79d34092 100644 --- a/unsloth/models/llama.py +++ b/unsloth/models/llama.py @@ -282,6 +282,7 @@ def LlamaAttention_fast_forward( # Attention module if (not HAS_FLASH_ATTENTION and attention_mask is None): + print(1) # Xformers memory efficient attention # Also has Flash Attention v2 dispatching Q = Q.transpose(1, 2) @@ -309,7 +310,7 @@ def LlamaAttention_fast_forward( V = V.transpose(1, 2) A = flash_attn_func(Q, K, V, causal = True) else: - print(attention_mask) + print(2) # Grouped query attention if n_groups != 1: K = K[:, :, None, :, :].expand(bsz, n_kv_heads, n_groups, kv_seq_len, head_dim)