From e94647c2dc2f0cf3de3c617983ea67359be5c06f Mon Sep 17 00:00:00 2001 From: Daniel Han-Chen Date: Wed, 7 Feb 2024 19:25:46 +1100 Subject: [PATCH] Update llama.py --- unsloth/models/llama.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/unsloth/models/llama.py b/unsloth/models/llama.py index f682eaf37c..5a45ad00a5 100644 --- a/unsloth/models/llama.py +++ b/unsloth/models/llama.py @@ -199,7 +199,7 @@ def LlamaAttention_fast_forward_inference( A = torch.matmul(Qn, Knn.transpose(2, 3)) #out = self.attention[:,:,:,:attention_size]) A *= self.scalar A[:] = torch.nn.functional.softmax(A, dim = -1, dtype = torch.float32)#.to(A.dtype) - A = torch.matmul(A, Vnn, out = Qn) + A = torch.matmul(A, Vnn)#, out = Qn) A = A.transpose(1, 2) A = A.reshape(bsz, 1, self.hidden_size) A = fast_linear_forward(self.o_proj, A, out = self.temp_QA[1])