Update llama.py

This commit is contained in:
Daniel Han-Chen 2024-02-24 03:38:25 +11:00
commit 1a2a10d028

View file

@ -203,7 +203,7 @@ def LlamaAttention_fast_forward_inference(
A = torch.matmul(A, Vnn, out = Qn)
A = A.transpose(1, 2)
A = A.reshape(bsz, 1, attention_size)
# print(self.temp_QA[1][:,:,self.hidden_size].shape)
print(self.temp_QA[1][:,:,self.hidden_size].shape)
A = fast_linear_forward(self.o_proj, A, out = self.temp_QA[1][:,:,self.hidden_size])
return A, (Kn, Vn)
pass