Update llama.py

This commit is contained in:
Daniel Han-Chen 2024-02-01 19:19:44 +11:00
commit e8ec80a4c2

View file

@ -286,15 +286,15 @@ def fast_mlp_inference(self, X):
# up = self.up_proj(X)
bsz, _, hd = X.shape
mlp_size = self.config.intermediate_size
# temp = torch.empty((2, bsz, 1, mlp_size), dtype = X.dtype, device = "cuda")
temp = torch.empty((2, bsz, 1, mlp_size), dtype = X.dtype, device = "cuda")
gate = fast_linear_forward(self.gate_proj, X)#, out = temp[0])
up = fast_linear_forward(self. up_proj, X)#, out = temp[1])
gate = fast_linear_forward(self.gate_proj, X, out = temp[0])
up = fast_linear_forward(self. up_proj, X, out = temp[1])
gate = torch.nn.functional.silu(gate, inplace = True)
gate *= up
# X = self.down_proj(gate)
down = fast_linear_forward(self.down_proj, gate)#, out = up[:,:,:hd])
down = fast_linear_forward(self.down_proj, gate, out = up[:,:,:hd])
return down
pass