Update rms_layernorm.py

This commit is contained in:
Daniel Han 2024-11-04 23:19:02 -08:00
commit 2d8d1e1e2d

View file

@ -144,7 +144,7 @@ class Fast_RMS_Layernorm(torch.autograd.Function):
n_rows, n_cols = X.shape
BLOCK_SIZE : int
num_warps : int
BLOCK_SIZE, num_warps = calculate_settings(n_cols)
# BLOCK_SIZE, num_warps = calculate_settings(n_cols)
Y = torch.empty((n_rows, n_cols), dtype = X.dtype, device = "cuda:0")
r = torch.empty(n_rows, dtype = torch.float32, device = "cuda:0")
@ -157,7 +157,7 @@ class Fast_RMS_Layernorm(torch.autograd.Function):
r, r.stride(0),
n_cols = int(n_cols),
eps = float(eps),
BLOCK_SIZE = BLOCK_SIZE,
BLOCK_SIZE = 4096,
num_warps = 16,
)
else: