#008
fused_add_rms_norm
bf16 vllm · both · vllm.csrc.layernorm_kernels · importance 1.1%
Reference Implementation
reference.py import torch
import torch.nn as nn
class Model(nn.Module):
def __init__(self, eps: float=1e-06) -> None:
super().__init__()
self.eps = float(eps)
def forward(self, x: torch.Tensor, residual: torch.Tensor, weight: torch.Tensor) -> dict[str, torch.Tensor]:
residual_out = (x + residual).to(residual.dtype)
residual_float = residual_out.float()
variance = residual_float.square().mean(dim=-1, keepdim=True)
normed = residual_float * torch.rsqrt(variance + self.eps) * weight.float()
return {'out': normed.to(x.dtype), 'residual': residual_out}
Shapes
TSOL hardware:
| # | token_count | hidden_size | TProd | S |
| 0 | 24752 | 1280 | 47.82 us | 105.20 us | 45.5% |
| 1 | 126 | 2048 | 0.39 us | 7.80 us | 5.0% |
| 2 | 257 | 2048 | 0.80 us | 8.40 us | 9.5% |
| 3 | 508 | 2048 | 1.57 us | 8.40 us | 18.7% |
| 4 | 1024 | 2048 | 3.17 us | 11.60 us | 27.3% |
| 5 | 2032 | 2048 | 6.28 us | 16.80 us | 37.4% |
| 6 | 4096 | 2048 | 12.66 us | 27.90 us | 45.4% |
| 7 | 141 | 5120 | 1.09 us | 8.10 us | 13.5% |
| 8 | 284 | 5120 | 2.20 us | 10.10 us | 21.8% |
| 9 | 558 | 5120 | 4.31 us | 13.80 us | 31.2% |
| 10 | 1024 | 5120 | 7.92 us | 20.70 us | 38.3% |
| 11 | 1720 | 5120 | 13.29 us | 31.20 us | 42.6% |
| 12 | 4205 | 5120 | 32.50 us | 66.10 us | 49.2% |
| 13 | 53 | 7168 | 0.58 us | 7.90 us | 7.3% |
| 14 | 141 | 7168 | 1.53 us | 8.80 us | 17.4% |
| 15 | 632 | 7168 | 6.84 us | 18.50 us | 37.0% |
| 16 | 917 | 7168 | 9.92 us | 24.70 us | 40.2% |
| 17 | 1688 | 7168 | 18.27 us | 40.00 us | 45.7% |
| 18 | 3804 | 7168 | 41.16 us | 81.20 us | 50.7% |
Input Generation
input.py import torch
def _make_inputs(token_count: int, hidden_size: int) -> dict[str, torch.Tensor]:
x = torch.randn(token_count, hidden_size, dtype=torch.bfloat16, device='cuda') * 0.02
residual = torch.randn_like(x)
weight = torch.randn(hidden_size, dtype=torch.bfloat16, device='cuda') * 0.02 + 1.0
return {'x': x, 'residual': residual, 'weight': weight}