#008

fused_add_rms_norm

bf16 vllm · both · vllm.csrc.layernorm_kernels · importance 1.1%

Reference Implementation

reference.py
import torch
import torch.nn as nn

class Model(nn.Module):

    def __init__(self, eps: float=1e-06) -> None:
        super().__init__()
        self.eps = float(eps)

    def forward(self, x: torch.Tensor, residual: torch.Tensor, weight: torch.Tensor) -> dict[str, torch.Tensor]:
        residual_out = (x + residual).to(residual.dtype)
        residual_float = residual_out.float()
        variance = residual_float.square().mean(dim=-1, keepdim=True)
        normed = residual_float * torch.rsqrt(variance + self.eps) * weight.float()
        return {'out': normed.to(x.dtype), 'residual': residual_out}

Shapes

TSOL hardware:
# token_counthidden_size TSOL(XPU-A)TProdS
0 247521280 47.82 us 105.20 us 45.5%
1 1262048 0.39 us 7.80 us 5.0%
2 2572048 0.80 us 8.40 us 9.5%
3 5082048 1.57 us 8.40 us 18.7%
4 10242048 3.17 us 11.60 us 27.3%
5 20322048 6.28 us 16.80 us 37.4%
6 40962048 12.66 us 27.90 us 45.4%
7 1415120 1.09 us 8.10 us 13.5%
8 2845120 2.20 us 10.10 us 21.8%
9 5585120 4.31 us 13.80 us 31.2%
10 10245120 7.92 us 20.70 us 38.3%
11 17205120 13.29 us 31.20 us 42.6%
12 42055120 32.50 us 66.10 us 49.2%
13 537168 0.58 us 7.90 us 7.3%
14 1417168 1.53 us 8.80 us 17.4%
15 6327168 6.84 us 18.50 us 37.0%
16 9177168 9.92 us 24.70 us 40.2%
17 16887168 18.27 us 40.00 us 45.7%
18 38047168 41.16 us 81.20 us 50.7%

Input Generation

input.py
import torch

def _make_inputs(token_count: int, hidden_size: int) -> dict[str, torch.Tensor]:
    x = torch.randn(token_count, hidden_size, dtype=torch.bfloat16, device='cuda') * 0.02
    residual = torch.randn_like(x)
    weight = torch.randn(hidden_size, dtype=torch.bfloat16, device='cuda') * 0.02 + 1.0
    return {'x': x, 'residual': residual, 'weight': weight}