zaydzuhri commited on May 11

Commit

66dbc1f

verified ·

1 Parent(s): 5379428

Add files using upload-large-folder tool

Browse files

This view is limited to 50 files because it contains too many changes. See raw diff

Files changed (50) hide show

configs/rectified_transformer_340M.json +19 -0
configs/scaled_vanilla_transformer_120M.json +19 -0
configs/transformer_1B.json +22 -0
configs/transformer_340M.json +18 -0
configs/vanilla_transformer_120M.json +19 -0
fla/__init__.py +114 -0
fla/__pycache__/__init__.cpython-312.pyc +0 -0
fla/__pycache__/utils.cpython-312.pyc +0 -0
fla/modules/feature_map.py +300 -0
fla/modules/fused_bitlinear.py +638 -0
fla/modules/fused_kl_div.py +323 -0
fla/modules/fused_linear_cross_entropy.py +570 -0
fla/modules/layernorm.py +1196 -0
fla/modules/parallel.py +37 -0
fla/modules/rotary.py +512 -0
fla/ops/__init__.py +46 -0
fla/ops/abc/__init__.py +7 -0
fla/ops/attn/__pycache__/__init__.cpython-312.pyc +0 -0
fla/ops/attn/__pycache__/naive_rectified.cpython-312.pyc +0 -0
fla/ops/attn/__pycache__/parallel.cpython-312.pyc +0 -0
fla/ops/attn/naive.py +28 -0
fla/ops/attn/parallel.py +629 -0
fla/ops/attn/parallel_rectified.py +643 -0
fla/ops/attn/parallel_softpick.py +650 -0
fla/ops/based/__pycache__/__init__.cpython-312.pyc +0 -0
fla/ops/based/__pycache__/parallel.cpython-312.pyc +0 -0
fla/ops/based/naive.py +72 -0
fla/ops/based/parallel.py +410 -0
fla/ops/common/__pycache__/__init__.cpython-312.pyc +0 -0
fla/ops/common/__pycache__/chunk_delta_h.cpython-312.pyc +0 -0
fla/ops/common/__pycache__/chunk_o.cpython-312.pyc +0 -0
fla/ops/common/__pycache__/fused_recurrent.cpython-312.pyc +0 -0
fla/ops/common/__pycache__/utils.cpython-312.pyc +0 -0
fla/ops/delta_rule/__pycache__/__init__.cpython-312.pyc +0 -0
fla/ops/delta_rule/__pycache__/wy_fast.cpython-312.pyc +0 -0
fla/ops/delta_rule/naive.py +120 -0
fla/ops/forgetting_attn/__pycache__/parallel.cpython-312.pyc +0 -0
fla/ops/forgetting_attn/parallel.py +708 -0
fla/ops/gated_delta_rule/__pycache__/__init__.cpython-312.pyc +0 -0
fla/ops/gated_delta_rule/__pycache__/fused_recurrent.cpython-312.pyc +0 -0
fla/ops/gated_delta_rule/__pycache__/wy_fast.cpython-312.pyc +0 -0
fla/ops/generalized_delta_rule/dplr/chunk_h_fwd.py +197 -0
fla/ops/gla/__pycache__/__init__.cpython-312.pyc +0 -0
fla/ops/gla/__pycache__/chunk.cpython-312.pyc +0 -0
fla/ops/gla/__pycache__/fused_recurrent.cpython-312.pyc +0 -0
fla/ops/gla/chunk.py +1486 -0
fla/ops/gsa/__pycache__/chunk.cpython-312.pyc +0 -0
fla/ops/hgrn/__pycache__/__init__.cpython-312.pyc +0 -0
fla/ops/hgrn/__pycache__/fused_recurrent.cpython-312.pyc +0 -0
fla/ops/lightning_attn/chunk.py +74 -0

configs/rectified_transformer_340M.json ADDED Viewed

	@@ -0,0 +1,19 @@

+{
+    "attention_bias": false,
+    "bos_token_id": 1,
+    "eos_token_id": 2,
+    "fuse_cross_entropy": true,
+    "fuse_norm": true,
+    "hidden_act": "swish",
+    "hidden_size": 1024,
+    "initializer_range": 0.006,
+    "max_position_embeddings": 8192,
+    "model_type": "transformer",
+    "num_heads": 16,
+    "num_hidden_layers": 24,
+    "norm_eps": 1e-06,
+    "tie_word_embeddings": false,
+    "use_cache": true,
+    "vocab_size": 32000,
+    "attn_impl": "parallel_rectified_attn"
+}

configs/scaled_vanilla_transformer_120M.json ADDED Viewed

	@@ -0,0 +1,19 @@

+{
+    "attention_bias": false,
+    "bos_token_id": 1,
+    "eos_token_id": 2,
+    "fuse_cross_entropy": true,
+    "fuse_norm": false,
+    "hidden_act": "swish",
+    "hidden_size": 768,
+    "initializer_range": 0.02,
+    "max_position_embeddings": 4096,
+    "model_type": "transformer",
+    "num_heads": 12,
+    "num_hidden_layers": 14,
+    "norm_eps": 1e-06,
+    "tie_word_embeddings": true,
+    "use_cache": true,
+    "vocab_size": 32000,
+    "attn_impl": "parallel_scaled_attn"
+}

configs/transformer_1B.json ADDED Viewed

	@@ -0,0 +1,22 @@

+{
+    "bos_token_id": 1,
+    "elementwise_affine": true,
+    "eos_token_id": 2,
+    "fuse_cross_entropy": true,
+    "fuse_norm": true,
+    "fuse_swiglu": true,
+    "hidden_act": "swish",
+    "hidden_ratio": 4,
+    "hidden_size": 2048,
+    "initializer_range": 0.006,
+    "intermediate_size": null,
+    "max_position_embeddings": 8192,
+    "model_type": "transformer",
+    "norm_eps": 1e-06,
+    "num_heads": 32,
+    "num_hidden_layers": 24,
+    "num_kv_heads": null,
+    "pad_token_id": 2,
+    "rope_theta": 10000.0,
+    "tie_word_embeddings": false
+}

configs/transformer_340M.json ADDED Viewed

	@@ -0,0 +1,18 @@

+{
+    "attention_bias": false,
+    "bos_token_id": 1,
+    "eos_token_id": 2,
+    "fuse_cross_entropy": true,
+    "fuse_norm": true,
+    "hidden_act": "swish",
+    "hidden_size": 1024,
+    "initializer_range": 0.006,
+    "max_position_embeddings": 8192,
+    "model_type": "transformer",
+    "num_heads": 16,
+    "num_hidden_layers": 24,
+    "norm_eps": 1e-06,
+    "tie_word_embeddings": false,
+    "use_cache": true,
+    "vocab_size": 32000
+}

configs/vanilla_transformer_120M.json ADDED Viewed

	@@ -0,0 +1,19 @@

+{
+    "attention_bias": false,
+    "bos_token_id": 1,
+    "eos_token_id": 2,
+    "fuse_cross_entropy": true,
+    "fuse_norm": false,
+    "hidden_act": "swish",
+    "hidden_size": 768,
+    "initializer_range": 0.02,
+    "max_position_embeddings": 4096,
+    "model_type": "transformer",
+    "num_heads": 12,
+    "num_hidden_layers": 14,
+    "norm_eps": 1e-06,
+    "tie_word_embeddings": true,
+    "use_cache": true,
+    "vocab_size": 32000,
+    "attn_impl": "naive_attn"
+}

fla/__init__.py ADDED Viewed

	@@ -0,0 +1,114 @@

+# -*- coding: utf-8 -*-
+from fla.layers import (
+    ABCAttention,
+    Attention,
+    BasedLinearAttention,
+    BitAttention,
+    DeltaNet,
+    GatedDeltaNet,
+    GatedDeltaProduct,
+    GatedLinearAttention,
+    GatedSlotAttention,
+    HGRN2Attention,
+    HGRNAttention,
+    LightNetAttention,
+    LinearAttention,
+    MultiScaleRetention,
+    NativeSparseAttention,
+    ReBasedLinearAttention,
+    RWKV6Attention,
+    RWKV7Attention,
+)
+from fla.models import (
+    ABCForCausalLM,
+    ABCModel,
+    BitNetForCausalLM,
+    BitNetModel,
+    DeltaNetForCausalLM,
+    DeltaNetModel,
+    GatedDeltaNetForCausalLM,
+    GatedDeltaNetModel,
+    GatedDeltaProductForCausalLM,
+    GatedDeltaProductModel,
+    GLAForCausalLM,
+    GLAModel,
+    GSAForCausalLM,
+    GSAModel,
+    HGRN2ForCausalLM,
+    HGRN2Model,
+    HGRNForCausalLM,
+    LightNetForCausalLM,
+    LightNetModel,
+    LinearAttentionForCausalLM,
+    LinearAttentionModel,
+    NSAForCausalLM,
+    NSAModel,
+    RetNetForCausalLM,
+    RetNetModel,
+    RWKV6ForCausalLM,
+    RWKV6Model,
+    RWKV7ForCausalLM,
+    RWKV7Model,
+    TransformerForCausalLM,
+    TransformerModel,
+    TransformerWithPruningForCausalLM,
+    TransformerWithPruningModel
+)
+__all__ = [
+    'ABCAttention',
+    'Attention',
+    'BasedLinearAttention',
+    'BitAttention',
+    'DeltaNet',
+    'GatedDeltaNet',
+    'GatedDeltaProduct',
+    'GatedLinearAttention',
+    'GatedSlotAttention',
+    'HGRNAttention',
+    'HGRN2Attention',
+    'LightNetAttention',
+    'LinearAttention',
+    'MultiScaleRetention',
+    'NativeSparseAttention',
+    'ReBasedLinearAttention',
+    'RWKV6Attention',
+    'RWKV7Attention',
+    'ABCForCausalLM',
+    'ABCModel',
+    'BitNetForCausalLM',
+    'BitNetModel',
+    'DeltaNetForCausalLM',
+    'DeltaNetModel',
+    'GatedDeltaNetForCausalLM',
+    'GatedDeltaNetModel',
+    'GatedDeltaProductForCausalLM',
+    'GatedDeltaProductModel',
+    'GLAForCausalLM',
+    'GLAModel',
+    'GSAForCausalLM',
+    'GSAModel',
+    'HGRNForCausalLM',
+    'HGRNModel',
+    'HGRN2ForCausalLM',
+    'HGRN2Model',
+    'LightNetForCausalLM',
+    'LightNetModel',
+    'LinearAttentionForCausalLM',
+    'LinearAttentionModel',
+    'NSAForCausalLM',
+    'NSAModel',
+    'RetNetForCausalLM',
+    'RetNetModel',
+    'RWKV6ForCausalLM',
+    'RWKV6Model',
+    'RWKV7ForCausalLM',
+    'RWKV7Model',
+    'TransformerForCausalLM',
+    'TransformerModel',
+    'TransformerWithPruningForCausalLM',
+    'TransformerWithPruningModel',
+]
+__version__ = '0.1.2'

fla/__pycache__/__init__.cpython-312.pyc ADDED Viewed

Binary file (1.95 kB). View file

fla/__pycache__/utils.cpython-312.pyc ADDED Viewed

Binary file (12.3 kB). View file

fla/modules/feature_map.py ADDED Viewed

	@@ -0,0 +1,300 @@

+# -*- coding: utf-8 -*-
+from __future__ import annotations
+import math
+from typing import Optional
+import torch
+import torch.nn.functional as F
+from torch import nn
+from fla.modules.activations import fast_gelu_impl, sigmoid, sqrelu, swish
+from fla.modules.layernorm import layer_norm
+from fla.utils import checkpoint
+@checkpoint
+def flatten_diag_outer_product(x, y):
+    z = torch.einsum("...i,...j->...ij", x, y)
+    N = z.size(-1)
+    indicies = torch.triu_indices(N, N)
+    return z[..., indicies[0], indicies[1]]
+@checkpoint
+def flatten_diag_outer_product_off1(x, y):
+    z = torch.einsum("...i,...j->...ij", x, y)
+    N = z.size(-1)
+    indicies = torch.triu_indices(N, N, 1)
+    indices2 = torch.arange(0, N)
+    return z[..., indicies[0], indicies[1]], z[..., indices2, indices2]
+def is_power_of_2(n):
+    return (n & (n - 1) == 0) and n != 0
+class HedgehogFeatureMap(nn.Module):
+    r"""
+    Hedgehog feature map as introduced in
+    `The Hedgehog & the Porcupine: Expressive Linear Attentions with Softmax Mimicry <https://arxiv.org/abs/2402.04347>`_
+    """
+    def __init__(
+        self,
+        head_dim: int
+    ) -> HedgehogFeatureMap:
+        super().__init__()
+        # Trainable map
+        self.layer = nn.Linear(head_dim, head_dim)
+        self.init_weights_()
+    def init_weights_(self):
+        """Initialize trainable map as identity"""
+        with torch.no_grad():
+            identity = torch.eye(*self.layer.weight.shape[-2:], dtype=torch.float)
+            self.layer.weight.copy_(identity.to(self.layer.weight))
+        nn.init.zeros_(self.layer.bias)
+    def forward(self, x: torch.Tensor):
+        x = self.layer(x)  # shape b, h, l, d
+        return torch.cat([2*x, -2*x], dim=-1).softmax(-1)
+class T2RFeatureMap(nn.Module):
+    r"""
+    Simple linear mapping feature map as in
+    `Finetuning Pretrained Transformers into RNNs <https://arxiv.org/abs/2103.13076>`_
+    """
+    def __init__(
+        self,
+        head_dim: int,
+        dot_dim: int = None,
+        bias: Optional[bool] = False
+    ) -> T2RFeatureMap:
+        super().__init__()
+        # Trainable map
+        if dot_dim is None:
+            dot_dim = head_dim
+        self.head_dim = head_dim
+        self.dot_dim = dot_dim
+        self.bias = bias
+        self.layer = nn.Linear(head_dim, dot_dim, bias=bias)
+    def __repr__(self) -> str:
+        return f"{self.__class__.__name__}(head_dim={self.head_dim}, dot_dim={self.dot_dim}, bias={self.bias})"
+    def forward(self, x: torch.Tensor):
+        return self.layer(x).relu()
+class DPFPFeatureMap(nn.Module):
+    r"""
+    Deterministic Parameter-Free Projection (DPFP) feature map in
+    `Linear Transformers Are Secretly Fast Weight Programmers <https://arxiv.org/abs/2102.11174>`_
+    """
+    def __init__(
+        self,
+        head_dim: int,
+        nu: int = 4
+    ) -> DPFPFeatureMap:
+        super().__init__()
+        self.nu = nu
+    def forward(self, x: torch.Tensor):
+        x = torch.cat([x.relu(), -x.relu()], dim=-1)
+        x_rolled = torch.cat([x.roll(shifts=j, dims=-1) for j in range(1, self.nu+1)], dim=-1)
+        x_repeat = torch.cat([x] * self.nu, dim=-1)
+        return x_repeat * x_rolled
+class HadamardFeatureMap(nn.Module):
+    def __init__(
+        self,
+        head_dim: int
+    ) -> HadamardFeatureMap:
+        super().__init__()
+        # Trainable map
+        self.layer1 = nn.Linear(head_dim, head_dim)
+        self.layer2 = nn.Linear(head_dim, head_dim)
+    def forward(self, x: torch.Tensor):
+        return self.layer1(x) * self.layer2(x)
+class LearnableOuterProductFeatureMap(nn.Module):
+    def __init__(
+        self,
+        head_dim: int,
+        feature_dim: int
+    ) -> LearnableOuterProductFeatureMap:
+        super().__init__()
+        # Trainable map
+        self.layer1 = nn.Linear(head_dim, feature_dim, bias=False)
+        self.layer2 = nn.Linear(head_dim, feature_dim, bias=False)
+        self.normalizer = feature_dim ** -0.5
+    def forward(self, x: torch.Tensor):
+        return flatten_diag_outer_product(self.layer1(x), self.layer2(x))
+class LearnablePolySketchNonNegativeFeatureMap(nn.Module):
+    def __init__(
+        self,
+        head_dim: int,
+        sketch_size: Optional[int] = None,
+        degree: Optional[int] = 2
+    ) -> LearnablePolySketchNonNegativeFeatureMap:
+        super().__init__()
+        assert is_power_of_2(degree) and degree >= 2, f"The degree {degree} must be a power of 2"
+        self.head_dim = head_dim
+        self.sketch_size = sketch_size if sketch_size is not None else head_dim
+        self.degree = degree
+        self.gamma = nn.Parameter(torch.ones(head_dim))
+        self.beta = nn.Parameter(torch.zeros(head_dim))
+        # NOTE: the sketch layers defined here are quite different from the original paper
+        # currently we simply use linear layers without any non-linear activations
+        self.sketches1 = nn.ModuleList([
+            nn.Linear(head_dim, sketch_size, bias=False),
+            *[nn.Linear(sketch_size, sketch_size, bias=False) for _ in range(int(math.log2(self.degree)) - 2)]
+        ])
+        self.sketches2 = nn.ModuleList([
+            nn.Linear(head_dim, sketch_size, bias=False),
+            *[nn.Linear(sketch_size, sketch_size, bias=False) for _ in range(int(math.log2(self.degree)) - 2)]
+        ])
+    def forward(self, x: torch.Tensor):
+        # Section 2.1
+        x = layer_norm(x, self.gamma, self.beta)
+        # first map the input to sketch size with learnable parameters
+        x = self.sketches1[0](x) * self.sketches2[0](x) * self.head_dim ** -0.5
+        for i in range(1, int(math.log2(self.degree)) - 1):
+            x = self.sketches1[i](x) * self.sketches2[i](x) * self.head_dim ** -0.5
+        # do sketch mapping for log2(p) - 1 times in total
+        # do p=2 mapping to ensure non-negativity
+        return flatten_diag_outer_product(x, x)
+class TaylorFeatureMap(nn.Module):
+    def __init__(
+        self,
+        head_dim: int
+    ) -> TaylorFeatureMap:
+        super().__init__()
+        self.head_dim = head_dim
+        self.r2 = math.sqrt(2)
+        self.rd = math.sqrt(self.head_dim)
+        self.rrd = math.sqrt(self.rd)
+    def forward(self, x: torch.Tensor):
+        x2_1, x2_2 = flatten_diag_outer_product_off1(x, x)
+        return torch.cat([torch.ones_like(x[..., 0:1]), x / self.rrd, x2_2 / (self.rd * self.r2), x2_1 / self.rd], dim=-1)
+class RebasedFeatureMap(nn.Module):
+    def __init__(
+        self,
+        head_dim: int,
+        use_gamma: Optional[bool] = True,
+        use_beta: Optional[bool] = True,
+        normalize: Optional[bool] = True
+    ) -> RebasedFeatureMap:
+        super().__init__()
+        self.head_dim = head_dim
+        self.use_gamma = use_gamma
+        self.use_beta = use_beta
+        self.normalize = normalize
+        self.gamma = None
+        self.beta = None
+        if use_gamma:
+            self.gamma = nn.Parameter(torch.ones(head_dim))
+        if use_beta:
+            self.beta = nn.Parameter(torch.zeros(head_dim))
+    def forward(self, x: torch.Tensor, flatten: Optional[bool] = True):
+        if self.use_beta and self.use_gamma and self.normalize:
+            x = layer_norm(x, self.gamma, self.beta)
+        elif self.normalize:
+            x = F.layer_norm(x, (self.head_dim,), self.gamma, self.beta)
+        elif self.use_gamma and self.use_beta:
+            x = torch.addcmul(self.beta, x, self.gamma)
+        elif self.use_gamma:
+            x = x.mul(self.gamma)
+        else:
+            raise RuntimeError(f"Not supported combination of `use_gamma`, `use_beta` and `normalize`, "
+                               f"which is currentlt set as (`{self.use_gamma}`, `{self.use_beta}`, `{self.normalize}`)")
+        if not flatten:
+            return x
+        x2_1, x2_2 = flatten_diag_outer_product_off1(x, x)
+        # rebased use learnable parameters to approximate any quadratic function
+        return torch.cat([x2_2 * self.head_dim ** -0.5, x2_1 * (2 / self.head_dim) ** 0.5], dim=-1)
+class ReLUFeatureMap(nn.Module):
+    def __init__(
+        self,
+    ) -> ReLUFeatureMap:
+        super().__init__()
+    def forward(self, x: torch.Tensor):
+        return F.relu(x)
+class SquaredReLUFeatureMap(nn.Module):
+    def __init__(
+        self,
+    ) -> SquaredReLUFeatureMap:
+        super().__init__()
+    def forward(self, x: torch.Tensor):
+        return sqrelu(x)
+class GELUFeatureMap(nn.Module):
+    def __init__(
+        self,
+    ) -> GELUFeatureMap:
+        super().__init__()
+    def forward(self, x: torch.Tensor):
+        return fast_gelu_impl(x)
+class SwishFeatureMap(nn.Module):
+    def __init__(
+        self,
+    ) -> SwishFeatureMap:
+        super().__init__()
+    def forward(self, x: torch.Tensor):
+        return swish(x)
+class SigmoidFeatureMap(nn.Module):
+    def __init__(
+        self,
+    ) -> SigmoidFeatureMap:
+        super().__init__()
+    def forward(self, x: torch.Tensor):
+        return sigmoid(x)

fla/modules/fused_bitlinear.py ADDED Viewed

	@@ -0,0 +1,638 @@

+# -*- coding: utf-8 -*-
+# Copyright (c) 2023-2025, Songlin Yang, Yu Zhang
+# Implementations of BitLinear layer with fused LayerNorm and quantized Linear layer.
+# [The Era of 1-bit LLMs: All Large Language Models are in 1.58 Bits](https://arxiv.org/abs/2402.17764)
+# [Scalable MatMul-free Language Modeling](https://arxiv.org/abs/2406.02528)
+# Code adapted from https://github.com/ridgerchu/matmulfreellm/
+from __future__ import annotations
+import math
+import torch
+import torch.nn as nn
+import torch.nn.functional as F
+import triton
+import triton.language as tl
+from fla.modules.layernorm import RMSNorm
+from fla.utils import get_multiprocessor_count, input_guard, require_version
+def activation_quant(x):
+    """
+    Per-token quantization to 8 bits. No grouping is needed for quantization.
+    Args:
+        x: An activation tensor with shape [n, d].
+    Returns:
+        A quantized activation tensor with shape [n, d].
+    """
+    # Compute the scale factor
+    scale = 127.0 / x.abs().max(dim=-1, keepdim=True).values.clamp_(min=1e-5)
+    # Quantize and then de-quantize the tensor
+    y = (x * scale).round().clamp_(-128, 127) / scale
+    return y
+def weight_quant(w):
+    """
+    Per-tensor quantization to 1.58 bits. No grouping is needed for quantization.
+    Args:
+        w: A weight tensor with shape [d, k].
+    Returns:
+        A quantized weight tensor with shape [d, k].
+    """
+    # Compute the scale factor
+    scale = 1.0 / w.abs().mean().clamp_(min=1e-5)
+    # Quantize and then de-quantize the tensor
+    u = (w * scale).round().clamp_(-1, 1) / scale
+    return u
+@triton.autotune(
+    configs=[
+        triton.Config({}, num_warps=1),
+        triton.Config({}, num_warps=2),
+        triton.Config({}, num_warps=4),
+        triton.Config({}, num_warps=8),
+        triton.Config({}, num_warps=16),
+        triton.Config({}, num_warps=32),
+    ],
+    key=["N", "HAS_RESIDUAL", "STORE_RESIDUAL_OUT", "IS_RMS_NORM", "HAS_BIAS"],
+)
+@triton.jit
+def layer_norm_fwd_kernel_quant(
+    X,  # pointer to the input
+    Y,  # pointer to the output
+    W,  # pointer to the weights
+    B,  # pointer to the biases
+    RESIDUAL,  # pointer to the residual
+    RESIDUAL_OUT,  # pointer to the residual
+    Mean,  # pointer to the mean
+    Rstd,  # pointer to the 1/std
+    stride_x_row,  # how much to increase the pointer when moving by 1 row
+    stride_y_row,
+    stride_res_row,
+    stride_res_out_row,
+    N,  # number of columns in X
+    eps,  # epsilon to avoid division by zero
+    IS_RMS_NORM: tl.constexpr,
+    BLOCK_N: tl.constexpr,
+    HAS_RESIDUAL: tl.constexpr,
+    STORE_RESIDUAL_OUT: tl.constexpr,
+    HAS_WEIGHT: tl.constexpr,
+    HAS_BIAS: tl.constexpr
+):
+    # Map the program id to the row of X and Y it should compute.
+    row = tl.program_id(0)
+    X += row * stride_x_row
+    Y += row * stride_y_row
+    if HAS_RESIDUAL:
+        RESIDUAL += row * stride_res_row
+    if STORE_RESIDUAL_OUT:
+        RESIDUAL_OUT += row * stride_res_out_row
+    # Compute mean and variance
+    cols = tl.arange(0, BLOCK_N)
+    x = tl.load(X + cols, mask=cols < N, other=0.0).to(tl.float32)
+    if HAS_RESIDUAL:
+        residual = tl.load(RESIDUAL + cols, mask=cols < N, other=0.0).to(tl.float32)
+        x += residual
+    if STORE_RESIDUAL_OUT:
+        tl.store(RESIDUAL_OUT + cols, x, mask=cols < N)
+    if not IS_RMS_NORM:
+        mean = tl.sum(x, axis=0) / N
+        tl.store(Mean + row, mean)
+        xbar = tl.where(cols < N, x - mean, 0.0)
+        var = tl.sum(xbar * xbar, axis=0) / N
+    else:
+        xbar = tl.where(cols < N, x, 0.0)
+        var = tl.sum(xbar * xbar, axis=0) / N
+    rstd = 1 / tl.sqrt(var + eps)
+    tl.store(Rstd + row, rstd)
+    # Normalize and apply linear transformation
+    mask = cols < N
+    if HAS_WEIGHT:
+        w = tl.load(W + cols, mask=mask).to(tl.float32)
+    if HAS_BIAS:
+        b = tl.load(B + cols, mask=mask).to(tl.float32)
+    x_hat = (x - mean) * rstd if not IS_RMS_NORM else x * rstd
+    y = x_hat * w if HAS_WEIGHT else x_hat
+    if HAS_BIAS:
+        y = y + b
+    # Aply quantization to the output
+    scale = 127.0 / tl.maximum(tl.max(tl.abs(y), 0), 1e-5)
+    # Quantize and then de-quantize the tensor
+    y = tl.extra.cuda.libdevice.round(y * scale)
+    y = tl.maximum(tl.minimum(y, 127), -128) / scale
+    # Write output
+    tl.store(Y + cols, y, mask=mask)
+def layer_norm_fwd_quant(
+    x: torch.Tensor,
+    weight: torch.Tensor,
+    bias: torch.Tensor,
+    eps: float,
+    residual: torch.Tensor = None,
+    out_dtype: torch.dtype = None,
+    residual_dtype: torch.dtype = None,
+    is_rms_norm: bool = False
+):
+    if residual is not None:
+        residual_dtype = residual.dtype
+    M, N = x.shape
+    # allocate output
+    y = torch.empty_like(x, dtype=x.dtype if out_dtype is None else out_dtype)
+    if residual is not None or (residual_dtype is not None and residual_dtype != x.dtype):
+        residual_out = torch.empty(M, N, device=x.device, dtype=residual_dtype)
+    else:
+        residual_out = None
+    mean = torch.empty((M,), dtype=torch.float32, device=x.device) if not is_rms_norm else None
+    rstd = torch.empty((M,), dtype=torch.float32, device=x.device)
+    # Less than 64KB per feature: enqueue fused kernel
+    MAX_FUSED_SIZE = 65536 // x.element_size()
+    BLOCK_N = min(MAX_FUSED_SIZE, triton.next_power_of_2(N))
+    if N > BLOCK_N:
+        raise RuntimeError("This layer norm doesn't support feature dim >= 64KB.")
+    # heuristics for number of warps
+    layer_norm_fwd_kernel_quant[(M,)](
+        x,
+        y,
+        weight,
+        bias,
+        residual,
+        residual_out,
+        mean,
+        rstd,
+        x.stride(0),
+        y.stride(0),
+        residual.stride(0) if residual is not None else 0,
+        residual_out.stride(0) if residual_out is not None else 0,
+        N,
+        eps,
+        is_rms_norm,
+        BLOCK_N,
+        residual is not None,
+        residual_out is not None,
+        weight is not None,
+        bias is not None,
+    )
+    # residual_out is None if residual is None and residual_dtype == input_dtype
+    return y, mean, rstd, residual_out if residual_out is not None else x
+@triton.heuristics({
+    "RECOMPUTE_OUTPUT": lambda args: args["Y"] is not None
+})
+@triton.autotune(
+    configs=[
+        triton.Config({}, num_warps=1),
+        triton.Config({}, num_warps=2),
+        triton.Config({}, num_warps=4),
+        triton.Config({}, num_warps=8),
+        triton.Config({}, num_warps=16),
+        triton.Config({}, num_warps=32),
+    ],
+    key=["N", "HAS_DRESIDUAL", "STORE_DRESIDUAL", "IS_RMS_NORM", "HAS_BIAS"],
+)
+@triton.jit
+def layer_norm_bwd_kernel(
+    X,  # pointer to the input
+    W,  # pointer to the weights
+    B,  # pointer to the biases
+    Y,  # pointer to the output to be recomputed
+    DY,  # pointer to the output gradient
+    DX,  # pointer to the input gradient
+    DW,  # pointer to the partial sum of weights gradient
+    DB,  # pointer to the partial sum of biases gradient
+    DRESIDUAL,
+    DRESIDUAL_IN,
+    Mean,  # pointer to the mean
+    Rstd,  # pointer to the 1/std
+    stride_x_row,  # how much to increase the pointer when moving by 1 row
+    stride_y_row,
+    stride_dy_row,
+    stride_dx_row,
+    stride_dres_row,
+    stride_dres_in_row,
+    M,  # number of rows in X
+    N,  # number of columns in X
+    eps,  # epsilon to avoid division by zero
+    rows_per_program,
+    IS_RMS_NORM: tl.constexpr,
+    BLOCK_N: tl.constexpr,
+    HAS_DRESIDUAL: tl.constexpr,
+    STORE_DRESIDUAL: tl.constexpr,
+    HAS_WEIGHT: tl.constexpr,
+    HAS_BIAS: tl.constexpr,
+    RECOMPUTE_OUTPUT: tl.constexpr,
+):
+    # Map the program id to the elements of X, DX, and DY it should compute.
+    row_block_id = tl.program_id(0)
+    row_start = row_block_id * rows_per_program
+    cols = tl.arange(0, BLOCK_N)
+    mask = cols < N
+    X += row_start * stride_x_row
+    if HAS_DRESIDUAL:
+        DRESIDUAL += row_start * stride_dres_row
+    if STORE_DRESIDUAL:
+        DRESIDUAL_IN += row_start * stride_dres_in_row
+    DY += row_start * stride_dy_row
+    DX += row_start * stride_dx_row
+    if RECOMPUTE_OUTPUT:
+        Y += row_start * stride_y_row
+    if HAS_WEIGHT:
+        w = tl.load(W + cols, mask=mask).to(tl.float32)
+        dw = tl.zeros((BLOCK_N,), dtype=tl.float32)
+    if RECOMPUTE_OUTPUT and HAS_BIAS:
+        b = tl.load(B + cols, mask=mask, other=0.0).to(tl.float32)
+    if HAS_BIAS:
+        db = tl.zeros((BLOCK_N,), dtype=tl.float32)
+    row_end = min((row_block_id + 1) * rows_per_program, M)
+    for row in range(row_start, row_end):
+        # Load data to SRAM
+        x = tl.load(X + cols, mask=mask, other=0).to(tl.float32)
+        dy = tl.load(DY + cols, mask=mask, other=0).to(tl.float32)
+        if not IS_RMS_NORM:
+            mean = tl.load(Mean + row)
+        rstd = tl.load(Rstd + row)
+        # Compute dx
+        xhat = (x - mean) * rstd if not IS_RMS_NORM else x * rstd
+        xhat = tl.where(mask, xhat, 0.0)
+        if RECOMPUTE_OUTPUT:
+            y = xhat * w if HAS_WEIGHT else xhat
+            if HAS_BIAS:
+                y = y + b
+            # Aply quantization to the output
+            scale = 127.0 / tl.maximum(tl.max(tl.abs(y), 0), 1e-5)
+            # Quantize and then de-quantize the tensor
+            y = tl.extra.cuda.libdevice.round(y * scale)
+            y = tl.maximum(tl.minimum(y, 127), -128) / scale
+            tl.store(Y + cols, y, mask=mask)
+        wdy = dy
+        if HAS_WEIGHT:
+            wdy = dy * w
+            dw += dy * xhat
+        if HAS_BIAS:
+            db += dy
+        if not IS_RMS_NORM:
+            c1 = tl.sum(xhat * wdy, axis=0) / N
+            c2 = tl.sum(wdy, axis=0) / N
+            dx = (wdy - (xhat * c1 + c2)) * rstd
+        else:
+            c1 = tl.sum(xhat * wdy, axis=0) / N
+            dx = (wdy - xhat * c1) * rstd
+        if HAS_DRESIDUAL:
+            dres = tl.load(DRESIDUAL + cols, mask=mask, other=0).to(tl.float32)
+            dx += dres
+        # Write dx
+        if STORE_DRESIDUAL:
+            tl.store(DRESIDUAL_IN + cols, dx, mask=mask)
+        tl.store(DX + cols, dx, mask=mask)
+        X += stride_x_row
+        if HAS_DRESIDUAL:
+            DRESIDUAL += stride_dres_row
+        if STORE_DRESIDUAL:
+            DRESIDUAL_IN += stride_dres_in_row
+        if RECOMPUTE_OUTPUT:
+            Y += stride_y_row
+        DY += stride_dy_row
+        DX += stride_dx_row
+    if HAS_WEIGHT:
+        tl.store(DW + row_block_id * N + cols, dw, mask=mask)
+    if HAS_BIAS:
+        tl.store(DB + row_block_id * N + cols, db, mask=mask)
+def layer_norm_bwd(
+    dy: torch.Tensor,
+    x: torch.Tensor,
+    weight: torch.Tensor,
+    bias: torch.Tensor,
+    eps: float,
+    mean: torch.Tensor,
+    rstd: torch.Tensor,
+    dresidual: torch.Tensor = None,
+    has_residual: bool = False,
+    is_rms_norm: bool = False,
+    x_dtype: torch.dtype = None,
+    recompute_output: bool = False,
+):
+    M, N = x.shape
+    # allocate output
+    dx = torch.empty_like(x) if x_dtype is None else torch.empty(M, N, dtype=x_dtype, device=x.device)
+    dresidual_in = torch.empty_like(x) if has_residual and dx.dtype != x.dtype else None
+    y = torch.empty(M, N, dtype=dy.dtype, device=dy.device) if recompute_output else None
+    # Less than 64KB per feature: enqueue fused kernel
+    MAX_FUSED_SIZE = 65536 // x.element_size()
+    BLOCK_N = min(MAX_FUSED_SIZE, triton.next_power_of_2(N))
+    if N > BLOCK_N:
+        raise RuntimeError("This layer norm doesn't support feature dim >= 64KB.")
+    sm_count = get_multiprocessor_count(x.device.index)
+    _dw = torch.empty((sm_count, N), dtype=torch.float32, device=weight.device) if weight is not None else None
+    _db = torch.empty((sm_count, N), dtype=torch.float32, device=bias.device) if bias is not None else None
+    rows_per_program = math.ceil(M / sm_count)
+    grid = (sm_count,)
+    layer_norm_bwd_kernel[grid](
+        x,
+        weight,
+        bias,
+        y,
+        dy,
+        dx,
+        _dw,
+        _db,
+        dresidual,
+        dresidual_in,
+        mean,
+        rstd,
+        x.stride(0),
+        0 if not recompute_output else y.stride(0),
+        dy.stride(0),
+        dx.stride(0),
+        dresidual.stride(0) if dresidual is not None else 0,
+        dresidual_in.stride(0) if dresidual_in is not None else 0,
+        M,
+        N,
+        eps,
+        rows_per_program,
+        is_rms_norm,
+        BLOCK_N,
+        dresidual is not None,
+        dresidual_in is not None,
+        weight is not None,
+        bias is not None,
+    )
+    dw = _dw.sum(0).to(weight.dtype) if weight is not None else None
+    db = _db.sum(0).to(bias.dtype) if bias is not None else None
+    # Don't need to compute dresidual_in separately in this case
+    if has_residual and dx.dtype == x.dtype:
+        dresidual_in = dx
+    return (dx, dw, db, dresidual_in) if not recompute_output else (dx, dw, db, dresidual_in, y)
+class LayerNormLinearQuantFn(torch.autograd.Function):
+    @staticmethod
+    @input_guard
+    def forward(
+        ctx,
+        x,
+        norm_weight,
+        norm_bias,
+        linear_weight,
+        linear_bias,
+        residual=None,
+        eps=1e-6,
+        prenorm=False,
+        residual_in_fp32=False,
+        is_rms_norm=False,
+    ):
+        x_shape_og = x.shape
+        # reshape input data into 2D tensor
+        x = x.reshape(-1, x.shape[-1])
+        if residual is not None:
+            assert residual.shape == x_shape_og
+            residual = residual.reshape(-1, residual.shape[-1])
+        residual_dtype = residual.dtype if residual is not None else (torch.float32 if residual_in_fp32 else None)
+        y, mean, rstd, residual_out = layer_norm_fwd_quant(
+            x,
+            norm_weight,
+            norm_bias,
+            eps,
+            residual,
+            out_dtype=None if not torch.is_autocast_enabled() else torch.get_autocast_gpu_dtype(),
+            residual_dtype=residual_dtype,
+            is_rms_norm=is_rms_norm,
+        )
+        y = y.reshape(x_shape_og)
+        dtype = torch.get_autocast_gpu_dtype() if torch.is_autocast_enabled() else y.dtype
+        linear_weight = weight_quant(linear_weight).to(dtype)
+        linear_bias = linear_bias.to(dtype) if linear_bias is not None else None
+        out = F.linear(y.to(linear_weight.dtype), linear_weight, linear_bias)
+        # We don't store y, will be recomputed in the backward pass to save memory
+        ctx.save_for_backward(residual_out, norm_weight, norm_bias, linear_weight, mean, rstd)
+        ctx.x_shape_og = x_shape_og
+        ctx.eps = eps
+        ctx.is_rms_norm = is_rms_norm
+        ctx.has_residual = residual is not None
+        ctx.prenorm = prenorm
+        ctx.x_dtype = x.dtype
+        ctx.linear_bias_is_none = linear_bias is None
+        return out if not prenorm else (out, residual_out.reshape(x_shape_og))
+    @staticmethod
+    @input_guard
+    def backward(ctx, dout, *args):
+        x, norm_weight, norm_bias, linear_weight, mean, rstd = ctx.saved_tensors
+        dout = dout.reshape(-1, dout.shape[-1])
+        dy = F.linear(dout, linear_weight.t())
+        dlinear_bias = None if ctx.linear_bias_is_none else dout.sum(0)
+        assert dy.shape == x.shape
+        if ctx.prenorm:
+            dresidual = args[0]
+            dresidual = dresidual.reshape(-1, dresidual.shape[-1])
+            assert dresidual.shape == x.shape
+        else:
+            dresidual = None
+        dx, dnorm_weight, dnorm_bias, dresidual_in, y = layer_norm_bwd(
+            dy,
+            x,
+            norm_weight,
+            norm_bias,
+            ctx.eps,
+            mean,
+            rstd,
+            dresidual,
+            ctx.has_residual,
+            ctx.is_rms_norm,
+            x_dtype=ctx.x_dtype,
+            recompute_output=True
+        )
+        dlinear_weight = torch.einsum("bo,bi->oi", dout, y)
+        return (
+            dx.reshape(ctx.x_shape_og),
+            dnorm_weight,
+            dnorm_bias,
+            dlinear_weight,
+            dlinear_bias,
+            dresidual_in.reshape(ctx.x_shape_og) if ctx.has_residual else None,
+            None,
+            None,
+            None,
+            None,
+        )
+def layer_norm_linear_quant_fn(
+    x,
+    norm_weight,
+    norm_bias,
+    linear_weight,
+    linear_bias,
+    residual=None,
+    eps=1e-6,
+    prenorm=False,
+    residual_in_fp32=False,
+    is_rms_norm=False,
+):
+    return LayerNormLinearQuantFn.apply(
+        x,
+        norm_weight,
+        norm_bias,
+        linear_weight,
+        linear_bias,
+        residual,
+        eps,
+        prenorm,
+        residual_in_fp32,
+        is_rms_norm,
+    )
+def rms_norm_linear_quant(
+    x: torch.Tensor,
+    norm_weight: torch.Tensor,
+    norm_bias: torch.Tensor,
+    linear_weight: torch.Tensor,
+    linear_bias: torch.Tensor,
+    residual: torch.Tensor = None,
+    eps: float = 1e-5,
+    prenorm: bool = False,
+    residual_in_fp32: bool = False
+):
+    return layer_norm_linear_quant_fn(
+        x=x,
+        norm_weight=norm_weight,
+        norm_bias=norm_bias,
+        linear_weight=linear_weight,
+        linear_bias=linear_bias,
+        residual=residual,
+        eps=eps,
+        prenorm=prenorm,
+        residual_in_fp32=residual_in_fp32,
+        is_rms_norm=True
+    )
+@require_version("triton>=3.0", "Triton >= 3.0 is required to do online quantization.")
+def bit_linear(x, weight, bias=None, norm_weight=None, norm_bias=None, eps=1e-8):
+    """
+    A functional version of BitLinear that applies quantization to activations and weights.
+    Args:
+        x: Input tensor with shape [n, d].
+        weight: Weight tensor with shape [out_features, in_features].
+        bias: Bias tensor with shape [out_features] (optional).
+        norm_weight: Weight tensor for RMS normalization with shape [in_features].
+        norm_bias: Bias tensor for RMS normalization with shape [in_features].
+        eps: A small constant for numerical stability in normalization.
+    Returns:
+        Output tensor with shape [n, out_features].
+    """
+    return layer_norm_linear_quant_fn(
+        x,
+        norm_weight,
+        norm_bias,
+        weight,
+        bias,
+        is_rms_norm=True
+    )
+class BitLinear(nn.Linear):
+    """
+    A custom linear layer that applies quantization on both activations and weights.
+    This is primarily for training; kernel optimization is needed for efficiency in deployment.
+    """
+    def __init__(
+        self,
+        in_features: int,
+        out_features: int,
+        bias: bool = False,
+        norm_eps: float = 1e-8
+    ):
+        """
+        Initializes the BitLinear layer.
+        Args:
+            in_features: Size of each input sample.
+            out_features: Size of each output sample.
+            bias: If set to False, the layer will not learn an additive bias. Default: True.
+        """
+        # Initialize the superclass nn.Linear with the given parameters
+        super(BitLinear, self).__init__(in_features, out_features, bias=bias)
+        self.norm = RMSNorm(in_features, eps=norm_eps)
+    def __repr__(self) -> str:
+        return f"{self.__class__.__name__}({super().extra_repr()}, norm_eps={self.norm.eps})"
+    def forward(self, x):
+        """
+        Overrides the forward pass to include quantization.
+        Args:
+            x: An input tensor with shape [n, d].
+        Returns:
+            An output tensor with shape [n, d].
+        """
+        # Weight tensor
+        w = self.weight
+        # Apply RMS normalization to the input
+        x_norm = self.norm(x)
+        # Apply quantization to both activations and weights
+        # Uses Straight-Through Estimator (STE) trick with .detach() for gradient flow
+        x_quant = x_norm + (activation_quant(x_norm) - x_norm).detach()
+        w_quant = w + (weight_quant(w) - w).detach()
+        # Perform linear operation with quantized values
+        y = F.linear(x_quant, w_quant)
+        return y
+class FusedBitLinear(BitLinear):
+    """
+    A custom linear layer that applies quantization on both activations and weights.
+    This is primarily for training; kernel optimization is needed for efficiency in deployment.
+    """
+    def __init__(self, in_features, out_features, bias=False):
+        """
+        Initializes the BitLinear layer.
+        Args:
+            in_features: Size of each input sample.
+            out_features: Size of each output sample.
+            bias: If set to False, the layer will not learn an additive bias. Default: True.
+        """
+        # Initialize the superclass nn.Linear with the given parameters
+        super(FusedBitLinear, self).__init__(in_features, out_features, bias=bias)
+    def forward(self, x):
+        return layer_norm_linear_quant_fn(
+            x,
+            self.norm.weight,
+            self.norm.bias,
+            self.weight,
+            self.bias,
+            is_rms_norm=True
+        )

fla/modules/fused_kl_div.py ADDED Viewed

	@@ -0,0 +1,323 @@

+# -*- coding: utf-8 -*-
+from typing import Tuple
+import torch
+import torch.nn as nn
+import torch.nn.functional as F
+import triton
+import triton.language as tl
+from fla.ops.utils.op import exp, log
+from fla.utils import input_guard
+# The hard limit of TRITON_MAX_TENSOR_NUMEL is 1048576
+# https://github.com/triton-lang/triton/blob/ba42a5c68fd0505f8c42f4202d53be0f8d9a5fe0/python/triton/language/core.py#L19
+# However, setting limit as 65536 as in LayerNorm tutorial is faster because of less register spilling
+# The optimal maximum block size depends on your hardware, your kernel, and your dtype
+MAX_FUSED_SIZE = 65536 // 2
+@triton.jit
+def kl_div_kernel(
+    logits,
+    target_logits,
+    loss,
+    s_logits,
+    s_loss,
+    reduction: tl.constexpr,
+    N: tl.constexpr,
+    V: tl.constexpr,
+    BV: tl.constexpr
+):
+    # https://github.com/triton-lang/triton/issues/1058
+    # If N*V is too large, i_n * stride will overflow out of int32, so we convert to int64
+    i_n = tl.program_id(0).to(tl.int64)
+    logits += i_n * s_logits
+    target_logits += i_n * s_logits
+    # m is the max value. use the notation from the paper
+    sm = float('-inf')
+    tm = float('-inf')
+    # d is the sum. use the notation from the paper
+    sd, td = 0.0, 0.0
+    NV = tl.cdiv(V, BV)
+    for iv in range(0, NV):
+        o_x = iv * BV + tl.arange(0, BV)
+        # for student
+        b_sl = tl.load(logits + o_x, mask=o_x < V, other=float('-inf'))
+        b_sm = tl.max(b_sl)
+        m_new = tl.maximum(sm, b_sm)
+        sd = sd * exp(sm - m_new) + tl.sum(exp(b_sl - m_new))
+        sm = m_new
+        # for teacher
+        b_tl = tl.load(target_logits + o_x, mask=o_x < V, other=float('-inf'))
+        b_tm = tl.max(b_tl)
+        m_new = tl.maximum(tm, b_tm)
+        td = td * exp(tm - m_new) + tl.sum(exp(b_tl - m_new))
+        tm = m_new
+    b_loss = 0.
+    # KL(y_true || y) = exp(y_true) * (log(y_true) - log(y))
+    for iv in range(0, NV):
+        o_x = iv * BV + tl.arange(0, BV)
+        b_sl = tl.load(logits + o_x, mask=o_x < V, other=float('-inf'))
+        b_tl = tl.load(target_logits + o_x, mask=o_x < V, other=float('-inf'))
+        b_sp_log = b_sl - sm - log(sd)
+        b_tp_log = b_tl - tm - log(td)
+        b_sp = exp(b_sp_log)
+        b_tp = exp(b_tp_log)
+        b_kl = tl.where(o_x < V, b_tp * (b_tp_log - b_sp_log), 0)
+        b_dl = -b_tp + b_sp
+        b_loss += tl.sum(b_kl)
+        if reduction == 'batchmean':
+            b_dl = b_dl / N
+        tl.store(logits + o_x, b_dl, mask=o_x < V)
+    # Normalize the loss by the number of elements if reduction is 'batchmean'
+    if reduction == 'batchmean':
+        b_loss = b_loss / N
+    tl.store(loss + i_n * s_loss, b_loss)
+@triton.jit
+def elementwise_mul_kernel(
+    x,
+    g,
+    N: tl.constexpr,
+    B: tl.constexpr
+):
+    """
+    This function multiplies each element of the tensor pointed by x with the value pointed by g.
+    The multiplication is performed in-place on the tensor pointed by x.
+    Parameters:
+    x:
+        Pointer to the input tensor.
+    g:
+        Pointer to the gradient output value.
+    N (int):
+        The number of columns in the input tensor.
+    B (int):
+        The block size for Triton operations.
+    """
+    # Get the program ID and convert it to int64 to avoid overflow
+    i_x = tl.program_id(0).to(tl.int64)
+    o_x = i_x * B + tl.arange(0, B)
+    # Load the gradient output value
+    b_g = tl.load(g)
+    b_x = tl.load(x + o_x, mask=o_x < N)
+    tl.store(x + o_x, b_x * b_g, mask=o_x < N)
+def fused_kl_div_forward(
+    x: torch.Tensor,
+    target_x: torch.Tensor,
+    weight: torch.Tensor,
+    target_weight: torch.Tensor,
+    reduction: str = 'batchmean'
+):
+    device = x.device
+    # ideally, we would like to achieve the same memory consumption as [N, H],
+    # so the expected chunk size should be:
+    # NC = ceil(V / H)
+    # C = ceil(N / NC)
+    # for ex: N = 4096*4, V = 32000, H = 4096 ==> NC = 8, C = ceil(N / NC) = 2048
+    N, H, V = *x.shape, weight.shape[0]
+    BV = min(MAX_FUSED_SIZE, triton.next_power_of_2(V))
+    # TODO: in real cases, we may need to limit the number of chunks NC to
+    # ensure the precisions of accumulated gradients
+    NC = min(8, triton.cdiv(V, H))
+    C = triton.next_power_of_2(triton.cdiv(N, NC))
+    NC = triton.cdiv(N, C)
+    dx = torch.zeros_like(x, device=device)
+    dw = torch.zeros_like(weight, device=device) if weight is not None else None
+    # we use fp32 for loss accumulator
+    loss = torch.zeros(N, dtype=torch.float32, device=device)
+    for ic in range(NC):
+        start, end = ic * C, min((ic + 1) * C, N)
+        # [C, N]
+        c_sx = x[start:end]
+        c_tx = target_x[start:end]
+        # when doing matmul, use the original precision
+        # [C, V]
+        c_sl = F.linear(c_sx, weight)
+        c_tl = F.linear(c_tx, target_weight)
+        # unreduced loss
+        c_loss = loss[start:end]
+        # Here we calculate the gradient of c_sx in place so we can save memory.
+        kl_div_kernel[(c_sx.shape[0],)](
+            logits=c_sl,
+            target_logits=c_tl,
+            loss=c_loss,
+            s_logits=c_sl.stride(-2),
+            s_loss=c_loss.stride(-1),
+            reduction=reduction,
+            N=N,
+            V=V,
+            BV=BV,
+            num_warps=32
+        )
+        # gradient of logits is computed in-place by the above triton kernel and is of shape: C x V
+        # thus dx[start: end] should be of shape: C x H
+        # additionally, since we are chunking the inputs, observe that the loss and gradients are calculated only
+        # on `n_non_ignore` tokens. However, the gradient of the input should be calculated for all tokens.
+        # Thus, we need an additional scaling factor of (n_non_ignore/total) to scale the gradients.
+        # [C, H]
+        dx[start:end] = torch.mm(c_sl, weight)
+        if weight is not None:
+            torch.addmm(input=dw, mat1=c_sl.t(), mat2=c_sx, out=dw)
+    loss = loss.sum()
+    return loss, dx, dw
+def fused_kl_div_backward(
+    do: torch.Tensor,
+    dx: torch.Tensor,
+    dw: torch.Tensor
+):
+    # If cross entropy is the last layer, do is 1.0. Skip the mul to save time
+    if torch.ne(do, torch.tensor(1.0, device=do.device)):
+        # We use a Triton kernel instead of a PyTorch operation because modifying inputs in-place
+        # for gradient storage and backward multiple times causes anomalies with PyTorch but not with Triton.
+        N, H = dx.shape
+        B = min(MAX_FUSED_SIZE, triton.next_power_of_2(H))
+        elementwise_mul_kernel[(triton.cdiv(N * H, B),)](
+            x=dx,
+            g=do,
+            N=N*H,
+            B=B,
+            num_warps=32,
+        )
+        # handle dw
+        if dw is not None:
+            V, H = dw.shape
+            elementwise_mul_kernel[(triton.cdiv(V * H, B),)](
+                x=dw,
+                g=do,
+                N=V*H,
+                B=B,
+                num_warps=32,
+            )
+    return dx, dw
+class FusedKLDivLossFunction(torch.autograd.Function):
+    @staticmethod
+    @input_guard
+    def forward(
+        ctx,
+        x: torch.Tensor,
+        target_x: torch.Tensor,
+        weight: torch.Tensor,
+        target_weight: torch.Tensor,
+        reduction: str
+    ):
+        loss, dx, dw = fused_kl_div_forward(
+            x=x,
+            target_x=target_x,
+            weight=weight,
+            target_weight=target_weight,
+            reduction=reduction
+        )
+        ctx.save_for_backward(dx, dw)
+        return loss
+    @staticmethod
+    @input_guard
+    def backward(ctx, do):
+        dx, dw = ctx.saved_tensors
+        dx, dw = fused_kl_div_backward(do, dx, dw)
+        return dx, None, dw, None, None
+def fused_kl_div_loss(
+    x: torch.Tensor,
+    target_x: torch.Tensor,
+    weight: torch.Tensor,
+    target_weight: torch.Tensor,
+    reduction: str = 'batchmean'
+) -> Tuple[torch.Tensor, torch.Tensor]:
+    """
+    Args:
+        x (torch.Tensor): [batch_size * seq_len, hidden_size]
+        target_x (torch.Tensor): [batch_size * seq_len, hidden_size]
+        weight (torch.Tensor): [vocab_size, hidden_size]
+            where `vocab_size` is the number of classes.
+        target_weight (torch.Tensor): [vocab_size, hidden_size]
+            where `vocab_size` is the number of classes.
+        reduction:
+            Specifies the reduction to apply to the output: 'batchmean'. Default: 'batchmean'.
+    Returns:
+        loss
+    """
+    return FusedKLDivLossFunction.apply(
+        x,
+        target_x,
+        weight,
+        target_weight,
+        reduction
+    )
+class FusedKLDivLoss(nn.Module):
+    def __init__(
+        self,
+        reduction: str = 'batchmean'
+    ):
+        """
+        Args:
+            reduction:
+                Specifies the reduction to apply to the output: 'batchmean'. Default: 'batchmean'.
+        """
+        super().__init__()
+        assert reduction in ['batchmean'], f"reduction: {reduction} is not supported"
+        self.reduction = reduction
+    def forward(
+        self,
+        x: torch.Tensor,
+        target_x: torch.Tensor,
+        weight: torch.Tensor,
+        target_weight: torch.Tensor
+    ):
+        """
+        Args:
+            x (torch.Tensor): [batch_size * seq_len, hidden_size]
+            target_x (torch.Tensor): [batch_size * seq_len, hidden_size]
+            weight (torch.Tensor): [vocab_size, hidden_size]
+                where `vocab_size` is the number of classes.
+            target_weight (torch.Tensor): [vocab_size, hidden_size]
+                where `vocab_size` is the number of classes.
+        Returns:
+            loss
+        """
+        loss = fused_kl_div_loss(
+            x=x,
+            target_x=target_x,
+            weight=weight,
+            target_weight=target_weight,
+            reduction=self.reduction
+        )
+        return loss

fla/modules/fused_linear_cross_entropy.py ADDED Viewed

	@@ -0,0 +1,570 @@

+# -*- coding: utf-8 -*-
+# Code adapted from
+# https://github.com/linkedin/Liger-Kernel/blob/main/src/liger_kernel/ops/fused_linear_cross_entropy.py
+from functools import partial
+from typing import Optional, Tuple
+import torch
+import torch.nn as nn
+import torch.nn.functional as F
+import triton
+import triton.language as tl
+from torch.distributed import DeviceMesh
+from torch.distributed.tensor import DTensor, Replicate, Shard, distribute_module
+from torch.distributed.tensor.parallel import ParallelStyle
+from fla.ops.utils import logsumexp_fwd
+from fla.ops.utils.op import exp
+from fla.utils import input_guard
+# The hard limit of TRITON_MAX_TENSOR_NUMEL is 1048576
+# https://github.com/triton-lang/triton/blob/ba42a5c68fd0505f8c42f4202d53be0f8d9a5fe0/python/triton/language/core.py#L19
+# However, setting limit as 65536 as in LayerNorm tutorial is faster because of less register spilling
+# The optimal maximum block size depends on your hardware, your kernel, and your dtype
+MAX_FUSED_SIZE = 65536 // 2
+@triton.jit
+def cross_entropy_kernel(
+    logits,
+    lse,
+    target,
+    loss,
+    total,
+    ignore_index,
+    label_smoothing: tl.constexpr,
+    logit_scale: tl.constexpr,
+    reduction: tl.constexpr,
+    V: tl.constexpr,
+    BV: tl.constexpr
+):
+    """
+    This kernel computes both cross entropy loss and the gradient of the input.
+    We only consider hard label + mean reduction for now.
+    Please refer to https://pytorch.org/docs/stable/generated/torch.nn.CrossEntropyLoss.html for the math.
+    Args:
+        logits:
+            Pointer to logits tensor.
+        lse:
+            Pointer to logsumexp tensor.
+        target: Pointer to target tensor.
+        loss:
+            Pointer to tensor to store the loss.
+        V (int):
+            The number of columns in the input tensor.
+        total (int):
+            The number of non-ignored classes.
+        ignore_index (int):
+            The index to ignore in the target.
+        label_smoothing (float):
+            The amount of smoothing when computing the loss, where 0.0 means no smoothing.
+        reduction (str):
+            The string for the reduction to apply
+        BV (int):
+            The block size for vocab.
+    """
+    # https://github.com/triton-lang/triton/issues/1058
+    # If B*T*V is too large, i_n * stride will overflow out of int32, so we convert to int64
+    i_n = tl.program_id(0).to(tl.int64)
+    NV = tl.cdiv(V, BV)
+    # 1. Load target first because if the target is ignore_index, we can return right away
+    b_y = tl.load(target + i_n)
+    # 2. locate the start index
+    logits += i_n * V
+    if b_y == ignore_index:
+        # set all x as 0
+        for i in range(0, V, BV):
+            o_v = i + tl.arange(0, BV)
+            tl.store(logits + o_v, 0.0, mask=o_v < V)
+        return
+    # Online softmax: 2 loads + 1 store (compared with 3 loads + 1 store for the safe softmax)
+    # Refer to Algorithm 3 in the paper: https://arxiv.org/pdf/1805.02867
+    # 3. [Online softmax] first pass: compute logsumexp
+    # we did this in anouter kernel
+    b_l = tl.load(logits + b_y) * logit_scale
+    b_lse = tl.load(lse + i_n)
+    # 4. Calculate the loss
+    # loss = lse - logits_l
+    b_loss = b_lse - b_l
+    # Label smoothing is a general case of normal cross entropy
+    # See the full derivation at https://github.com/linkedin/Liger-Kernel/pull/198#issue-2503665310
+    b_z = 0.0
+    eps = label_smoothing / V
+    # We need tl.debug_barrier() as mentioned in
+    # https://github.com/triton-lang/triton/blob/ba42a5c68fd0505f8c42f4202d53be0f8d9a5fe0/python/triton/ops/cross_entropy.py#L34
+    tl.debug_barrier()
+    # 5. [Online Softmax] Second pass: compute gradients
+    # For 'mean' reduction, gradients are normalized by number of non-ignored elements
+    # dx_y = (softmax(x_y) - 1) / N
+    # dx_i = softmax(x_i) / N, i != y
+    # For label smoothing:
+    # dx_i = (softmax(x_y) - label_smoothing / V) / N, i != y
+    # dx_y = (softmax(x_y) - label_smoothing / V - (1 - label_smoothing)) / N
+    #      = dx_i - (1 - label_smoothing) / N
+    for iv in range(0, NV):
+        o_v = iv * BV + tl.arange(0, BV)
+        b_logits = tl.load(logits + o_v, mask=o_v < V, other=float('-inf')) * logit_scale
+        if label_smoothing > 0:
+            # scale X beforehand to avoid overflow
+            b_z += tl.sum(tl.where(o_v < V, -eps * b_logits, 0.0))
+        b_p = (exp(b_logits - b_lse) - eps) * logit_scale
+        if reduction == "mean":
+            b_p = b_p / total
+        tl.store(logits + o_v, b_p, mask=o_v < V)
+        tl.debug_barrier()
+    # Orginal loss = H(q, p),  with label smoothing regularization = H(q', p) and (label_smoothing / V) = eps
+    # H(q', p) = (1 - label_smoothing) * H(q, p) + label_smoothing * H(u, p)
+    #          = (1 - label_smoothing) * H(q, p) + eps * sum(logsoftmax(x_i))
+    # By using m (global max of xi) and d (sum of e^(xi-m)), we can simplify as:
+    #          = (1 - label_smoothing) * H(q, p) + (-sum(x_i * eps) + label_smoothing * (m + logd))
+    # Refer to H(q', p) in section 7 of the paper:
+    # https://arxiv.org/pdf/1512.00567
+    # pytorch:
+    # https://github.com/pytorch/pytorch/blob/2981534f54d49fa3a9755c9b0855e7929c2527f0/aten/src/ATen/native/LossNLL.cpp#L516
+    # See full derivation at https://github.com/linkedin/Liger-Kernel/pull/198#issuecomment-2333753087
+    if label_smoothing > 0:
+        b_loss = b_loss * (1 - label_smoothing) + (b_z + label_smoothing * b_lse)
+    # 6. Specially handle the i==y case where `dx_y = (softmax(x_y) - (1 - label_smoothing) / N`
+    b_l = tl.load(logits + b_y)
+    # Normalize the loss by the number of non-ignored elements if reduction is "mean"
+    if reduction == 'mean':
+        b_loss = b_loss / total
+        b_l += (label_smoothing - 1) / total * logit_scale
+    else:
+        b_l += (label_smoothing - 1) * logit_scale
+    tl.store(loss + i_n, b_loss)
+    tl.store(logits + b_y, b_l)
+@triton.jit
+def elementwise_mul_kernel(
+    x,
+    g,
+    N: tl.constexpr,
+    B: tl.constexpr
+):
+    """
+    This function multiplies each element of the tensor pointed by x with the value pointed by g.
+    The multiplication is performed in-place on the tensor pointed by x.
+    Parameters:
+    x:
+        Pointer to the input tensor.
+    g:
+        Pointer to the gradient output value.
+    N (int):
+        The number of columns in the input tensor.
+    B (int):
+        The block size for Triton operations.
+    """
+    # Get the program ID and convert it to int64 to avoid overflow
+    i_x = tl.program_id(0).to(tl.int64)
+    o_x = i_x * B + tl.arange(0, B)
+    # Load the gradient output value
+    b_g = tl.load(g)
+    b_x = tl.load(x + o_x, mask=o_x < N)
+    tl.store(x + o_x, b_x * b_g, mask=o_x < N)
+def fused_linear_cross_entropy_forward(
+    x: torch.Tensor,
+    target: torch.LongTensor,
+    weight: torch.Tensor,
+    bias: torch.Tensor = None,
+    ignore_index: int = -100,
+    label_smoothing: float = 0.0,
+    logit_scale: float = 1.0,
+    num_chunks: int = 8,
+    reduction: str = "mean"
+):
+    device = x.device
+    # inputs have shape: [N, H]
+    # materialized activations will have shape: [N, V]
+    # the increase in memory = [N, V]
+    # reduction can be achieved by partitioning the number of tokens N into smaller chunks.
+    # ideally, we would like to achieve the same memory consumption as [N, H],
+    # so the expected chunk size should be:
+    # NC = ceil(V / H)
+    # C = ceil(N / NC)
+    # for ex: N = 4096*4, V = 32000, H = 4096 ==> NC = 8, C = ceil(N / NC) = 2048
+    N, H, V = *x.shape, weight.shape[0]
+    BV = min(MAX_FUSED_SIZE, triton.next_power_of_2(V))
+    # TODO: in real cases, we may need to limit the number of chunks NC to
+    # ensure the precisions of accumulated gradients
+    NC = min(num_chunks, triton.cdiv(V, H))
+    C = triton.next_power_of_2(triton.cdiv(N, NC))
+    NC = triton.cdiv(N, C)
+    # [N, H]
+    dx = torch.zeros_like(x, device=device)
+    # [V, H]
+    dw = torch.zeros_like(weight, device=device, dtype=torch.float) if weight is not None else None
+    # [V]
+    db = torch.zeros_like(bias, device=device, dtype=torch.float) if bias is not None else None
+    # [N]
+    loss = torch.zeros(N, device=device, dtype=torch.float)
+    total = target.ne(ignore_index).sum().item()
+    for ic in range(NC):
+        start, end = ic * C, min((ic + 1) * C, N)
+        # [C, N]
+        c_x = x[start:end]
+        # when doing matmul, use the original precision
+        # [C, V]
+        c_logits = F.linear(c_x, weight, bias)
+        c_target = target[start:end]
+        # [C]
+        # keep lse in fp32 to maintain precision
+        c_lse = logsumexp_fwd(c_logits, scale=logit_scale, dtype=torch.float)
+        # unreduced loss
+        c_loss = loss[start:end]
+        # Here we calculate the gradient of c_logits in place so we can save memory.
+        cross_entropy_kernel[(c_logits.shape[0],)](
+            logits=c_logits,
+            lse=c_lse,
+            target=c_target,
+            loss=c_loss,
+            total=total,
+            ignore_index=ignore_index,
+            label_smoothing=label_smoothing,
+            logit_scale=logit_scale,
+            reduction=reduction,
+            V=V,
+            BV=BV,
+            num_warps=32
+        )
+        # gradient of logits is computed in-place by the above triton kernel and is of shape: C x V
+        # thus dx should be of shape: C x H
+        dx[start:end] = torch.mm(c_logits, weight)
+        # keep dw in fp32 to maintain precision
+        if weight is not None:
+            dw += c_logits.t() @ c_x
+        if bias is not None:
+            torch.add(input=db, other=c_logits.sum(0), out=db)
+    loss = loss.sum()
+    if dw is not None:
+        dw = dw.to(weight)
+    if db is not None:
+        db = db.to(bias)
+    return loss, dx, dw, db
+def fused_linear_cross_entropy_backward(
+    do: torch.Tensor,
+    dx: torch.Tensor,
+    dw: torch.Tensor,
+    db: torch.Tensor
+):
+    # If cross entropy is the last layer, do is 1.0. Skip the mul to save time
+    if torch.ne(do, torch.tensor(1.0, device=do.device)):
+        # We use a Triton kernel instead of a PyTorch operation because modifying inputs in-place
+        # for gradient storage and backward multiple times causes anomalies with PyTorch but not with Triton.
+        N, H = dx.shape
+        B = min(MAX_FUSED_SIZE, triton.next_power_of_2(H))
+        elementwise_mul_kernel[(triton.cdiv(N * H, B),)](
+            x=dx,
+            g=do,
+            N=N*H,
+            B=B,
+            num_warps=32,
+        )
+        # handle dw
+        if dw is not None:
+            V, H = dw.shape
+            elementwise_mul_kernel[(triton.cdiv(V * H, B),)](
+                x=dw,
+                g=do,
+                N=V*H,
+                B=B,
+                num_warps=32,
+            )
+        if db is not None:
+            V = db.shape[0]
+            elementwise_mul_kernel[(triton.cdiv(V, B),)](
+                x=db,
+                g=do,
+                N=V,
+                B=B,
+                num_warps=32,
+            )
+    return dx, dw, db
+class FusedLinearCrossEntropyFunction(torch.autograd.Function):
+    @staticmethod
+    @input_guard
+    def forward(
+        ctx,
+        x: torch.Tensor,
+        target: torch.LongTensor,
+        weight: torch.Tensor,
+        bias: torch.Tensor = None,
+        ignore_index: int = -100,
+        label_smoothing: float = 0.0,
+        logit_scale: float = 1.0,
+        num_chunks: int = 8,
+        reduction: str = "mean"
+    ):
+        """
+        Fusing the last linear layer with cross-entropy loss
+            Reference: https://github.com/mgmalek/efficient_cross_entropy
+        Handle the forward and backward pass of the final linear layer via cross-entropy loss by avoiding
+        the materialization of the large logits tensor. Since Cross Entropy Loss is the last layer, we can
+        compute the gradient at the forward pass. By doing so, we don't have to store the x and target
+        for the backward pass.
+        x (torch.Tensor): [batch_size * seq_len, hidden_size]
+        target (torch.LongTensor): [batch_size * seq_len]
+            where each value is in [0, vocab_size).
+        weight (torch.Tensor): [vocab_size, hidden_size]
+            where `vocab_size` is the number of classes.
+        bias (Optional[torch.Tensor]): [vocab_size]
+            where `vocab_size` is the number of classes.
+        ignore_index:
+            the index to ignore in the target.
+        label_smoothing:
+            the amount of smoothing when computing the loss, where 0.0 means no smoothing.
+        logit_scale: float = 1.0,
+            A scaling factor applied to the logits. Default: 1.0
+        num_chunks: int
+            The number of chunks to split the input tensor into for processing.
+            This can help optimize memory usage and computation speed.
+            Default: 8
+        reduction:
+            Specifies the reduction to apply to the output: 'mean' | 'sum'.
+            'mean': the weighted mean of the output is taken,
+            'sum': the output will be summed.
+            Default: 'mean'.
+        """
+        loss, dx, dw, db = fused_linear_cross_entropy_forward(
+            x,
+            target,
+            weight,
+            bias,
+            ignore_index,
+            label_smoothing,
+            logit_scale,
+            num_chunks,
+            reduction
+        )
+        # downcast to dtype and store for backward
+        ctx.save_for_backward(
+            dx.detach(),
+            dw.detach() if weight is not None else None,
+            db.detach() if bias is not None else None,
+        )
+        return loss
+    @staticmethod
+    @input_guard
+    def backward(ctx, do):
+        dx, dw, db = ctx.saved_tensors
+        dx, dw, db = fused_linear_cross_entropy_backward(do, dx, dw, db)
+        return dx, None, dw, db, None, None, None, None, None
+def fused_linear_cross_entropy_loss(
+    x: torch.Tensor,
+    target: torch.LongTensor,
+    weight: torch.Tensor,
+    bias: torch.Tensor = None,
+    ignore_index: int = -100,
+    label_smoothing: float = 0.0,
+    logit_scale: float = 1.0,
+    num_chunks: int = 8,
+    reduction: str = "mean"
+) -> Tuple[torch.Tensor, torch.Tensor]:
+    """
+    Args:
+        x (torch.Tensor): [batch_size * seq_len, hidden_size]
+        target (torch.LongTensor): [batch_size * seq_len]
+            where each value is in [0, vocab_size).
+        weight (torch.Tensor): [vocab_size, hidden_size]
+            where `vocab_size` is the number of classes.
+        bias (Optional[torch.Tensor]): [vocab_size]
+            where `vocab_size` is the number of classes.
+        ignore_index: int.
+            If target == ignore_index, the loss is set to 0.0.
+        label_smoothing: float
+        logit_scale: float
+            A scaling factor applied to the logits. Default: 1.0
+        num_chunks: int
+            The number of chunks to split the input tensor into for processing.
+            This can help optimize memory usage and computation speed.
+            Default: 8
+        reduction:
+            Specifies the reduction to apply to the output: 'mean' | 'sum'.
+            'mean': the weighted mean of the output is taken,
+            'sum': the output will be summed.
+            Default: 'mean'.
+    Returns:
+        losses: [batch,], float
+    """
+    return FusedLinearCrossEntropyFunction.apply(
+        x,
+        target,
+        weight,
+        bias,
+        ignore_index,
+        label_smoothing,
+        logit_scale,
+        num_chunks,
+        reduction
+    )
+class FusedLinearCrossEntropyLoss(nn.Module):
+    def __init__(
+        self,
+        ignore_index: int = -100,
+        label_smoothing: float = 0.0,
+        logit_scale: float = 1.0,
+        num_chunks: int = 8,
+        reduction: str = "mean"
+    ):
+        """
+        Args:
+            ignore_index: int.
+                If target == ignore_index, the loss is set to 0.0.
+            label_smoothing: float
+            logit_scale: float
+                A scaling factor applied to the logits. Default: 1.0
+            num_chunks: int
+                The number of chunks to split the input tensor into for processing.
+                This can help optimize memory usage and computation speed.
+                Default: 8
+            reduction:
+                Specifies the reduction to apply to the output: 'mean' | 'sum'.
+                'mean': the weighted mean of the output is taken,
+                'sum': the output will be summed.
+                Default: 'mean'.
+        """
+        super().__init__()
+        assert reduction in ["mean", "sum"], f"reduction: {reduction} is not supported"
+        self.ignore_index = ignore_index
+        self.label_smoothing = label_smoothing
+        self.logit_scale = logit_scale
+        self.num_chunks = num_chunks
+        self.reduction = reduction
+    @torch.compiler.disable
+    def forward(
+        self,
+        x: torch.Tensor,
+        target: torch.LongTensor,
+        weight: torch.Tensor,
+        bias: Optional[torch.Tensor] = None
+    ):
+        """
+        Args:
+            x (torch.Tensor): [batch_size, seq_len, hidden_size]
+            target (torch.LongTensor): [batch_size, seq_len]
+                where each value is in [0, V).
+            weight (torch.Tensor): [vocab_size, hidden_size]
+                where `vocab_size` is the number of classes.
+            bias (Optional[torch.Tensor]): [vocab_size]
+                where `vocab_size` is the number of classes.
+        Returns:
+            loss
+        """
+        loss = fused_linear_cross_entropy_loss(
+            x.view(-1, x.shape[-1]),
+            target.view(-1),
+            weight=weight,
+            bias=bias,
+            ignore_index=self.ignore_index,
+            label_smoothing=self.label_smoothing,
+            logit_scale=self.logit_scale,
+            num_chunks=self.num_chunks,
+            reduction=self.reduction
+        )
+        return loss
+class LinearLossParallel(ParallelStyle):
+    def __init__(
+        self,
+        *,
+        sequence_dim: int = 1,
+        use_local_output: bool = False,
+    ):
+        super().__init__()
+        self.sequence_sharding = (Shard(sequence_dim),)
+        self.use_local_output = use_local_output
+    @staticmethod
+    def _prepare_input_fn(sequence_sharding, mod, inputs, device_mesh):
+        x, target, weight, bias = inputs
+        if not isinstance(x, DTensor):
+            # assume the input passed in already sharded on the sequence dim and create the DTensor
+            x = DTensor.from_local(x, device_mesh, sequence_sharding)
+        if x.placements != sequence_sharding:
+            x = x.redistribute(placements=sequence_sharding, async_op=True)
+        if not isinstance(target, DTensor):
+            target = DTensor.from_local(target, device_mesh, [Replicate()])
+        if target.placements != sequence_sharding:
+            target = target.redistribute(placements=sequence_sharding, async_op=True)
+        if not isinstance(weight, DTensor):
+            weight = DTensor.from_local(weight, device_mesh, [Replicate()])
+        if weight.placements != [Replicate()]:
+            # we replicate the weight/bias in FLCE
+            weight = weight.redistribute(placements=[Replicate()], async_op=True)
+        if bias is not None and not isinstance(bias, DTensor):
+            bias = DTensor.from_local(bias, device_mesh, [Replicate()])
+        if bias is not None and bias.placements != [Replicate()]:
+            bias = bias.redistribute(placements=[Replicate()], async_op=True)
+        return x.to_local(), target.to_local(), weight.to_local(), bias.to_local() if bias is not None else bias
+    @staticmethod
+    def _prepare_output_fn(use_local_output, mod, outputs, device_mesh):
+        return outputs.to_local() if use_local_output else outputs
+    def _apply(self, module: nn.Module, device_mesh: DeviceMesh) -> nn.Module:
+        return distribute_module(
+            module,
+            device_mesh,
+            partition_fn=None,
+            input_fn=partial(self._prepare_input_fn, self.sequence_sharding),
+            output_fn=partial(self._prepare_output_fn, self.use_local_output)
+        )

fla/modules/layernorm.py ADDED Viewed

	@@ -0,0 +1,1196 @@

+# -*- coding: utf-8 -*-
+# Copyright (c) 2023, Tri Dao.
+# https://github.com/state-spaces/mamba/blob/fb7b5310fa865dbd62aa059b1e26f2b431363e2a/mamba_ssm/ops/triton/layernorm.py
+# Implement residual + layer_norm / rms_norm.
+# Based on the Triton LayerNorm tutorial: https://triton-lang.org/main/getting-started/tutorials/05-layer-norm.html
+# For the backward pass, we keep weight_grad and bias_grad in registers and accumulate.
+# This is faster for dimensions up to 8k, but after that it's much slower due to register spilling.
+# The models we train have hidden dim up to 8k anyway (e.g. Llama 70B), so this is fine.
+from __future__ import annotations
+from functools import partial
+import torch
+import torch.nn as nn
+import torch.nn.functional as F
+import triton
+import triton.language as tl
+from einops import rearrange
+from torch.distributed import DeviceMesh
+from torch.distributed.tensor import DTensor, Replicate, Shard, distribute_module
+from torch.distributed.tensor.parallel import ParallelStyle
+from fla.utils import get_multiprocessor_count, input_guard
+def layer_norm_ref(
+    x: torch.Tensor,
+    weight: torch.Tensor,
+    bias: torch.Tensor,
+    residual: torch.Tensor = None,
+    eps: float = 1e-5,
+    prenorm: bool = False,
+    upcast: bool = False
+):
+    dtype = x.dtype
+    if upcast:
+        weight = weight.float()
+        bias = bias.float() if bias is not None else None
+    if upcast:
+        x = x.float()
+        residual = residual.float() if residual is not None else residual
+    if residual is not None:
+        x = (x + residual).to(x.dtype)
+    out = F.layer_norm(x.to(weight.dtype), x.shape[-1:], weight=weight, bias=bias, eps=eps).to(
+        dtype
+    )
+    return out if not prenorm else (out, x)
+def rms_norm_ref(
+    x: torch.Tensor,
+    weight: torch.Tensor,
+    bias: torch.Tensor,
+    residual: torch.Tensor = None,
+    eps: float = 1e-5,
+    prenorm: bool = False,
+    upcast: bool = False
+):
+    dtype = x.dtype
+    if upcast:
+        weight = weight.float()
+        bias = bias.float() if bias is not None else None
+    if upcast:
+        x = x.float()
+        residual = residual.float() if residual is not None else residual
+    if residual is not None:
+        x = (x + residual).to(x.dtype)
+    rstd = 1 / torch.sqrt((x.square()).mean(dim=-1, keepdim=True) + eps)
+    out = (x * rstd * weight) + bias if bias is not None else (x * rstd * weight)
+    out = out.to(dtype)
+    return out if not prenorm else (out, x)
+def group_norm_ref(
+    x: torch.Tensor,
+    weight: torch.Tensor,
+    bias: torch.Tensor,
+    num_groups: int,
+    residual: torch.Tensor = None,
+    eps: float = 1e-5,
+    is_rms_norm: bool = False,
+    prenorm: bool = False,
+    upcast: bool = False
+):
+    dtype = x.dtype
+    if upcast:
+        weight = weight.float()
+        bias = bias.float() if bias is not None else None
+    if upcast:
+        x = x.float()
+        residual = residual.float() if residual is not None else residual
+    if residual is not None:
+        x = (x + residual).to(x.dtype)
+    residual = x
+    x, weight = [
+        rearrange(data, "... (g d) -> ... g d", g=num_groups) for data in (x, weight)
+    ]
+    if bias is not None:
+        bias = rearrange(bias, '... (g d) -> ... g d', g=num_groups)
+    if not is_rms_norm:
+        mean = x.mean(dim=-1, keepdim=True)
+        x = x - mean
+    rstd = 1 / torch.sqrt((x.square()).mean(dim=-1, keepdim=True) + eps)
+    out = (x * rstd * weight) + bias if bias is not None else (x * rstd * weight)
+    out = rearrange(out, "... g d -> ... (g d)")
+    out = out.to(dtype)
+    return out if not prenorm else (out, residual)
+class GroupNormRef(nn.Module):
+    def __init__(
+        self,
+        num_groups: int,
+        hidden_size: int,
+        elementwise_affine: bool = True,
+        bias: bool = False,
+        eps: float = 1e-5,
+        is_rms_norm: bool = False
+    ) -> GroupNormRef:
+        super().__init__()
+        if hidden_size % num_groups != 0:
+            raise ValueError('num_channels must be divisible by num_groups')
+        self.num_groups = num_groups
+        self.hidden_size = hidden_size
+        self.elementwise_affine = elementwise_affine
+        self.eps = eps
+        self.is_rms_norm = is_rms_norm
+        self.register_parameter("weight", None)
+        self.register_parameter("bias", None)
+        if elementwise_affine:
+            self.weight = nn.Parameter(torch.empty(hidden_size))
+            if bias:
+                self.bias = nn.Parameter(torch.empty(hidden_size))
+        self.reset_parameters()
+    def reset_parameters(self):
+        if self.elementwise_affine:
+            nn.init.ones_(self.weight)
+            if self.bias is not None:
+                nn.init.zeros_(self.bias)
+    def __repr__(self) -> str:
+        s = f"{self.__class__.__name__}({self.num_groups}, {self.hidden_size}"
+        if not self.elementwise_affine:
+            s += f", elementwise_affine={self.elementwise_affine}"
+        if self.is_rms_norm:
+            s += f", is_rms_norm={self.is_rms_norm}"
+        s += f", eps={self.eps}"
+        s += ")"
+        return s
+    def forward(self, x, residual=None, prenorm=False):
+        return group_norm_ref(
+            x,
+            self.weight,
+            self.bias,
+            num_groups=self.num_groups,
+            residual=residual,
+            eps=self.eps,
+            is_rms_norm=self.is_rms_norm,
+            prenorm=prenorm,
+            upcast=True
+        )
+@triton.autotune(
+    configs=[
+        triton.Config({}, num_warps=num_warps, num_stages=num_stages)
+        for num_warps in [1, 2, 4, 8, 16, 32]
+        for num_stages in [2, 3, 4]
+    ],
+    key=["N", "HAS_RESIDUAL", "STORE_RESIDUAL_OUT", "IS_RMS_NORM", "HAS_BIAS"],
+)
+@triton.jit
+def layer_norm_fwd_kernel(
+    X,  # pointer to the input
+    Y,  # pointer to the output
+    W,  # pointer to the weights
+    B,  # pointer to the biases
+    RESIDUAL,  # pointer to the residual
+    RESIDUAL_OUT,  # pointer to the residual
+    Mean,  # pointer to the mean
+    Rstd,  # pointer to the 1/std
+    N,  # number of columns in X
+    G,  # number of groups
+    eps,  # epsilon to avoid division by zero
+    IS_RMS_NORM: tl.constexpr,
+    BLOCK_N: tl.constexpr,
+    HAS_RESIDUAL: tl.constexpr,
+    STORE_RESIDUAL_OUT: tl.constexpr,
+    HAS_WEIGHT: tl.constexpr,
+    HAS_BIAS: tl.constexpr
+):
+    # Map the program id to the row of X and Y it should compute.
+    row = tl.program_id(0)
+    group = row % G
+    X += row * N
+    Y += row * N
+    if HAS_RESIDUAL:
+        RESIDUAL += row * N
+    if STORE_RESIDUAL_OUT:
+        RESIDUAL_OUT += row * N
+    # Compute mean and variance
+    cols = tl.arange(0, BLOCK_N)
+    x = tl.load(X + cols, mask=cols < N, other=0.0).to(tl.float32)
+    if HAS_RESIDUAL:
+        residual = tl.load(RESIDUAL + cols, mask=cols < N, other=0.0).to(tl.float32)
+        x += residual
+    if STORE_RESIDUAL_OUT:
+        tl.store(RESIDUAL_OUT + cols, x, mask=cols < N)
+    if not IS_RMS_NORM:
+        mean = tl.sum(x, axis=0) / N
+        tl.store(Mean + row, mean)
+        xbar = tl.where(cols < N, x - mean, 0.0)
+        var = tl.sum(xbar * xbar, axis=0) / N
+    else:
+        xbar = tl.where(cols < N, x, 0.0)
+        var = tl.sum(xbar * xbar, axis=0) / N
+    rstd = 1 / tl.sqrt(var + eps)
+    tl.store(Rstd + row, rstd)
+    # Normalize and apply linear transformation
+    mask = cols < N
+    if HAS_WEIGHT:
+        w = tl.load(W + group * N + cols, mask=mask).to(tl.float32)
+    if HAS_BIAS:
+        b = tl.load(B + group * N + cols, mask=mask).to(tl.float32)
+    x_hat = (x - mean) * rstd if not IS_RMS_NORM else x * rstd
+    y = tl.fma(x_hat, w, b) if HAS_WEIGHT and HAS_BIAS else \
+        x_hat * w if HAS_WEIGHT else \
+        x_hat + b if HAS_BIAS else x_hat
+    # Write output
+    y = tl.cast(y, dtype=Y.dtype.element_ty, fp_downcast_rounding="rtne")
+    tl.store(Y + cols, y, mask=mask)
+def layer_norm_fwd(
+    x: torch.Tensor,
+    weight: torch.Tensor,
+    bias: torch.Tensor,
+    eps: float,
+    residual: torch.Tensor = None,
+    out_dtype: torch.dtype = None,
+    residual_dtype: torch.dtype = None,
+    is_rms_norm: bool = False,
+    num_groups: int = 1
+):
+    if residual is not None:
+        residual_dtype = residual.dtype
+    M, N, G = *x.shape, num_groups
+    if residual is not None:
+        assert residual.shape == (M, N)
+    if weight is not None:
+        assert weight.shape == (G * N,)
+    if bias is not None:
+        assert bias.shape == (G * N,)
+    # allocate output
+    y = torch.empty_like(x, dtype=x.dtype if out_dtype is None else out_dtype)
+    if residual is not None or (residual_dtype is not None and residual_dtype != x.dtype):
+        residual_out = torch.empty(M, N, device=x.device, dtype=residual_dtype)
+    else:
+        residual_out = None
+    mean = torch.empty((M,), dtype=torch.float32, device=x.device) if not is_rms_norm else None
+    rstd = torch.empty((M,), dtype=torch.float32, device=x.device)
+    # Less than 64KB per feature: enqueue fused kernel
+    MAX_FUSED_SIZE = 65536 // x.element_size()
+    BLOCK_N = min(MAX_FUSED_SIZE, triton.next_power_of_2(N))
+    if N > BLOCK_N:
+        raise RuntimeError("This layer norm doesn't support feature dim >= 64KB.")
+    # heuristics for number of warps
+    layer_norm_fwd_kernel[(M,)](
+        x,
+        y,
+        weight,
+        bias,
+        residual,
+        residual_out,
+        mean,
+        rstd,
+        N,
+        G,
+        eps,
+        is_rms_norm,
+        BLOCK_N,
+        residual is not None,
+        residual_out is not None,
+        weight is not None,
+        bias is not None,
+    )
+    # residual_out is None if residual is None and residual_dtype == input_dtype
+    return y, mean, rstd, residual_out if residual_out is not None else x
+@triton.heuristics({
+    "RECOMPUTE_OUTPUT": lambda args: args["Y"] is not None
+})
+@triton.autotune(
+    configs=[
+        triton.Config({}, num_warps=num_warps, num_stages=num_stages)
+        for num_warps in [1, 2, 4, 8, 16, 32]
+        for num_stages in [2, 3, 4]
+    ],
+    key=["N", "HAS_DRESIDUAL", "STORE_DRESIDUAL", "IS_RMS_NORM", "HAS_BIAS"],
+)
+@triton.jit
+def layer_norm_bwd_kernel(
+    X,  # pointer to the input
+    W,  # pointer to the weights
+    B,  # pointer to the biases
+    Y,  # pointer to the output to be recomputed
+    DY,  # pointer to the output gradient
+    DX,  # pointer to the input gradient
+    DW,  # pointer to the partial sum of weights gradient
+    DB,  # pointer to the partial sum of biases gradient
+    DRESIDUAL,
+    DRESIDUAL_IN,
+    Mean,  # pointer to the mean
+    Rstd,  # pointer to the 1/std
+    M,  # number of rows in X
+    N,  # number of columns in X
+    G,  # number of groups
+    rows_per_program,
+    programs_per_group,
+    IS_RMS_NORM: tl.constexpr,
+    BLOCK_N: tl.constexpr,
+    HAS_DRESIDUAL: tl.constexpr,
+    STORE_DRESIDUAL: tl.constexpr,
+    HAS_WEIGHT: tl.constexpr,
+    HAS_BIAS: tl.constexpr,
+    RECOMPUTE_OUTPUT: tl.constexpr,
+):
+    row_block_id = tl.program_id(0)
+    group_id, program_id_in_group = row_block_id // programs_per_group, row_block_id % programs_per_group
+    row_start = group_id + program_id_in_group * G * rows_per_program
+    row_end = min(row_start + G * rows_per_program, M)
+    cols = tl.arange(0, BLOCK_N)
+    mask = cols < N
+    if HAS_WEIGHT:
+        w = tl.load(W + group_id * N + cols, mask=mask).to(tl.float32)
+        dw = tl.zeros((BLOCK_N,), dtype=tl.float32)
+    if RECOMPUTE_OUTPUT and HAS_BIAS:
+        b = tl.load(B + group_id * N + cols, mask=mask, other=0.0).to(tl.float32)
+    if HAS_BIAS:
+        db = tl.zeros((BLOCK_N,), dtype=tl.float32)
+    for row in range(row_start, row_end, G):
+        # Load data to SRAM
+        x = tl.load(X + row * N + cols, mask=mask, other=0).to(tl.float32)
+        dy = tl.load(DY + row * N + cols, mask=mask, other=0).to(tl.float32)
+        if not IS_RMS_NORM:
+            mean = tl.load(Mean + row)
+        rstd = tl.load(Rstd + row)
+        # Compute dx
+        xhat = (x - mean) * rstd if not IS_RMS_NORM else x * rstd
+        xhat = tl.where(mask, xhat, 0.0)
+        if RECOMPUTE_OUTPUT:
+            y = xhat * w if HAS_WEIGHT else xhat
+            if HAS_BIAS:
+                y = y + b
+            tl.store(Y + row * N + cols, y, mask=mask)
+        wdy = dy
+        if HAS_WEIGHT:
+            wdy = dy * w
+            dw += dy * xhat
+        if HAS_BIAS:
+            db += dy
+        if not IS_RMS_NORM:
+            c1 = tl.sum(xhat * wdy, axis=0) / N
+            c2 = tl.sum(wdy, axis=0) / N
+            dx = (wdy - (xhat * c1 + c2)) * rstd
+        else:
+            c1 = tl.sum(xhat * wdy, axis=0) / N
+            dx = (wdy - xhat * c1) * rstd
+        if HAS_DRESIDUAL:
+            dres = tl.load(DRESIDUAL + row * N + cols, mask=mask, other=0).to(tl.float32)
+            dx += dres
+        # Write dx
+        dx = tl.cast(dx, dtype=DX.dtype.element_ty, fp_downcast_rounding="rtne")
+        if STORE_DRESIDUAL:
+            tl.store(DRESIDUAL_IN + row * N + cols, dx, mask=mask)
+        tl.store(DX + row * N + cols, dx, mask=mask)
+    if HAS_WEIGHT:
+        tl.store(DW + row_block_id * N + cols, dw, mask=mask)
+    if HAS_BIAS:
+        tl.store(DB + row_block_id * N + cols, db, mask=mask)
+def layer_norm_bwd(
+    dy: torch.Tensor,
+    x: torch.Tensor,
+    weight: torch.Tensor,
+    bias: torch.Tensor,
+    eps: float,
+    mean: torch.Tensor,
+    rstd: torch.Tensor,
+    dresidual: torch.Tensor = None,
+    has_residual: bool = False,
+    is_rms_norm: bool = False,
+    x_dtype: torch.dtype = None,
+    recompute_output: bool = False,
+    num_groups: int = 1
+):
+    M, N, G = *x.shape, num_groups
+    assert dy.shape == (M, N)
+    if dresidual is not None:
+        assert dresidual.shape == (M, N)
+    if weight is not None:
+        assert weight.shape == (G * N,)
+    if bias is not None:
+        assert bias.shape == (G * N,)
+    # allocate output
+    dx = torch.empty_like(x) if x_dtype is None else torch.empty(M, N, dtype=x_dtype, device=x.device)
+    dresidual_in = torch.empty_like(x) if has_residual and dx.dtype != x.dtype else None
+    y = torch.empty(M, N, dtype=dy.dtype, device=dy.device) if recompute_output else None
+    # Less than 64KB per feature: enqueue fused kernel
+    MAX_FUSED_SIZE = 65536 // x.element_size()
+    BLOCK_N = min(MAX_FUSED_SIZE, triton.next_power_of_2(N))
+    if N > BLOCK_N:
+        raise RuntimeError("This layer norm doesn't support feature dim >= 64KB.")
+    # each program handles one group only
+    S = triton.cdiv(get_multiprocessor_count(x.device.index), G) * G
+    dw = torch.empty((S, N), dtype=torch.float32, device=weight.device) if weight is not None else None
+    db = torch.empty((S, N), dtype=torch.float32, device=bias.device) if bias is not None else None
+    rows_per_program = triton.cdiv(M, S)
+    programs_per_group = S // G
+    grid = (S,)
+    layer_norm_bwd_kernel[grid](
+        x,
+        weight,
+        bias,
+        y,
+        dy,
+        dx,
+        dw,
+        db,
+        dresidual,
+        dresidual_in,
+        mean,
+        rstd,
+        M,
+        N,
+        G,
+        rows_per_program,
+        programs_per_group,
+        is_rms_norm,
+        BLOCK_N,
+        dresidual is not None,
+        dresidual_in is not None,
+        weight is not None,
+        bias is not None,
+    )
+    dw = dw.view(G, -1, N).sum(1).to(weight).view_as(weight) if weight is not None else None
+    db = db.view(G, -1, N).sum(1).to(bias).view_as(bias) if bias is not None else None
+    # Don't need to compute dresidual_in separately in this case
+    if has_residual and dx.dtype == x.dtype:
+        dresidual_in = dx
+    return (dx, dw, db, dresidual_in) if not recompute_output else (dx, dw, db, dresidual_in, y)
+class LayerNormFunction(torch.autograd.Function):
+    @staticmethod
+    @input_guard
+    def forward(
+        ctx,
+        x,
+        weight,
+        bias,
+        residual=None,
+        eps=1e-5,
+        prenorm=False,
+        residual_in_fp32=False,
+        is_rms_norm=False,
+        num_groups=1
+    ):
+        x_shape_og = x.shape
+        if x.shape[-1] % num_groups != 0:
+            raise ValueError('num_channels must be divisible by num_groups')
+        # reshape input data into 2D tensor
+        x = x.reshape(-1, (x.shape[-1] // num_groups))
+        if residual is not None:
+            assert residual.shape == x_shape_og
+            residual = residual.reshape_as(x)
+        residual_dtype = (
+            residual.dtype
+            if residual is not None
+            else (torch.float32 if residual_in_fp32 else None)
+        )
+        y, mean, rstd, residual_out = layer_norm_fwd(
+            x,
+            weight,
+            bias,
+            eps,
+            residual,
+            residual_dtype=residual_dtype,
+            is_rms_norm=is_rms_norm,
+            num_groups=num_groups
+        )
+        ctx.save_for_backward(residual_out, weight, bias, mean, rstd)
+        ctx.x_shape_og = x_shape_og
+        ctx.eps = eps
+        ctx.is_rms_norm = is_rms_norm
+        ctx.num_groups = num_groups
+        ctx.has_residual = residual is not None
+        ctx.prenorm = prenorm
+        ctx.x_dtype = x.dtype
+        y = y.reshape(x_shape_og)
+        return y if not prenorm else (y, residual_out.reshape(x_shape_og))
+    @staticmethod
+    @input_guard
+    def backward(ctx, dy, *args):
+        x, weight, bias, mean, rstd = ctx.saved_tensors
+        dy = dy.reshape(-1, (dy.shape[-1] // ctx.num_groups))
+        assert dy.shape == x.shape
+        if ctx.prenorm:
+            dresidual = args[0]
+            dresidual = dresidual.reshape(-1, x.shape[-1])
+            assert dresidual.shape == x.shape
+        else:
+            dresidual = None
+        dx, dw, db, dresidual_in = layer_norm_bwd(
+            dy,
+            x,
+            weight,
+            bias,
+            ctx.eps,
+            mean,
+            rstd,
+            dresidual,
+            ctx.has_residual,
+            ctx.is_rms_norm,
+            x_dtype=ctx.x_dtype,
+            num_groups=ctx.num_groups
+        )
+        return (
+            dx.reshape(ctx.x_shape_og),
+            dw,
+            db,
+            dresidual_in.reshape(ctx.x_shape_og) if ctx.has_residual else None,
+            None,
+            None,
+            None,
+            None,
+            None
+        )
+def layer_norm(
+    x: torch.Tensor,
+    weight: torch.Tensor,
+    bias: torch.Tensor,
+    residual: torch.Tensor = None,
+    eps: float = 1e-5,
+    prenorm: bool = False,
+    residual_in_fp32: bool = False,
+    is_rms_norm: bool = False
+):
+    return LayerNormFunction.apply(
+        x,
+        weight,
+        bias,
+        residual,
+        eps,
+        prenorm,
+        residual_in_fp32,
+        is_rms_norm
+    )
+def group_norm(
+    x: torch.Tensor,
+    weight: torch.Tensor,
+    bias: torch.Tensor,
+    residual: torch.Tensor = None,
+    eps: float = 1e-5,
+    prenorm: bool = False,
+    residual_in_fp32: bool = False,
+    is_rms_norm: bool = False,
+    num_groups: int = 1
+):
+    return LayerNormFunction.apply(
+        x,
+        weight,
+        bias,
+        residual,
+        eps,
+        prenorm,
+        residual_in_fp32,
+        is_rms_norm,
+        num_groups
+    )
+def rms_norm(
+    x: torch.Tensor,
+    weight: torch.Tensor,
+    bias: torch.Tensor,
+    residual: torch.Tensor = None,
+    eps: float = 1e-5,
+    prenorm: bool = False,
+    residual_in_fp32: bool = False
+):
+    return LayerNormFunction.apply(
+        x,
+        weight,
+        bias,
+        residual,
+        eps,
+        prenorm,
+        residual_in_fp32,
+        True
+    )
+def layer_norm_linear(
+    x: torch.Tensor,
+    norm_weight: torch.Tensor,
+    norm_bias: torch.Tensor,
+    linear_weight: torch.Tensor,
+    linear_bias: torch.Tensor,
+    residual: torch.Tensor = None,
+    eps: float = 1e-5,
+    prenorm: bool = False,
+    residual_in_fp32: bool = False,
+    is_rms_norm: bool = False,
+    num_groups: int = 1
+):
+    return LayerNormLinearFunction.apply(
+        x,
+        norm_weight,
+        norm_bias,
+        linear_weight,
+        linear_bias,
+        residual,
+        eps,
+        prenorm,
+        residual_in_fp32,
+        is_rms_norm,
+        num_groups
+    )
+def rms_norm_linear(
+    x: torch.Tensor,
+    norm_weight: torch.Tensor,
+    norm_bias: torch.Tensor,
+    linear_weight: torch.Tensor,
+    linear_bias: torch.Tensor,
+    residual: torch.Tensor = None,
+    eps: float = 1e-5,
+    prenorm: bool = False,
+    residual_in_fp32: bool = False
+):
+    return layer_norm_linear(
+        x=x,
+        norm_weight=norm_weight,
+        norm_bias=norm_bias,
+        linear_weight=linear_weight,
+        linear_bias=linear_bias,
+        residual=residual,
+        eps=eps,
+        prenorm=prenorm,
+        residual_in_fp32=residual_in_fp32,
+        is_rms_norm=True
+    )
+def group_norm_linear(
+    x: torch.Tensor,
+    norm_weight: torch.Tensor,
+    norm_bias: torch.Tensor,
+    linear_weight: torch.Tensor,
+    linear_bias: torch.Tensor,
+    residual: torch.Tensor = None,
+    eps: float = 1e-5,
+    prenorm: bool = False,
+    residual_in_fp32: bool = False,
+    is_rms_norm: bool = False,
+    num_groups: int = 1
+):
+    return layer_norm_linear(
+        x=x,
+        norm_weight=norm_weight,
+        norm_bias=norm_bias,
+        linear_weight=linear_weight,
+        linear_bias=linear_bias,
+        residual=residual,
+        eps=eps,
+        prenorm=prenorm,
+        residual_in_fp32=residual_in_fp32,
+        is_rms_norm=is_rms_norm,
+        num_groups=num_groups
+    )
+class LayerNorm(nn.Module):
+    def __init__(
+        self,
+        hidden_size: int,
+        elementwise_affine: bool = True,
+        bias: bool = False,
+        eps: float = 1e-5
+    ) -> LayerNorm:
+        super().__init__()
+        self.hidden_size = hidden_size
+        self.elementwise_affine = elementwise_affine
+        self.eps = eps
+        self.register_parameter("weight", None)
+        self.register_parameter("bias", None)
+        if elementwise_affine:
+            self.weight = nn.Parameter(torch.empty(hidden_size))
+            if bias:
+                self.bias = nn.Parameter(torch.empty(hidden_size))
+        self.reset_parameters()
+    def reset_parameters(self):
+        if self.elementwise_affine:
+            nn.init.ones_(self.weight)
+            if self.bias is not None:
+                nn.init.zeros_(self.bias)
+    def __repr__(self) -> str:
+        s = f"{self.__class__.__name__}({self.hidden_size}"
+        if not self.elementwise_affine:
+            s += f", elementwise_affine={self.elementwise_affine}"
+        s += f", eps={self.eps}"
+        s += ")"
+        return s
+    def forward(self, x, residual=None, prenorm=False, residual_in_fp32=False):
+        return layer_norm(
+            x,
+            self.weight,
+            self.bias,
+            residual=residual,
+            eps=self.eps,
+            prenorm=prenorm,
+            residual_in_fp32=residual_in_fp32
+        )
+class GroupNorm(nn.Module):
+    def __init__(
+        self,
+        num_groups: int,
+        hidden_size: int,
+        elementwise_affine: bool = True,
+        bias: bool = False,
+        eps: float = 1e-5,
+        is_rms_norm: bool = False
+    ) -> GroupNorm:
+        super().__init__()
+        if hidden_size % num_groups != 0:
+            raise ValueError('num_channels must be divisible by num_groups')
+        self.num_groups = num_groups
+        self.hidden_size = hidden_size
+        self.elementwise_affine = elementwise_affine
+        self.eps = eps
+        self.is_rms_norm = is_rms_norm
+        self.register_parameter("weight", None)
+        self.register_parameter("bias", None)
+        if elementwise_affine:
+            self.weight = nn.Parameter(torch.empty(hidden_size))
+            if bias:
+                self.bias = nn.Parameter(torch.empty(hidden_size))
+        self.reset_parameters()
+    def reset_parameters(self):
+        if self.elementwise_affine:
+            nn.init.ones_(self.weight)
+            if self.bias is not None:
+                nn.init.zeros_(self.bias)
+    def __repr__(self) -> str:
+        s = f"{self.__class__.__name__}({self.num_groups}, {self.hidden_size}"
+        if not self.elementwise_affine:
+            s += f", elementwise_affine={self.elementwise_affine}"
+        if self.is_rms_norm:
+            s += f", is_rms_norm={self.is_rms_norm}"
+        s += f", eps={self.eps}"
+        s += ")"
+        return s
+    def forward(self, x, residual=None, prenorm=False, residual_in_fp32=False):
+        return group_norm(
+            x,
+            self.weight,
+            self.bias,
+            residual=residual,
+            eps=self.eps,
+            prenorm=prenorm,
+            residual_in_fp32=residual_in_fp32,
+            is_rms_norm=self.is_rms_norm,
+            num_groups=self.num_groups
+        )
+class RMSNorm(nn.Module):
+    def __init__(
+        self,
+        hidden_size: int,
+        elementwise_affine: bool = True,
+        bias: bool = False,
+        eps: float = 1e-5
+    ) -> RMSNorm:
+        super().__init__()
+        self.hidden_size = hidden_size
+        self.elementwise_affine = elementwise_affine
+        self.eps = eps
+        self.register_parameter("weight", None)
+        self.register_parameter("bias", None)
+        if elementwise_affine:
+            self.weight = nn.Parameter(torch.empty(hidden_size))
+            if bias:
+                self.bias = nn.Parameter(torch.empty(hidden_size))
+        self.reset_parameters()
+    def reset_parameters(self):
+        if self.elementwise_affine:
+            nn.init.ones_(self.weight)
+            if self.bias is not None:
+                nn.init.zeros_(self.bias)
+    def __repr__(self) -> str:
+        s = f"{self.__class__.__name__}({self.hidden_size}"
+        if not self.elementwise_affine:
+            s += f", elementwise_affine={self.elementwise_affine}"
+        s += f", eps={self.eps}"
+        s += ")"
+        return s
+    def forward(self, x, residual=None, prenorm=False, residual_in_fp32=False):
+        return rms_norm(
+            x,
+            self.weight,
+            self.bias,
+            residual=residual,
+            eps=self.eps,
+            prenorm=prenorm,
+            residual_in_fp32=residual_in_fp32,
+        )
+class LayerNormLinearFunction(torch.autograd.Function):
+    @staticmethod
+    @input_guard
+    def forward(
+        ctx,
+        x,
+        norm_weight,
+        norm_bias,
+        linear_weight,
+        linear_bias,
+        residual=None,
+        eps=1e-5,
+        prenorm=False,
+        residual_in_fp32=False,
+        is_rms_norm=False,
+        num_groups=1
+    ):
+        x_shape_og = x.shape
+        if x.shape[-1] % num_groups != 0:
+            raise ValueError('num_channels must be divisible by num_groups')
+        # reshape input data into 2D tensor
+        x = x.reshape(-1, (x.shape[-1] // num_groups))
+        if residual is not None:
+            assert residual.shape == x_shape_og
+            residual = residual.reshape_as(x)
+        residual_dtype = (
+            residual.dtype
+            if residual is not None
+            else (torch.float32 if residual_in_fp32 else None)
+        )
+        y, mean, rstd, residual_out = layer_norm_fwd(
+            x,
+            norm_weight,
+            norm_bias,
+            eps,
+            residual,
+            out_dtype=None if not torch.is_autocast_enabled() else torch.get_autocast_gpu_dtype(),
+            residual_dtype=residual_dtype,
+            is_rms_norm=is_rms_norm,
+            num_groups=num_groups
+        )
+        y = y.reshape(x_shape_og)
+        dtype = torch.get_autocast_gpu_dtype() if torch.is_autocast_enabled() else y.dtype
+        linear_weight = linear_weight.to(dtype)
+        linear_bias = linear_bias.to(dtype) if linear_bias is not None else None
+        out = F.linear(y.to(linear_weight.dtype), linear_weight, linear_bias)
+        # We don't store y, will be recomputed in the backward pass to save memory
+        ctx.save_for_backward(residual_out, norm_weight, norm_bias, linear_weight, mean, rstd)
+        ctx.x_shape_og = x_shape_og
+        ctx.eps = eps
+        ctx.is_rms_norm = is_rms_norm
+        ctx.num_groups = num_groups
+        ctx.has_residual = residual is not None
+        ctx.prenorm = prenorm
+        ctx.x_dtype = x.dtype
+        ctx.linear_bias_is_none = linear_bias is None
+        return out if not prenorm else (out, residual_out.reshape(x_shape_og))
+    @staticmethod
+    @input_guard
+    def backward(ctx, dout, *args):
+        x, norm_weight, norm_bias, linear_weight, mean, rstd = ctx.saved_tensors
+        dout = dout.reshape(-1, dout.shape[-1])
+        dy = F.linear(dout, linear_weight.t())
+        dy = dy.reshape(-1, (dy.shape[-1] // ctx.num_groups))
+        dlinear_bias = None if ctx.linear_bias_is_none else dout.sum(0)
+        assert dy.shape == x.shape
+        if ctx.prenorm:
+            dresidual = args[0]
+            dresidual = dresidual.reshape(-1, x.shape[-1])
+            assert dresidual.shape == x.shape
+        else:
+            dresidual = None
+        dx, dnorm_weight, dnorm_bias, dresidual_in, y = layer_norm_bwd(
+            dy,
+            x,
+            norm_weight,
+            norm_bias,
+            ctx.eps,
+            mean,
+            rstd,
+            dresidual,
+            ctx.has_residual,
+            ctx.is_rms_norm,
+            x_dtype=ctx.x_dtype,
+            recompute_output=True,
+            num_groups=ctx.num_groups
+        )
+        dlinear_weight = torch.einsum("bo,bi->oi", dout, y.view(-1, linear_weight.shape[-1]))
+        return (
+            dx.reshape(ctx.x_shape_og),
+            dnorm_weight,
+            dnorm_bias,
+            dlinear_weight,
+            dlinear_bias,
+            dresidual_in.reshape(ctx.x_shape_og) if ctx.has_residual else None,
+            None,
+            None,
+            None,
+            None,
+            None
+        )
+class LayerNormLinear(nn.Module):
+    def __init__(
+        self,
+        hidden_size,
+        elementwise_affine: bool = True,
+        bias: bool = False,
+        eps: float = 1e-5
+    ) -> LayerNormLinear:
+        super().__init__()
+        self.hidden_size = hidden_size
+        self.elementwise_affine = elementwise_affine
+        self.eps = eps
+        self.register_parameter("weight", None)
+        self.register_parameter("bias", None)
+        if elementwise_affine:
+            self.weight = nn.Parameter(torch.empty(hidden_size))
+            if bias:
+                self.bias = nn.Parameter(torch.empty(hidden_size))
+        self.reset_parameters()
+    def reset_parameters(self):
+        if self.elementwise_affine:
+            nn.init.ones_(self.weight)
+            if self.bias is not None:
+                nn.init.zeros_(self.bias)
+    def __repr__(self) -> str:
+        s = f"{self.__class__.__name__}({self.hidden_size}"
+        if not self.elementwise_affine:
+            s += f", elementwise_affine={self.elementwise_affine}"
+        s += f", eps={self.eps}"
+        s += ")"
+        return s
+    def forward(self, x, weight, bias, residual=None, prenorm=False, residual_in_fp32=False):
+        return layer_norm_linear(
+            x=x,
+            norm_weight=self.weight,
+            norm_bias=self.bias,
+            linear_weight=weight,
+            linear_bias=bias,
+            residual=residual,
+            eps=self.eps,
+            prenorm=prenorm,
+            residual_in_fp32=residual_in_fp32,
+            is_rms_norm=False
+        )
+class GroupNormLinear(nn.Module):
+    def __init__(
+        self,
+        num_groups: int,
+        hidden_size: int,
+        elementwise_affine: bool = True,
+        bias: bool = False,
+        eps: float = 1e-5,
+        is_rms_norm: bool = False
+    ) -> GroupNormLinear:
+        super().__init__()
+        if hidden_size % num_groups != 0:
+            raise ValueError('num_channels must be divisible by num_groups')
+        self.num_groups = num_groups
+        self.hidden_size = hidden_size
+        self.elementwise_affine = elementwise_affine
+        self.eps = eps
+        self.is_rms_norm = is_rms_norm
+        self.register_parameter("weight", None)
+        self.register_parameter("bias", None)
+        if elementwise_affine:
+            self.weight = nn.Parameter(torch.empty(hidden_size))
+            if bias:
+                self.bias = nn.Parameter(torch.empty(hidden_size))
+        self.reset_parameters()
+    def reset_parameters(self):
+        if self.elementwise_affine:
+            nn.init.ones_(self.weight)
+            if self.bias is not None:
+                nn.init.zeros_(self.bias)
+    def __repr__(self) -> str:
+        s = f"{self.__class__.__name__}({self.num_groups}, {self.hidden_size}"
+        if not self.elementwise_affine:
+            s += f", elementwise_affine={self.elementwise_affine}"
+        if self.is_rms_norm:
+            s += f", is_rms_norm={self.is_rms_norm}"
+        s += f", eps={self.eps}"
+        s += ")"
+        return s
+    def forward(self, x, weight, bias, residual=None, prenorm=False, residual_in_fp32=False):
+        return layer_norm_linear(
+            x=x,
+            norm_weight=self.weight,
+            norm_bias=self.bias,
+            linear_weight=weight,
+            linear_bias=bias,
+            residual=residual,
+            eps=self.eps,
+            prenorm=prenorm,
+            residual_in_fp32=residual_in_fp32,
+            is_rms_norm=self.is_rms_norm,
+            num_groups=self.num_groups
+        )
+class RMSNormLinear(nn.Module):
+    def __init__(
+        self,
+        hidden_size,
+        elementwise_affine: bool = True,
+        bias: bool = False,
+        eps: float = 1e-5
+    ) -> RMSNormLinear:
+        super().__init__()
+        self.hidden_size = hidden_size
+        self.elementwise_affine = elementwise_affine
+        self.eps = eps
+        self.register_parameter("weight", None)
+        self.register_parameter("bias", None)
+        if elementwise_affine:
+            self.weight = nn.Parameter(torch.empty(hidden_size))
+            if bias:
+                self.bias = nn.Parameter(torch.empty(hidden_size))
+        self.reset_parameters()
+    def reset_parameters(self):
+        if self.elementwise_affine:
+            nn.init.ones_(self.weight)
+            if self.bias is not None:
+                nn.init.zeros_(self.bias)
+    def __repr__(self) -> str:
+        s = f"{self.__class__.__name__}({self.hidden_size}"
+        if not self.elementwise_affine:
+            s += f", elementwise_affine={self.elementwise_affine}"
+        s += f", eps={self.eps}"
+        s += ")"
+        return s
+    def forward(self, x, weight, bias, residual=None, prenorm=False, residual_in_fp32=False):
+        return layer_norm_linear(
+            x=x,
+            norm_weight=self.weight,
+            norm_bias=self.bias,
+            linear_weight=weight,
+            linear_bias=bias,
+            residual=residual,
+            eps=self.eps,
+            prenorm=prenorm,
+            residual_in_fp32=residual_in_fp32,
+            is_rms_norm=True
+        )
+class NormParallel(ParallelStyle):
+    def __init__(self, *, sequence_dim: int = 1, use_local_output: bool = False):
+        super().__init__()
+        self.sequence_sharding = (Shard(sequence_dim),)
+        self.use_local_output = use_local_output
+    def _replicate_module_fn(
+        self, name: str, module: nn.Module, device_mesh: DeviceMesh
+    ):
+        for p_name, param in module.named_parameters():
+            # simple replication with fixed ones_ init from LayerNorm/RMSNorm, which allow
+            # us to simply just use from_local
+            replicated_param = torch.nn.Parameter(
+                DTensor.from_local(param, device_mesh, [Replicate()], run_check=False)
+            )
+            module.register_parameter(p_name, replicated_param)
+    @staticmethod
+    def _prepare_input_fn(sequence_sharding, mod, inputs, device_mesh):
+        input_tensor = inputs[0]
+        if isinstance(input_tensor, DTensor):
+            # if the passed in input DTensor is not sharded on the sequence dim, we need to redistribute it
+            if input_tensor.placements != sequence_sharding:
+                input_tensor = input_tensor.redistribute(
+                    placements=sequence_sharding, async_op=True
+                )
+            return input_tensor
+        elif isinstance(input_tensor, torch.Tensor):
+            # assume the input passed in already sharded on the sequence dim and create the DTensor
+            return DTensor.from_local(
+                input_tensor, device_mesh, sequence_sharding, run_check=False
+            )
+        else:
+            raise ValueError(
+                f"expecting input of {mod} to be a torch.Tensor or DTensor, but got {input_tensor}"
+            )
+    @staticmethod
+    def _prepare_output_fn(use_local_output, mod, outputs, device_mesh):
+        return outputs.to_local() if use_local_output else outputs
+    def _apply(self, module: nn.Module, device_mesh: DeviceMesh) -> nn.Module:
+        return distribute_module(
+            module,
+            device_mesh,
+            self._replicate_module_fn,
+            partial(self._prepare_input_fn, self.sequence_sharding),
+            partial(self._prepare_output_fn, self.use_local_output),
+        )

fla/modules/parallel.py ADDED Viewed

	@@ -0,0 +1,37 @@

+# -*- coding: utf-8 -*-
+# Copyright (c) 2023-2025, Songlin Yang, Yu Zhang
+from typing import Optional
+import torch.nn as nn
+from torch.distributed import DeviceMesh
+from torch.distributed.tensor import DTensor, distribute_module
+from torch.distributed.tensor.parallel import ParallelStyle
+from torch.distributed.tensor.placement_types import Placement
+class PrepareModuleWeight(ParallelStyle):
+    def __init__(self, *, layouts: Optional[Placement] = None):
+        super().__init__()
+        self.layouts = layouts
+    def _replicate_module_fn(
+        self,
+        name: str,
+        module: nn.Module,
+        device_mesh: DeviceMesh
+    ):
+        for p_name, param in module.named_parameters():
+            replicated_param = nn.Parameter(
+                DTensor.from_local(param, device_mesh, [self.layouts], run_check=False)
+            )
+            module.register_parameter(p_name, replicated_param)
+    def _apply(self, module: nn.Module, device_mesh: DeviceMesh) -> nn.Module:
+        return distribute_module(
+            module,
+            device_mesh,
+            partition_fn=self._replicate_module_fn,
+            input_fn=None,
+            output_fn=None
+        )

fla/modules/rotary.py ADDED Viewed

	@@ -0,0 +1,512 @@

+# -*- coding: utf-8 -*-
+# Copyright (c) 2023, Tri Dao.
+# https://github.com/Dao-AILab/flash-attention/blob/main/flash_attn/ops/triton/rotary.py
+from typing import Optional, Tuple, Union
+import torch
+import torch.nn as nn
+import triton
+import triton.language as tl
+from einops import rearrange, repeat
+from fla.utils import get_multiprocessor_count, input_guard
+def rotate_half(x, interleaved=False):
+    if not interleaved:
+        x1, x2 = x.chunk(2, dim=-1)
+        return torch.cat((-x2, x1), dim=-1)
+    else:
+        x1, x2 = x[..., ::2], x[..., 1::2]
+        return rearrange(torch.stack((-x2, x1), dim=-1), '... d two -> ... (d two)', two=2)
+def rotary_embedding_ref(x, cos, sin, interleaved=False):
+    ro_dim = cos.shape[-1] * 2
+    assert ro_dim <= x.shape[-1]
+    cos = repeat(cos, '... d -> ... 1 (2 d)' if not interleaved else '... d -> ... 1 (d 2)')
+    sin = repeat(sin, '... d -> ... 1 (2 d)' if not interleaved else '... d -> ... 1 (d 2)')
+    return torch.cat([x[..., :ro_dim] * cos + rotate_half(x[..., :ro_dim], interleaved) * sin, x[..., ro_dim:]], -1)
+@triton.autotune(
+    configs=[
+        triton.Config({}, num_warps=num_warps, num_stages=num_stages)
+        for num_warps in [2, 4, 8, 16, 32]
+        for num_stages in [2, 3, 4]
+    ],
+    key=['B', 'H', 'D', 'INTERLEAVED'],
+)
+@triton.jit
+def rotary_embedding_kernel(
+    x,
+    cos,
+    sin,
+    y,
+    cu_seqlens,
+    seq_offsets,  # this could be int or a pointer
+    # Matrix dimensions
+    B: tl.constexpr,
+    T: tl.constexpr,
+    H: tl.constexpr,
+    D: tl.constexpr,
+    R: tl.constexpr,
+    TR: tl.constexpr,
+    BT: tl.constexpr,
+    BD: tl.constexpr,
+    IS_SEQLEN_OFFSETS_TENSOR: tl.constexpr,
+    IS_VARLEN: tl.constexpr,
+    INTERLEAVED: tl.constexpr,
+    CONJUGATE: tl.constexpr
+):
+    i_t, i_b, i_h = tl.program_id(0), tl.program_id(1), tl.program_id(2)
+    if not IS_VARLEN:
+        x = x + i_b * T*H*D + i_h * D
+        y = y + i_b * T*H*D + i_h * D
+    else:
+        bos, eos = tl.load(cu_seqlens + i_b), tl.load(cu_seqlens + i_b + 1)
+        T = eos - bos
+        x = x + bos * H*D + i_h * D
+        y = y + bos * H*D + i_h * D
+    if i_t * BT >= T:
+        return
+    o_t = i_t * BT + tl.arange(0, BT)
+    if not IS_SEQLEN_OFFSETS_TENSOR:
+        o_cs = o_t + seq_offsets
+    else:
+        o_cs = o_t + tl.load(seq_offsets + i_b)
+    if not INTERLEAVED:
+        # Load the 1st and 2nd halves of x, do calculation, then store to 1st and 2nd halves of out
+        o_r = tl.arange(0, BD // 2)
+        p_x = x + o_t[:, None] * H*D + o_r[None, :]
+        p_cos = cos + (o_cs[:, None] * R + o_r[None, :])
+        p_sin = sin + (o_cs[:, None] * R + o_r[None, :])
+        mask = (o_t[:, None] >= 0) & (o_t[:, None] < T) & (o_r[None, :] < R)
+        b_cos = tl.load(p_cos, mask=mask, other=1.0).to(tl.float32)
+        b_sin = tl.load(p_sin, mask=mask, other=0.0).to(tl.float32)
+        b_x0 = tl.load(p_x, mask=mask, other=0.0).to(tl.float32)
+        b_x1 = tl.load(p_x + R, mask=mask, other=0.0).to(tl.float32)
+        if CONJUGATE:
+            b_sin = -b_sin
+        b_o0 = b_x0 * b_cos - b_x1 * b_sin
+        b_o1 = b_x0 * b_sin + b_x1 * b_cos
+        # write back result
+        p_y = y + (o_t[:, None] * H*D + o_r[None, :])
+        tl.store(p_y, b_o0, mask=mask)
+        tl.store(p_y + R, b_o1, mask=mask)
+    else:
+        # We don't want to load x[0, 2, 4, ...] and x[1, 3, 5, ...] separately since both are slow.
+        # Instead, we load x0 = x[0, 1, 2, 3, ...] and x1 = x[1, 0, 3, 2, ...].
+        # Loading x0 will be fast but x1 will be slow.
+        # Then we load cos = cos[0, 0, 1, 1, ...] and sin = sin[0, 0, 1, 1, ...].
+        # Then we do the calculation and use tl.where to pick put the right outputs for the even
+        # and for the odd indices.
+        o_d = tl.arange(0, BD)
+        o_d_swap = o_d + ((o_d + 1) % 2) * 2 - 1  # 1, 0, 3, 2, 5, 4, ...
+        o_d_repeat = tl.arange(0, BD) // 2
+        p_x0 = x + o_t[:, None] * H*D + o_d[None, :]
+        p_x1 = x + o_t[:, None] * H*D + o_d_swap[None, :]
+        p_cos = cos + (o_cs[:, None] * R + o_d_repeat[None, :])
+        p_sin = sin + (o_cs[:, None] * R + o_d_repeat[None, :])
+        mask = (o_cs[:, None] >= 0) & (o_cs[:, None] < TR) & (o_d_repeat[None, :] < R)
+        b_cos = tl.load(p_cos, mask=mask, other=1.0).to(tl.float32)
+        b_sin = tl.load(p_sin, mask=mask, other=0.0).to(tl.float32)
+        b_x0 = tl.load(p_x0, mask=mask, other=0.0).to(tl.float32)
+        b_x1 = tl.load(p_x1, mask=mask, other=0.0).to(tl.float32)
+        if CONJUGATE:
+            b_sin = -b_sin
+        b_o0 = b_x0 * b_cos
+        b_o1 = b_x1 * b_sin
+        b_y = tl.where(o_d[None, :] % 2 == 0, b_o0 - b_o1, b_o0 + b_o1)
+        p_y = y + (o_t[:, None] * H*D + o_d[None, :])
+        tl.store(p_y, b_y, mask=mask)
+def rotary_embedding_fwdbwd(
+    x: torch.Tensor,
+    cos: torch.Tensor,
+    sin: torch.Tensor,
+    seqlen_offsets: Union[int, torch.Tensor] = 0,
+    cu_seqlens: Optional[torch.Tensor] = None,
+    max_seqlen: Optional[int] = None,
+    interleaved: bool = False,
+    inplace: bool = False,
+    conjugate: bool = False
+) -> torch.Tensor:
+    """
+    Args:
+        x: [B, T, H, D].
+        cos: [TR, R / 2]
+        sin: [TR, R / 2]
+        seqlen_offsets: integer or integer tensor of size (N,)
+        cu_seqlens: (N + 1,) or None
+        max_seqlen: int
+    Returns:
+        y: [B, T, H, D]
+    """
+    is_varlen = cu_seqlens is not None
+    B, T, H, D = x.shape
+    if not is_varlen:
+        N = B
+    else:
+        assert max_seqlen is not None, "If cu_seqlens is passed in, then max_seqlen must be passed"
+        N, T = cu_seqlens.shape[0] - 1, max_seqlen
+    TR, R = cos.shape
+    assert sin.shape == cos.shape
+    R2 = R * 2
+    assert D <= 256, "Only support D <= 256"
+    assert TR >= T, "TR must be >= T"
+    assert cos.dtype == sin.dtype, f"cos and sin must have the same dtype, got {cos.dtype} and {sin.dtype}"
+    assert x.dtype == cos.dtype, f"Input and cos/sin must have the same dtype, got {x.dtype} and {cos.dtype}"
+    if isinstance(seqlen_offsets, torch.Tensor):
+        assert seqlen_offsets.shape == (N,)
+        assert seqlen_offsets.dtype in [torch.int32, torch.int64]
+    else:
+        assert seqlen_offsets + T <= TR
+    y = torch.empty_like(x) if not inplace else x
+    if R2 < D and not inplace:
+        y[..., R2:].copy_(x[..., R2:])
+    BD = triton.next_power_of_2(R2)
+    BT = min(128, triton.next_power_of_2(triton.cdiv(T, get_multiprocessor_count(x.device.index))))
+    def grid(meta): return (triton.cdiv(T, meta['BT']), N, H)  # noqa
+    rotary_embedding_kernel[grid](
+        x,
+        cos,
+        sin,
+        y,
+        cu_seqlens,
+        seqlen_offsets,
+        B=B,
+        T=T,
+        H=H,
+        D=D,
+        R=R,
+        TR=TR,
+        BT=BT,
+        BD=BD,
+        IS_SEQLEN_OFFSETS_TENSOR=isinstance(seqlen_offsets, torch.Tensor),
+        IS_VARLEN=is_varlen,
+        INTERLEAVED=interleaved,
+        CONJUGATE=conjugate
+    )
+    return y
+class RotaryEmbeddingFunction(torch.autograd.Function):
+    @staticmethod
+    @input_guard
+    def forward(
+        ctx,
+        x,
+        cos,
+        sin,
+        interleaved=False,
+        inplace=False,
+        seqlen_offsets: Union[int, torch.Tensor] = 0,
+        cu_seqlens: Optional[torch.Tensor] = None,
+        max_seqlen: Optional[int] = None,
+    ):
+        y = rotary_embedding_fwdbwd(
+            x,
+            cos,
+            sin,
+            seqlen_offsets=seqlen_offsets,
+            cu_seqlens=cu_seqlens,
+            max_seqlen=max_seqlen,
+            interleaved=interleaved,
+            inplace=inplace,
+        )
+        if isinstance(seqlen_offsets, int):
+            # Can't save int with save_for_backward
+            ctx.save_for_backward(cos, sin, cu_seqlens)
+            ctx.seqlen_offsets = seqlen_offsets
+        else:
+            ctx.save_for_backward(cos, sin, cu_seqlens, seqlen_offsets)
+            ctx.seqlen_offsets = None
+        ctx.interleaved = interleaved
+        ctx.inplace = inplace
+        ctx.max_seqlen = max_seqlen
+        return y if not inplace else x
+    @staticmethod
+    @input_guard
+    def backward(ctx, do):
+        seqlen_offsets = ctx.seqlen_offsets
+        if seqlen_offsets is None:
+            cos, sin, cu_seqlens, seqlen_offsets = ctx.saved_tensors
+        else:
+            cos, sin, cu_seqlens = ctx.saved_tensors
+        # TD [2023-09-02]: For some reason Triton (2.0.0.post1) errors with
+        # "[CUDA]: invalid device context", and cloning makes it work. Idk why. Triton 2.1.0 works.
+        if not ctx.interleaved and not ctx.inplace:
+            do = do.clone()
+        dx = rotary_embedding_fwdbwd(
+            do,
+            cos,
+            sin,
+            seqlen_offsets=seqlen_offsets,
+            cu_seqlens=cu_seqlens,
+            max_seqlen=ctx.max_seqlen,
+            interleaved=ctx.interleaved,
+            inplace=ctx.inplace,
+            conjugate=True,
+        )
+        return dx, None, None, None, None, None, None, None
+def rotary_embedding(
+    x,
+    cos,
+    sin,
+    interleaved=False,
+    inplace=False,
+    seqlen_offsets: Union[int, torch.Tensor] = 0,
+    cu_seqlens: Optional[torch.Tensor] = None,
+    max_seqlen: Optional[int] = None,
+):
+    """
+    Args:
+        x: [B, T, H, D]
+        cos, sin: [TR, R//2]
+        interleaved:
+            If True, rotate pairs of even and odd dimensions (GPT-J style) instead of 1st half and 2nd half (GPT-NeoX style).
+        inplace:
+            If True, apply rotary embedding in-place.
+        seqlen_offsets: [N,] or int.
+            Each sequence in x is shifted by this amount.
+            Most commonly used in inference when we have KV cache.
+        cu_seqlens: [N + 1,] or None
+        max_seqlen: int
+    Returns:
+        out: [B, T, H, D]
+    """
+    return RotaryEmbeddingFunction.apply(
+        x,
+        cos,
+        sin,
+        interleaved,
+        inplace,
+        seqlen_offsets,
+        cu_seqlens,
+        max_seqlen
+    )
+class RotaryEmbedding(nn.Module):
+    """
+    The rotary position embeddings from RoFormer_ (Su et. al).
+    A crucial insight from the method is that the query and keys are
+    transformed by rotation matrices which depend on the relative positions.
+    Other implementations are available in the Rotary Transformer repo_ and in
+    GPT-NeoX_, GPT-NeoX was an inspiration
+    .. _RoFormer: https://arxiv.org/abs/2104.09864
+    .. _repo: https://github.com/ZhuiyiTechnology/roformer
+    .. _GPT-NeoX: https://github.com/EleutherAI/gpt-neox
+    If scale_base is not None, this implements XPos (Sun et al., https://arxiv.org/abs/2212.10554).
+    A recommended value for scale_base is 512: https://github.com/HazyResearch/flash-attention/issues/96
+    Reference: https://github.com/sunyt32/torchscale/blob/main/torchscale/component/xpos_relative_position.py
+    """
+    def __init__(
+        self,
+        dim: int,
+        base: float = 10000.0,
+        scale_base: Optional[float] = None,
+        interleaved: bool = False,
+        pos_idx_in_fp32: bool = True,
+        device: Optional[torch.device] = None,
+    ):
+        """
+        interleaved:
+            If True, rotate pairs of even and odd dimensions (GPT-J style) instead of 1st half and 2nd half (GPT-NeoX style).
+        pos_idx_in_fp32:
+            If True, the position indices [0.0, ..., seqlen - 1] are in fp32, otherwise they might be in lower precision.
+            This option was added because previously (before 2023-07-02), when we construct
+            the position indices, we use the dtype of self.inv_freq.
+            In most cases this would be fp32, but if the model is trained in pure bf16 (not mixed precision), then
+            self.inv_freq would be bf16, and the position indices are also in bf16.
+            Because of the limited precision of bf16 (e.g. 1995.0 is rounded to 2000.0), the
+            embeddings for some positions will coincide.
+            To maintain compatibility with models previously trained in pure bf16, we add this option.
+        """
+        super().__init__()
+        self.dim = dim
+        self.base = float(base)
+        self.scale_base = scale_base
+        self.interleaved = interleaved
+        self.pos_idx_in_fp32 = pos_idx_in_fp32
+        self.device = device
+        # Generate and save the inverse frequency buffer (non trainable)
+        self.register_buffer("inv_freq", torch.empty(-(dim // -2), dtype=torch.float32, device=device), persistent=False)
+        scale = None
+        if scale_base is not None:
+            scale = torch.empty(-(dim // -2), dtype=torch.float32, device=device)
+        self.register_buffer("scale", scale, persistent=False)
+        self._seq_len_cached = 0
+        self._cos_cached = None
+        self._sin_cached = None
+        self._cos_k_cached = None
+        self._sin_k_cached = None
+        self.reset_parameters()
+    def reset_parameters(self):
+        with torch.no_grad():
+            self.inv_freq.copy_(self._compute_inv_freq(device=self.inv_freq.device))
+            if self.scale_base is not None:
+                self.scale.copy_(self._compute_scale(device=self.scale.device))
+    def __repr__(self):
+        s = f"{self.__class__.__name__}("
+        s += f"dim={self.dim}, "
+        s += f"base={self.base}, "
+        s += f"interleaved={self.interleaved}, "
+        if self.scale_base is not None:
+            s += f"scale_base={self.scale_base}, "
+        s += f"pos_idx_in_fp32={self.pos_idx_in_fp32})"
+        return s
+    def _compute_inv_freq(self, device=None):
+        return 1.0 / (
+            self.base
+            ** (torch.arange(0, self.dim, 2, device=device, dtype=torch.float32) / self.dim)
+        )
+    def _compute_scale(self, device=None):
+        return (torch.arange(0, self.dim, 2, device=device, dtype=torch.float32) + 0.4 * self.dim) / (1.4 * self.dim)
+    def _update_cos_sin_cache(self, seqlen, device=None, dtype=None):
+        # Reset the tables if the sequence length has changed,
+        # if we're on a new device (possibly due to tracing for instance),
+        # or if we're switching from inference mode to training
+        if (
+            seqlen > self._seq_len_cached
+            or self._cos_cached is None
+            or self._cos_cached.device != device
+            or self._cos_cached.dtype != dtype
+            or (self.training and self._cos_cached.is_inference())
+        ):
+            self._seq_len_cached = seqlen
+            # We want fp32 here, not self.inv_freq.dtype, since the model could be loaded in bf16
+            # And the output of arange can be quite large, so bf16 would lose a lot of precision.
+            # However, for compatibility reason, we add an option to use the dtype of self.inv_freq.
+            if self.pos_idx_in_fp32:
+                t = torch.arange(seqlen, device=device, dtype=torch.float32)
+                # We want fp32 here as well since inv_freq will be multiplied with t, and the output
+                # will be large. Having it in bf16 will lose a lot of precision and cause the
+                # cos & sin output to change significantly.
+                # We want to recompute self.inv_freq if it was not loaded in fp32
+                if self.inv_freq.dtype != torch.float32:
+                    inv_freq = self._compute_inv_freq(device=device)
+                else:
+                    inv_freq = self.inv_freq
+            else:
+                t = torch.arange(seqlen, device=device, dtype=self.inv_freq.dtype)
+                inv_freq = self.inv_freq
+            # Don't do einsum, it converts fp32 to fp16 under AMP
+            # freqs = torch.einsum("i,j->ij", t, self.inv_freq)
+            freqs = torch.outer(t, inv_freq)
+            if self.scale is None:
+                self._cos_cached = torch.cos(freqs).to(dtype)
+                self._sin_cached = torch.sin(freqs).to(dtype)
+            else:
+                power = (
+                    torch.arange(seqlen, dtype=self.scale.dtype, device=self.scale.device)
+                    - seqlen // 2
+                ) / self.scale_base
+                scale = self.scale.to(device=power.device) ** rearrange(power, "s -> s 1")
+                # We want the multiplication by scale to happen in fp32
+                self._cos_cached = (torch.cos(freqs) * scale).to(dtype)
+                self._sin_cached = (torch.sin(freqs) * scale).to(dtype)
+                self._cos_k_cached = (torch.cos(freqs) / scale).to(dtype)
+                self._sin_k_cached = (torch.sin(freqs) / scale).to(dtype)
+    def forward(
+        self,
+        q: torch.Tensor,
+        k: torch.Tensor,
+        seqlen_offset: Union[int, torch.Tensor] = 0,
+        cu_seqlens: Optional[torch.Tensor] = None,
+        max_seqlen: Optional[int] = None,
+    ) -> Union[torch.Tensor, Tuple[torch.Tensor, torch.Tensor]]:
+        """
+        q: [B, T, H, D]
+        k: [B, T, H, D]
+        seqlen_offset:
+            (N,) or int. Each sequence in x is shifted by this amount.
+            Most commonly used in inference when we have KV cache.
+            If it's a tensor of shape (N,), then to update the cos / sin cache, one
+            should pass in max_seqlen, which will update the cos / sin cache up to that length.
+        cu_seqlens: (N + 1,) or None
+        max_seqlen: int
+        """
+        if max_seqlen is not None:
+            self._update_cos_sin_cache(max_seqlen, device=q.device, dtype=q.dtype)
+        elif isinstance(seqlen_offset, int):
+            self._update_cos_sin_cache(q.shape[1] + seqlen_offset, device=q.device, dtype=q.dtype)
+        if self.scale is None:
+            q = rotary_embedding(
+                q,
+                self._cos_cached,
+                self._sin_cached,
+                interleaved=self.interleaved,
+                seqlen_offsets=seqlen_offset,
+                cu_seqlens=cu_seqlens,
+                max_seqlen=max_seqlen
+            )
+            k = rotary_embedding(
+                k,
+                self._cos_cached,
+                self._sin_cached,
+                interleaved=self.interleaved,
+                seqlen_offsets=seqlen_offset,
+                cu_seqlens=cu_seqlens,
+                max_seqlen=max_seqlen
+            )
+        else:
+            q = rotary_embedding(
+                q,
+                self._cos_cached,
+                self._sin_cached,
+                interleaved=self.interleaved,
+                seqlen_offsets=seqlen_offset,
+                cu_seqlens=cu_seqlens,
+                max_seqlen=max_seqlen
+            )
+            k = rotary_embedding(
+                k,
+                self._cos_k_cached,
+                self._sin_k_cached,
+                interleaved=self.interleaved,
+                seqlen_offsets=seqlen_offset,
+                cu_seqlens=cu_seqlens,
+                max_seqlen=max_seqlen
+            )
+        return q, k

fla/ops/__init__.py ADDED Viewed

	@@ -0,0 +1,46 @@

+# -*- coding: utf-8 -*-
+from .abc import chunk_abc
+from .attn import parallel_attn, parallel_rectified_attn, parallel_softpick_attn, naive_attn, naive_rectified_attn, naive_softpick_attn
+from .based import fused_chunk_based, parallel_based
+from .delta_rule import chunk_delta_rule, fused_chunk_delta_rule, fused_recurrent_delta_rule
+from .forgetting_attn import parallel_forgetting_attn
+from .gated_delta_rule import chunk_gated_delta_rule, fused_recurrent_gated_delta_rule
+from .generalized_delta_rule import (
+    chunk_dplr_delta_rule,
+    chunk_iplr_delta_rule,
+    fused_recurrent_dplr_delta_rule,
+    fused_recurrent_iplr_delta_rule
+)
+from .gla import chunk_gla, fused_chunk_gla, fused_recurrent_gla
+from .gsa import chunk_gsa, fused_recurrent_gsa
+from .hgrn import fused_recurrent_hgrn
+from .lightning_attn import chunk_lightning_attn, fused_recurrent_lightning_attn
+from .linear_attn import chunk_linear_attn, fused_chunk_linear_attn, fused_recurrent_linear_attn
+from .nsa import parallel_nsa
+from .retention import chunk_retention, fused_chunk_retention, fused_recurrent_retention, parallel_retention
+from .rwkv6 import chunk_rwkv6, fused_recurrent_rwkv6
+from .rwkv7 import chunk_rwkv7, fused_recurrent_rwkv7
+from .simple_gla import chunk_simple_gla, fused_recurrent_simple_gla, parallel_simple_gla
+__all__ = [
+    'chunk_abc',
+    'parallel_attn', 'parallel_rectified_attn', 'parallel_softpick_attn',
+    'naive_attn', 'naive_rectified_attn', 'naive_softpick_attn',
+    'fused_chunk_based', 'parallel_based',
+    'chunk_delta_rule', 'fused_chunk_delta_rule', 'fused_recurrent_delta_rule',
+    'parallel_forgetting_attn',
+    'chunk_gated_delta_rule', 'fused_recurrent_gated_delta_rule',
+    'chunk_dplr_delta_rule', 'chunk_iplr_delta_rule',
+    'fused_recurrent_dplr_delta_rule', 'fused_recurrent_iplr_delta_rule',
+    'chunk_gla', 'fused_chunk_gla', 'fused_recurrent_gla',
+    'chunk_gsa', 'fused_recurrent_gsa',
+    'fused_recurrent_hgrn',
+    'chunk_lightning_attn', 'fused_recurrent_lightning_attn',
+    'chunk_linear_attn', 'fused_chunk_linear_attn', 'fused_recurrent_linear_attn',
+    'parallel_nsa',
+    'chunk_retention', 'fused_chunk_retention', 'fused_recurrent_retention', 'parallel_retention',
+    'chunk_rwkv6', 'fused_recurrent_rwkv6',
+    'chunk_rwkv7', 'fused_recurrent_rwkv7',
+    'chunk_simple_gla', 'fused_recurrent_simple_gla', 'parallel_simple_gla',
+]

fla/ops/abc/__init__.py ADDED Viewed

	@@ -0,0 +1,7 @@

+# -*- coding: utf-8 -*-
+from .chunk import chunk_abc
+__all__ = [
+    'chunk_abc'
+]

fla/ops/attn/__pycache__/__init__.cpython-312.pyc ADDED Viewed

Binary file (540 Bytes). View file

fla/ops/attn/__pycache__/naive_rectified.cpython-312.pyc ADDED Viewed

Binary file (2.24 kB). View file

fla/ops/attn/__pycache__/parallel.cpython-312.pyc ADDED Viewed

Binary file (33.1 kB). View file

fla/ops/attn/naive.py ADDED Viewed

	@@ -0,0 +1,28 @@

+import torch
+from typing import Optional
+from einops import rearrange
+def naive_attn(
+    q: torch.Tensor,
+    k: torch.Tensor,
+    v: torch.Tensor,
+    scale: Optional[float] = None,
+    cu_seqlens: Optional[torch.LongTensor] = None,
+    head_first: bool = False
+) -> torch.Tensor:
+    head_dim = q.shape[-1]
+    if scale is None:
+        scale = 1.0 / (head_dim ** 0.5)
+    if not head_first:
+        q, k, v = map(lambda x: rearrange(x, 'b t h d -> b h t d'), (q, k, v))
+    q_len = q.shape[-2]
+    k_len = k.shape[-2]
+    mask = torch.tril(torch.ones(k_len, k_len, device=q.device))
+    wei = torch.matmul(q, k.transpose(2, 3)) # shape: (batch_size, num_heads, q_len, k_len)
+    wei = wei * scale
+    wei = wei.masked_fill(mask[k_len-q_len:k_len, :k_len] == 0, float('-inf'))
+    wei = torch.softmax(wei.float(), dim=-1).to(q.dtype)
+    o = torch.matmul(wei, v) # shape: (batch_size, num_heads, q_len, head_dim)
+    if not head_first:
+        o = rearrange(o, 'b h t d -> b t h d')
+    return o, wei

fla/ops/attn/parallel.py ADDED Viewed

	@@ -0,0 +1,629 @@

+# -*- coding: utf-8 -*-
+# Copyright (c) 2023-2025, Songlin Yang, Yu Zhang
+from typing import Optional
+import torch
+import triton
+import triton.language as tl
+from einops import rearrange, reduce
+from fla.ops.common.utils import prepare_chunk_indices
+from fla.ops.utils.op import exp, log
+from fla.utils import autocast_custom_bwd, autocast_custom_fwd, check_shared_mem, contiguous
+@triton.heuristics({
+    'USE_OFFSETS': lambda args: args['offsets'] is not None
+})
+@triton.autotune(
+    configs=[
+        triton.Config({}, num_warps=num_warps, num_stages=num_stages)
+        for num_warps in [1, 2, 4] + ([8] if check_shared_mem('hopper') else [])
+        for num_stages in [2, 3, 4, 5]
+    ],
+    key=['B', 'H', 'G', 'K', 'V', 'BK', 'BV'],
+)
+@triton.jit
+def parallel_attn_fwd_kernel(
+    q,
+    k,
+    v,
+    o,
+    lse,
+    scale,
+    offsets,
+    indices,
+    T,
+    B: tl.constexpr,
+    H: tl.constexpr,
+    HQ: tl.constexpr,
+    G: tl.constexpr,
+    K: tl.constexpr,
+    V: tl.constexpr,
+    BT: tl.constexpr,
+    BS: tl.constexpr,
+    BK: tl.constexpr,
+    BV: tl.constexpr,
+    USE_OFFSETS: tl.constexpr
+):
+    i_v, i_t, i_bh = tl.program_id(0), tl.program_id(1), tl.program_id(2)
+    i_b, i_hq = i_bh // HQ, i_bh % HQ
+    i_h = i_hq // G
+    if USE_OFFSETS:
+        i_n, i_t = tl.load(indices + i_t * 2).to(tl.int32), tl.load(indices + i_t * 2 + 1).to(tl.int32)
+        bos, eos = tl.load(offsets + i_n).to(tl.int32), tl.load(offsets + i_n + 1).to(tl.int32)
+        T = eos - bos
+    else:
+        i_n = i_b
+        bos, eos = i_n * T, i_n * T + T
+    p_q = tl.make_block_ptr(q + (bos * HQ + i_hq) * K, (T, K), (HQ*K, 1), (i_t * BT, 0), (BT, BK), (1, 0))
+    p_o = tl.make_block_ptr(o + (bos * HQ + i_hq) * V, (T, V), (HQ*V, 1), (i_t * BT, i_v * BV), (BT, BV), (1, 0))
+    p_lse = tl.make_block_ptr(lse + bos * HQ + i_hq, (T,), (HQ,), (i_t * BT,), (BT,), (0,))
+    # the Q block is kept in the shared memory throughout the whole kernel
+    # [BT, BK]
+    b_q = tl.load(p_q, boundary_check=(0, 1))
+    b_q = (b_q * scale).to(b_q.dtype)
+    # [BT, BV]
+    b_o = tl.zeros([BT, BV], dtype=tl.float32)
+    b_m = tl.full([BT], float('-inf'), dtype=tl.float32)
+    b_acc = tl.zeros([BT], dtype=tl.float32)
+    for i_s in range(0, i_t * BT, BS):
+        p_k = tl.make_block_ptr(k + (bos * H + i_h) * K, (K, T), (1, H*K), (0, i_s), (BK, BS), (0, 1))
+        p_v = tl.make_block_ptr(v + (bos * H + i_h) * V, (T, V), (H*V, 1), (i_s, i_v * BV), (BS, BV), (1, 0))
+        # [BK, BS]
+        b_k = tl.load(p_k, boundary_check=(0, 1))
+        # [BS, BV]
+        b_v = tl.load(p_v, boundary_check=(0, 1))
+        # [BT, BS]
+        b_s = tl.dot(b_q, b_k)
+        # [BT, BS]
+        b_m, b_mp = tl.maximum(b_m, tl.max(b_s, 1)), b_m
+        b_r = exp(b_mp - b_m)
+        # [BT, BS]
+        b_p = exp(b_s - b_m[:, None])
+        # [BT]
+        b_acc = b_acc * b_r + tl.sum(b_p, 1)
+        # [BT, BV]
+        b_o = b_o * b_r[:, None] + tl.dot(b_p.to(b_q.dtype), b_v)
+        b_mp = b_m
+    # [BT]
+    o_q = i_t * BT + tl.arange(0, BT)
+    for i_s in range(i_t * BT, min((i_t + 1) * BT, T), BS):
+        p_k = tl.make_block_ptr(k + (bos * H + i_h) * K, (K, T), (1, H*K), (0, i_s), (BK, BS), (0, 1))
+        p_v = tl.make_block_ptr(v + (bos * H + i_h) * V, (T, V), (H*V, 1), (i_s, i_v * BV), (BS, BV), (1, 0))
+        # [BS]
+        o_k = i_s + tl.arange(0, BS)
+        # [BK, BS]
+        b_k = tl.load(p_k, boundary_check=(0, 1))
+        # [BS, BV]
+        b_v = tl.load(p_v, boundary_check=(0, 1))
+        # [BT, BS]
+        b_s = tl.dot(b_q, b_k)
+        b_s = tl.where(o_q[:, None] >= o_k[None, :], b_s, float('-inf'))
+        # [BT]
+        b_m, b_mp = tl.maximum(b_m, tl.max(b_s, 1)), b_m
+        b_r = exp(b_mp - b_m)
+        # [BT, BS]
+        b_p = exp(b_s - b_m[:, None])
+        # [BT]
+        b_acc = b_acc * b_r + tl.sum(b_p, 1)
+        # [BT, BV]
+        b_o = b_o * b_r[:, None] + tl.dot(b_p.to(b_q.dtype), b_v)
+        b_mp = b_m
+    b_o = b_o / b_acc[:, None]
+    b_m += log(b_acc)
+    tl.store(p_o, b_o.to(p_o.dtype.element_ty), boundary_check=(0, 1))
+    tl.store(p_lse, b_m.to(p_lse.dtype.element_ty), boundary_check=(0,))
+@triton.jit
+def parallel_attn_bwd_kernel_preprocess(
+    o,
+    do,
+    delta,
+    B: tl.constexpr,
+    V: tl.constexpr
+):
+    i_n = tl.program_id(0)
+    o_d = tl.arange(0, B)
+    m_d = o_d < V
+    b_o = tl.load(o + i_n * V + o_d, mask=m_d, other=0)
+    b_do = tl.load(do + i_n * V + o_d, mask=m_d, other=0).to(tl.float32)
+    b_delta = tl.sum(b_o * b_do)
+    tl.store(delta + i_n, b_delta.to(delta.dtype.element_ty))
+@triton.heuristics({
+    'USE_OFFSETS': lambda args: args['offsets'] is not None
+})
+@triton.autotune(
+    configs=[
+        triton.Config({}, num_warps=num_warps, num_stages=num_stages)
+        for num_warps in [1, 2, 4] + ([8] if check_shared_mem('hopper') else [])
+        for num_stages in [2, 3, 4, 5]
+    ],
+    key=['B', 'H', 'G', 'K', 'V', 'BK', 'BV'],
+)
+@triton.jit(do_not_specialize=['T'])
+def parallel_attn_bwd_kernel_dq(
+    q,
+    k,
+    v,
+    lse,
+    delta,
+    do,
+    dq,
+    scale,
+    offsets,
+    indices,
+    T,
+    B: tl.constexpr,
+    H: tl.constexpr,
+    HQ: tl.constexpr,
+    G: tl.constexpr,
+    K: tl.constexpr,
+    V: tl.constexpr,
+    BT: tl.constexpr,
+    BS: tl.constexpr,
+    BK: tl.constexpr,
+    BV: tl.constexpr,
+    USE_OFFSETS: tl.constexpr
+):
+    i_v, i_t, i_bh = tl.program_id(0), tl.program_id(1), tl.program_id(2)
+    i_b, i_hq = i_bh // HQ, i_bh % HQ
+    i_h = i_hq // G
+    if USE_OFFSETS:
+        i_n, i_t = tl.load(indices + i_t * 2).to(tl.int32), tl.load(indices + i_t * 2 + 1).to(tl.int32)
+        bos, eos = tl.load(offsets + i_n).to(tl.int32), tl.load(offsets + i_n + 1).to(tl.int32)
+        T = eos - bos
+    else:
+        i_n = i_b
+        bos, eos = i_n * T, i_n * T + T
+    p_q = tl.make_block_ptr(q + (bos * HQ + i_hq) * K, (T, K), (HQ*K, 1), (i_t * BT, 0), (BT, BK), (1, 0))
+    p_dq = tl.make_block_ptr(dq + (bos * HQ + i_hq) * K, (T, K), (HQ*K, 1), (i_t * BT, 0), (BT, BK), (1, 0))
+    p_do = tl.make_block_ptr(do + (bos * HQ + i_hq) * V, (T, V), (HQ*V, 1), (i_t * BT, i_v * BV), (BT, BV), (1, 0))
+    p_lse = tl.make_block_ptr(lse + bos * HQ + i_hq, (T,), (HQ,), (i_t * BT,), (BT,), (0,))
+    p_delta = tl.make_block_ptr(delta + bos * HQ + i_hq, (T,), (HQ,), (i_t * BT,), (BT,), (0,))
+    # [BT, BK]
+    b_q = tl.load(p_q, boundary_check=(0, 1))
+    b_q = (b_q * scale).to(b_q.dtype)
+    # [BT, BV]
+    b_do = tl.load(p_do, boundary_check=(0, 1))
+    # [BT]
+    b_lse = tl.load(p_lse, boundary_check=(0,))
+    b_delta = tl.load(p_delta, boundary_check=(0,))
+    # [BT, BK]
+    b_dq = tl.zeros([BT, BK], dtype=tl.float32)
+    for i_s in range(0, i_t * BT, BS):
+        p_k = tl.make_block_ptr(k + (bos * H + i_h) * K, (K, T), (1, H*K), (0, i_s), (BK, BS), (0, 1))
+        p_v = tl.make_block_ptr(v + (bos * H + i_h) * V, (V, T), (1, H*V), (i_v * BV, i_s), (BV, BS), (0, 1))
+        # [BK, BS]
+        b_k = tl.load(p_k, boundary_check=(0, 1))
+        # [BV, BS]
+        b_v = tl.load(p_v, boundary_check=(0, 1))
+        # [BT, BS]
+        b_s = tl.dot(b_q, b_k)
+        b_p = exp(b_s - b_lse[:, None])
+        # [BT, BV] @ [BV, BS] -> [BT, BS]
+        b_dp = tl.dot(b_do, b_v)
+        b_ds = b_p * (b_dp.to(tl.float32) - b_delta[:, None])
+        # [BT, BS] @ [BS, BK] -> [BT, BK]
+        b_dq += tl.dot(b_ds.to(b_k.dtype), tl.trans(b_k))
+    # [BT]
+    o_q = i_t * BT + tl.arange(0, BT)
+    for i_s in range(i_t * BT, min((i_t + 1) * BT, T), BS):
+        p_k = tl.make_block_ptr(k + (bos * H + i_h) * K, (K, T), (1, H*K), (0, i_s), (BK, BS), (0, 1))
+        p_v = tl.make_block_ptr(v + (bos * H + i_h) * V, (V, T), (1, H*V), (i_v * BV, i_s), (BV, BS), (0, 1))
+        # [BS]
+        o_k = i_s + tl.arange(0, BS)
+        # [BK, BS]
+        b_k = tl.load(p_k, boundary_check=(0, 1))
+        # [BV, BS]
+        b_v = tl.load(p_v, boundary_check=(0, 1))
+        # [BT, BS]
+        b_s = tl.dot(b_q, b_k)
+        b_p = exp(b_s - b_lse[:, None])
+        b_p = tl.where(o_q[:, None] >= o_k[None, :], b_p, 0)
+        # [BT, BV] @ [BV, BS] -> [BT, BS]
+        b_dp = tl.dot(b_do, b_v)
+        b_ds = b_p * (b_dp.to(tl.float32) - b_delta[:, None])
+        # [BT, BS] @ [BS, BK] -> [BT, BK]
+        b_dq += tl.dot(b_ds.to(b_k.dtype), tl.trans(b_k))
+    b_dq *= scale
+    tl.store(p_dq, b_dq.to(p_dq.dtype.element_ty), boundary_check=(0, 1))
+@triton.heuristics({
+    'USE_OFFSETS': lambda args: args['offsets'] is not None
+})
+@triton.autotune(
+    configs=[
+        triton.Config({}, num_warps=num_warps, num_stages=num_stages)
+        for num_warps in [1, 2, 4] + ([8] if check_shared_mem('hopper') else [])
+        for num_stages in [2, 3, 4, 5]
+    ],
+    key=['B', 'H', 'G', 'K', 'V', 'BK', 'BV'],
+)
+@triton.jit(do_not_specialize=['T'])
+def parallel_attn_bwd_kernel_dkv(
+    q,
+    k,
+    v,
+    lse,
+    delta,
+    do,
+    dk,
+    dv,
+    offsets,
+    indices,
+    scale,
+    T,
+    B: tl.constexpr,
+    H: tl.constexpr,
+    HQ: tl.constexpr,
+    G: tl.constexpr,
+    K: tl.constexpr,
+    V: tl.constexpr,
+    BT: tl.constexpr,
+    BS: tl.constexpr,
+    BK: tl.constexpr,
+    BV: tl.constexpr,
+    USE_OFFSETS: tl.constexpr
+):
+    i_v, i_t, i_bh = tl.program_id(0), tl.program_id(1), tl.program_id(2)
+    i_b, i_hq = i_bh // HQ, i_bh % HQ
+    i_h = i_hq // G
+    if USE_OFFSETS:
+        i_n, i_t = tl.load(indices + i_t * 2).to(tl.int32), tl.load(indices + i_t * 2 + 1).to(tl.int32)
+        bos, eos = tl.load(offsets + i_n).to(tl.int32), tl.load(offsets + i_n + 1).to(tl.int32)
+        T = eos - bos
+    else:
+        i_n = i_b
+        bos, eos = i_n * T, i_n * T + T
+    p_k = tl.make_block_ptr(k + (bos * H + i_h) * K, (T, K), (H*K, 1), (i_t * BT, 0), (BT, BK), (1, 0))
+    p_v = tl.make_block_ptr(v + (bos * H + i_h) * V, (T, V), (H*V, 1), (i_t * BT, i_v * BV), (BT, BV), (1, 0))
+    p_dk = tl.make_block_ptr(dk + (bos * HQ + i_hq) * K, (T, K), (HQ*K, 1), (i_t * BT, 0), (BT, BK), (1, 0))
+    p_dv = tl.make_block_ptr(dv + (bos * HQ + i_hq) * V, (T, V), (HQ*V, 1), (i_t * BT, i_v * BV), (BT, BV), (1, 0))
+    # [BT, BK]
+    b_k = tl.load(p_k, boundary_check=(0, 1))
+    b_dk = tl.zeros([BT, BK], dtype=tl.float32)
+    # [BT, BV]
+    b_v = tl.load(p_v, boundary_check=(0, 1))
+    b_dv = tl.zeros([BT, BV], dtype=tl.float32)
+    o_k = i_t * BT + tl.arange(0, BT)
+    for i_s in range(i_t * BT, min((i_t + 1) * BT, T), BS):
+        p_q = tl.make_block_ptr(q + (bos * HQ + i_hq) * K, (T, K), (HQ*K, 1), (i_s, 0), (BS, BK), (1, 0))
+        p_do = tl.make_block_ptr(do + (bos * HQ + i_hq) * V, (T, V), (HQ*V, 1), (i_s, i_v * BV), (BS, BV), (1, 0))
+        p_lse = tl.make_block_ptr(lse + bos * HQ + i_hq, (T,), (HQ,), (i_s,), (BS,), (0,))
+        p_delta = tl.make_block_ptr(delta + bos * HQ + i_hq, (T,), (HQ,), (i_s,), (BS,), (0,))
+        # [BS]
+        o_q = i_s + tl.arange(0, BS)
+        # [BS, BK]
+        b_q = tl.load(p_q, boundary_check=(0, 1))
+        b_q = (b_q * scale).to(b_q.dtype)
+        # [BS, BV]
+        b_do = tl.load(p_do, boundary_check=(0, 1))
+        # [BS]
+        b_lse = tl.load(p_lse, boundary_check=(0,))
+        b_delta = tl.load(p_delta, boundary_check=(0,))
+        # [BT, BS]
+        b_s = tl.dot(b_k, tl.trans(b_q))
+        b_p = exp(b_s - b_lse[None, :])
+        b_p = tl.where(o_k[:, None] <= o_q[None, :], b_p, 0)
+        # [BT, BS] @ [BS, BV] -> [BT, BV]
+        b_dv += tl.dot(b_p.to(b_do.dtype), b_do)
+        # [BT, BV] @ [BV, BS] -> [BT, BS]
+        b_dp = tl.dot(b_v, tl.trans(b_do))
+        # [BT, BS]
+        b_ds = b_p * (b_dp - b_delta[None, :])
+        # [BT, BS] @ [BS, BK] -> [BT, BK]
+        b_dk += tl.dot(b_ds.to(b_q.dtype), b_q)
+    for i_s in range((i_t + 1) * BT, tl.cdiv(T, BS) * BS, BS):
+        p_q = tl.make_block_ptr(q + (bos * HQ + i_hq) * K, (T, K), (HQ*K, 1), (i_s, 0), (BS, BK), (1, 0))
+        p_do = tl.make_block_ptr(do + (bos * HQ + i_hq) * V, (T, V), (HQ*V, 1), (i_s, i_v * BV), (BS, BV), (1, 0))
+        p_lse = tl.make_block_ptr(lse + bos * HQ + i_hq, (T,), (HQ,), (i_s,), (BS,), (0,))
+        p_delta = tl.make_block_ptr(delta + bos * HQ + i_hq, (T,), (HQ,), (i_s,), (BS,), (0,))
+        # [BS]
+        o_q = i_s + tl.arange(0, BS)
+        # [BS, BK]
+        b_q = tl.load(p_q, boundary_check=(0, 1))
+        b_q = (b_q * scale).to(b_q.dtype)
+        # [BS, BV]
+        b_do = tl.load(p_do, boundary_check=(0, 1))
+        # [BS]
+        b_lse = tl.load(p_lse, boundary_check=(0,))
+        b_delta = tl.load(p_delta, boundary_check=(0,))
+        # [BT, BS]
+        b_s = tl.dot(b_k, tl.trans(b_q))
+        b_p = exp(b_s - b_lse[None, :])
+        # [BT, BS] @ [BS, BV] -> [BT, BV]
+        b_dv += tl.dot(b_p.to(b_do.dtype), b_do)
+        # [BT, BV] @ [BV, BS] -> [BT, BS]
+        b_dp = tl.dot(b_v, tl.trans(b_do))
+        # [BT, BS]
+        b_ds = b_p * (b_dp - b_delta[None, :])
+        # [BT, BS] @ [BS, BK] -> [BT, BK]
+        b_dk += tl.dot(b_ds.to(b_q.dtype), b_q)
+    tl.store(p_dk, b_dk.to(p_dk.dtype.element_ty), boundary_check=(0, 1))
+    tl.store(p_dv, b_dv.to(p_dv.dtype.element_ty), boundary_check=(0, 1))
+def parallel_attn_fwd(
+    q: torch.Tensor,
+    k: torch.Tensor,
+    v: torch.Tensor,
+    scale: float,
+    chunk_size: int = 128,
+    offsets: Optional[torch.LongTensor] = None,
+    indices: Optional[torch.LongTensor] = None,
+):
+    B, T, H, K, V = *k.shape, v.shape[-1]
+    HQ = q.shape[2]
+    G = HQ // H
+    BT = chunk_size
+    if check_shared_mem('hopper', q.device.index):
+        BS = min(64, max(16, triton.next_power_of_2(T)))
+        BK = min(256, max(16, triton.next_power_of_2(K)))
+        BV = min(256, max(16, triton.next_power_of_2(V)))
+    elif check_shared_mem('ampere', q.device.index):
+        BS = min(32, max(16, triton.next_power_of_2(T)))
+        BK = min(256, max(16, triton.next_power_of_2(K)))
+        BV = min(128, max(16, triton.next_power_of_2(V)))
+    else:
+        BS = min(32, max(16, triton.next_power_of_2(T)))
+        BK = min(256, max(16, triton.next_power_of_2(K)))
+        BV = min(64, max(16, triton.next_power_of_2(V)))
+    NK = triton.cdiv(K, BK)
+    NV = triton.cdiv(V, BV)
+    NT = triton.cdiv(T, BT) if offsets is None else len(indices)
+    assert NK == 1, "The key dimension can not be larger than 256"
+    o = torch.empty(B, T, HQ, V, dtype=v.dtype, device=q.device)
+    lse = torch.empty(B, T, HQ, dtype=torch.float, device=q.device)
+    grid = (NV, NT, B * HQ)
+    parallel_attn_fwd_kernel[grid](
+        q=q,
+        k=k,
+        v=v,
+        o=o,
+        lse=lse,
+        scale=scale,
+        offsets=offsets,
+        indices=indices,
+        B=B,
+        T=T,
+        H=H,
+        HQ=HQ,
+        G=G,
+        K=K,
+        V=V,
+        BT=BT,
+        BS=BS,
+        BK=BK,
+        BV=BV,
+    )
+    return o, lse
+def parallel_attn_bwd_preprocess(
+    o: torch.Tensor,
+    do: torch.Tensor
+):
+    V = o.shape[-1]
+    delta = torch.empty_like(o[..., 0], dtype=torch.float32)
+    parallel_attn_bwd_kernel_preprocess[(delta.numel(),)](
+        o=o,
+        do=do,
+        delta=delta,
+        B=triton.next_power_of_2(V),
+        V=V,
+    )
+    return delta
+def parallel_attn_bwd(
+    q: torch.Tensor,
+    k: torch.Tensor,
+    v: torch.Tensor,
+    o: torch.Tensor,
+    lse: torch.Tensor,
+    do: torch.Tensor,
+    scale: float = None,
+    chunk_size: int = 128,
+    offsets: Optional[torch.LongTensor] = None,
+    indices: Optional[torch.LongTensor] = None,
+):
+    B, T, H, K, V = *k.shape, v.shape[-1]
+    HQ = q.shape[2]
+    G = HQ // H
+    BT = chunk_size
+    BS = max(16, triton.next_power_of_2(T))
+    BS = min(32, BS) if check_shared_mem('ampere') else min(16, BS)
+    BK = max(16, triton.next_power_of_2(K))
+    BV = max(16, triton.next_power_of_2(V))
+    NV = triton.cdiv(V, BV)
+    NT = triton.cdiv(T, BT) if offsets is None else len(indices)
+    delta = parallel_attn_bwd_preprocess(o, do)
+    dq = torch.empty(B, T, HQ, K, dtype=k.dtype if H == HQ else torch.float, device=q.device)
+    dk = torch.empty(B, T, HQ, K, dtype=k.dtype if H == HQ else torch.float, device=q.device)
+    dv = torch.empty(B, T, HQ, V, dtype=v.dtype if H == HQ else torch.float, device=q.device)
+    grid = (NV, NT, B * HQ)
+    parallel_attn_bwd_kernel_dq[grid](
+        q=q,
+        k=k,
+        v=v,
+        lse=lse,
+        delta=delta,
+        do=do,
+        dq=dq,
+        offsets=offsets,
+        indices=indices,
+        scale=scale,
+        T=T,
+        B=B,
+        H=H,
+        HQ=HQ,
+        G=G,
+        K=K,
+        V=V,
+        BT=BT,
+        BS=BS,
+        BK=BK,
+        BV=BV
+    )
+    parallel_attn_bwd_kernel_dkv[grid](
+        q=q,
+        k=k,
+        v=v,
+        lse=lse,
+        delta=delta,
+        do=do,
+        dk=dk,
+        dv=dv,
+        offsets=offsets,
+        indices=indices,
+        scale=scale,
+        T=T,
+        B=B,
+        H=H,
+        HQ=HQ,
+        G=G,
+        K=K,
+        V=V,
+        BT=BT,
+        BS=BS,
+        BK=BK,
+        BV=BV
+    )
+    dk = reduce(dk, 'b t (h g) k -> b t h k', g=G, reduction='sum')
+    dv = reduce(dv, 'b t (h g) v -> b t h v', g=G, reduction='sum')
+    return dq, dk, dv
+@torch.compile
+class ParallelAttentionFunction(torch.autograd.Function):
+    @staticmethod
+    @contiguous
+    @autocast_custom_fwd
+    def forward(ctx, q, k, v, scale, offsets):
+        ctx.dtype = q.dtype
+        chunk_size = min(128, max(16, triton.next_power_of_2(q.shape[1])))
+        # 2-d indices denoting the offsets of chunks in each sequence
+        # for example, if the passed `offsets` is [0, 100, 356] and `chunk_size` is 64,
+        # then there are 2 and 4 chunks in the 1st and 2nd sequences respectively, and `indices` will be
+        # [[0, 0], [0, 1], [1, 0], [1, 1], [1, 2], [1, 3]]
+        indices = prepare_chunk_indices(offsets, chunk_size) if offsets is not None else None
+        o, lse = parallel_attn_fwd(
+            q=q,
+            k=k,
+            v=v,
+            scale=scale,
+            chunk_size=chunk_size,
+            offsets=offsets,
+            indices=indices
+        )
+        ctx.save_for_backward(q, k, v, o, lse)
+        ctx.chunk_size = chunk_size
+        ctx.offsets = offsets
+        ctx.indices = indices
+        ctx.scale = scale
+        return o.to(q.dtype)
+    @staticmethod
+    @contiguous
+    @autocast_custom_bwd
+    def backward(ctx, do):
+        q, k, v, o, lse = ctx.saved_tensors
+        dq, dk, dv = parallel_attn_bwd(
+            q=q,
+            k=k,
+            v=v,
+            o=o,
+            lse=lse,
+            do=do,
+            scale=ctx.scale,
+            chunk_size=ctx.chunk_size,
+            offsets=ctx.offsets,
+            indices=ctx.indices
+        )
+        return dq.to(q), dk.to(k), dv.to(v), None, None, None, None, None, None, None, None
+def parallel_attn(
+    q: torch.Tensor,
+    k: torch.Tensor,
+    v: torch.Tensor,
+    scale: Optional[float] = None,
+    cu_seqlens: Optional[torch.LongTensor] = None,
+    head_first: bool = False
+) -> torch.Tensor:
+    r"""
+    Args:
+        q (torch.Tensor):
+            queries of shape `[B, T, HQ, K]` if `head_first=False` else `[B, HQ, T, K]`.
+        k (torch.Tensor):
+            keys of shape `[B, T, H, K]` if `head_first=False` else `[B, H, T, K]`.
+            GQA will be applied if HQ is divisible by H.
+        v (torch.Tensor):
+            values of shape `[B, T, H, V]` if `head_first=False` else `[B, H, T, V]`.
+        scale (Optional[int]):
+            Scale factor for attention scores.
+            If not provided, it will default to `1 / sqrt(K)`. Default: `None`.
+        cu_seqlens (torch.LongTensor):
+            Cumulative sequence lengths of shape `[N+1]` used for variable-length training,
+            consistent with the FlashAttention API.
+        head_first (Optional[bool]):
+            Whether the inputs are in the head-first format. Default: `False`.
+    Returns:
+        o (torch.Tensor):
+            Outputs of shape `[B, T, HQ, V]` if `head_first=False` else `[B, HQ, T, V]`.
+    """
+    if scale is None:
+        scale = k.shape[-1] ** -0.5
+    if cu_seqlens is not None:
+        assert q.shape[0] == 1, "batch size must be 1 when cu_seqlens are provided"
+    if head_first:
+        q, k, v = map(lambda x: rearrange(x, 'b h t d -> b t h d'), (q, k, v))
+    o = ParallelAttentionFunction.apply(q, k, v, scale, cu_seqlens)
+    if head_first:
+        o = rearrange(o, 'b t h d -> b h t d')
+    return o

fla/ops/attn/parallel_rectified.py ADDED Viewed

	@@ -0,0 +1,643 @@

+# -*- coding: utf-8 -*-
+# Copyright (c) 2023-2025, Songlin Yang, Yu Zhang
+from typing import Optional
+import torch
+import triton
+import triton.language as tl
+from einops import rearrange, reduce
+from fla.ops.common.utils import prepare_chunk_indices
+from fla.ops.utils.op import exp, log
+from fla.utils import autocast_custom_bwd, autocast_custom_fwd, check_shared_mem, contiguous
+@triton.heuristics({
+    'USE_OFFSETS': lambda args: args['offsets'] is not None
+})
+@triton.autotune(
+    configs=[
+        triton.Config({}, num_warps=num_warps, num_stages=num_stages)
+        for num_warps in [1, 2, 4] + ([8] if check_shared_mem('hopper') else [])
+        for num_stages in [2, 3, 4, 5]
+    ],
+    key=['B', 'H', 'G', 'K', 'V', 'BK', 'BV'],
+)
+@triton.jit
+def parallel_rect_attn_fwd_kernel(
+    q,
+    k,
+    v,
+    o,
+    lse,
+    scale,
+    offsets,
+    indices,
+    T,
+    B: tl.constexpr,
+    H: tl.constexpr,
+    HQ: tl.constexpr,
+    G: tl.constexpr,
+    K: tl.constexpr,
+    V: tl.constexpr,
+    BT: tl.constexpr,
+    BS: tl.constexpr,
+    BK: tl.constexpr,
+    BV: tl.constexpr,
+    USE_OFFSETS: tl.constexpr
+):
+    i_v, i_t, i_bh = tl.program_id(0), tl.program_id(1), tl.program_id(2)
+    i_b, i_hq = i_bh // HQ, i_bh % HQ
+    i_h = i_hq // G
+    if USE_OFFSETS:
+        i_n, i_t = tl.load(indices + i_t * 2).to(tl.int32), tl.load(indices + i_t * 2 + 1).to(tl.int32)
+        bos, eos = tl.load(offsets + i_n).to(tl.int32), tl.load(offsets + i_n + 1).to(tl.int32)
+        T = eos - bos
+    else:
+        i_n = i_b
+        bos, eos = i_n * T, i_n * T + T
+    p_q = tl.make_block_ptr(q + (bos * HQ + i_hq) * K, (T, K), (HQ*K, 1), (i_t * BT, 0), (BT, BK), (1, 0))
+    p_o = tl.make_block_ptr(o + (bos * HQ + i_hq) * V, (T, V), (HQ*V, 1), (i_t * BT, i_v * BV), (BT, BV), (1, 0))
+    p_lse = tl.make_block_ptr(lse + bos * HQ + i_hq, (T,), (HQ,), (i_t * BT,), (BT,), (0,))
+    # the Q block is kept in the shared memory throughout the whole kernel
+    # [BT, BK]
+    b_q = tl.load(p_q, boundary_check=(0, 1))
+    b_q = (b_q * scale).to(b_q.dtype)
+    # [BT, BV]
+    b_o = tl.zeros([BT, BV], dtype=tl.float32)
+    b_m = tl.full([BT], float('-inf'), dtype=tl.float32)
+    b_acc = tl.zeros([BT], dtype=tl.float32)
+    for i_s in range(0, i_t * BT, BS):
+        p_k = tl.make_block_ptr(k + (bos * H + i_h) * K, (K, T), (1, H*K), (0, i_s), (BK, BS), (0, 1))
+        p_v = tl.make_block_ptr(v + (bos * H + i_h) * V, (T, V), (H*V, 1), (i_s, i_v * BV), (BS, BV), (1, 0))
+        # [BK, BS]
+        b_k = tl.load(p_k, boundary_check=(0, 1))
+        # [BS, BV]
+        b_v = tl.load(p_v, boundary_check=(0, 1))
+        # [BT, BS]
+        b_s = tl.dot(b_q, b_k)
+        # [BT, BS]
+        b_m, b_mp = tl.maximum(b_m, tl.max(b_s, 1)), b_m
+        b_r = exp(b_mp - b_m)
+        # [BT, BS]
+        # b_p = exp(b_s - b_m[:, None])
+        # b_p = exp(tl.where(b_s < 0, float('-inf'), b_s - b_m[:, None]))
+        b_p = tl.where(b_s < 0, 0.0, exp(b_s - b_m[:, None])) # Just do this
+        # [BT]
+        b_acc = b_acc * b_r + tl.sum(b_p, 1)
+        # [BT, BV]
+        b_o = b_o * b_r[:, None] + tl.dot(b_p.to(b_q.dtype), b_v)
+        b_mp = b_m
+    # [BT]
+    o_q = i_t * BT + tl.arange(0, BT)
+    for i_s in range(i_t * BT, min((i_t + 1) * BT, T), BS):
+        p_k = tl.make_block_ptr(k + (bos * H + i_h) * K, (K, T), (1, H*K), (0, i_s), (BK, BS), (0, 1))
+        p_v = tl.make_block_ptr(v + (bos * H + i_h) * V, (T, V), (H*V, 1), (i_s, i_v * BV), (BS, BV), (1, 0))
+        # [BS]
+        o_k = i_s + tl.arange(0, BS)
+        # [BK, BS]
+        b_k = tl.load(p_k, boundary_check=(0, 1))
+        # [BS, BV]
+        b_v = tl.load(p_v, boundary_check=(0, 1))
+        # [BT, BS]
+        b_s = tl.dot(b_q, b_k)
+        b_s = tl.where(o_q[:, None] >= o_k[None, :], b_s, float('-inf'))
+        # [BT]
+        b_m, b_mp = tl.maximum(b_m, tl.max(b_s, 1)), b_m
+        b_r = exp(b_mp - b_m)
+        # [BT, BS]
+        # b_p = exp(b_s - b_m[:, None])
+        # b_p = exp(tl.where(b_s < 0, float('-inf'), b_s - b_m[:, None]))
+        b_p = tl.where(b_s < 0, 0.0, exp(b_s - b_m[:, None]))
+        # [BT]
+        b_acc = b_acc * b_r + tl.sum(b_p, 1)
+        # [BT, BV]
+        b_o = b_o * b_r[:, None] + tl.dot(b_p.to(b_q.dtype), b_v)
+        b_mp = b_m
+    # b_o = b_o / b_acc[:, None]
+    b_o = tl.where(b_acc[:, None] == 0, 0.0, b_o / b_acc[:, None])
+    # b_m += tl.log(b_acc)
+    b_m = tl.where(b_acc == 0, 0.0, b_m + tl.log(b_acc))
+    tl.store(p_o, b_o.to(p_o.dtype.element_ty), boundary_check=(0, 1))
+    tl.store(p_lse, b_m.to(p_lse.dtype.element_ty), boundary_check=(0,))
+@triton.jit
+def parallel_rect_attn_bwd_kernel_preprocess(
+    o,
+    do,
+    delta,
+    B: tl.constexpr,
+    V: tl.constexpr
+):
+    i_n = tl.program_id(0)
+    o_d = tl.arange(0, B)
+    m_d = o_d < V
+    b_o = tl.load(o + i_n * V + o_d, mask=m_d, other=0)
+    b_do = tl.load(do + i_n * V + o_d, mask=m_d, other=0).to(tl.float32)
+    b_delta = tl.sum(b_o * b_do)
+    tl.store(delta + i_n, b_delta.to(delta.dtype.element_ty))
+@triton.heuristics({
+    'USE_OFFSETS': lambda args: args['offsets'] is not None
+})
+@triton.autotune(
+    configs=[
+        triton.Config({}, num_warps=num_warps, num_stages=num_stages)
+        for num_warps in [1, 2, 4] + ([8] if check_shared_mem('hopper') else [])
+        for num_stages in [2, 3, 4, 5]
+    ],
+    key=['B', 'H', 'G', 'K', 'V', 'BK', 'BV'],
+)
+@triton.jit(do_not_specialize=['T'])
+def parallel_rect_attn_bwd_kernel_dq(
+    q,
+    k,
+    v,
+    lse,
+    delta,
+    do,
+    dq,
+    scale,
+    offsets,
+    indices,
+    T,
+    B: tl.constexpr,
+    H: tl.constexpr,
+    HQ: tl.constexpr,
+    G: tl.constexpr,
+    K: tl.constexpr,
+    V: tl.constexpr,
+    BT: tl.constexpr,
+    BS: tl.constexpr,
+    BK: tl.constexpr,
+    BV: tl.constexpr,
+    USE_OFFSETS: tl.constexpr
+):
+    i_v, i_t, i_bh = tl.program_id(0), tl.program_id(1), tl.program_id(2)
+    i_b, i_hq = i_bh // HQ, i_bh % HQ
+    i_h = i_hq // G
+    if USE_OFFSETS:
+        i_n, i_t = tl.load(indices + i_t * 2).to(tl.int32), tl.load(indices + i_t * 2 + 1).to(tl.int32)
+        bos, eos = tl.load(offsets + i_n).to(tl.int32), tl.load(offsets + i_n + 1).to(tl.int32)
+        T = eos - bos
+    else:
+        i_n = i_b
+        bos, eos = i_n * T, i_n * T + T
+    p_q = tl.make_block_ptr(q + (bos * HQ + i_hq) * K, (T, K), (HQ*K, 1), (i_t * BT, 0), (BT, BK), (1, 0))
+    p_dq = tl.make_block_ptr(dq + (bos * HQ + i_hq) * K, (T, K), (HQ*K, 1), (i_t * BT, 0), (BT, BK), (1, 0))
+    p_do = tl.make_block_ptr(do + (bos * HQ + i_hq) * V, (T, V), (HQ*V, 1), (i_t * BT, i_v * BV), (BT, BV), (1, 0))
+    p_lse = tl.make_block_ptr(lse + bos * HQ + i_hq, (T,), (HQ,), (i_t * BT,), (BT,), (0,))
+    p_delta = tl.make_block_ptr(delta + bos * HQ + i_hq, (T,), (HQ,), (i_t * BT,), (BT,), (0,))
+    # [BT, BK]
+    b_q = tl.load(p_q, boundary_check=(0, 1))
+    b_q = (b_q * scale).to(b_q.dtype)
+    # [BT, BV]
+    b_do = tl.load(p_do, boundary_check=(0, 1))
+    # [BT]
+    b_lse = tl.load(p_lse, boundary_check=(0,))
+    b_delta = tl.load(p_delta, boundary_check=(0,))
+    # [BT, BK]
+    b_dq = tl.zeros([BT, BK], dtype=tl.float32)
+    for i_s in range(0, i_t * BT, BS):
+        p_k = tl.make_block_ptr(k + (bos * H + i_h) * K, (K, T), (1, H*K), (0, i_s), (BK, BS), (0, 1))
+        p_v = tl.make_block_ptr(v + (bos * H + i_h) * V, (V, T), (1, H*V), (i_v * BV, i_s), (BV, BS), (0, 1))
+        # [BK, BS]
+        b_k = tl.load(p_k, boundary_check=(0, 1))
+        # [BV, BS]
+        b_v = tl.load(p_v, boundary_check=(0, 1))
+        # [BT, BS]
+        b_s = tl.dot(b_q, b_k)
+        # b_p = exp(b_s - b_lse[:, None])
+        # b_p = exp(tl.where(b_s < 0, float('-inf'), b_s - b_lse[:, None]))
+        b_p = tl.where(b_s < 0, 0.0, exp(b_s - b_lse[:, None]))
+        # [BT, BV] @ [BV, BS] -> [BT, BS]
+        b_dp = tl.dot(b_do, b_v)
+        b_ds = b_p * (b_dp.to(tl.float32) - b_delta[:, None])
+        # [BT, BS] @ [BS, BK] -> [BT, BK]
+        b_dq += tl.dot(b_ds.to(b_k.dtype), tl.trans(b_k))
+    # [BT]
+    o_q = i_t * BT + tl.arange(0, BT)
+    for i_s in range(i_t * BT, min((i_t + 1) * BT, T), BS):
+        p_k = tl.make_block_ptr(k + (bos * H + i_h) * K, (K, T), (1, H*K), (0, i_s), (BK, BS), (0, 1))
+        p_v = tl.make_block_ptr(v + (bos * H + i_h) * V, (V, T), (1, H*V), (i_v * BV, i_s), (BV, BS), (0, 1))
+        # [BS]
+        o_k = i_s + tl.arange(0, BS)
+        # [BK, BS]
+        b_k = tl.load(p_k, boundary_check=(0, 1))
+        # [BV, BS]
+        b_v = tl.load(p_v, boundary_check=(0, 1))
+        # [BT, BS]
+        b_s = tl.dot(b_q, b_k)
+        # b_p = exp(b_s - b_lse[:, None])
+        # b_p = exp(tl.where(b_s < 0, float('-inf'), b_s - b_lse[:, None]))
+        b_p = tl.where(b_s < 0, 0.0, exp(b_s - b_lse[:, None]))
+        b_p = tl.where(o_q[:, None] >= o_k[None, :], b_p, 0)
+        # [BT, BV] @ [BV, BS] -> [BT, BS]
+        b_dp = tl.dot(b_do, b_v)
+        b_ds = b_p * (b_dp.to(tl.float32) - b_delta[:, None])
+        # [BT, BS] @ [BS, BK] -> [BT, BK]
+        b_dq += tl.dot(b_ds.to(b_k.dtype), tl.trans(b_k))
+    b_dq *= scale
+    tl.store(p_dq, b_dq.to(p_dq.dtype.element_ty), boundary_check=(0, 1))
+@triton.heuristics({
+    'USE_OFFSETS': lambda args: args['offsets'] is not None
+})
+@triton.autotune(
+    configs=[
+        triton.Config({}, num_warps=num_warps, num_stages=num_stages)
+        for num_warps in [1, 2, 4] + ([8] if check_shared_mem('hopper') else [])
+        for num_stages in [2, 3, 4, 5]
+    ],
+    key=['B', 'H', 'G', 'K', 'V', 'BK', 'BV'],
+)
+@triton.jit(do_not_specialize=['T'])
+def parallel_rect_attn_bwd_kernel_dkv(
+    q,
+    k,
+    v,
+    lse,
+    delta,
+    do,
+    dk,
+    dv,
+    offsets,
+    indices,
+    scale,
+    T,
+    B: tl.constexpr,
+    H: tl.constexpr,
+    HQ: tl.constexpr,
+    G: tl.constexpr,
+    K: tl.constexpr,
+    V: tl.constexpr,
+    BT: tl.constexpr,
+    BS: tl.constexpr,
+    BK: tl.constexpr,
+    BV: tl.constexpr,
+    USE_OFFSETS: tl.constexpr
+):
+    i_v, i_t, i_bh = tl.program_id(0), tl.program_id(1), tl.program_id(2)
+    i_b, i_hq = i_bh // HQ, i_bh % HQ
+    i_h = i_hq // G
+    if USE_OFFSETS:
+        i_n, i_t = tl.load(indices + i_t * 2).to(tl.int32), tl.load(indices + i_t * 2 + 1).to(tl.int32)
+        bos, eos = tl.load(offsets + i_n).to(tl.int32), tl.load(offsets + i_n + 1).to(tl.int32)
+        T = eos - bos
+    else:
+        i_n = i_b
+        bos, eos = i_n * T, i_n * T + T
+    p_k = tl.make_block_ptr(k + (bos * H + i_h) * K, (T, K), (H*K, 1), (i_t * BT, 0), (BT, BK), (1, 0))
+    p_v = tl.make_block_ptr(v + (bos * H + i_h) * V, (T, V), (H*V, 1), (i_t * BT, i_v * BV), (BT, BV), (1, 0))
+    p_dk = tl.make_block_ptr(dk + (bos * HQ + i_hq) * K, (T, K), (HQ*K, 1), (i_t * BT, 0), (BT, BK), (1, 0))
+    p_dv = tl.make_block_ptr(dv + (bos * HQ + i_hq) * V, (T, V), (HQ*V, 1), (i_t * BT, i_v * BV), (BT, BV), (1, 0))
+    # [BT, BK]
+    b_k = tl.load(p_k, boundary_check=(0, 1))
+    b_dk = tl.zeros([BT, BK], dtype=tl.float32)
+    # [BT, BV]
+    b_v = tl.load(p_v, boundary_check=(0, 1))
+    b_dv = tl.zeros([BT, BV], dtype=tl.float32)
+    o_k = i_t * BT + tl.arange(0, BT)
+    for i_s in range(i_t * BT, min((i_t + 1) * BT, T), BS):
+        p_q = tl.make_block_ptr(q + (bos * HQ + i_hq) * K, (T, K), (HQ*K, 1), (i_s, 0), (BS, BK), (1, 0))
+        p_do = tl.make_block_ptr(do + (bos * HQ + i_hq) * V, (T, V), (HQ*V, 1), (i_s, i_v * BV), (BS, BV), (1, 0))
+        p_lse = tl.make_block_ptr(lse + bos * HQ + i_hq, (T,), (HQ,), (i_s,), (BS,), (0,))
+        p_delta = tl.make_block_ptr(delta + bos * HQ + i_hq, (T,), (HQ,), (i_s,), (BS,), (0,))
+        # [BS]
+        o_q = i_s + tl.arange(0, BS)
+        # [BS, BK]
+        b_q = tl.load(p_q, boundary_check=(0, 1))
+        b_q = (b_q * scale).to(b_q.dtype)
+        # [BS, BV]
+        b_do = tl.load(p_do, boundary_check=(0, 1))
+        # [BS]
+        b_lse = tl.load(p_lse, boundary_check=(0,))
+        b_delta = tl.load(p_delta, boundary_check=(0,))
+        # [BT, BS]
+        b_s = tl.dot(b_k, tl.trans(b_q))
+        # b_p = exp(b_s - b_lse[None, :])
+        # b_p = exp(tl.where(b_s < 0, float('-inf'), b_s - b_lse[None, :]))
+        b_p = tl.where(b_s < 0, 0.0, exp(b_s - b_lse[None, :]))
+        b_p = tl.where(o_k[:, None] <= o_q[None, :], b_p, 0)
+        # [BT, BS] @ [BS, BV] -> [BT, BV]
+        b_dv += tl.dot(b_p.to(b_do.dtype), b_do)
+        # [BT, BV] @ [BV, BS] -> [BT, BS]
+        b_dp = tl.dot(b_v, tl.trans(b_do))
+        # [BT, BS]
+        b_ds = b_p * (b_dp - b_delta[None, :])
+        # [BT, BS] @ [BS, BK] -> [BT, BK]
+        b_dk += tl.dot(b_ds.to(b_q.dtype), b_q)
+    for i_s in range((i_t + 1) * BT, tl.cdiv(T, BS) * BS, BS):
+        p_q = tl.make_block_ptr(q + (bos * HQ + i_hq) * K, (T, K), (HQ*K, 1), (i_s, 0), (BS, BK), (1, 0))
+        p_do = tl.make_block_ptr(do + (bos * HQ + i_hq) * V, (T, V), (HQ*V, 1), (i_s, i_v * BV), (BS, BV), (1, 0))
+        p_lse = tl.make_block_ptr(lse + bos * HQ + i_hq, (T,), (HQ,), (i_s,), (BS,), (0,))
+        p_delta = tl.make_block_ptr(delta + bos * HQ + i_hq, (T,), (HQ,), (i_s,), (BS,), (0,))
+        # [BS]
+        o_q = i_s + tl.arange(0, BS)
+        # [BS, BK]
+        b_q = tl.load(p_q, boundary_check=(0, 1))
+        b_q = (b_q * scale).to(b_q.dtype)
+        # [BS, BV]
+        b_do = tl.load(p_do, boundary_check=(0, 1))
+        # [BS]
+        b_lse = tl.load(p_lse, boundary_check=(0,))
+        b_delta = tl.load(p_delta, boundary_check=(0,))
+        # [BT, BS]
+        b_s = tl.dot(b_k, tl.trans(b_q))
+        # b_p = exp(b_s - b_lse[None, :])
+        # b_p = exp(tl.where(b_s < 0, float('-inf'), b_s - b_lse[None, :]))
+        b_p = tl.where(b_s < 0, 0.0, exp(b_s - b_lse[None, :]))
+        # [BT, BS] @ [BS, BV] -> [BT, BV]
+        b_dv += tl.dot(b_p.to(b_do.dtype), b_do)
+        # [BT, BV] @ [BV, BS] -> [BT, BS]
+        b_dp = tl.dot(b_v, tl.trans(b_do))
+        # [BT, BS]
+        b_ds = b_p * (b_dp - b_delta[None, :])
+        # [BT, BS] @ [BS, BK] -> [BT, BK]
+        b_dk += tl.dot(b_ds.to(b_q.dtype), b_q)
+    tl.store(p_dk, b_dk.to(p_dk.dtype.element_ty), boundary_check=(0, 1))
+    tl.store(p_dv, b_dv.to(p_dv.dtype.element_ty), boundary_check=(0, 1))
+def parallel_rect_attn_fwd(
+    q: torch.Tensor,
+    k: torch.Tensor,
+    v: torch.Tensor,
+    scale: float,
+    chunk_size: int = 128,
+    offsets: Optional[torch.LongTensor] = None,
+    indices: Optional[torch.LongTensor] = None,
+):
+    B, T, H, K, V = *k.shape, v.shape[-1]
+    HQ = q.shape[2]
+    G = HQ // H
+    BT = chunk_size
+    if check_shared_mem('hopper', q.device.index):
+        BS = min(64, max(16, triton.next_power_of_2(T)))
+        BK = min(256, max(16, triton.next_power_of_2(K)))
+        BV = min(256, max(16, triton.next_power_of_2(V)))
+    elif check_shared_mem('ampere', q.device.index):
+        BS = min(32, max(16, triton.next_power_of_2(T)))
+        BK = min(256, max(16, triton.next_power_of_2(K)))
+        BV = min(128, max(16, triton.next_power_of_2(V)))
+    else:
+        BS = min(32, max(16, triton.next_power_of_2(T)))
+        BK = min(256, max(16, triton.next_power_of_2(K)))
+        BV = min(64, max(16, triton.next_power_of_2(V)))
+    NK = triton.cdiv(K, BK)
+    NV = triton.cdiv(V, BV)
+    NT = triton.cdiv(T, BT) if offsets is None else len(indices)
+    assert NK == 1, "The key dimension can not be larger than 256"
+    o = torch.empty(B, T, HQ, V, dtype=v.dtype, device=q.device)
+    lse = torch.empty(B, T, HQ, dtype=torch.float, device=q.device)
+    grid = (NV, NT, B * HQ)
+    parallel_rect_attn_fwd_kernel[grid](
+        q=q,
+        k=k,
+        v=v,
+        o=o,
+        lse=lse,
+        scale=scale,
+        offsets=offsets,
+        indices=indices,
+        B=B,
+        T=T,
+        H=H,
+        HQ=HQ,
+        G=G,
+        K=K,
+        V=V,
+        BT=BT,
+        BS=BS,
+        BK=BK,
+        BV=BV,
+    )
+    return o, lse
+def parallel_rect_attn_bwd_preprocess(
+    o: torch.Tensor,
+    do: torch.Tensor
+):
+    V = o.shape[-1]
+    delta = torch.empty_like(o[..., 0], dtype=torch.float32)
+    parallel_rect_attn_bwd_kernel_preprocess[(delta.numel(),)](
+        o=o,
+        do=do,
+        delta=delta,
+        B=triton.next_power_of_2(V),
+        V=V,
+    )
+    return delta
+def parallel_rect_attn_bwd(
+    q: torch.Tensor,
+    k: torch.Tensor,
+    v: torch.Tensor,
+    o: torch.Tensor,
+    lse: torch.Tensor,
+    do: torch.Tensor,
+    scale: float = None,
+    chunk_size: int = 128,
+    offsets: Optional[torch.LongTensor] = None,
+    indices: Optional[torch.LongTensor] = None,
+):
+    B, T, H, K, V = *k.shape, v.shape[-1]
+    HQ = q.shape[2]
+    G = HQ // H
+    BT = chunk_size
+    BS = max(16, triton.next_power_of_2(T))
+    BS = min(32, BS) if check_shared_mem('ampere') else min(16, BS)
+    BK = max(16, triton.next_power_of_2(K))
+    BV = max(16, triton.next_power_of_2(V))
+    NV = triton.cdiv(V, BV)
+    NT = triton.cdiv(T, BT) if offsets is None else len(indices)
+    delta = parallel_rect_attn_bwd_preprocess(o, do)
+    dq = torch.empty(B, T, HQ, K, dtype=k.dtype if H == HQ else torch.float, device=q.device)
+    dk = torch.empty(B, T, HQ, K, dtype=k.dtype if H == HQ else torch.float, device=q.device)
+    dv = torch.empty(B, T, HQ, V, dtype=v.dtype if H == HQ else torch.float, device=q.device)
+    grid = (NV, NT, B * HQ)
+    parallel_rect_attn_bwd_kernel_dq[grid](
+        q=q,
+        k=k,
+        v=v,
+        lse=lse,
+        delta=delta,
+        do=do,
+        dq=dq,
+        offsets=offsets,
+        indices=indices,
+        scale=scale,
+        T=T,
+        B=B,
+        H=H,
+        HQ=HQ,
+        G=G,
+        K=K,
+        V=V,
+        BT=BT,
+        BS=BS,
+        BK=BK,
+        BV=BV
+    )
+    parallel_rect_attn_bwd_kernel_dkv[grid](
+        q=q,
+        k=k,
+        v=v,
+        lse=lse,
+        delta=delta,
+        do=do,
+        dk=dk,
+        dv=dv,
+        offsets=offsets,
+        indices=indices,
+        scale=scale,
+        T=T,
+        B=B,
+        H=H,
+        HQ=HQ,
+        G=G,
+        K=K,
+        V=V,
+        BT=BT,
+        BS=BS,
+        BK=BK,
+        BV=BV
+    )
+    dk = reduce(dk, 'b t (h g) k -> b t h k', g=G, reduction='sum')
+    dv = reduce(dv, 'b t (h g) v -> b t h v', g=G, reduction='sum')
+    return dq, dk, dv
+@torch.compile
+class ParallelRectifiedAttentionFunction(torch.autograd.Function):
+    @staticmethod
+    @contiguous
+    @autocast_custom_fwd
+    def forward(ctx, q, k, v, scale, offsets):
+        ctx.dtype = q.dtype
+        chunk_size = min(128, max(16, triton.next_power_of_2(q.shape[1])))
+        # 2-d indices denoting the offsets of chunks in each sequence
+        # for example, if the passed `offsets` is [0, 100, 356] and `chunk_size` is 64,
+        # then there are 2 and 4 chunks in the 1st and 2nd sequences respectively, and `indices` will be
+        # [[0, 0], [0, 1], [1, 0], [1, 1], [1, 2], [1, 3]]
+        indices = prepare_chunk_indices(offsets, chunk_size) if offsets is not None else None
+        o, lse = parallel_rect_attn_fwd(
+            q=q,
+            k=k,
+            v=v,
+            scale=scale,
+            chunk_size=chunk_size,
+            offsets=offsets,
+            indices=indices
+        )
+        ctx.save_for_backward(q, k, v, o, lse)
+        ctx.chunk_size = chunk_size
+        ctx.offsets = offsets
+        ctx.indices = indices
+        ctx.scale = scale
+        return o.to(q.dtype)
+    @staticmethod
+    @contiguous
+    @autocast_custom_bwd
+    def backward(ctx, do):
+        q, k, v, o, lse = ctx.saved_tensors
+        dq, dk, dv = parallel_rect_attn_bwd(
+            q=q,
+            k=k,
+            v=v,
+            o=o,
+            lse=lse,
+            do=do,
+            scale=ctx.scale,
+            chunk_size=ctx.chunk_size,
+            offsets=ctx.offsets,
+            indices=ctx.indices
+        )
+        return dq.to(q), dk.to(k), dv.to(v), None, None, None, None, None, None, None, None
+def parallel_rectified_attn(
+    q: torch.Tensor,
+    k: torch.Tensor,
+    v: torch.Tensor,
+    scale: Optional[float] = None,
+    cu_seqlens: Optional[torch.LongTensor] = None,
+    head_first: bool = False
+) -> torch.Tensor:
+    r"""
+    Args:
+        q (torch.Tensor):
+            queries of shape `[B, T, HQ, K]` if `head_first=False` else `[B, HQ, T, K]`.
+        k (torch.Tensor):
+            keys of shape `[B, T, H, K]` if `head_first=False` else `[B, H, T, K]`.
+            GQA will be applied if HQ is divisible by H.
+        v (torch.Tensor):
+            values of shape `[B, T, H, V]` if `head_first=False` else `[B, H, T, V]`.
+        scale (Optional[int]):
+            Scale factor for attention scores.
+            If not provided, it will default to `1 / sqrt(K)`. Default: `None`.
+        cu_seqlens (torch.LongTensor):
+            Cumulative sequence lengths of shape `[N+1]` used for variable-length training,
+            consistent with the FlashAttention API.
+        head_first (Optional[bool]):
+            Whether the inputs are in the head-first format. Default: `False`.
+    Returns:
+        o (torch.Tensor):
+            Outputs of shape `[B, T, HQ, V]` if `head_first=False` else `[B, HQ, T, V]`.
+    """
+    if scale is None:
+        scale = k.shape[-1] ** -0.5
+    if cu_seqlens is not None:
+        assert q.shape[0] == 1, "batch size must be 1 when cu_seqlens are provided"
+    if head_first:
+        q, k, v = map(lambda x: rearrange(x, 'b h t d -> b t h d'), (q, k, v))
+    o = ParallelRectifiedAttentionFunction.apply(q, k, v, scale, cu_seqlens)
+    if head_first:
+        o = rearrange(o, 'b t h d -> b h t d')
+    return o

fla/ops/attn/parallel_softpick.py ADDED Viewed

	@@ -0,0 +1,650 @@

+# -*- coding: utf-8 -*-
+# Copyright (c) 2023-2025, Songlin Yang, Yu Zhang
+from typing import Optional
+import torch
+import triton
+import triton.language as tl
+from einops import rearrange, reduce
+from fla.ops.common.utils import prepare_chunk_indices
+from fla.ops.utils.op import exp, log
+from fla.utils import autocast_custom_bwd, autocast_custom_fwd, check_shared_mem, contiguous
+@triton.heuristics({
+    'USE_OFFSETS': lambda args: args['offsets'] is not None
+})
+@triton.autotune(
+    configs=[
+        triton.Config({}, num_warps=num_warps, num_stages=num_stages)
+        for num_warps in [1, 2, 4] + ([8] if check_shared_mem('hopper') else [])
+        for num_stages in [2, 3, 4, 5]
+    ],
+    key=['B', 'H', 'G', 'K', 'V', 'BK', 'BV'],
+)
+@triton.jit
+def parallel_softpick_attn_fwd_kernel(
+    q,
+    k,
+    v,
+    o,
+    lse,
+    scale,
+    offsets,
+    indices,
+    T,
+    B: tl.constexpr,
+    H: tl.constexpr,
+    HQ: tl.constexpr,
+    G: tl.constexpr,
+    K: tl.constexpr,
+    V: tl.constexpr,
+    BT: tl.constexpr,
+    BS: tl.constexpr,
+    BK: tl.constexpr,
+    BV: tl.constexpr,
+    USE_OFFSETS: tl.constexpr
+):
+    i_v, i_t, i_bh = tl.program_id(0), tl.program_id(1), tl.program_id(2)
+    i_b, i_hq = i_bh // HQ, i_bh % HQ
+    i_h = i_hq // G
+    if USE_OFFSETS:
+        i_n, i_t = tl.load(indices + i_t * 2).to(tl.int32), tl.load(indices + i_t * 2 + 1).to(tl.int32)
+        bos, eos = tl.load(offsets + i_n).to(tl.int32), tl.load(offsets + i_n + 1).to(tl.int32)
+        T = eos - bos
+    else:
+        i_n = i_b
+        bos, eos = i_n * T, i_n * T + T
+    p_q = tl.make_block_ptr(q + (bos * HQ + i_hq) * K, (T, K), (HQ*K, 1), (i_t * BT, 0), (BT, BK), (1, 0))
+    p_o = tl.make_block_ptr(o + (bos * HQ + i_hq) * V, (T, V), (HQ*V, 1), (i_t * BT, i_v * BV), (BT, BV), (1, 0))
+    p_lse = tl.make_block_ptr(lse + bos * HQ + i_hq, (T,), (HQ,), (i_t * BT,), (BT,), (0,))
+    # the Q block is kept in the shared memory throughout the whole kernel
+    # [BT, BK]
+    b_q = tl.load(p_q, boundary_check=(0, 1))
+    b_q = (b_q * scale).to(b_q.dtype)
+    # [BT, BV]
+    b_o = tl.zeros([BT, BV], dtype=tl.float32)
+    b_m = tl.full([BT], float('-inf'), dtype=tl.float32)
+    b_acc = tl.zeros([BT], dtype=tl.float32)
+    for i_s in range(0, i_t * BT, BS):
+        p_k = tl.make_block_ptr(k + (bos * H + i_h) * K, (K, T), (1, H*K), (0, i_s), (BK, BS), (0, 1))
+        p_v = tl.make_block_ptr(v + (bos * H + i_h) * V, (T, V), (H*V, 1), (i_s, i_v * BV), (BS, BV), (1, 0))
+        # [BK, BS]
+        b_k = tl.load(p_k, boundary_check=(0, 1))
+        # [BS, BV]
+        b_v = tl.load(p_v, boundary_check=(0, 1))
+        # [BT, BS]
+        b_s = tl.dot(b_q, b_k)
+        # [BT, BS]
+        b_m, b_mp = tl.maximum(b_m, tl.max(b_s, 1)), b_m
+        b_r = exp(b_mp - b_m)
+        # [BT, BS]
+        b_p = exp(b_s - b_m[:, None]) - exp(-b_m[:, None])
+        b_p_r = tl.maximum(b_p, 0.0)
+        b_p_a = tl.abs(b_p)
+        # [BT]
+        b_acc = b_acc * b_r + tl.sum(b_p_a, 1)
+        # [BT, BV]
+        b_o = b_o * b_r[:, None] + tl.dot(b_p_r.to(b_q.dtype), b_v)
+        b_mp = b_m
+    # [BT]
+    o_q = i_t * BT + tl.arange(0, BT)
+    for i_s in range(i_t * BT, min((i_t + 1) * BT, T), BS):
+        p_k = tl.make_block_ptr(k + (bos * H + i_h) * K, (K, T), (1, H*K), (0, i_s), (BK, BS), (0, 1))
+        p_v = tl.make_block_ptr(v + (bos * H + i_h) * V, (T, V), (H*V, 1), (i_s, i_v * BV), (BS, BV), (1, 0))
+        # [BS]
+        o_k = i_s + tl.arange(0, BS)
+        # [BK, BS]
+        b_k = tl.load(p_k, boundary_check=(0, 1))
+        # [BS, BV]
+        b_v = tl.load(p_v, boundary_check=(0, 1))
+        # [BT, BS]
+        b_s = tl.dot(b_q, b_k)
+        b_s = tl.where(o_q[:, None] >= o_k[None, :], b_s, float('-inf'))
+        # [BT]
+        b_m, b_mp = tl.maximum(b_m, tl.max(b_s, 1)), b_m
+        b_r = exp(b_mp - b_m)
+        # [BT, BS]
+        b_p = exp(b_s - b_m[:, None]) - exp(-b_m[:, None])
+        b_p_r = tl.maximum(b_p, 0.0)
+        b_p_a = tl.abs(b_p)
+        b_p_a = tl.where(o_q[:, None] >= o_k[None, :], b_p_a, 0)
+        # [BT]
+        b_acc = b_acc * b_r + tl.sum(b_p_a, 1)
+        # [BT, BV]
+        b_o = b_o * b_r[:, None] + tl.dot(b_p_r.to(b_q.dtype), b_v)
+        b_mp = b_m
+    b_acc += 1e-6 # harcoded epsilon... sorry
+    b_o = b_o / b_acc[:, None]
+    b_m += log(b_acc)
+    tl.store(p_o, b_o.to(p_o.dtype.element_ty), boundary_check=(0, 1))
+    tl.store(p_lse, b_m.to(p_lse.dtype.element_ty), boundary_check=(0,))
+@triton.jit
+def parallel_softpick_attn_bwd_kernel_preprocess(
+    o,
+    do,
+    delta,
+    B: tl.constexpr,
+    V: tl.constexpr
+):
+    i_n = tl.program_id(0)
+    o_d = tl.arange(0, B)
+    m_d = o_d < V
+    b_o = tl.load(o + i_n * V + o_d, mask=m_d, other=0)
+    b_do = tl.load(do + i_n * V + o_d, mask=m_d, other=0).to(tl.float32)
+    b_delta = tl.sum(b_o * b_do)
+    tl.store(delta + i_n, b_delta.to(delta.dtype.element_ty))
+@triton.heuristics({
+    'USE_OFFSETS': lambda args: args['offsets'] is not None
+})
+@triton.autotune(
+    configs=[
+        triton.Config({}, num_warps=num_warps, num_stages=num_stages)
+        for num_warps in [1, 2, 4] + ([8] if check_shared_mem('hopper') else [])
+        for num_stages in [2, 3, 4, 5]
+    ],
+    key=['B', 'H', 'G', 'K', 'V', 'BK', 'BV'],
+)
+@triton.jit(do_not_specialize=['T'])
+def parallel_softpick_attn_bwd_kernel_dq(
+    q,
+    k,
+    v,
+    lse,
+    delta,
+    do,
+    dq,
+    scale,
+    offsets,
+    indices,
+    T,
+    B: tl.constexpr,
+    H: tl.constexpr,
+    HQ: tl.constexpr,
+    G: tl.constexpr,
+    K: tl.constexpr,
+    V: tl.constexpr,
+    BT: tl.constexpr,
+    BS: tl.constexpr,
+    BK: tl.constexpr,
+    BV: tl.constexpr,
+    USE_OFFSETS: tl.constexpr
+):
+    i_v, i_t, i_bh = tl.program_id(0), tl.program_id(1), tl.program_id(2)
+    i_b, i_hq = i_bh // HQ, i_bh % HQ
+    i_h = i_hq // G
+    if USE_OFFSETS:
+        i_n, i_t = tl.load(indices + i_t * 2).to(tl.int32), tl.load(indices + i_t * 2 + 1).to(tl.int32)
+        bos, eos = tl.load(offsets + i_n).to(tl.int32), tl.load(offsets + i_n + 1).to(tl.int32)
+        T = eos - bos
+    else:
+        i_n = i_b
+        bos, eos = i_n * T, i_n * T + T
+    p_q = tl.make_block_ptr(q + (bos * HQ + i_hq) * K, (T, K), (HQ*K, 1), (i_t * BT, 0), (BT, BK), (1, 0))
+    p_dq = tl.make_block_ptr(dq + (bos * HQ + i_hq) * K, (T, K), (HQ*K, 1), (i_t * BT, 0), (BT, BK), (1, 0))
+    p_do = tl.make_block_ptr(do + (bos * HQ + i_hq) * V, (T, V), (HQ*V, 1), (i_t * BT, i_v * BV), (BT, BV), (1, 0))
+    p_lse = tl.make_block_ptr(lse + bos * HQ + i_hq, (T,), (HQ,), (i_t * BT,), (BT,), (0,))
+    p_delta = tl.make_block_ptr(delta + bos * HQ + i_hq, (T,), (HQ,), (i_t * BT,), (BT,), (0,))
+    # [BT, BK]
+    b_q = tl.load(p_q, boundary_check=(0, 1))
+    b_q = (b_q * scale).to(b_q.dtype)
+    # [BT, BV]
+    b_do = tl.load(p_do, boundary_check=(0, 1))
+    # [BT]
+    b_lse = tl.load(p_lse, boundary_check=(0,))
+    b_delta = tl.load(p_delta, boundary_check=(0,))
+    # [BT, BK]
+    b_dq = tl.zeros([BT, BK], dtype=tl.float32)
+    for i_s in range(0, i_t * BT, BS):
+        p_k = tl.make_block_ptr(k + (bos * H + i_h) * K, (K, T), (1, H*K), (0, i_s), (BK, BS), (0, 1))
+        p_v = tl.make_block_ptr(v + (bos * H + i_h) * V, (V, T), (1, H*V), (i_v * BV, i_s), (BV, BS), (0, 1))
+        # [BK, BS]
+        b_k = tl.load(p_k, boundary_check=(0, 1))
+        # [BV, BS]
+        b_v = tl.load(p_v, boundary_check=(0, 1))
+        # [BT, BS]
+        b_s = tl.dot(b_q, b_k)
+        b_e = exp(b_s - b_lse[:, None])
+        # [BT, BV] @ [BV, BS] -> [BT, BS]
+        b_dp = tl.dot(b_do, b_v)
+        # [BT, BS]
+        b_step = tl.where(b_s > 0, b_dp, 0)
+        b_sign = tl.where(b_s > 0, b_delta[:, None], -b_delta[:, None])
+        b_ds = b_e * (b_step.to(tl.float32) - b_sign)
+        # [BT, BS] @ [BS, BK] -> [BT, BK]
+        b_dq += tl.dot(b_ds.to(b_k.dtype), tl.trans(b_k))
+    # [BT]
+    o_q = i_t * BT + tl.arange(0, BT)
+    for i_s in range(i_t * BT, min((i_t + 1) * BT, T), BS):
+        p_k = tl.make_block_ptr(k + (bos * H + i_h) * K, (K, T), (1, H*K), (0, i_s), (BK, BS), (0, 1))
+        p_v = tl.make_block_ptr(v + (bos * H + i_h) * V, (V, T), (1, H*V), (i_v * BV, i_s), (BV, BS), (0, 1))
+        # [BS]
+        o_k = i_s + tl.arange(0, BS)
+        # [BK, BS]
+        b_k = tl.load(p_k, boundary_check=(0, 1))
+        # [BV, BS]
+        b_v = tl.load(p_v, boundary_check=(0, 1))
+        # [BT, BS]
+        b_s = tl.dot(b_q, b_k)
+        b_e = exp(b_s - b_lse[:, None])
+        # [BT, BV] @ [BV, BS] -> [BT, BS]
+        b_dp = tl.dot(b_do, b_v)
+        # [BT, BS]
+        b_e = tl.where(o_q[:, None] >= o_k[None, :], b_e, 0)
+        b_step = tl.where(b_s > 0, b_dp, 0)
+        b_sign = tl.where(b_s > 0, b_delta[:, None], -b_delta[:, None])
+        b_ds = b_e * (b_step.to(tl.float32) - b_sign)
+        # [BT, BS] @ [BS, BK] -> [BT, BK]
+        b_dq += tl.dot(b_ds.to(b_k.dtype), tl.trans(b_k))
+    b_dq *= scale
+    tl.store(p_dq, b_dq.to(p_dq.dtype.element_ty), boundary_check=(0, 1))
+@triton.heuristics({
+    'USE_OFFSETS': lambda args: args['offsets'] is not None
+})
+@triton.autotune(
+    configs=[
+        triton.Config({}, num_warps=num_warps, num_stages=num_stages)
+        for num_warps in [1, 2, 4] + ([8] if check_shared_mem('hopper') else [])
+        for num_stages in [2, 3, 4, 5]
+    ],
+    key=['B', 'H', 'G', 'K', 'V', 'BK', 'BV'],
+)
+@triton.jit(do_not_specialize=['T'])
+def parallel_softpick_attn_bwd_kernel_dkv(
+    q,
+    k,
+    v,
+    lse,
+    delta,
+    do,
+    dk,
+    dv,
+    offsets,
+    indices,
+    scale,
+    T,
+    B: tl.constexpr,
+    H: tl.constexpr,
+    HQ: tl.constexpr,
+    G: tl.constexpr,
+    K: tl.constexpr,
+    V: tl.constexpr,
+    BT: tl.constexpr,
+    BS: tl.constexpr,
+    BK: tl.constexpr,
+    BV: tl.constexpr,
+    USE_OFFSETS: tl.constexpr
+):
+    i_v, i_t, i_bh = tl.program_id(0), tl.program_id(1), tl.program_id(2)
+    i_b, i_hq = i_bh // HQ, i_bh % HQ
+    i_h = i_hq // G
+    if USE_OFFSETS:
+        i_n, i_t = tl.load(indices + i_t * 2).to(tl.int32), tl.load(indices + i_t * 2 + 1).to(tl.int32)
+        bos, eos = tl.load(offsets + i_n).to(tl.int32), tl.load(offsets + i_n + 1).to(tl.int32)
+        T = eos - bos
+    else:
+        i_n = i_b
+        bos, eos = i_n * T, i_n * T + T
+    p_k = tl.make_block_ptr(k + (bos * H + i_h) * K, (T, K), (H*K, 1), (i_t * BT, 0), (BT, BK), (1, 0))
+    p_v = tl.make_block_ptr(v + (bos * H + i_h) * V, (T, V), (H*V, 1), (i_t * BT, i_v * BV), (BT, BV), (1, 0))
+    p_dk = tl.make_block_ptr(dk + (bos * HQ + i_hq) * K, (T, K), (HQ*K, 1), (i_t * BT, 0), (BT, BK), (1, 0))
+    p_dv = tl.make_block_ptr(dv + (bos * HQ + i_hq) * V, (T, V), (HQ*V, 1), (i_t * BT, i_v * BV), (BT, BV), (1, 0))
+    # [BT, BK]
+    b_k = tl.load(p_k, boundary_check=(0, 1))
+    b_dk = tl.zeros([BT, BK], dtype=tl.float32)
+    # [BT, BV]
+    b_v = tl.load(p_v, boundary_check=(0, 1))
+    b_dv = tl.zeros([BT, BV], dtype=tl.float32)
+    o_k = i_t * BT + tl.arange(0, BT)
+    for i_s in range(i_t * BT, min((i_t + 1) * BT, T), BS):
+        p_q = tl.make_block_ptr(q + (bos * HQ + i_hq) * K, (T, K), (HQ*K, 1), (i_s, 0), (BS, BK), (1, 0))
+        p_do = tl.make_block_ptr(do + (bos * HQ + i_hq) * V, (T, V), (HQ*V, 1), (i_s, i_v * BV), (BS, BV), (1, 0))
+        p_lse = tl.make_block_ptr(lse + bos * HQ + i_hq, (T,), (HQ,), (i_s,), (BS,), (0,))
+        p_delta = tl.make_block_ptr(delta + bos * HQ + i_hq, (T,), (HQ,), (i_s,), (BS,), (0,))
+        # [BS]
+        o_q = i_s + tl.arange(0, BS)
+        # [BS, BK]
+        b_q = tl.load(p_q, boundary_check=(0, 1))
+        b_q = (b_q * scale).to(b_q.dtype)
+        # [BS, BV]
+        b_do = tl.load(p_do, boundary_check=(0, 1))
+        # [BS]
+        b_lse = tl.load(p_lse, boundary_check=(0,))
+        b_delta = tl.load(p_delta, boundary_check=(0,))
+        # [BT, BS]
+        b_s = tl.dot(b_k, tl.trans(b_q))
+        b_e = exp(b_s - b_lse[None, :])
+        b_p = b_e - exp(-b_lse[None, :])
+        b_p_r = tl.maximum(b_p, 0.0)
+        b_p_r = tl.where(o_k[:, None] <= o_q[None, :], b_p_r, 0)
+        # [BT, BS] @ [BS, BV] -> [BT, BV]
+        b_dv += tl.dot(b_p_r.to(b_do.dtype), b_do)
+        # [BT, BV] @ [BV, BS] -> [BT, BS]
+        b_dp = tl.dot(b_v, tl.trans(b_do))
+        # [BT, BS]
+        b_e = tl.where(o_k[:, None] <= o_q[None, :], b_e, 0)
+        b_step = tl.where(b_s > 0, b_dp, 0)
+        b_sign = tl.where(b_s > 0, b_delta[None, :], -b_delta[None, :])
+        b_ds = b_e * (b_step - b_sign)
+        # [BT, BS] @ [BS, BK] -> [BT, BK]
+        b_dk += tl.dot(b_ds.to(b_q.dtype), b_q)
+    for i_s in range((i_t + 1) * BT, tl.cdiv(T, BS) * BS, BS):
+        p_q = tl.make_block_ptr(q + (bos * HQ + i_hq) * K, (T, K), (HQ*K, 1), (i_s, 0), (BS, BK), (1, 0))
+        p_do = tl.make_block_ptr(do + (bos * HQ + i_hq) * V, (T, V), (HQ*V, 1), (i_s, i_v * BV), (BS, BV), (1, 0))
+        p_lse = tl.make_block_ptr(lse + bos * HQ + i_hq, (T,), (HQ,), (i_s,), (BS,), (0,))
+        p_delta = tl.make_block_ptr(delta + bos * HQ + i_hq, (T,), (HQ,), (i_s,), (BS,), (0,))
+        # [BS]
+        o_q = i_s + tl.arange(0, BS)
+        # [BS, BK]
+        b_q = tl.load(p_q, boundary_check=(0, 1))
+        b_q = (b_q * scale).to(b_q.dtype)
+        # [BS, BV]
+        b_do = tl.load(p_do, boundary_check=(0, 1))
+        # [BS]
+        b_lse = tl.load(p_lse, boundary_check=(0,))
+        b_delta = tl.load(p_delta, boundary_check=(0,))
+        # [BT, BS]
+        b_s = tl.dot(b_k, tl.trans(b_q))
+        b_e = exp(b_s - b_lse[None, :])
+        b_p = b_e - exp(-b_lse[None, :])
+        b_p_r = tl.maximum(b_p, 0.0)
+        # [BT, BS] @ [BS, BV] -> [BT, BV]
+        b_dv += tl.dot(b_p_r.to(b_do.dtype), b_do)
+        # [BT, BV] @ [BV, BS] -> [BT, BS]
+        b_dp = tl.dot(b_v, tl.trans(b_do))
+        # [BT, BS]
+        b_step = tl.where(b_s > 0, b_dp, 0)
+        b_sign = tl.where(b_s > 0, b_delta[None, :], -b_delta[None, :])
+        b_ds = b_e * (b_step - b_sign)
+        # [BT, BS] @ [BS, BK] -> [BT, BK]
+        b_dk += tl.dot(b_ds.to(b_q.dtype), b_q)
+    tl.store(p_dk, b_dk.to(p_dk.dtype.element_ty), boundary_check=(0, 1))
+    tl.store(p_dv, b_dv.to(p_dv.dtype.element_ty), boundary_check=(0, 1))
+def parallel_softpick_attn_fwd(
+    q: torch.Tensor,
+    k: torch.Tensor,
+    v: torch.Tensor,
+    scale: float,
+    chunk_size: int = 128,
+    offsets: Optional[torch.LongTensor] = None,
+    indices: Optional[torch.LongTensor] = None,
+):
+    B, T, H, K, V = *k.shape, v.shape[-1]
+    HQ = q.shape[2]
+    G = HQ // H
+    BT = chunk_size
+    if check_shared_mem('hopper', q.device.index):
+        BS = min(64, max(16, triton.next_power_of_2(T)))
+        BK = min(256, max(16, triton.next_power_of_2(K)))
+        BV = min(256, max(16, triton.next_power_of_2(V)))
+    elif check_shared_mem('ampere', q.device.index):
+        BS = min(32, max(16, triton.next_power_of_2(T)))
+        BK = min(256, max(16, triton.next_power_of_2(K)))
+        BV = min(128, max(16, triton.next_power_of_2(V)))
+    else:
+        BS = min(32, max(16, triton.next_power_of_2(T)))
+        BK = min(256, max(16, triton.next_power_of_2(K)))
+        BV = min(64, max(16, triton.next_power_of_2(V)))
+    NK = triton.cdiv(K, BK)
+    NV = triton.cdiv(V, BV)
+    NT = triton.cdiv(T, BT) if offsets is None else len(indices)
+    assert NK == 1, "The key dimension can not be larger than 256"
+    o = torch.empty(B, T, HQ, V, dtype=v.dtype, device=q.device)
+    lse = torch.empty(B, T, HQ, dtype=torch.float, device=q.device)
+    grid = (NV, NT, B * HQ)
+    parallel_softpick_attn_fwd_kernel[grid](
+        q=q,
+        k=k,
+        v=v,
+        o=o,
+        lse=lse,
+        scale=scale,
+        offsets=offsets,
+        indices=indices,
+        B=B,
+        T=T,
+        H=H,
+        HQ=HQ,
+        G=G,
+        K=K,
+        V=V,
+        BT=BT,
+        BS=BS,
+        BK=BK,
+        BV=BV,
+    )
+    return o, lse
+def parallel_softpick_attn_bwd_preprocess(
+    o: torch.Tensor,
+    do: torch.Tensor
+):
+    V = o.shape[-1]
+    delta = torch.empty_like(o[..., 0], dtype=torch.float32)
+    parallel_softpick_attn_bwd_kernel_preprocess[(delta.numel(),)](
+        o=o,
+        do=do,
+        delta=delta,
+        B=triton.next_power_of_2(V),
+        V=V,
+    )
+    return delta
+def parallel_softpick_attn_bwd(
+    q: torch.Tensor,
+    k: torch.Tensor,
+    v: torch.Tensor,
+    o: torch.Tensor,
+    lse: torch.Tensor,
+    do: torch.Tensor,
+    scale: float = None,
+    chunk_size: int = 128,
+    offsets: Optional[torch.LongTensor] = None,
+    indices: Optional[torch.LongTensor] = None,
+):
+    B, T, H, K, V = *k.shape, v.shape[-1]
+    HQ = q.shape[2]
+    G = HQ // H
+    BT = chunk_size
+    BS = max(16, triton.next_power_of_2(T))
+    BS = min(32, BS) if check_shared_mem('ampere') else min(16, BS)
+    BK = max(16, triton.next_power_of_2(K))
+    BV = max(16, triton.next_power_of_2(V))
+    NV = triton.cdiv(V, BV)
+    NT = triton.cdiv(T, BT) if offsets is None else len(indices)
+    delta = parallel_softpick_attn_bwd_preprocess(o, do)
+    dq = torch.empty(B, T, HQ, K, dtype=k.dtype if H == HQ else torch.float, device=q.device)
+    dk = torch.empty(B, T, HQ, K, dtype=k.dtype if H == HQ else torch.float, device=q.device)
+    dv = torch.empty(B, T, HQ, V, dtype=v.dtype if H == HQ else torch.float, device=q.device)
+    grid = (NV, NT, B * HQ)
+    parallel_softpick_attn_bwd_kernel_dq[grid](
+        q=q,
+        k=k,
+        v=v,
+        lse=lse,
+        delta=delta,
+        do=do,
+        dq=dq,
+        offsets=offsets,
+        indices=indices,
+        scale=scale,
+        T=T,
+        B=B,
+        H=H,
+        HQ=HQ,
+        G=G,
+        K=K,
+        V=V,
+        BT=BT,
+        BS=BS,
+        BK=BK,
+        BV=BV
+    )
+    parallel_softpick_attn_bwd_kernel_dkv[grid](
+        q=q,
+        k=k,
+        v=v,
+        lse=lse,
+        delta=delta,
+        do=do,
+        dk=dk,
+        dv=dv,
+        offsets=offsets,
+        indices=indices,
+        scale=scale,
+        T=T,
+        B=B,
+        H=H,
+        HQ=HQ,
+        G=G,
+        K=K,
+        V=V,
+        BT=BT,
+        BS=BS,
+        BK=BK,
+        BV=BV
+    )
+    dk = reduce(dk, 'b t (h g) k -> b t h k', g=G, reduction='sum')
+    dv = reduce(dv, 'b t (h g) v -> b t h v', g=G, reduction='sum')
+    return dq, dk, dv
+@torch.compile
+class ParallelSoftpickAttentionFunction(torch.autograd.Function):
+    @staticmethod
+    @contiguous
+    @autocast_custom_fwd
+    def forward(ctx, q, k, v, scale, offsets):
+        ctx.dtype = q.dtype
+        chunk_size = min(128, max(16, triton.next_power_of_2(q.shape[1])))
+        # 2-d indices denoting the offsets of chunks in each sequence
+        # for example, if the passed `offsets` is [0, 100, 356] and `chunk_size` is 64,
+        # then there are 2 and 4 chunks in the 1st and 2nd sequences respectively, and `indices` will be
+        # [[0, 0], [0, 1], [1, 0], [1, 1], [1, 2], [1, 3]]
+        indices = prepare_chunk_indices(offsets, chunk_size) if offsets is not None else None
+        o, lse = parallel_softpick_attn_fwd(
+            q=q,
+            k=k,
+            v=v,
+            scale=scale,
+            chunk_size=chunk_size,
+            offsets=offsets,
+            indices=indices
+        )
+        ctx.save_for_backward(q, k, v, o, lse)
+        ctx.chunk_size = chunk_size
+        ctx.offsets = offsets
+        ctx.indices = indices
+        ctx.scale = scale
+        return o.to(q.dtype)
+    @staticmethod
+    @contiguous
+    @autocast_custom_bwd
+    def backward(ctx, do):
+        q, k, v, o, lse = ctx.saved_tensors
+        dq, dk, dv = parallel_softpick_attn_bwd(
+            q=q,
+            k=k,
+            v=v,
+            o=o,
+            lse=lse,
+            do=do,
+            scale=ctx.scale,
+            chunk_size=ctx.chunk_size,
+            offsets=ctx.offsets,
+            indices=ctx.indices
+        )
+        return dq.to(q), dk.to(k), dv.to(v), None, None, None, None, None, None, None, None
+def parallel_softpick_attn(
+    q: torch.Tensor,
+    k: torch.Tensor,
+    v: torch.Tensor,
+    scale: Optional[float] = None,
+    cu_seqlens: Optional[torch.LongTensor] = None,
+    head_first: bool = False
+) -> torch.Tensor:
+    r"""
+    Args:
+        q (torch.Tensor):
+            queries of shape `[B, T, HQ, K]` if `head_first=False` else `[B, HQ, T, K]`.
+        k (torch.Tensor):
+            keys of shape `[B, T, H, K]` if `head_first=False` else `[B, H, T, K]`.
+            GQA will be applied if HQ is divisible by H.
+        v (torch.Tensor):
+            values of shape `[B, T, H, V]` if `head_first=False` else `[B, H, T, V]`.
+        scale (Optional[int]):
+            Scale factor for attention scores.
+            If not provided, it will default to `1 / sqrt(K)`. Default: `None`.
+        cu_seqlens (torch.LongTensor):
+            Cumulative sequence lengths of shape `[N+1]` used for variable-length training,
+            consistent with the FlashAttention API.
+        head_first (Optional[bool]):
+            Whether the inputs are in the head-first format. Default: `False`.
+    Returns:
+        o (torch.Tensor):
+            Outputs of shape `[B, T, HQ, V]` if `head_first=False` else `[B, HQ, T, V]`.
+    """
+    if scale is None:
+        scale = k.shape[-1] ** -0.5
+    if cu_seqlens is not None:
+        assert q.shape[0] == 1, "batch size must be 1 when cu_seqlens are provided"
+    if head_first:
+        q, k, v = map(lambda x: rearrange(x, 'b h t d -> b t h d'), (q, k, v))
+    o = ParallelSoftpickAttentionFunction.apply(q, k, v, scale, cu_seqlens)
+    if head_first:
+        o = rearrange(o, 'b t h d -> b h t d')
+    return o

fla/ops/based/__pycache__/__init__.cpython-312.pyc ADDED Viewed

Binary file (289 Bytes). View file

fla/ops/based/__pycache__/parallel.cpython-312.pyc ADDED Viewed

Binary file (22.6 kB). View file

fla/ops/based/naive.py ADDED Viewed

	@@ -0,0 +1,72 @@

+# -*- coding: utf-8 -*-
+from typing import Optional
+import torch
+from einops import rearrange
+def naive_parallel_based(
+    q: torch.Tensor,
+    k: torch.Tensor,
+    v: torch.Tensor,
+    scale: Optional[float] = None,
+    use_norm: bool = True
+):
+    if scale is None:
+        scale = q.shape[-1] ** -0.5
+    q = q * scale
+    attn = q @ k.transpose(-2, -1)
+    attn = 1 + attn + 1/2 * (attn ** 2)
+    attn.masked_fill_(~torch.tril(torch.ones(
+        q.shape[-2], q.shape[-2], dtype=torch.bool, device=q.device)), 0)
+    o = attn @ v
+    if use_norm:
+        z = attn.sum(-1)
+        return o / (z[..., None] + 1e-6)
+    else:
+        return o
+def naive_chunk_based(q, k, v, chunk_size=256):
+    q = q * (q.shape[-1] ** -0.5)
+    # compute normalizer.
+    k_cumsum = torch.cumsum(k, dim=-2)
+    kk_cumsum = torch.cumsum(k.unsqueeze(-1) * k.unsqueeze(-2), dim=-3)
+    # first
+    z = (q * k_cumsum).sum(-1)
+    # second order
+    z += (q.unsqueeze(-1) * q.unsqueeze(-2) * kk_cumsum).sum((-1, -2)) * 0.5
+    # zero-th order
+    z += (torch.arange(0, q.shape[-2]).to(z.device) * 1.0 + 1.0)[None, None, :]
+    # compute o
+    # constant term
+    _o = v.cumsum(-2)
+    q = rearrange(q, 'b h (n c) d -> b h n c d', c=chunk_size)
+    k = rearrange(k, 'b h (n c) d -> b h n c d', c=chunk_size)
+    v = rearrange(v, 'b h (n c) d -> b h n c d', c=chunk_size)
+    intra_chunk_attn = q @ k.transpose(-2, -1)
+    intra_chunk_attn = intra_chunk_attn + 1/2 * (intra_chunk_attn ** 2)
+    intra_chunk_attn.masked_fill_(~torch.tril(torch.ones(chunk_size, chunk_size, dtype=torch.bool, device=q.device)), 0)
+    o = intra_chunk_attn @ v
+    # quadractic term
+    kv = torch.einsum('b h n c x, b h n c y, b h n c z -> b h n x y z', k, k, v)
+    kv = kv.cumsum(2)
+    kv = torch.cat([torch.zeros_like(kv[:, :, :1]), kv[:, :, :-1]], dim=2)
+    o += 0.5 * torch.einsum('b h n x y z, b h n c x, b h n c y -> b h n c z', kv, q, q)
+    # linear term
+    kv = torch.einsum('b h n c x, b h n c y -> b h n x y', k, v)
+    kv = kv.cumsum(2)
+    kv = torch.cat([torch.zeros_like(kv[:, :, :1]), kv[:, :, :-1]], dim=2)
+    o += torch.einsum('b h n x y, b h n c x -> b h n c y', kv, q)
+    o = rearrange(o, 'b h n c d -> b h (n c) d')
+    o = o + _o
+    return o / (z[..., None] + 1e-6)

fla/ops/based/parallel.py ADDED Viewed

	@@ -0,0 +1,410 @@

+# -*- coding: utf-8 -*-
+# Copyright (c) 2023-2025, Songlin Yang, Yu Zhang
+from typing import Optional
+import torch
+import triton
+import triton.language as tl
+from fla.utils import autocast_custom_bwd, autocast_custom_fwd, input_guard
+# Based: An Educational and Effective Sequence Mixer
+# https://hazyresearch.stanford.edu/blog/2023-12-11-zoology2-based
+@triton.jit(do_not_specialize=['T'])
+def parallel_based_fwd_kernel(
+    q,
+    k,
+    v,
+    o,
+    z,
+    scale,
+    T,
+    B: tl.constexpr,
+    H: tl.constexpr,
+    K: tl.constexpr,
+    V: tl.constexpr,
+    BTL: tl.constexpr,
+    BTS: tl.constexpr,
+    BK: tl.constexpr,
+    BV: tl.constexpr,
+):
+    # i_c: chunk index. used for sequence parallelism
+    i_kv, i_c, i_bh = tl.program_id(0), tl.program_id(1), tl.program_id(2)
+    NV = tl.cdiv(V, BV)
+    i_k = i_kv // (NV)
+    i_v = i_kv % (NV)
+    p_q = tl.make_block_ptr(q + i_bh * T*K, (T, K), (K, 1), (i_c * BTL, i_k * BK), (BTL, BK), (1, 0))
+    p_k = tl.make_block_ptr(k + i_bh * T*K, (K, T), (1, K), (i_k * BK, 0), (BK, BTS), (0, 1))
+    p_v = tl.make_block_ptr(v + i_bh * T*V, (T, V), (V, 1), (0, i_v * BV), (BTS, BV), (1, 0))
+    # [BQ, BD] block Q, in the shared memory throughout the whole kernel
+    b_q = tl.load(p_q, boundary_check=(0, 1))
+    b_q = (b_q * scale).to(b_q.dtype)
+    b_o = tl.zeros([BTL, BV], dtype=tl.float32)
+    b_z = tl.zeros([BTL], dtype=tl.float32)
+    # Q block and K block have no overlap
+    # no need for mask, thereby saving flops
+    for _ in range(0, i_c * BTL, BTS):
+        # [BK, BTS]
+        b_k = tl.load(p_k, boundary_check=(0, 1))
+        # [BTS, BV]
+        b_v = tl.load(p_v, boundary_check=(0, 1))
+        # [BTL, BTS]
+        b_s = tl.dot(b_q, (b_k), allow_tf32=False)
+        b_s = 1 + b_s + 0.5 * b_s * b_s
+        b_z += tl.sum(b_s, axis=1)
+        # [BQ, BD]
+        b_o = b_o + tl.dot(b_s.to(b_v.dtype), b_v, allow_tf32=False)
+        p_k = tl.advance(p_k, (0, BTS))
+        p_v = tl.advance(p_v, (BTS, 0))
+    # # rescale interchunk output
+    tl.debug_barrier()
+    o_q = tl.arange(0, BTL)
+    # # sync threads, easy for compiler to optimize
+    # tl.debug_barrier()
+    o_k = tl.arange(0, BTS)
+    p_k = tl.make_block_ptr(k + i_bh * T*K, (K, T), (1, K), (i_k * BK, i_c * BTL), (BK, BTS), (0, 1))
+    p_v = tl.make_block_ptr(v + i_bh * T*V, (T, V), (V, 1), (i_c * BTL, i_v * BV), (BTS, BV), (1, 0))
+    # Q block and K block have overlap. masks required
+    for _ in range(i_c * BTL, (i_c + 1) * BTL, BTS):
+        # [BK, BTS]
+        b_k = tl.load(p_k, boundary_check=(0, 1))
+        # [BTS, BV]
+        b_v = tl.load(p_v, boundary_check=(0, 1))
+        # [BTL, BTS]
+        m_s = o_q[:, None] >= o_k[None, :]
+        b_s = tl.dot(b_q, b_k, allow_tf32=False)
+        b_s = 1 + b_s + 0.5 * b_s * b_s
+        b_s = tl.where(m_s, b_s, 0)
+        b_z += tl.sum(b_s, axis=1)
+        # [BTL, BV]
+        b_o += tl.dot(b_s.to(b_q.dtype), b_v, allow_tf32=False)
+        p_k = tl.advance(p_k, (0, BTS))
+        p_v = tl.advance(p_v, (BTS, 0))
+        o_k += BTS
+    p_o = tl.make_block_ptr(o + (i_bh + B * H * i_k) * T*V, (T, V), (V, 1), (i_c*BTL, i_v*BV), (BTL, BV), (1, 0))
+    p_z = z + (i_bh + B * H * i_k) * T + i_c * BTL + tl.arange(0, BTL)
+    tl.store(p_o, b_o.to(p_o.dtype.element_ty), boundary_check=(0, 1))
+    tl.store(p_z, b_z.to(p_z.dtype.element_ty), mask=((i_c * BTL + tl.arange(0, BTL)) < T))
+@triton.jit
+def _parallel_based_bwd_dq(
+    i_bh,
+    i_c,
+    i_k,
+    i_v,
+    q,
+    k,
+    v,
+    do,
+    dz,
+    dq,
+    scale,
+    T,
+    B: tl.constexpr,
+    H: tl.constexpr,
+    BTL: tl.constexpr,
+    BTS: tl.constexpr,
+    BK: tl.constexpr,
+    BV: tl.constexpr,
+    K: tl.constexpr,
+    V: tl.constexpr,
+):
+    p_do = tl.make_block_ptr(do + i_bh * T*V, (T, V), (V, 1), (i_c * BTL, i_v * BV), (BTL, BV), (1, 0))
+    p_q = tl.make_block_ptr(q + (i_bh) * T*K, (T, K), (K, 1), (i_c*BTL, i_k*BK), (BTL, BK), (1, 0))
+    b_q = tl.load(p_q, boundary_check=(0, 1))
+    b_q = (b_q * scale).to(b_q.dtype)
+    b_do = tl.load(p_do, boundary_check=(0, 1)).to(b_q.dtype)
+    b_dq = tl.zeros([BTL, BK], dtype=tl.float32)
+    p_k = tl.make_block_ptr(k + i_bh * T*K, (T, K), (K, 1), (0, i_k * BK), (BTS, BK), (1, 0))
+    p_v = tl.make_block_ptr(v + i_bh * T*V, (V, T), (1, V), (i_v * BV, 0), (BV, BTS), (0, 1))
+    p_dz = dz + i_bh * T + i_c * BTL + tl.arange(0, BTL)
+    b_dz = tl.load(p_dz, mask=(i_c * BTL + tl.arange(0, BTL)) < T)
+    for _ in range(0, i_c * BTL, BTS):
+        # [BTS, BK]
+        b_k = tl.load(p_k, boundary_check=(0, 1))
+        # [BV, BTS]
+        b_v = tl.load(p_v, boundary_check=(0, 1))
+        # [BTL, BTS]
+        b_ds = tl.dot(b_do, b_v, allow_tf32=False)
+        if i_v == 0:
+            b_ds += b_dz[:, None]
+        else:
+            b_ds = b_ds
+        b_s = tl.dot(b_q, tl.trans(b_k), allow_tf32=False)
+        # [BQ, BD]
+        b_dq += tl.dot((b_ds * (1 + b_s)).to(b_v.dtype), b_k, allow_tf32=False)
+        p_k = tl.advance(p_k, (BTS, 0))
+        p_v = tl.advance(p_v, (0, BTS))
+    b_dq *= scale
+    o_q = tl.arange(0, BTL)
+    o_k = tl.arange(0, BTS)
+    p_k = tl.make_block_ptr(k + i_bh * T*K, (T, K), (K, 1), (i_c * BTL, i_k * BK), (BTS, BK), (1, 0))
+    p_v = tl.make_block_ptr(v + i_bh * T*V, (V, T), (1, V), (i_v * BV, i_c * BTL), (BV, BTS), (0, 1))
+    # Q block and K block have overlap. masks required
+    for _ in range(i_c * BTL, (i_c + 1) * BTL, BTS):
+        # [BTS, BK]
+        b_k = tl.load(p_k, boundary_check=(0, 1))
+        # [BV, BTS]
+        b_v = tl.load(p_v, boundary_check=(0, 1))
+        # [BTL, BTS]
+        m_s = o_q[:, None] >= o_k[None, :]
+        b_ds = tl.dot(b_do, b_v, allow_tf32=False)
+        if i_v == 0:
+            b_ds += b_dz[:, None]
+        else:
+            b_ds = b_ds
+        b_ds = tl.where(m_s, b_ds, 0) * scale
+        b_s = tl.dot(b_q, tl.trans(b_k), allow_tf32=False)
+        b_s = tl.where(m_s, b_s, 0)
+        # [BTL, BK]
+        b_dq += tl.dot((b_ds + b_ds * b_s).to(b_k.dtype), b_k, allow_tf32=False)
+        p_k = tl.advance(p_k, (BTS, 0))
+        p_v = tl.advance(p_v, (0, BTS))
+        o_k += BTS
+    p_dq = tl.make_block_ptr(dq + (i_bh + B * H * i_v) * T*K, (T, K), (K, 1), (i_c*BTL, i_k*BK), (BTL, BK), (1, 0))
+    tl.store(p_dq, b_dq.to(p_dq.dtype.element_ty), boundary_check=(0, 1))
+    return
+@triton.jit
+def _parallel_based_bwd_dkv(
+    i_bh,
+    i_c,
+    i_k,
+    i_v,
+    q,
+    k,
+    v,
+    do,
+    dz,
+    dk,
+    dv,
+    scale,
+    T,
+    B: tl.constexpr,
+    H: tl.constexpr,
+    BTL: tl.constexpr,
+    BTS: tl.constexpr,
+    BK: tl.constexpr,
+    BV: tl.constexpr,
+    K: tl.constexpr,
+    V: tl.constexpr,
+):
+    # compute dk dv
+    p_k = tl.make_block_ptr(k + i_bh * T*K, (T, K), (K, 1), (i_c * BTL, i_k * BK), (BTL, BK), (1, 0))
+    p_v = tl.make_block_ptr(v + i_bh * T*V, (T, V), (V, 1), (i_c * BTL, i_v * BV), (BTL, BV), (1, 0))
+    b_k, b_v = tl.load(p_k, boundary_check=(0, 1)), tl.load(p_v, boundary_check=(0, 1))
+    b_dk, b_dv = tl.zeros([BTL, BK], dtype=tl.float32), tl.zeros([BTL, BV], dtype=tl.float32)
+    for i in range((tl.cdiv(T, BTS) * BTS)-BTS, (i_c + 1) * BTL - BTS, -BTS):
+        p_q = tl.make_block_ptr(q + i_bh * T*K, (K, T), (1, K), (i_k * BK, i), (BK, BTS), (0, 1))
+        p_do = tl.make_block_ptr(do + i_bh * T*V, (V, T), (1, V), (i_v * BV, i), (BV, BTS), (0, 1))
+        p_dz = dz + i_bh * T + i + tl.arange(0, BTS)
+        b_q = tl.load(p_q, boundary_check=(0, 1))  # [BK, BTS]
+        b_do = tl.load(p_do, boundary_check=(0, 1)).to(b_q.dtype)  # [BV, BTS]
+        b_dz = tl.load(p_dz, mask=(i + tl.arange(0, BTS)) < T)
+        b_s = tl.dot(b_k.to(b_q.dtype), b_q, allow_tf32=False) * scale  # [BTL, BTS]
+        b_s2 = 1 + b_s + 0.5 * b_s * b_s
+        b_dv += tl.dot(b_s2.to(b_q.dtype), tl.trans(b_do), allow_tf32=False)
+        b_ds = tl.dot(b_v, b_do, allow_tf32=False) * scale
+        if i_v == 0:
+            b_ds += b_dz[None, :] * scale
+        else:
+            b_ds = b_ds
+        b_dk += tl.dot((b_ds + b_ds * b_s).to(b_q.dtype), tl.trans(b_q), allow_tf32=False)
+    tl.debug_barrier()
+    o_q, o_k = tl.arange(0, BTS), tl.arange(0, BTL)
+    for i in range(i_c*BTL, (i_c+1)*BTL, BTS):
+        p_q = tl.make_block_ptr(q + i_bh * T*K, (K, T), (1, K), (i_k * BK, i), (BK, BTS), (0, 1))
+        p_do = tl.make_block_ptr(do + i_bh * T*V, (V, T), (1, V), (i_v * BV, i), (BV, BTS), (0, 1))
+        p_dz = dz + i_bh * T + i + tl.arange(0, BTS)
+        b_q = tl.load(p_q, boundary_check=(0, 1))  # [BD, BQ]
+        b_do = tl.load(p_do, boundary_check=(0, 1)).to(b_q.dtype)
+        b_dz = tl.load(p_dz, mask=(i + tl.arange(0, BTS)) < T)
+        # [BK, BQ]
+        m_s = o_k[:, None] <= o_q[None, :]
+        b_s = tl.dot(b_k, b_q, allow_tf32=False) * scale
+        b_s2 = 1 + b_s + 0.5 * b_s * b_s
+        b_s = tl.where(m_s, b_s, 0)
+        b_s2 = tl.where(m_s, b_s2, 0)
+        b_ds = tl.dot(b_v, b_do, allow_tf32=False)
+        if i_v == 0:
+            b_ds += b_dz[None, :]
+        else:
+            b_ds = b_ds
+        b_ds = tl.where(m_s, b_ds, 0) * scale
+        # [BK, BD]
+        b_dv += tl.dot(b_s2.to(b_q.dtype), tl.trans(b_do), allow_tf32=False)
+        b_dk += tl.dot((b_ds + b_ds * b_s).to(b_q.dtype), tl.trans(b_q), allow_tf32=False)
+        o_q += BTS
+    p_dk = tl.make_block_ptr(dk + (i_bh + B * H * i_v) * T*K, (T, K), (K, 1), (i_c*BTL, i_k*BK), (BTL, BK), (1, 0))
+    p_dv = tl.make_block_ptr(dv + (i_bh + B * H * i_k) * T*V, (T, V), (V, 1), (i_c*BTL, i_v*BV), (BTL, BV), (1, 0))
+    tl.store(p_dk, b_dk.to(p_dk.dtype.element_ty), boundary_check=(0, 1))
+    tl.store(p_dv, b_dv.to(p_dv.dtype.element_ty), boundary_check=(0, 1))
+    return
+@triton.jit(do_not_specialize=['T'])
+def parallel_based_bwd_kernel(
+    q,
+    k,
+    v,
+    do,
+    dz,
+    dq,
+    dk,
+    dv,
+    scale,
+    T,
+    B: tl.constexpr,
+    H: tl.constexpr,
+    K: tl.constexpr,
+    V: tl.constexpr,
+    BTL: tl.constexpr,
+    BTS: tl.constexpr,
+    BK: tl.constexpr,
+    BV: tl.constexpr,
+):
+    i_kv, i_c, i_bh = tl.program_id(0), tl.program_id(1), tl.program_id(2)
+    NV = tl.cdiv(V, BV)
+    i_k = i_kv // (NV)
+    i_v = i_kv % NV
+    _parallel_based_bwd_dq(
+        i_bh, i_c, i_k, i_v,
+        q, k, v, do, dz, dq,
+        scale, T, B, H, BTL, BTS, BK, BV, K, V
+    )
+    tl.debug_barrier()
+    _parallel_based_bwd_dkv(
+        i_bh, i_c, i_k, i_v,
+        q, k, v, do, dz, dk, dv,
+        scale, T, B, H, BTL, BTS, BK, BV, K, V
+    )
+class ParallelBasedFunction(torch.autograd.Function):
+    @staticmethod
+    @input_guard
+    @autocast_custom_fwd
+    def forward(ctx, q, k, v, scale):
+        BTL, BTS = 128, 32
+        assert BTL % BTS == 0
+        # assert q.shape[-1] % 16 == 0
+        BK = min(128, triton.next_power_of_2(k.shape[-1]))
+        BV = min(128, triton.next_power_of_2(v.shape[-1]))
+        BK, BV = max(BK, 16), max(BV, 16)
+        B, H, T, K, V = *k.shape, v.shape[-1]
+        num_stages = 2
+        num_warps = 4
+        NK = triton.cdiv(K, BK)
+        NV = triton.cdiv(V, BV)
+        grid = (NK * NV, triton.cdiv(T, BTL), B * H)
+        assert NK == 1, "will encounter some synchronization issue if not."
+        o = torch.empty(NK, B, H, T, V, device=q.device)
+        z = torch.empty(NK, B, H, T, device=q.device)
+        parallel_based_fwd_kernel[grid](
+            q, k, v, o, z,
+            scale,
+            B=B,
+            H=H,
+            T=T,
+            K=K,
+            V=V,
+            BTL=BTL,
+            BTS=BTS,
+            BK=BK,
+            BV=BV,
+            num_warps=num_warps,
+            num_stages=num_stages
+        )
+        ctx.save_for_backward(q, k, v)
+        ctx.scale = scale
+        return o.sum(0).to(q.dtype), z.sum(0).to(q.dtype)
+    @staticmethod
+    @input_guard
+    @autocast_custom_bwd
+    def backward(ctx, do, dz):
+        q, k, v = ctx.saved_tensors
+        scale = ctx.scale
+        BTL, BTS = 64, 32
+        assert BTL % BTS == 0
+        BK = min(128, triton.next_power_of_2(k.shape[-1]))
+        BV = min(128, triton.next_power_of_2(v.shape[-1]))
+        BK, BV = max(BK, 16), max(BV, 16)
+        B, H, T, K, V = *k.shape, v.shape[-1]
+        num_stages = 2
+        num_warps = 4
+        NK = triton.cdiv(K, BK)
+        NV = triton.cdiv(V, BV)
+        grid = (NK * NV, triton.cdiv(T, BTL), B * H)
+        assert NK == 1, "will encounter some synchronization issue if not"
+        dq = torch.empty(NV, B, H, T, K, dtype=q.dtype, device=q.device)
+        dk = torch.empty(NV, B, H, T, K, dtype=q.dtype, device=q.device)
+        dv = torch.empty(NK, B, H, T, V, dtype=q.dtype, device=q.device)
+        parallel_based_bwd_kernel[grid](
+            q, k, v, do, dz, dq, dk, dv,
+            scale,
+            B=B,
+            H=H,
+            T=T,
+            K=K,
+            V=V,
+            BTL=BTL,
+            BTS=BTS,
+            BK=BK,
+            BV=BV,
+            num_warps=num_warps,
+            num_stages=num_stages
+        )
+        return dq.sum(0).to(q.dtype), dk.sum(0).to(k.dtype), dv.sum(0).to(v.dtype), None
+triton_parallel_based = ParallelBasedFunction.apply
+def parallel_based(
+    q: torch.Tensor,
+    k: torch.Tensor,
+    v: torch.Tensor,
+    scale: Optional[float] = None,
+    use_norm: bool = True,
+    head_first: bool = True
+):
+    assert q.shape[-1] <= 128, "only support feature dim up to 128"
+    if scale is None:
+        scale = q.shape[-1] ** -0.5
+    if not head_first:
+        q, k, v = map(lambda x: x.transpose(1, 2), (q, k, v))
+    o, z = triton_parallel_based(q, k, v, scale)
+    if use_norm:
+        o = o / (z[..., None] + 1e-6)
+    if not head_first:
+        o = o.transpose(1, 2)
+    return o.to(q.dtype)

fla/ops/common/__pycache__/__init__.cpython-312.pyc ADDED Viewed

Binary file (142 Bytes). View file

fla/ops/common/__pycache__/chunk_delta_h.cpython-312.pyc ADDED Viewed

Binary file (23.9 kB). View file

fla/ops/common/__pycache__/chunk_o.cpython-312.pyc ADDED Viewed

Binary file (37 kB). View file

fla/ops/common/__pycache__/fused_recurrent.cpython-312.pyc ADDED Viewed

Binary file (32.4 kB). View file

fla/ops/common/__pycache__/utils.cpython-312.pyc ADDED Viewed

Binary file (4.42 kB). View file

fla/ops/delta_rule/__pycache__/__init__.cpython-312.pyc ADDED Viewed

Binary file (364 Bytes). View file

fla/ops/delta_rule/__pycache__/wy_fast.cpython-312.pyc ADDED Viewed

Binary file (20.5 kB). View file

fla/ops/delta_rule/naive.py ADDED Viewed

	@@ -0,0 +1,120 @@

+# -*- coding: utf-8 -*-
+import torch
+from einops import rearrange
+def delta_rule_recurrence(q, k, v, beta, initial_state=None, output_final_state=True):
+    orig_dtype = q.dtype
+    b, h, l, d_k = q.shape
+    q, k, v, beta = map(lambda x: x.float(), [q, k, v, beta])
+    d_v = v.shape[-1]
+    o = torch.zeros_like(v)
+    S = torch.zeros(b, h, d_k, d_v).to(v)
+    q = q * (d_k ** -0.5)
+    if beta.ndim < v.ndim:
+        beta = beta[..., None]
+    if initial_state is not None:
+        S += initial_state
+    for i in range(l):
+        _k = k[:, :, i]
+        _q = q[:, :, i]
+        _v = v[:, :, i].clone()
+        beta_i = beta[:, :, i]
+        _v = _v - (S.clone() * _k[..., None]).sum(-2)
+        _v = _v * beta_i
+        S = S.clone() + _k.unsqueeze(-1) * _v.unsqueeze(-2)
+        o[:, :, i] = torch.einsum('bhd,bhdm->bhm', _q, S)
+    S = None if output_final_state is False else S
+    return o.to(orig_dtype), S
+def delta_rule_chunkwise(q, k, v, beta, chunk_size=32):
+    b, h, l, d_k = q.shape
+    d_v = v.shape[-1]
+    q = q * (d_k ** -0.5)
+    v = v * beta[..., None]
+    k_beta = k * beta[..., None]
+    assert l % chunk_size == 0
+    # compute (I - tri(diag(beta) KK^T))^{-1}
+    mask = torch.triu(torch.ones(chunk_size, chunk_size, dtype=torch.bool, device=q.device), diagonal=0)
+    q, k, v, k_beta = map(lambda x: rearrange(x, 'b h (n c) d -> b h n c d', c=chunk_size), [q, k, v, k_beta])
+    attn = -(k_beta @ k.transpose(-1, -2)).masked_fill(mask, 0)
+    for i in range(1, chunk_size):
+        attn[..., i, :i] = attn[..., i, :i] + (attn[..., i, :, None].clone() * attn[..., :, :i].clone()).sum(-2)
+    attn = attn + torch.eye(chunk_size, dtype=torch.float, device=q.device)
+    u = attn @ v
+    w = attn @ k_beta
+    S = k.new_zeros(b, h, d_k, d_v)
+    o = torch.zeros_like(v)
+    mask = torch.triu(torch.ones(chunk_size, chunk_size, dtype=torch.bool, device=q.device), diagonal=1)
+    for i in range(0, l // chunk_size):
+        q_i, k_i = q[:, :, i], k[:, :, i]
+        attn = (q_i @ k_i.transpose(-1, -2)).masked_fill_(mask, 0)
+        u_i = u[:, :, i] - w[:, :, i] @ S
+        o_inter = q_i @ S
+        o[:, :, i] = o_inter + attn @ u_i
+        S = S + k_i.transpose(-1, -2) @ u_i
+    return rearrange(o, 'b h n c d -> b h (n c) d'), S
+def delta_rule_parallel(q, k, v, beta, BM=128, BN=32):
+    b, h, l, d_k = q.shape
+    # d_v = v.shape[-1]
+    q = q * (d_k ** -0.5)
+    v = v * beta[..., None]
+    k_beta = k * beta[..., None]
+    # compute (I - tri(diag(beta) KK^T))^{-1}
+    q, k, v, k_beta = map(lambda x: rearrange(x, 'b h (n c) d -> b h n c d', c=BN), [q, k, v, k_beta])
+    mask = torch.triu(torch.ones(BN, BN, dtype=torch.bool, device=q.device), diagonal=0)
+    T = -(k_beta @ k.transpose(-1, -2)).masked_fill(mask, 0)
+    for i in range(1, BN):
+        T[..., i, :i] = T[..., i, :i].clone() + (T[..., i, :, None].clone() * T[..., :, :i].clone()).sum(-2)
+    T = T + torch.eye(BN, dtype=torch.float, device=q.device)
+    mask2 = torch.triu(torch.ones(BN, BN, dtype=torch.bool, device=q.device), diagonal=1)
+    A_local = (q @ k.transpose(-1, -2)).masked_fill(mask2, 0) @ T
+    o_intra = A_local @ v
+    # apply cumprod transition matrices on k to the last position within the chunk
+    k = k - ((k @ k.transpose(-1, -2)).masked_fill(mask, 0) @ T).transpose(-1, -2) @ k_beta
+    # apply cumprod transition matrices on q to the first position within the chunk
+    q = q - A_local @ k_beta
+    o_intra = A_local @ v
+    A = torch.zeros(b, h, l, l, device=q.device)
+    q, k, v, k_beta, o_intra = map(lambda x: rearrange(x, 'b h n c d -> b h (n c) d'), [q, k, v, k_beta, o_intra])
+    o = torch.empty_like(v)
+    for i in range(0, l, BM):
+        q_i = q[:, :, i:i+BM]
+        o_i = o_intra[:, :, i:i+BM]
+        # intra block
+        for j in range(i + BM - 2 * BN, i-BN, -BN):
+            k_j = k[:, :, j:j+BN]
+            A_ij = q_i @ k_j.transpose(-1, -2)
+            mask = torch.arange(i, i+BM) >= (j + BN)
+            A_ij = A_ij.masked_fill_(~mask[:, None].to(A_ij.device), 0)
+            A[:, :, i:i+BM, j:j+BN] = A_ij
+            q_i = q_i - A_ij @ k_beta[:, :, j:j+BN]
+            o_i += A_ij @ v[:, :, j:j+BN]
+        # inter block
+        for j in range(i - BN, -BN, -BN):
+            k_j = k[:, :, j:j+BN]
+            A_ij = q_i @ k_j.transpose(-1, -2)
+            A[:, :, i:i+BM, j:j+BN] = A_ij
+            q_i = q_i - A_ij @ k_beta[:, :, j:j+BN]
+            o_i += A_ij @ v[:, :, j:j+BN]
+        o[:, :, i:i+BM] = o_i
+    for i in range(0, l//BN):
+        A[:, :, i*BN:i*BN+BN, i*BN:i*BN+BN] = A_local[:, :, i]
+    return o, A

fla/ops/forgetting_attn/__pycache__/parallel.cpython-312.pyc ADDED Viewed

Binary file (39 kB). View file

fla/ops/forgetting_attn/parallel.py ADDED Viewed

	@@ -0,0 +1,708 @@

+# -*- coding: utf-8 -*-
+# Copyright (c) 2023-2025, Songlin Yang, Yu Zhang
+from typing import Optional
+import torch
+import triton
+import triton.language as tl
+from einops import rearrange, reduce
+from fla.ops.common.utils import prepare_chunk_indices
+from fla.ops.utils import chunk_global_cumsum, chunk_local_cumsum
+from fla.ops.utils.op import div, exp, log
+from fla.utils import autocast_custom_bwd, autocast_custom_fwd, check_shared_mem, input_guard
+@triton.heuristics({
+    'USE_OFFSETS': lambda args: args['offsets'] is not None
+})
+@triton.autotune(
+    configs=[
+        triton.Config({}, num_warps=num_warps, num_stages=num_stages)
+        for num_warps in [1, 2, 4] + ([8] if check_shared_mem('hopper') else [])
+        for num_stages in [2, 3, 4, 5]
+    ],
+    key=['B', 'H', 'G', 'K', 'V', 'BK', 'BV'],
+)
+@triton.jit
+def parallel_forgetting_attn_fwd_kernel(
+    q,
+    k,
+    v,
+    g,
+    o,
+    lse,
+    scale,
+    offsets,
+    indices,
+    T,
+    B: tl.constexpr,
+    H: tl.constexpr,
+    HQ: tl.constexpr,
+    G: tl.constexpr,
+    K: tl.constexpr,
+    V: tl.constexpr,
+    BT: tl.constexpr,
+    BS: tl.constexpr,
+    BK: tl.constexpr,
+    BV: tl.constexpr,
+    USE_OFFSETS: tl.constexpr
+):
+    i_v, i_t, i_bh = tl.program_id(0), tl.program_id(1), tl.program_id(2)
+    i_b, i_hq = i_bh // HQ, i_bh % HQ
+    i_h = i_hq // G
+    if USE_OFFSETS:
+        i_n, i_t = tl.load(indices + i_t * 2).to(tl.int32), tl.load(indices + i_t * 2 + 1).to(tl.int32)
+        bos, eos = tl.load(offsets + i_n).to(tl.int32), tl.load(offsets + i_n + 1).to(tl.int32)
+        T = eos - bos
+    else:
+        i_n = i_b
+        bos, eos = i_n * T, i_n * T + T
+    p_q = tl.make_block_ptr(q + (bos * HQ + i_hq) * K, (T, K), (HQ*K, 1), (i_t * BT, 0), (BT, BK), (1, 0))
+    p_g = tl.make_block_ptr(g + bos * HQ + i_hq, (T,), (HQ,), (i_t * BT,), (BT,), (0,))
+    p_o = tl.make_block_ptr(o + (bos * HQ + i_hq) * V, (T, V), (HQ*V, 1), (i_t * BT, i_v * BV), (BT, BV), (1, 0))
+    p_lse = tl.make_block_ptr(lse + bos * HQ + i_hq, (T,), (HQ,), (i_t * BT,), (BT,), (0,))
+    # the Q block is kept in the shared memory throughout the whole kernel
+    # [BT, BK]
+    b_q = tl.load(p_q, boundary_check=(0, 1))
+    b_q = (b_q * scale).to(b_q.dtype)
+    # [BT,]
+    b_gq = tl.load(p_g, boundary_check=(0,)).to(tl.float32)
+    # [BT, BV]
+    b_o = tl.zeros([BT, BV], dtype=tl.float32)
+    b_m = tl.full([BT], float('-inf'), dtype=tl.float32)
+    b_acc = tl.zeros([BT], dtype=tl.float32)
+    # [BT]
+    o_q = i_t * BT + tl.arange(0, BT)
+    for i_s in range(i_t * BT, min((i_t + 1) * BT, T), BS):
+        p_k = tl.make_block_ptr(k + (bos * H + i_h) * K, (K, T), (1, H*K), (0, i_s), (BK, BS), (0, 1))
+        p_v = tl.make_block_ptr(v + (bos * H + i_h) * V, (T, V), (H*V, 1), (i_s, i_v * BV), (BS, BV), (1, 0))
+        p_gk = tl.make_block_ptr(g + bos * HQ + i_hq, (T,), (HQ,), (i_s,), (BS,), (0,))
+        # [BS]
+        o_k = i_s + tl.arange(0, BS)
+        # [BK, BS]
+        b_k = tl.load(p_k, boundary_check=(0, 1))
+        # [BS, BV]
+        b_v = tl.load(p_v, boundary_check=(0, 1))
+        # [BS,]
+        b_gk = tl.load(p_gk, boundary_check=(0,))
+        # [BT, BS]
+        b_s = tl.dot(b_q, b_k) + b_gq[:, None] - b_gk[None, :]
+        b_s = tl.where(o_q[:, None] >= o_k[None, :], b_s, float('-inf'))
+        # [BT]
+        b_m, b_mp = tl.maximum(b_m, tl.max(b_s, 1)), b_m
+        b_r = exp(b_mp - b_m)
+        # [BT, BS]
+        b_p = exp(b_s - b_m[:, None])
+        # [BT]
+        b_acc = b_acc * b_r + tl.sum(b_p, 1)
+        # [BT, BV]
+        b_o = b_o * b_r[:, None] + tl.dot(b_p.to(b_q.dtype), b_v)
+        b_mp = b_m
+    for i_s in range(i_t * BT - BS, -BS, -BS):
+        p_k = tl.make_block_ptr(k + (bos * H + i_h) * K, (K, T), (1, H*K), (0, i_s), (BK, BS), (0, 1))
+        p_v = tl.make_block_ptr(v + (bos * H + i_h) * V, (T, V), (H*V, 1), (i_s, i_v * BV), (BS, BV), (1, 0))
+        p_gk = tl.make_block_ptr(g + bos * HQ + i_hq, (T,), (HQ,), (i_s,), (BS,), (0,))
+        # [BK, BS]
+        b_k = tl.load(p_k, boundary_check=(0, 1))
+        # [BS, BV]
+        b_v = tl.load(p_v, boundary_check=(0, 1))
+        # [BS,]
+        b_gk = tl.load(p_gk, boundary_check=(0,)).to(tl.float32)
+        b_gn = tl.load(g + (bos + min(i_s + BS, T) - 1) * HQ + i_hq).to(tl.float32)
+        b_gp = tl.load(g + (bos + i_s - 1) * HQ + i_hq).to(tl.float32) if i_s % BT > 0 else 0.
+        # [BT, BS]
+        b_s = tl.dot(b_q, b_k) + b_gq[:, None] + (b_gn - b_gk)[None, :]
+        b_gq += b_gn - b_gp
+        b_m, b_mp = tl.maximum(b_m, tl.max(b_s, 1)), b_m
+        b_r = exp(b_mp - b_m)
+        # [BT, BS]
+        b_p = exp(b_s - b_m[:, None])
+        # [BT]
+        b_acc = b_acc * b_r + tl.sum(b_p, 1)
+        # [BT, BV]
+        b_o = b_o * b_r[:, None] + tl.dot(b_p.to(b_q.dtype), b_v)
+        b_mp = b_m
+    b_o = div(b_o, b_acc[:, None])
+    b_m += log(b_acc)
+    tl.store(p_o, b_o.to(p_o.dtype.element_ty), boundary_check=(0, 1))
+    tl.store(p_lse, b_m.to(p_lse.dtype.element_ty), boundary_check=(0,))
+@triton.jit
+def parallel_forgetting_attn_bwd_kernel_preprocess(
+    o,
+    do,
+    delta,
+    B: tl.constexpr,
+    V: tl.constexpr
+):
+    i_n = tl.program_id(0)
+    o_d = tl.arange(0, B)
+    m_d = o_d < V
+    b_o = tl.load(o + i_n * V + o_d, mask=m_d, other=0)
+    b_do = tl.load(do + i_n * V + o_d, mask=m_d, other=0).to(tl.float32)
+    b_delta = tl.sum(b_o * b_do)
+    tl.store(delta + i_n, b_delta.to(delta.dtype.element_ty))
+@triton.heuristics({
+    'USE_OFFSETS': lambda args: args['offsets'] is not None
+})
+@triton.autotune(
+    configs=[
+        triton.Config({}, num_warps=num_warps, num_stages=num_stages)
+        for num_warps in [1, 2, 4] + ([8] if check_shared_mem('hopper') else [])
+        for num_stages in [2, 3, 4]
+    ],
+    key=['B', 'H', 'G', 'K', 'V', 'BK', 'BV'],
+)
+@triton.jit(do_not_specialize=['T'])
+def parallel_forgetting_attn_bwd_kernel_dq(
+    q,
+    k,
+    v,
+    g,
+    lse,
+    delta,
+    do,
+    dq,
+    dg,
+    scale,
+    offsets,
+    indices,
+    T,
+    B: tl.constexpr,
+    H: tl.constexpr,
+    HQ: tl.constexpr,
+    G: tl.constexpr,
+    K: tl.constexpr,
+    V: tl.constexpr,
+    BT: tl.constexpr,
+    BS: tl.constexpr,
+    BK: tl.constexpr,
+    BV: tl.constexpr,
+    USE_OFFSETS: tl.constexpr
+):
+    i_v, i_t, i_bh = tl.program_id(0), tl.program_id(1), tl.program_id(2)
+    i_b, i_hq = i_bh // HQ, i_bh % HQ
+    i_h = i_hq // G
+    if USE_OFFSETS:
+        i_n, i_t = tl.load(indices + i_t * 2).to(tl.int32), tl.load(indices + i_t * 2 + 1).to(tl.int32)
+        bos, eos = tl.load(offsets + i_n).to(tl.int32), tl.load(offsets + i_n + 1).to(tl.int32)
+        T = eos - bos
+    else:
+        i_n = i_b
+        bos, eos = i_n * T, i_n * T + T
+    p_q = tl.make_block_ptr(q + (bos * HQ + i_hq) * K, (T, K), (HQ*K, 1), (i_t * BT, 0), (BT, BK), (1, 0))
+    p_g = tl.make_block_ptr(g + bos * HQ + i_hq, (T,), (HQ,), (i_t * BT,), (BT,), (0,))
+    p_dq = tl.make_block_ptr(dq + (bos * HQ + i_hq) * K, (T, K), (HQ*K, 1), (i_t * BT, 0), (BT, BK), (1, 0))
+    p_dg = tl.make_block_ptr(dg + (bos * HQ + i_hq), (T,), (HQ,), (i_t * BT,), (BT,), (0,))
+    p_do = tl.make_block_ptr(do + (bos * HQ + i_hq) * V, (T, V), (HQ*V, 1), (i_t * BT, i_v * BV), (BT, BV), (1, 0))
+    p_lse = tl.make_block_ptr(lse + bos * HQ + i_hq, (T,), (HQ,), (i_t * BT,), (BT,), (0,))
+    p_delta = tl.make_block_ptr(delta + bos * HQ + i_hq, (T,), (HQ,), (i_t * BT,), (BT,), (0,))
+    # [BT, BK]
+    b_q = tl.load(p_q, boundary_check=(0, 1))
+    b_q = (b_q * scale).to(b_q.dtype)
+    # [BT, BV]
+    b_do = tl.load(p_do, boundary_check=(0, 1))
+    # [BT]
+    b_gq = tl.load(p_g, boundary_check=(0,)).to(tl.float32)
+    b_lse = tl.load(p_lse, boundary_check=(0,))
+    b_delta = tl.load(p_delta, boundary_check=(0,))
+    # [BT]
+    o_q = i_t * BT + tl.arange(0, BT)
+    # [BT, BK]
+    b_dq = tl.zeros([BT, BK], dtype=tl.float32)
+    # [BT]
+    b_dg = tl.zeros([BT,], dtype=tl.float32)
+    for i_s in range(i_t * BT, min((i_t + 1) * BT, T), BS):
+        p_k = tl.make_block_ptr(k + (bos * H + i_h) * K, (K, T), (1, H*K), (0, i_s), (BK, BS), (0, 1))
+        p_v = tl.make_block_ptr(v + (bos * H + i_h) * V, (V, T), (1, H*V), (i_v * BV, i_s), (BV, BS), (0, 1))
+        p_gk = tl.make_block_ptr(g + bos * HQ + i_hq, (T,), (HQ,), (i_s,), (BS,), (0,))
+        # [BS]
+        o_k = i_s + tl.arange(0, BS)
+        # [BK, BS]
+        b_k = tl.load(p_k, boundary_check=(0, 1))
+        # [BV, BS]
+        b_v = tl.load(p_v, boundary_check=(0, 1))
+        # [BS,]
+        b_gk = tl.load(p_gk, boundary_check=(0,))
+        # [BT, BS]
+        b_s = tl.dot(b_q, b_k) + (b_gq - b_lse)[:, None] - b_gk[None, :]
+        b_p = exp(tl.where(o_q[:, None] >= o_k[None, :], b_s, float('-inf')))
+        # [BT, BV] @ [BV, BS] -> [BT, BS]
+        b_dp = tl.dot(b_do, b_v)
+        b_ds = b_p * (b_dp.to(tl.float32) - b_delta[:, None])
+        # [BT, BS] @ [BS, BK] -> [BT, BK]
+        b_dq += tl.dot(b_ds.to(b_k.dtype), tl.trans(b_k))
+        # [BT]
+        b_dg += tl.sum(b_ds, 1)
+    for i_s in range(i_t * BT - BS, -BS, -BS):
+        p_k = tl.make_block_ptr(k + (bos * H + i_h) * K, (K, T), (1, H*K), (0, i_s), (BK, BS), (0, 1))
+        p_v = tl.make_block_ptr(v + (bos * H + i_h) * V, (V, T), (1, H*V), (i_v * BV, i_s), (BV, BS), (0, 1))
+        p_gk = tl.make_block_ptr(g + bos * HQ + i_hq, (T,), (HQ,), (i_s,), (BS,), (0,))
+        # [BK, BS]
+        b_k = tl.load(p_k, boundary_check=(0, 1))
+        # [BV, BS]
+        b_v = tl.load(p_v, boundary_check=(0, 1))
+        # [BS,]
+        b_gk = tl.load(p_gk, boundary_check=(0,)).to(tl.float32)
+        b_gn = tl.load(g + (bos + min(i_s + BS, T) - 1) * HQ + i_hq).to(tl.float32)
+        b_gp = tl.load(g + (bos + i_s - 1) * HQ + i_hq).to(tl.float32) if i_s % BT > 0 else 0.
+        # [BT, BS]
+        b_s = tl.dot(b_q, b_k) + (b_gq - b_lse)[:, None] + (b_gn - b_gk)[None, :]
+        b_p = exp(b_s)
+        # [BT, BV] @ [BV, BS] -> [BT, BS]
+        b_dp = tl.dot(b_do, b_v)
+        b_ds = b_p * (b_dp - b_delta[:, None])
+        # [BT, BS] @ [BS, BK] -> [BT, BK]
+        b_dq += tl.dot(b_ds.to(b_k.dtype), tl.trans(b_k))
+        # [BT]
+        b_dg += tl.sum(b_ds, 1)
+        b_gq += b_gn - b_gp
+    b_dq *= scale
+    tl.store(p_dq, b_dq.to(p_dq.dtype.element_ty), boundary_check=(0, 1))
+    tl.store(p_dg, b_dg.to(p_dg.dtype.element_ty), boundary_check=(0,))
+@triton.heuristics({
+    'USE_OFFSETS': lambda args: args['offsets'] is not None
+})
+@triton.autotune(
+    configs=[
+        triton.Config({}, num_warps=num_warps, num_stages=num_stages)
+        for num_warps in [1, 2, 4, 8]
+        for num_stages in [2, 3, 4]
+    ],
+    key=['B', 'H', 'G', 'K', 'V', 'BK', 'BV'],
+)
+@triton.jit(do_not_specialize=['T'])
+def parallel_forgetting_attn_bwd_kernel_dkv(
+    q,
+    k,
+    v,
+    g,
+    lse,
+    delta,
+    do,
+    dk,
+    dv,
+    dg,
+    offsets,
+    indices,
+    scale,
+    T,
+    B: tl.constexpr,
+    H: tl.constexpr,
+    HQ: tl.constexpr,
+    G: tl.constexpr,
+    K: tl.constexpr,
+    V: tl.constexpr,
+    BT: tl.constexpr,
+    BS: tl.constexpr,
+    BK: tl.constexpr,
+    BV: tl.constexpr,
+    USE_OFFSETS: tl.constexpr
+):
+    i_v, i_t, i_bh = tl.program_id(0), tl.program_id(1), tl.program_id(2)
+    i_b, i_hq = i_bh // HQ, i_bh % HQ
+    i_h = i_hq // G
+    if USE_OFFSETS:
+        i_n, i_t = tl.load(indices + i_t * 2).to(tl.int32), tl.load(indices + i_t * 2 + 1).to(tl.int32)
+        bos, eos = tl.load(offsets + i_n).to(tl.int32), tl.load(offsets + i_n + 1).to(tl.int32)
+        T = eos - bos
+    else:
+        i_n = i_b
+        bos, eos = i_n * T, i_n * T + T
+    p_k = tl.make_block_ptr(k + (bos * H + i_h) * K, (T, K), (H*K, 1), (i_t * BT, 0), (BT, BK), (1, 0))
+    p_v = tl.make_block_ptr(v + (bos * H + i_h) * V, (T, V), (H*V, 1), (i_t * BT, i_v * BV), (BT, BV), (1, 0))
+    p_gk = tl.make_block_ptr(g + bos * HQ + i_hq, (T,), (HQ,), (i_t * BT,), (BT,), (0,))
+    p_dk = tl.make_block_ptr(dk + (bos * HQ + i_hq) * K, (T, K), (HQ*K, 1), (i_t * BT, 0), (BT, BK), (1, 0))
+    p_dv = tl.make_block_ptr(dv + (bos * HQ + i_hq) * V, (T, V), (HQ*V, 1), (i_t * BT, i_v * BV), (BT, BV), (1, 0))
+    p_dg = tl.make_block_ptr(dg + (bos * HQ + i_hq), (T,), (HQ,), (i_t * BT,), (BT,), (0,))
+    # [BT, BK]
+    b_k = tl.load(p_k, boundary_check=(0, 1))
+    b_dk = tl.zeros([BT, BK], dtype=tl.float32)
+    # [BT, BV]
+    b_v = tl.load(p_v, boundary_check=(0, 1))
+    b_dv = tl.zeros([BT, BV], dtype=tl.float32)
+    # [BT]
+    b_gk = tl.load(p_gk, boundary_check=(0,)).to(tl.float32)
+    b_dg = tl.zeros([BT,], dtype=tl.float32)
+    o_k = i_t * BT + tl.arange(0, BT)
+    m_k = o_k < T
+    for i_s in range(i_t * BT, min((i_t + 1) * BT, T), BS):
+        p_q = tl.make_block_ptr(q + (bos * HQ + i_hq) * K, (T, K), (HQ*K, 1), (i_s, 0), (BS, BK), (1, 0))
+        p_do = tl.make_block_ptr(do + (bos * HQ + i_hq) * V, (T, V), (HQ*V, 1), (i_s, i_v * BV), (BS, BV), (1, 0))
+        p_lse = tl.make_block_ptr(lse + bos * HQ + i_hq, (T,), (HQ,), (i_s,), (BS,), (0,))
+        p_delta = tl.make_block_ptr(delta + bos * HQ + i_hq, (T,), (HQ,), (i_s,), (BS,), (0,))
+        p_gq = tl.make_block_ptr(g + bos * HQ + i_hq, (T,), (HQ,), (i_s,), (BS,), (0,))
+        # [BS]
+        o_q = i_s + tl.arange(0, BS)
+        # [BS, BK]
+        b_q = tl.load(p_q, boundary_check=(0, 1))
+        b_q = (b_q * scale).to(b_q.dtype)
+        # [BS, BV]
+        b_do = tl.load(p_do, boundary_check=(0, 1))
+        # [BS]
+        b_lse = tl.load(p_lse, boundary_check=(0,))
+        b_delta = tl.load(p_delta, boundary_check=(0,))
+        b_gq = tl.load(p_gq, boundary_check=(0,)).to(tl.float32)
+        m_q = o_q < T
+        m_s = (o_k[:, None] <= o_q[None, :]) & m_k[:, None] & m_q[None, :]
+        # [BT, BS]
+        b_s = tl.dot(b_k, tl.trans(b_q)) - b_gk[:, None] + (b_gq - b_lse)[None, :]
+        b_p = tl.where(m_s, exp(b_s), 0)
+        # [BT, BS] @ [BS, BV] -> [BT, BV]
+        b_dv += tl.dot(b_p.to(b_do.dtype), b_do)
+        # [BT, BV] @ [BV, BS] -> [BT, BS]
+        b_dp = tl.dot(b_v, tl.trans(b_do))
+        # [BT, BS]
+        b_ds = b_p * (b_dp - b_delta[None, :])
+        # [BT, BS] @ [BS, BK] -> [BT, BK]
+        b_dk += tl.dot(b_ds.to(b_q.dtype), b_q)
+        # [BT]
+        b_dg -= tl.sum(b_ds, 1)
+    b_gk -= tl.load(g + (bos + min((i_t + 1) * BT, T) - 1) * HQ + i_hq).to(tl.float32)
+    for i_s in range((i_t + 1) * BT, T, BS):
+        p_q = tl.make_block_ptr(q + (bos * HQ + i_hq) * K, (T, K), (HQ*K, 1), (i_s, 0), (BS, BK), (1, 0))
+        p_do = tl.make_block_ptr(do + (bos * HQ + i_hq) * V, (T, V), (HQ*V, 1), (i_s, i_v * BV), (BS, BV), (1, 0))
+        p_lse = tl.make_block_ptr(lse + bos * HQ + i_hq, (T,), (HQ,), (i_s,), (BS,), (0,))
+        p_delta = tl.make_block_ptr(delta + bos * HQ + i_hq, (T,), (HQ,), (i_s,), (BS,), (0,))
+        p_gq = tl.make_block_ptr(g + bos * HQ + i_hq, (T,), (HQ,), (i_s,), (BS,), (0,))
+        # [BS]
+        o_q = i_s + tl.arange(0, BS)
+        # [BS, BK]
+        b_q = tl.load(p_q, boundary_check=(0, 1))
+        b_q = (b_q * scale).to(b_q.dtype)
+        # [BS, BV]
+        b_do = tl.load(p_do, boundary_check=(0, 1))
+        # [BS]
+        b_lse = tl.load(p_lse, boundary_check=(0,))
+        b_delta = tl.load(p_delta, boundary_check=(0,))
+        b_gq = tl.load(p_gq, boundary_check=(0,)).to(tl.float32)
+        b_gn = tl.load(g + (bos + min(i_s + BS, T) - 1) * HQ + i_hq).to(tl.float32)
+        b_gp = tl.load(g + (bos + i_s - 1) * HQ + i_hq).to(tl.float32) if i_s % BT > 0 else 0.
+        # [BT, BS]
+        b_s = tl.dot(b_k, tl.trans(b_q)) - (b_gk + b_gp)[:, None] + (b_gq - b_lse)[None, :]
+        b_p = exp(b_s)
+        # [BT, BS] @ [BS, BV] -> [BT, BV]
+        b_dv += tl.dot(b_p.to(b_do.dtype), b_do)
+        # [BT, BV] @ [BV, BS] -> [BT, BS]
+        b_dp = tl.dot(b_v, tl.trans(b_do))
+        # [BT, BS]
+        b_ds = b_p * (b_dp - b_delta[None, :])
+        # [BT, BS] @ [BS, BK] -> [BT, BK]
+        b_dk += tl.dot(b_ds.to(b_q.dtype), b_q)
+        # [BT]
+        b_dg -= tl.sum(b_ds, 1)
+        b_gk -= b_gn - b_gp
+    tl.store(p_dk, b_dk.to(p_dk.dtype.element_ty), boundary_check=(0, 1))
+    tl.store(p_dv, b_dv.to(p_dv.dtype.element_ty), boundary_check=(0, 1))
+    tl.store(p_dg, b_dg.to(p_dg.dtype.element_ty), boundary_check=(0,))
+def parallel_forgetting_attn_fwd(
+    q: torch.Tensor,
+    k: torch.Tensor,
+    v: torch.Tensor,
+    g: torch.Tensor,
+    scale: float,
+    chunk_size: int = 128,
+    offsets: Optional[torch.LongTensor] = None,
+    indices: Optional[torch.LongTensor] = None,
+):
+    B, T, H, K, V = *k.shape, v.shape[-1]
+    HQ = q.shape[2]
+    G = HQ // H
+    BT = chunk_size
+    BK = max(16, triton.next_power_of_2(K))
+    assert V <= 256, "V must be less than or equal to 256"
+    if check_shared_mem('hopper'):
+        BS = min(64, max(16, triton.next_power_of_2(T)))
+    else:
+        BS = min(32, max(16, triton.next_power_of_2(T)))
+    BV = min(256, max(16, triton.next_power_of_2(V)))
+    NV = triton.cdiv(V, BV)
+    NT = triton.cdiv(T, BT) if offsets is None else len(indices)
+    o = torch.empty(B, T, HQ, V, dtype=v.dtype, device=q.device)
+    lse = torch.empty(B, T, HQ, dtype=torch.float, device=q.device)
+    grid = (NV, NT, B * HQ)
+    parallel_forgetting_attn_fwd_kernel[grid](
+        q=q,
+        k=k,
+        v=v,
+        g=g,
+        o=o,
+        lse=lse,
+        scale=scale,
+        offsets=offsets,
+        indices=indices,
+        B=B,
+        T=T,
+        H=H,
+        HQ=HQ,
+        G=G,
+        K=K,
+        V=V,
+        BT=BT,
+        BS=BS,
+        BK=BK,
+        BV=BV,
+    )
+    return o, lse
+def parallel_forgetting_attn_bwd_preprocess(
+    o: torch.Tensor,
+    do: torch.Tensor
+):
+    V = o.shape[-1]
+    delta = torch.empty_like(o[..., 0], dtype=torch.float)
+    parallel_forgetting_attn_bwd_kernel_preprocess[(delta.numel(),)](
+        o=o,
+        do=do,
+        delta=delta,
+        B=triton.next_power_of_2(V),
+        V=V,
+    )
+    return delta
+def parallel_forgetting_attn_bwd(
+    q: torch.Tensor,
+    k: torch.Tensor,
+    v: torch.Tensor,
+    g: torch.Tensor,
+    o: torch.Tensor,
+    lse: torch.Tensor,
+    do: torch.Tensor,
+    scale: float = None,
+    chunk_size: int = 128,
+    offsets: Optional[torch.LongTensor] = None,
+    indices: Optional[torch.LongTensor] = None,
+):
+    B, T, H, K, V = *k.shape, v.shape[-1]
+    HQ = q.shape[2]
+    G = HQ // H
+    BT = chunk_size
+    BS = min(32, max(16, triton.next_power_of_2(T)))
+    BK = max(16, triton.next_power_of_2(K))
+    BV = max(16, triton.next_power_of_2(V))
+    NV = triton.cdiv(V, BV)
+    NT = triton.cdiv(T, BT) if offsets is None else len(indices)
+    delta = parallel_forgetting_attn_bwd_preprocess(o, do)
+    dq = q.new_empty(B, T, HQ, K, dtype=q.dtype)
+    dk = q.new_empty(B, T, HQ, K, dtype=k.dtype if H == HQ else torch.float)
+    dv = q.new_empty(B, T, HQ, V, dtype=v.dtype if H == HQ else torch.float)
+    dg = q.new_empty(g.shape, dtype=torch.float)
+    # NOTE: the original `dg` can be destroyed during autotuning
+    # this is [a known triton issue](https://github.com/triton-lang/triton/issues/5082), which will be fixed in 3.3 (?)
+    # so we need to make a copy of `dg`
+    dg2 = q.new_empty(g.shape, dtype=torch.float)
+    grid = (NV, NT, B * HQ)
+    parallel_forgetting_attn_bwd_kernel_dq[grid](
+        q=q,
+        k=k,
+        v=v,
+        g=g,
+        lse=lse,
+        delta=delta,
+        do=do,
+        dq=dq,
+        dg=dg,
+        offsets=offsets,
+        indices=indices,
+        scale=scale,
+        T=T,
+        B=B,
+        H=H,
+        HQ=HQ,
+        G=G,
+        K=K,
+        V=V,
+        BT=BT,
+        BS=BS,
+        BK=BK,
+        BV=BV
+    )
+    parallel_forgetting_attn_bwd_kernel_dkv[grid](
+        q=q,
+        k=k,
+        v=v,
+        g=g,
+        lse=lse,
+        delta=delta,
+        do=do,
+        dk=dk,
+        dv=dv,
+        dg=dg2,
+        offsets=offsets,
+        indices=indices,
+        scale=scale,
+        T=T,
+        B=B,
+        H=H,
+        HQ=HQ,
+        G=G,
+        K=K,
+        V=V,
+        BT=BT,
+        BS=BS,
+        BK=BK,
+        BV=BV
+    )
+    dk = reduce(dk, 'b t (h g) k -> b t h k', g=G, reduction='sum')
+    dv = reduce(dv, 'b t (h g) v -> b t h v', g=G, reduction='sum')
+    dg = dg.add_(dg2)
+    return dq, dk, dv, dg
+@torch.compile
+class ParallelForgettingAttentionFunction(torch.autograd.Function):
+    @staticmethod
+    @input_guard
+    @autocast_custom_fwd
+    def forward(ctx, q, k, v, g, scale, offsets):
+        ctx.dtype = q.dtype
+        if check_shared_mem('hopper'):
+            chunk_size = min(128, max(16, triton.next_power_of_2(q.shape[1])))
+        else:
+            chunk_size = min(64, max(16, triton.next_power_of_2(q.shape[1])))
+        # 2-d indices denoting the offsets of chunks in each sequence
+        # for example, if the passed `offsets` is [0, 100, 356] and `chunk_size` is 64,
+        # then there are 2 and 4 chunks in the 1st and 2nd sequences respectively, and `indices` will be
+        # [[0, 0], [0, 1], [1, 0], [1, 1], [1, 2], [1, 3]]
+        indices = prepare_chunk_indices(offsets, chunk_size) if offsets is not None else None
+        g = chunk_local_cumsum(g, chunk_size, offsets=offsets, indices=indices, head_first=False)
+        o, lse = parallel_forgetting_attn_fwd(
+            q=q,
+            k=k,
+            v=v,
+            g=g,
+            scale=scale,
+            chunk_size=chunk_size,
+            offsets=offsets,
+            indices=indices
+        )
+        ctx.save_for_backward(q, k, v, g, o, lse)
+        ctx.chunk_size = chunk_size
+        ctx.offsets = offsets
+        ctx.indices = indices
+        ctx.scale = scale
+        return o.to(q.dtype)
+    @staticmethod
+    @input_guard
+    @autocast_custom_bwd
+    def backward(ctx, do):
+        q, k, v, g, o, lse = ctx.saved_tensors
+        dq, dk, dv, dg = parallel_forgetting_attn_bwd(
+            q=q,
+            k=k,
+            v=v,
+            g=g,
+            o=o,
+            lse=lse,
+            do=do,
+            scale=ctx.scale,
+            chunk_size=ctx.chunk_size,
+            offsets=ctx.offsets,
+            indices=ctx.indices
+        )
+        dg = chunk_global_cumsum(dg, reverse=True, head_first=False, offsets=ctx.offsets)
+        return dq.to(q), dk.to(k), dv.to(v), dg.to(g), None, None, None, None, None, None, None, None
+def parallel_forgetting_attn(
+    q: torch.Tensor,
+    k: torch.Tensor,
+    v: torch.Tensor,
+    g: torch.Tensor,
+    scale: Optional[float] = None,
+    cu_seqlens: Optional[torch.LongTensor] = None,
+    head_first: bool = False
+) -> torch.Tensor:
+    r"""
+    Args:
+        q (torch.Tensor):
+            queries of shape `[B, T, HQ, K]` if `head_first=False` else `[B, HQ, T, K]`.
+        k (torch.Tensor):
+            keys of shape `[B, T, H, K]` if `head_first=False` else `[B, H, T, K]`.
+            GQA will be applied if HQ is divisible by H.
+        v (torch.Tensor):
+            values of shape `[B, T, H, V]` if `head_first=False` else `[B, H, T, V]`.
+        g (torch.Tensor):
+            Forget gates (in **log space**) of shape `[B, T, HQ]` if `head_first=False` else `[B, HQ, T]`.
+        scale (Optional[int]):
+            Scale factor for attention scores.
+            If not provided, it will default to `1 / sqrt(K)`. Default: `None`.
+        cu_seqlens (torch.LongTensor):
+            Cumulative sequence lengths of shape `[N+1]` used for variable-length training,
+            consistent with the FlashAttention API.
+        head_first (Optional[bool]):
+            Whether the inputs are in the head-first format. Default: `False`.
+    Returns:
+        o (torch.Tensor):
+            Outputs of shape `[B, T, HQ, V]` if `head_first=False` else `[B, HQ, T, V]`.
+    """
+    if scale is None:
+        scale = k.shape[-1] ** -0.5
+    if cu_seqlens is not None:
+        assert q.shape[0] == 1, "batch size must be 1 when cu_seqlens are provided"
+    if g is not None:
+        g = g.float()
+    if head_first:
+        q, k, v = map(lambda x: rearrange(x, 'b h t d -> b t h d'), (q, k, v))
+        g = rearrange(g, 'b h t -> b t h')
+    o = ParallelForgettingAttentionFunction.apply(q, k, v, g, scale, cu_seqlens)
+    if head_first:
+        o = rearrange(o, 'b t h d -> b h t d')
+    return o

fla/ops/gated_delta_rule/__pycache__/__init__.cpython-312.pyc ADDED Viewed

Binary file (322 Bytes). View file

fla/ops/gated_delta_rule/__pycache__/fused_recurrent.cpython-312.pyc ADDED Viewed

Binary file (15.1 kB). View file

fla/ops/gated_delta_rule/__pycache__/wy_fast.cpython-312.pyc ADDED Viewed

Binary file (45.1 kB). View file

fla/ops/generalized_delta_rule/dplr/chunk_h_fwd.py ADDED Viewed

	@@ -0,0 +1,197 @@

+# -*- coding: utf-8 -*-
+# Copyright (c) 2023-2025, Songlin Yang, Yu Zhang
+from typing import Optional, Tuple
+import torch
+import triton
+import triton.language as tl
+from fla.ops.common.utils import prepare_chunk_offsets
+from fla.ops.utils.op import exp
+from fla.utils import check_shared_mem, use_cuda_graph
+@triton.heuristics({
+    'USE_INITIAL_STATE': lambda args: args['h0'] is not None,
+    'STORE_FINAL_STATE': lambda args: args['ht'] is not None,
+    'USE_OFFSETS': lambda args: args['offsets'] is not None,
+})
+@triton.autotune(
+    configs=[
+        triton.Config({}, num_warps=num_warps, num_stages=num_stages)
+        for num_warps in [2, 4, 8, 16, 32]
+        for num_stages in [2, 3, 4]
+    ],
+    key=['BT', 'BK', 'BV'],
+    use_cuda_graph=use_cuda_graph,
+)
+@triton.jit(do_not_specialize=['T'])
+def chunk_dplr_fwd_kernel_h(
+    kg,
+    v,
+    w,
+    bg,
+    u,
+    v_new,
+    gk,
+    h,
+    h0,
+    ht,
+    offsets,
+    chunk_offsets,
+    T,
+    H: tl.constexpr,
+    K: tl.constexpr,
+    V: tl.constexpr,
+    BT: tl.constexpr,
+    BC: tl.constexpr,
+    BK: tl.constexpr,
+    BV: tl.constexpr,
+    NT: tl.constexpr,
+    USE_INITIAL_STATE: tl.constexpr,
+    STORE_FINAL_STATE: tl.constexpr,
+    USE_OFFSETS: tl.constexpr,
+    HEAD_FIRST: tl.constexpr,
+):
+    i_k, i_v, i_nh = tl.program_id(0), tl.program_id(1), tl.program_id(2)
+    i_n, i_h = i_nh // H, i_nh % H
+    if USE_OFFSETS:
+        bos, eos = tl.load(offsets + i_n).to(tl.int32), tl.load(offsets + i_n + 1).to(tl.int32)
+        T = eos - bos
+        NT = tl.cdiv(T, BT)
+        boh = tl.load(chunk_offsets + i_n).to(tl.int32)
+    else:
+        bos, eos = i_n * T, i_n * T + T
+        NT = tl.cdiv(T, BT)
+        boh = i_n * NT
+    # [BK, BV]
+    b_h = tl.zeros([BK, BV], dtype=tl.float32)
+    if USE_INITIAL_STATE:
+        p_h0 = tl.make_block_ptr(h0 + i_nh * K*V, (K, V), (V, 1), (i_k * BK, i_v * BV), (BK, BV), (1, 0))
+        b_h = tl.load(p_h0, boundary_check=(0, 1)).to(tl.float32)
+    for i_t in range(NT):
+        if HEAD_FIRST:
+            p_h = tl.make_block_ptr(h + (i_nh * NT + i_t) * K*V, (K, V), (V, 1), (i_k * BK, i_v * BV), (BK, BV), (1, 0))
+        else:
+            p_h = tl.make_block_ptr(h + ((boh + i_t) * H + i_h) * K*V, (K, V), (V, 1), (i_k * BK, i_v * BV), (BK, BV), (1, 0))
+        tl.store(p_h, b_h.to(p_h.dtype.element_ty), boundary_check=(0, 1))
+        b_hc = tl.zeros([BK, BV], dtype=tl.float32)
+        # since we need to make all DK in the SRAM. we face serve SRAM memory burden. By subchunking we allievate such burden
+        for i_c in range(tl.cdiv(min(BT, T - i_t * BT), BC)):
+            if HEAD_FIRST:
+                p_kg = tl.make_block_ptr(kg + i_nh * T*K, (K, T), (1, K), (i_k * BK, i_t * BT + i_c * BC), (BK, BC), (0, 1))
+                p_bg = tl.make_block_ptr(bg + i_nh * T*K, (K, T), (1, K), (i_k * BK, i_t * BT + i_c * BC), (BK, BC), (0, 1))
+                p_w = tl.make_block_ptr(w + i_nh * T*K, (T, K), (K, 1), (i_t * BT + i_c * BC, i_k * BK), (BC, BK), (1, 0))
+                p_v = tl.make_block_ptr(v + i_nh * T*V, (T, V), (V, 1), (i_t * BT + i_c * BC, i_v * BV), (BC, BV), (1, 0))
+                p_u = tl.make_block_ptr(u + i_nh * T*V, (T, V), (V, 1), (i_t * BT + i_c * BC, i_v * BV), (BC, BV), (1, 0))
+                p_v_new = tl.make_block_ptr(v_new+i_nh*T*V, (T, V), (V, 1), (i_t * BT + i_c * BC, i_v * BV), (BC, BV), (1, 0))
+            else:
+                p_kg = tl.make_block_ptr(kg+(bos*H+i_h)*K, (K, T), (1, H*K), (i_k * BK, i_t * BT + i_c * BC), (BK, BC), (0, 1))
+                p_bg = tl.make_block_ptr(bg+(bos*H+i_h)*K, (K, T), (1, H*K), (i_k * BK, i_t * BT + i_c * BC), (BK, BC), (0, 1))
+                p_w = tl.make_block_ptr(w+(bos*H+i_h)*K, (T, K), (H*K, 1), (i_t * BT + i_c * BC, i_k * BK), (BC, BK), (1, 0))
+                p_v = tl.make_block_ptr(v+(bos*H+i_h)*V, (T, V), (H*V, 1), (i_t * BT + i_c * BC, i_v * BV), (BC, BV), (1, 0))
+                p_u = tl.make_block_ptr(u+(bos*H+i_h)*V, (T, V), (H*V, 1), (i_t * BT + i_c * BC, i_v * BV), (BC, BV), (1, 0))
+                p_v_new = tl.make_block_ptr(v_new+(bos*H+i_h)*V, (T, V), (H*V, 1), (i_t*BT+i_c*BC, i_v * BV), (BC, BV), (1, 0))
+            # [BK, BC]
+            b_kg = tl.load(p_kg, boundary_check=(0, 1))
+            b_v = tl.load(p_v, boundary_check=(0, 1))
+            b_w = tl.load(p_w, boundary_check=(0, 1))
+            b_bg = tl.load(p_bg, boundary_check=(0, 1))
+            b_v2 = tl.dot(b_w, b_h.to(b_w.dtype)) + tl.load(p_u, boundary_check=(0, 1))
+            b_hc += tl.dot(b_kg, b_v)
+            b_hc += tl.dot(b_bg.to(b_hc.dtype), b_v2)
+            tl.store(p_v_new, b_v2.to(p_v_new.dtype.element_ty), boundary_check=(0, 1))
+        last_idx = min((i_t + 1) * BT, T) - 1
+        if HEAD_FIRST:
+            b_g_last = tl.load(gk + i_nh * T * K + last_idx * K + tl.arange(0, BK), mask=tl.arange(0, BK) < K).to(tl.float32)
+        else:
+            b_g_last = tl.load(gk + (bos + last_idx) * H * K + i_h * K +
+                               tl.arange(0, BK), mask=tl.arange(0, BK) < K).to(tl.float32)
+        b_h *= exp(b_g_last[:, None])
+        b_h += b_hc
+    if STORE_FINAL_STATE:
+        p_ht = tl.make_block_ptr(ht + i_nh * K*V, (K, V), (V, 1), (i_k * BK, i_v * BV), (BK, BV), (1, 0))
+        tl.store(p_ht, b_h.to(p_ht.dtype.element_ty, fp_downcast_rounding="rtne"), boundary_check=(0, 1))
+def chunk_dplr_fwd_h(
+    kg: torch.Tensor,
+    v: torch.Tensor,
+    w: torch.Tensor,
+    u: torch.Tensor,
+    bg: torch.Tensor,
+    gk: torch.Tensor,
+    initial_state: Optional[torch.Tensor] = None,
+    output_final_state: bool = False,
+    offsets: Optional[torch.LongTensor] = None,
+    indices: Optional[torch.LongTensor] = None,
+    head_first: bool = True,
+    chunk_size: int = 64
+) -> Tuple[torch.Tensor, torch.Tensor]:
+    if head_first:
+        B, H, T, K, V = *kg.shape, u.shape[-1]
+    else:
+        B, T, H, K, V = *kg.shape, u.shape[-1]
+    BT = min(chunk_size, max(triton.next_power_of_2(T), 16))
+    # N: the actual number of sequences in the batch with either equal or variable lengths
+    if offsets is None:
+        N, NT, chunk_offsets = B, triton.cdiv(T, BT), None
+    else:
+        N, NT, chunk_offsets = len(offsets) - 1, len(indices), prepare_chunk_offsets(offsets, BT)
+    BK = triton.next_power_of_2(K)
+    assert BK <= 256, "current kernel does not support head dimension larger than 256."
+    # H100 can have larger block size
+    if check_shared_mem('hopper', kg.device.index):
+        BV = 64
+        BC = 64 if K <= 128 else 32
+    elif check_shared_mem('ampere', kg.device.index):  # A100
+        BV = 32
+        BC = 32
+    else:
+        BV = 16
+        BC = 16
+    BC = min(BT, BC)
+    NK = triton.cdiv(K, BK)
+    NV = triton.cdiv(V, BV)
+    assert NK == 1, 'NK > 1 is not supported because it involves time-consuming synchronization'
+    if head_first:
+        h = kg.new_empty(B, H, NT, K, V)
+    else:
+        h = kg.new_empty(B, NT, H, K, V)
+    final_state = kg.new_empty(N, H, K, V, dtype=torch.float32) if output_final_state else None
+    v_new = torch.empty_like(u)
+    grid = (NK, NV, N * H)
+    chunk_dplr_fwd_kernel_h[grid](
+        kg=kg,
+        v=v,
+        w=w,
+        bg=bg,
+        u=u,
+        v_new=v_new,
+        h=h,
+        gk=gk,
+        h0=initial_state,
+        ht=final_state,
+        offsets=offsets,
+        chunk_offsets=chunk_offsets,
+        T=T,
+        H=H,
+        K=K,
+        V=V,
+        BT=BT,
+        BC=BC,
+        BK=BK,
+        BV=BV,
+        NT=NT,
+        HEAD_FIRST=head_first
+    )
+    return h, v_new, final_state

fla/ops/gla/__pycache__/__init__.cpython-312.pyc ADDED Viewed

Binary file (336 Bytes). View file

fla/ops/gla/__pycache__/chunk.cpython-312.pyc ADDED Viewed

Binary file (81.8 kB). View file

fla/ops/gla/__pycache__/fused_recurrent.cpython-312.pyc ADDED Viewed

Binary file (5.69 kB). View file

fla/ops/gla/chunk.py ADDED Viewed

	@@ -0,0 +1,1486 @@

+# -*- coding: utf-8 -*-
+# Copyright (c) 2023-2025, Songlin Yang, Yu Zhang
+from typing import Optional, Tuple
+import torch
+import triton
+import triton.language as tl
+from fla.ops.common.chunk_h import chunk_bwd_dh, chunk_fwd_h
+from fla.ops.common.utils import prepare_chunk_indices
+from fla.ops.utils import chunk_local_cumsum
+from fla.ops.utils.op import exp, safe_exp
+from fla.utils import check_shared_mem, input_guard
+BK_LIST = [32, 64] if check_shared_mem() else [16, 32]
+BV_LIST = [64, 128] if check_shared_mem('ampere') else [16, 32]
+@triton.heuristics({
+    'USE_OFFSETS': lambda args: args['offsets'] is not None
+})
+@triton.autotune(
+    configs=[
+        triton.Config({'BK': BK}, num_warps=num_warps, num_stages=num_stages)
+        for BK in [32, 64]
+        for num_warps in [1, 2, 4, 8]
+        for num_stages in [2, 3, 4]
+    ],
+    key=["BC"]
+)
+@triton.jit(do_not_specialize=['T'])
+def chunk_gla_fwd_A_kernel_intra_sub_inter(
+    q,
+    k,
+    g,
+    A,
+    offsets,
+    indices,
+    scale,
+    T,
+    H: tl.constexpr,
+    K: tl.constexpr,
+    BT: tl.constexpr,
+    BC: tl.constexpr,
+    BK: tl.constexpr,
+    NC: tl.constexpr,
+    USE_OFFSETS: tl.constexpr,
+    HEAD_FIRST: tl.constexpr
+):
+    i_t, i_c, i_bh = tl.program_id(0), tl.program_id(1), tl.program_id(2)
+    i_b, i_h = i_bh // H, i_bh % H
+    i_i, i_j = i_c // NC, i_c % NC
+    if USE_OFFSETS:
+        i_n, i_t = tl.load(indices + i_t * 2).to(tl.int32), tl.load(indices + i_t * 2 + 1).to(tl.int32)
+        bos, eos = tl.load(offsets + i_n).to(tl.int32), tl.load(offsets + i_n + 1).to(tl.int32)
+        T = eos - bos
+    else:
+        bos, eos = i_b * T, i_b * T + T
+    if i_t * BT + i_i * BC >= T:
+        return
+    if i_i <= i_j:
+        return
+    b_A = tl.zeros([BC, BC], dtype=tl.float32)
+    for i_k in range(tl.cdiv(K, BK)):
+        o_k = i_k * BK + tl.arange(0, BK)
+        m_k = o_k < K
+        if HEAD_FIRST:
+            p_q = tl.make_block_ptr(q + i_bh * T*K, (T, K), (K, 1), (i_t * BT + i_i * BC, i_k * BK), (BC, BK), (1, 0))
+            p_g = tl.make_block_ptr(g + i_bh * T*K, (T, K), (K, 1), (i_t * BT + i_i * BC, i_k * BK), (BC, BK), (1, 0))
+            p_k = tl.make_block_ptr(k + i_bh * T*K, (K, T), (1, K), (i_k * BK, i_t * BT + i_j * BC), (BK, BC), (0, 1))
+            p_gk = tl.make_block_ptr(g + i_bh * T*K, (K, T), (1, K), (i_k * BK, i_t * BT + i_j * BC), (BK, BC), (0, 1))
+            p_gn = tl.max_contiguous(tl.multiple_of(g + (i_bh * T + i_t * BT + i_i * BC) * K + o_k, BK), BK)
+        else:
+            p_q = tl.make_block_ptr(q + (bos*H+i_h)*K, (T, K), (H*K, 1), (i_t * BT + i_i * BC, i_k * BK), (BC, BK), (1, 0))
+            p_g = tl.make_block_ptr(g + (bos*H+i_h)*K, (T, K), (H*K, 1), (i_t * BT + i_i * BC, i_k * BK), (BC, BK), (1, 0))
+            p_k = tl.make_block_ptr(k + (bos*H+i_h)*K, (K, T), (1, H*K), (i_k * BK, i_t * BT + i_j * BC), (BK, BC), (0, 1))
+            p_gk = tl.make_block_ptr(g + (bos*H+i_h)*K, (K, T), (1, H*K), (i_k * BK, i_t * BT + i_j * BC), (BK, BC), (0, 1))
+            p_gn = g + (bos + i_t * BT + i_i * BC) * H*K + i_h * K + o_k
+        # [BK,]
+        b_gn = tl.load(p_gn, mask=m_k, other=0)
+        # [BC, BK]
+        b_q = tl.load(p_q, boundary_check=(0, 1))
+        b_g = tl.load(p_g, boundary_check=(0, 1))
+        b_qg = b_q * exp(b_g - b_gn[None, :]) * scale
+        # [BK, BC]
+        b_k = tl.load(p_k, boundary_check=(0, 1))
+        b_gk = tl.load(p_gk, boundary_check=(0, 1))
+        b_kg = b_k * exp(b_gn[:, None] - b_gk)
+        # [BC, BC] using tf32 to improve precision here.
+        b_A += tl.dot(b_qg, b_kg)
+    if HEAD_FIRST:
+        p_A = tl.make_block_ptr(A + i_bh*T*BT, (T, BT), (BT, 1), (i_t * BT + i_i * BC, i_j * BC), (BC, BC), (1, 0))
+    else:
+        p_A = tl.make_block_ptr(A + (bos*H + i_h)*BT, (T, BT), (H*BT, 1), (i_t * BT + i_i * BC, i_j * BC), (BC, BC), (1, 0))
+    tl.store(p_A, b_A.to(A.dtype.element_ty), boundary_check=(0, 1))
+@triton.heuristics({
+    'USE_OFFSETS': lambda args: args['offsets'] is not None
+})
+@triton.autotune(
+    configs=[
+        triton.Config({}, num_warps=1),
+        triton.Config({}, num_warps=2),
+        triton.Config({}, num_warps=4),
+        triton.Config({}, num_warps=8),
+    ],
+    key=["BK", "BT"]
+)
+@triton.jit(do_not_specialize=['T'])
+def chunk_gla_fwd_A_kernel_intra_sub_intra(
+    q,
+    k,
+    g,
+    A,
+    offsets,
+    indices,
+    scale,
+    T,
+    H: tl.constexpr,
+    K: tl.constexpr,
+    BT: tl.constexpr,
+    BC: tl.constexpr,
+    BK: tl.constexpr,
+    USE_OFFSETS: tl.constexpr,
+    HEAD_FIRST: tl.constexpr
+):
+    i_t, i_i, i_bh = tl.program_id(0), tl.program_id(1), tl.program_id(2)
+    i_b, i_h = i_bh // H, i_bh % H
+    i_j = i_i
+    if USE_OFFSETS:
+        i_n, i_t = tl.load(indices + i_t * 2).to(tl.int32), tl.load(indices + i_t * 2 + 1).to(tl.int32)
+        bos, eos = tl.load(offsets + i_n).to(tl.int32), tl.load(offsets + i_n + 1).to(tl.int32)
+        T = eos - bos
+    else:
+        bos, eos = i_b * T, i_b * T + T
+    if i_t * BT + i_i * BC >= T:
+        return
+    o_i = tl.arange(0, BC)
+    o_k = tl.arange(0, BK)
+    m_k = o_k < K
+    m_A = (i_t * BT + i_i * BC + tl.arange(0, BC)) < T
+    if HEAD_FIRST:
+        o_A = i_bh * T*BT + (i_t * BT + i_i * BC + tl.arange(0, BC)) * BT + i_j * BC
+        p_q = tl.make_block_ptr(q + i_bh * T*K, (T, K), (K, 1), (i_t * BT + i_i * BC, 0), (BC, BK), (1, 0))
+        p_g = tl.make_block_ptr(g + i_bh * T*K, (T, K), (K, 1), (i_t * BT + i_i * BC, 0), (BC, BK), (1, 0))
+        p_k = tl.max_contiguous(tl.multiple_of(k + (i_bh * T + i_t * BT + i_j * BC) * K + o_k, BK), BK)
+        p_gk = tl.max_contiguous(tl.multiple_of(g + (i_bh * T + i_t * BT + i_j * BC) * K + o_k, BK), BK)
+    else:
+        o_A = (bos + i_t * BT + i_i * BC + tl.arange(0, BC)) * H*BT + i_h * BT + i_j * BC
+        p_q = tl.make_block_ptr(q + (bos * H + i_h) * K, (T, K), (H*K, 1), (i_t * BT + i_i * BC, 0), (BC, BK), (1, 0))
+        p_g = tl.make_block_ptr(g + (bos * H + i_h) * K, (T, K), (H*K, 1), (i_t * BT + i_i * BC, 0), (BC, BK), (1, 0))
+        p_k = k + (bos + i_t * BT + i_j * BC) * H*K + i_h * K + o_k
+        p_gk = g + (bos + i_t * BT + i_j * BC) * H*K + i_h * K + o_k
+    b_q = tl.load(p_q, boundary_check=(0, 1))
+    b_g = tl.load(p_g, boundary_check=(0, 1))
+    for j in range(0, min(BC, T - i_t * BT - i_i * BC)):
+        b_k = tl.load(p_k, mask=m_k, other=0).to(tl.float32)
+        b_gk = tl.load(p_gk, mask=m_k, other=0).to(tl.float32)
+        b_A = tl.sum(b_q * b_k[None, :] * exp(b_g - b_gk[None, :]), 1)
+        b_A = tl.where(o_i >= j, b_A * scale, 0.)
+        tl.store(A + o_A + j, b_A, mask=m_A)
+        p_k += K if HEAD_FIRST else H*K
+        p_gk += K if HEAD_FIRST else H*K
+@triton.heuristics({
+    'USE_OFFSETS': lambda args: args['offsets'] is not None
+})
+@triton.autotune(
+    configs=[
+        triton.Config({}, num_warps=1),
+        triton.Config({}, num_warps=2),
+        triton.Config({}, num_warps=4),
+        triton.Config({}, num_warps=8),
+    ],
+    key=['BC', 'BK']
+)
+@triton.jit(do_not_specialize=['T'])
+def chunk_gla_fwd_A_kernel_intra_sub_intra_split(
+    q,
+    k,
+    g,
+    A,
+    offsets,
+    indices,
+    scale,
+    T,
+    B: tl.constexpr,
+    H: tl.constexpr,
+    K: tl.constexpr,
+    BT: tl.constexpr,
+    BC: tl.constexpr,
+    BK: tl.constexpr,
+    NC: tl.constexpr,
+    USE_OFFSETS: tl.constexpr,
+    HEAD_FIRST: tl.constexpr
+):
+    i_k, i_tc, i_bh = tl.program_id(0), tl.program_id(1), tl.program_id(2)
+    i_b, i_h = i_bh // H, i_bh % H
+    i_t, i_i = i_tc // NC, i_tc % NC
+    i_j = i_i
+    if USE_OFFSETS:
+        i_n, i_t = tl.load(indices + i_t * 2).to(tl.int32), tl.load(indices + i_t * 2 + 1).to(tl.int32)
+        bos, eos = tl.load(offsets + i_n).to(tl.int32), tl.load(offsets + i_n + 1).to(tl.int32)
+        all = T
+        T = eos - bos
+    else:
+        bos, eos = i_b * T, i_b * T + T
+        all = B * T
+    if i_t * BT + i_i * BC >= T:
+        return
+    o_i = tl.arange(0, BC)
+    o_k = i_k * BK + tl.arange(0, BK)
+    m_k = o_k < K
+    m_A = (i_t * BT + i_i * BC + tl.arange(0, BC)) < T
+    if HEAD_FIRST:
+        o_A = (i_k * B*H + i_bh) * T * BC + (i_t * BT + i_i * BC + tl.arange(0, BC)) * BC
+        p_q = tl.make_block_ptr(q + i_bh * T*K, (T, K), (K, 1), (i_t * BT + i_i * BC, i_k * BK), (BC, BK), (1, 0))
+        p_g = tl.make_block_ptr(g + i_bh * T*K, (T, K), (K, 1), (i_t * BT + i_i * BC, i_k * BK), (BC, BK), (1, 0))
+        p_k = tl.max_contiguous(tl.multiple_of(k + (i_bh * T + i_t * BT + i_j * BC) * K + o_k, BK), BK)
+        p_gk = tl.max_contiguous(tl.multiple_of(g + (i_bh * T + i_t * BT + i_j * BC) * K + o_k, BK), BK)
+    else:
+        o_A = (i_k * all + bos + i_t * BT + i_i * BC + tl.arange(0, BC)) * H*BC + i_h * BC
+        p_q = tl.make_block_ptr(q + (bos * H + i_h) * K, (T, K), (H*K, 1), (i_t * BT + i_i * BC, i_k * BK), (BC, BK), (1, 0))
+        p_g = tl.make_block_ptr(g + (bos * H + i_h) * K, (T, K), (H*K, 1), (i_t * BT + i_i * BC, i_k * BK), (BC, BK), (1, 0))
+        p_k = k + (bos + i_t * BT + i_j * BC) * H*K + i_h * K + o_k
+        p_gk = g + (bos + i_t * BT + i_j * BC) * H*K + i_h * K + o_k
+    b_q = tl.load(p_q, boundary_check=(0, 1))
+    b_g = tl.load(p_g, boundary_check=(0, 1))
+    for j in range(0, min(BC, T - i_t * BT - i_i * BC)):
+        b_A = tl.zeros([BC], dtype=tl.float32)
+        b_k = tl.load(p_k, mask=m_k, other=0).to(tl.float32)
+        b_gk = tl.load(p_gk, mask=m_k, other=0).to(tl.float32)
+        b_A += tl.sum(b_q * b_k[None, :] * exp(b_g - b_gk[None, :]), 1)
+        b_A = tl.where(o_i >= j, b_A * scale, 0.)
+        tl.store(A + o_A + j, b_A, mask=m_A)
+        p_k += K if HEAD_FIRST else H*K
+        p_gk += K if HEAD_FIRST else H*K
+@triton.heuristics({
+    'USE_OFFSETS': lambda args: args['offsets'] is not None
+})
+@triton.autotune(
+    configs=[
+        triton.Config({}, num_warps=1),
+        triton.Config({}, num_warps=2),
+        triton.Config({}, num_warps=4),
+        triton.Config({}, num_warps=8),
+    ],
+    key=['BC']
+)
+@triton.jit(do_not_specialize=['T'])
+def chunk_gla_fwd_A_kernel_intra_sub_intra_merge(
+    A,
+    A2,
+    offsets,
+    indices,
+    T,
+    B: tl.constexpr,
+    H: tl.constexpr,
+    BT: tl.constexpr,
+    BC: tl.constexpr,
+    NK: tl.constexpr,
+    USE_OFFSETS: tl.constexpr,
+    HEAD_FIRST: tl.constexpr
+):
+    i_t, i_c, i_bh = tl.program_id(0), tl.program_id(1), tl.program_id(2)
+    i_b, i_h = i_bh // H, i_bh % H
+    if USE_OFFSETS:
+        i_n, i_t = tl.load(indices + i_t * 2).to(tl.int32), tl.load(indices + i_t * 2 + 1).to(tl.int32)
+        bos, eos = tl.load(offsets + i_n).to(tl.int32), tl.load(offsets + i_n + 1).to(tl.int32)
+        all = T
+        T = eos - bos
+    else:
+        bos, eos = i_b * T, i_b * T + T
+        all = B * T
+    if i_t * BT + i_c * BC >= T:
+        return
+    b_A = tl.zeros([BC, BC], dtype=tl.float32)
+    for i_k in range(0, NK):
+        if HEAD_FIRST:
+            p_A = tl.make_block_ptr(A + (i_k*B*H+i_bh)*T*BC, (T, BC), (BC, 1), (i_t*BT + i_c*BC, 0), (BC, BC), (1, 0))
+        else:
+            p_A = tl.make_block_ptr(A + (i_k*all+bos)*H*BC+i_h*BC, (T, BC), (H*BC, 1), (i_t*BT + i_c*BC, 0), (BC, BC), (1, 0))
+        b_A += tl.load(p_A, boundary_check=(0, 1))
+    if HEAD_FIRST:
+        p_A2 = tl.make_block_ptr(A2 + i_bh*T*BT, (T, BT), (BT, 1), (i_t * BT + i_c * BC, i_c * BC), (BC, BC), (1, 0))
+    else:
+        p_A2 = tl.make_block_ptr(A2 + (bos*H+i_h)*BT, (T, BT), (H*BT, 1), (i_t * BT + i_c * BC, i_c * BC), (BC, BC), (1, 0))
+    tl.store(p_A2, b_A.to(A2.dtype.element_ty), boundary_check=(0, 1))
+@triton.heuristics({
+    'USE_OFFSETS': lambda args: args['offsets'] is not None
+})
+@triton.autotune(
+    configs=[
+        triton.Config({'BK': BK, 'BV': BV}, num_warps=num_warps)
+        for BK in [32, 64]
+        for BV in [64, 128]
+        for num_warps in [2, 4, 8]
+    ],
+    key=['BT'],
+)
+@triton.jit(do_not_specialize=['T'])
+def chunk_gla_fwd_kernel_o(
+    q,
+    v,
+    g,
+    h,
+    o,
+    A,
+    offsets,
+    indices,
+    scale,
+    T,
+    H: tl.constexpr,
+    K: tl.constexpr,
+    V: tl.constexpr,
+    BT: tl.constexpr,
+    BK: tl.constexpr,
+    BV: tl.constexpr,
+    USE_OFFSETS: tl.constexpr,
+    HEAD_FIRST: tl.constexpr
+):
+    i_v, i_t, i_bh = tl.program_id(0), tl.program_id(1), tl.program_id(2)
+    i_b, i_h = i_bh // H, i_bh % H
+    if USE_OFFSETS:
+        i_tg = i_t
+        i_n, i_t = tl.load(indices + i_t * 2).to(tl.int32), tl.load(indices + i_t * 2 + 1).to(tl.int32)
+        bos, eos = tl.load(offsets + i_n).to(tl.int32), tl.load(offsets + i_n + 1).to(tl.int32)
+        T = eos - bos
+        NT = tl.cdiv(T, BT)
+    else:
+        NT = tl.cdiv(T, BT)
+        i_tg = i_b * NT + i_t
+        bos, eos = i_b * T, i_b * T + T
+    m_s = tl.arange(0, BT)[:, None] >= tl.arange(0, BT)[None, :]
+    b_o = tl.zeros([BT, BV], dtype=tl.float32)
+    for i_k in range(tl.cdiv(K, BK)):
+        if HEAD_FIRST:
+            p_q = tl.make_block_ptr(q + i_bh * T*K, (T, K), (K, 1), (i_t * BT, i_k * BK), (BT, BK), (1, 0))
+            p_g = tl.make_block_ptr(g + i_bh * T*K, (T, K), (K, 1), (i_t * BT, i_k * BK), (BT, BK), (1, 0))
+            p_h = tl.make_block_ptr(h + (i_bh * NT + i_t) * K*V, (K, V), (V, 1), (i_k * BK, i_v * BV), (BK, BV), (1, 0))
+        else:
+            p_q = tl.make_block_ptr(q + (bos * H + i_h) * K, (T, K), (H*K, 1), (i_t * BT, i_k * BK), (BT, BK), (1, 0))
+            p_g = tl.make_block_ptr(g + (bos * H + i_h) * K, (T, K), (H*K, 1), (i_t * BT, i_k * BK), (BT, BK), (1, 0))
+            p_h = tl.make_block_ptr(h + (i_tg * H + i_h) * K*V, (K, V), (V, 1), (i_k * BK, i_v * BV), (BK, BV), (1, 0))
+        # [BT, BK]
+        b_q = tl.load(p_q, boundary_check=(0, 1))
+        b_q = (b_q * scale).to(b_q.dtype)
+        # [BT, BK]
+        b_g = tl.load(p_g, boundary_check=(0, 1))
+        # [BT, BK]
+        b_qg = (b_q * exp(b_g)).to(b_q.dtype)
+        # [BK, BV]
+        b_h = tl.load(p_h, boundary_check=(0, 1))
+        # works but dkw, owing to divine benevolence
+        # [BT, BV]
+        if i_k >= 0:
+            b_o += tl.dot(b_qg, b_h.to(b_qg.dtype))
+    if HEAD_FIRST:
+        p_v = tl.make_block_ptr(v + i_bh * T*V, (T, V), (V, 1), (i_t * BT, i_v * BV), (BT, BV), (1, 0))
+        p_o = tl.make_block_ptr(o + i_bh * T*V, (T, V), (V, 1), (i_t * BT, i_v * BV), (BT, BV), (1, 0))
+        p_A = tl.make_block_ptr(A + i_bh * T*BT, (T, BT), (BT, 1), (i_t * BT, 0), (BT, BT), (1, 0))
+    else:
+        p_v = tl.make_block_ptr(v + (bos * H + i_h) * V, (T, V), (H*V, 1), (i_t * BT, i_v * BV), (BT, BV), (1, 0))
+        p_o = tl.make_block_ptr(o + (bos * H + i_h) * V, (T, V), (H*V, 1), (i_t * BT, i_v * BV), (BT, BV), (1, 0))
+        p_A = tl.make_block_ptr(A + (bos * H + i_h) * BT, (T, BT), (H*BT, 1), (i_t * BT, 0), (BT, BT), (1, 0))
+    # [BT, BV]
+    b_v = tl.load(p_v, boundary_check=(0, 1))
+    # [BT, BT]
+    b_A = tl.load(p_A, boundary_check=(0, 1))
+    b_A = tl.where(m_s, b_A, 0.).to(b_v.dtype)
+    b_o += tl.dot(b_A, b_v, allow_tf32=False)
+    tl.store(p_o, b_o.to(p_o.dtype.element_ty), boundary_check=(0, 1))
+@triton.heuristics({
+    'USE_OFFSETS': lambda args: args['offsets'] is not None
+})
+@triton.autotune(
+    configs=[
+        triton.Config({}, num_warps=1),
+        triton.Config({}, num_warps=2),
+        triton.Config({}, num_warps=4),
+        triton.Config({}, num_warps=8),
+    ],
+    key=['BK', 'NC', 'BT'],
+)
+@triton.jit(do_not_specialize=['T'])
+def chunk_gla_bwd_kernel_intra(
+    q,
+    k,
+    g,
+    dA,
+    dq,
+    dk,
+    offsets,
+    indices,
+    T,
+    H: tl.constexpr,
+    K: tl.constexpr,
+    BT: tl.constexpr,
+    BC: tl.constexpr,
+    BK: tl.constexpr,
+    NC: tl.constexpr,
+    USE_OFFSETS: tl.constexpr,
+    HEAD_FIRST: tl.constexpr
+):
+    i_k, i_c, i_bh = tl.program_id(0), tl.program_id(1), tl.program_id(2)
+    i_b, i_h = i_bh // H, i_bh % H
+    i_t, i_i = i_c // NC, i_c % NC
+    if USE_OFFSETS:
+        i_n, i_t = tl.load(indices + i_t * 2).to(tl.int32), tl.load(indices + i_t * 2 + 1).to(tl.int32)
+        bos, eos = tl.load(offsets + i_n).to(tl.int32), tl.load(offsets + i_n + 1).to(tl.int32)
+    else:
+        bos, eos = i_b * T, i_b * T + T
+    T = eos - bos
+    if i_t * BT + i_i * BC >= T:
+        return
+    o_k = i_k * BK + tl.arange(0, BK)
+    m_k = o_k < K
+    if HEAD_FIRST:
+        p_g = tl.make_block_ptr(g + i_bh * T*K, (T, K), (K, 1), (i_t * BT + i_i * BC, i_k * BK), (BC, BK), (1, 0))
+    else:
+        p_g = tl.make_block_ptr(g + (bos*H + i_h) * K, (T, K), (H*K, 1), (i_t * BT + i_i * BC, i_k * BK), (BC, BK), (1, 0))
+    # [BC, BK]
+    b_g = tl.load(p_g, boundary_check=(0, 1))
+    b_dq = tl.zeros([BC, BK], dtype=tl.float32)
+    if i_i > 0:
+        if HEAD_FIRST:
+            p_gn = g + (i_bh * T + i_t * BT + i_i * BC) * K + o_k
+            p_gn = tl.max_contiguous(tl.multiple_of(p_gn, BK), BK)
+        else:
+            p_gn = g + (bos + i_t * BT + i_i * BC) * H*K + i_h*K + o_k
+        # [BK,]
+        b_gn = tl.load(p_gn, mask=m_k, other=0)
+        for i_j in range(0, i_i):
+            if HEAD_FIRST:
+                p_k = tl.make_block_ptr(k + i_bh * T*K, (T, K), (K, 1), (i_t * BT + i_j * BC, i_k * BK), (BC, BK), (1, 0))
+                p_gk = tl.make_block_ptr(g + i_bh * T*K, (T, K), (K, 1), (i_t * BT + i_j * BC, i_k * BK), (BC, BK), (1, 0))
+                p_dA = tl.make_block_ptr(dA + i_bh * T*BT, (T, BT), (BT, 1), (i_t * BT + i_i * BC, i_j * BC), (BC, BC), (1, 0))
+            else:
+                p_k = tl.make_block_ptr(k+(bos*H+i_h)*K, (T, K), (H*K, 1), (i_t*BT+i_j*BC, i_k * BK), (BC, BK), (1, 0))
+                p_gk = tl.make_block_ptr(g+(bos*H+i_h)*K, (T, K), (H*K, 1), (i_t*BT+i_j*BC, i_k * BK), (BC, BK), (1, 0))
+                p_dA = tl.make_block_ptr(dA+(bos*H+i_h)*BT, (T, BT), (H*BT, 1), (i_t*BT+i_i*BC, i_j * BC), (BC, BC), (1, 0))
+            # [BC, BK]
+            b_k = tl.load(p_k, boundary_check=(0, 1))
+            b_gk = tl.load(p_gk, boundary_check=(0, 1))
+            b_kg = (b_k * exp(b_gn[None, :] - b_gk))
+            # [BC, BC]
+            b_dA = tl.load(p_dA, boundary_check=(0, 1))
+            # [BC, BK]
+            b_dq += tl.dot(b_dA, b_kg)
+        b_dq *= exp(b_g - b_gn[None, :])
+    o_i = tl.arange(0, BC)
+    m_dA = (i_t * BT + i_i * BC + tl.arange(0, BC)) < T
+    if HEAD_FIRST:
+        o_dA = i_bh * T*BT + (i_t * BT + i_i * BC + tl.arange(0, BC)) * BT + i_i * BC
+        p_kj = tl.max_contiguous(tl.multiple_of(k + (i_bh * T + i_t * BT + i_i * BC) * K + o_k, BK), BK)
+        p_gkj = tl.max_contiguous(tl.multiple_of(g + (i_bh * T + i_t * BT + i_i * BC) * K + o_k, BK), BK)
+        p_dq = tl.make_block_ptr(dq + i_bh * T*K, (T, K), (K, 1), (i_t * BT + i_i * BC, i_k * BK), (BC, BK), (1, 0))
+    else:
+        o_dA = bos*H*BT + (i_t * BT + i_i * BC + tl.arange(0, BC)) * H*BT + i_h * BT + i_i * BC
+        p_kj = k + (bos + i_t * BT + i_i * BC) * H*K + i_h * K + o_k
+        p_gkj = g + (bos + i_t * BT + i_i * BC) * H*K + i_h * K + o_k
+        p_dq = tl.make_block_ptr(dq + (bos*H + i_h) * K, (T, K), (H*K, 1), (i_t * BT + i_i * BC, i_k * BK), (BC, BK), (1, 0))
+    for j in range(0, min(BC, T - i_t * BT - i_i * BC)):
+        # [BC,]
+        b_dA = tl.load(dA + o_dA + j, mask=m_dA, other=0)
+        # [BK,]
+        b_kj = tl.load(p_kj, mask=m_k, other=0).to(tl.float32)
+        b_gkj = tl.load(p_gkj, mask=m_k, other=0).to(tl.float32)
+        # [BC, BK]
+        m_i = o_i[:, None] >= j
+        # [BC, BK]
+        # (SY 09/17) important to not use bf16 here to have a good precision.
+        b_dq += tl.where(m_i, b_dA[:, None] * b_kj[None, :] * exp(b_g - b_gkj[None, :]), 0.)
+        p_kj += K if HEAD_FIRST else H*K
+        p_gkj += K if HEAD_FIRST else H*K
+    tl.store(p_dq, b_dq.to(p_dq.dtype.element_ty), boundary_check=(0, 1))
+    tl.debug_barrier()
+    if HEAD_FIRST:
+        p_k = tl.make_block_ptr(k + i_bh * T*K, (T, K), (K, 1), (i_t * BT + i_i * BC, i_k * BK), (BC, BK), (1, 0))
+        p_gk = tl.make_block_ptr(g + i_bh * T*K, (T, K), (K, 1), (i_t * BT + i_i * BC, i_k * BK), (BC, BK), (1, 0))
+    else:
+        p_k = tl.make_block_ptr(k + (bos*H + i_h) * K, (T, K), (H*K, 1), (i_t * BT + i_i * BC, i_k * BK), (BC, BK), (1, 0))
+        p_gk = tl.make_block_ptr(g + (bos*H + i_h) * K, (T, K), (H*K, 1), (i_t * BT + i_i * BC, i_k * BK), (BC, BK), (1, 0))
+    # [BC, BK]
+    b_k = tl.load(p_k, boundary_check=(0, 1))
+    b_gk = tl.load(p_gk, boundary_check=(0, 1))
+    b_dk = tl.zeros([BC, BK], dtype=tl.float32)
+    NC = min(NC, tl.cdiv(T - i_t * BT, BC))
+    if i_i < NC - 1:
+        if HEAD_FIRST:
+            p_gn = g + (i_bh * T + min(i_t * BT + i_i * BC + BC, T) - 1) * K + o_k
+            p_gn = tl.max_contiguous(tl.multiple_of(p_gn, BK), BK)
+        else:
+            p_gn = g + (bos + min(i_t * BT + i_i * BC + BC, T) - 1) * H*K + i_h * K + o_k
+        # [BK,]
+        b_gn = tl.load(p_gn, mask=m_k, other=0)
+        for i_j in range(i_i + 1, NC):
+            if HEAD_FIRST:
+                p_q = tl.make_block_ptr(q + i_bh * T*K, (T, K), (K, 1), (i_t*BT + i_j*BC, i_k*BK), (BC, BK), (1, 0))
+                p_gq = tl.make_block_ptr(g + i_bh * T*K, (T, K), (K, 1), (i_t*BT + i_j*BC, i_k*BK), (BC, BK), (1, 0))
+                p_dA = tl.make_block_ptr(dA + i_bh * T * BT, (BT, T), (1, BT), (i_i*BC, i_t*BT + i_j*BC), (BC, BC), (0, 1))
+            else:
+                p_q = tl.make_block_ptr(q + (bos*H+i_h)*K, (T, K), (H*K, 1), (i_t*BT+i_j*BC, i_k*BK), (BC, BK), (1, 0))
+                p_gq = tl.make_block_ptr(g + (bos*H+i_h)*K, (T, K), (H*K, 1), (i_t*BT+i_j*BC, i_k*BK), (BC, BK), (1, 0))
+                p_dA = tl.make_block_ptr(dA + (bos*H+i_h)*BT, (BT, T), (1, H*BT), (i_i*BC, i_t*BT + i_j*BC), (BC, BC), (0, 1))
+            # [BC, BK]
+            b_q = tl.load(p_q, boundary_check=(0, 1))
+            b_gq = tl.load(p_gq, boundary_check=(0, 1))
+            b_qg = b_q * safe_exp(b_gq - b_gn[None, :])
+            # [BC, BC]
+            b_dA = tl.load(p_dA, boundary_check=(0, 1))
+            # [BC, BK]
+            # (SY 09/17) important to not use bf16 here to have a good precision.
+            b_dk += tl.dot(b_dA, b_qg)
+        b_dk *= exp(b_gn[None, :] - b_gk)
+    if HEAD_FIRST:
+        o_dA = i_bh * T * BT + (i_t * BT + i_i * BC) * BT + i_i * BC + tl.arange(0, BC)
+        p_qj = tl.max_contiguous(tl.multiple_of(q + (i_bh * T + i_t * BT + i_i * BC) * K + o_k, BK), BK)
+        p_gqj = tl.max_contiguous(tl.multiple_of(g + (i_bh * T + i_t * BT + i_i * BC) * K + o_k, BK), BK)
+        p_dk = tl.make_block_ptr(dk + i_bh*T*K, (T, K), (K, 1), (i_t * BT + i_i * BC, i_k * BK), (BC, BK), (1, 0))
+    else:
+        o_dA = bos*H*BT + (i_t * BT + i_i * BC) * H*BT + i_h * BT + i_i * BC + tl.arange(0, BC)
+        p_qj = q + (bos + i_t * BT + i_i * BC) * H*K + i_h * K + o_k
+        p_gqj = g + (bos + i_t * BT + i_i * BC) * H*K + i_h * K + o_k
+        p_dk = tl.make_block_ptr(dk + (bos*H+i_h)*K, (T, K), (H*K, 1), (i_t * BT + i_i * BC, i_k * BK), (BC, BK), (1, 0))
+    for j in range(0, min(BC, T - i_t * BT - i_i * BC)):
+        # [BC,]
+        b_dA = tl.load(dA + o_dA + j * (1 if HEAD_FIRST else H) * BT)
+        # [BK,]
+        b_qj = tl.load(p_qj, mask=m_k, other=0).to(tl.float32)
+        b_gqj = tl.load(p_gqj, mask=m_k, other=0).to(tl.float32)
+        # [BC, BK]
+        m_i = o_i[:, None] <= j
+        b_dk += tl.where(m_i, b_dA[:, None] * b_qj[None, :] * exp(b_gqj[None, :] - b_gk), 0.)
+        p_qj += K if HEAD_FIRST else H*K
+        p_gqj += K if HEAD_FIRST else H*K
+    tl.store(p_dk, b_dk.to(p_dk.dtype.element_ty), boundary_check=(0, 1))
+@triton.heuristics({
+    'USE_OFFSETS': lambda args: args['offsets'] is not None
+})
+@triton.autotune(
+    configs=[
+        triton.Config({}, num_warps=1),
+        triton.Config({}, num_warps=2),
+        triton.Config({}, num_warps=4),
+        triton.Config({}, num_warps=8),
+    ],
+    key=['BV', 'BT'],
+)
+@triton.jit(do_not_specialize=['T'])
+def chunk_gla_bwd_kernel_dA(
+    v,
+    do,
+    dA,
+    offsets,
+    indices,
+    scale,
+    T,
+    H: tl.constexpr,
+    V: tl.constexpr,
+    BT: tl.constexpr,
+    BV: tl.constexpr,
+    USE_OFFSETS: tl.constexpr,
+    HEAD_FIRST: tl.constexpr
+):
+    i_t, i_bh = tl.program_id(0), tl.program_id(1)
+    i_b, i_h = i_bh // H, i_bh % H
+    if USE_OFFSETS:
+        i_n, i_t = tl.load(indices + i_t * 2).to(tl.int32), tl.load(indices + i_t * 2 + 1).to(tl.int32)
+        bos, eos = tl.load(offsets + i_n).to(tl.int32), tl.load(offsets + i_n + 1).to(tl.int32)
+    else:
+        bos, eos = i_b * T, i_b * T + T
+    T = eos - bos
+    b_dA = tl.zeros([BT, BT], dtype=tl.float32)
+    for i_v in range(tl.cdiv(V, BV)):
+        if HEAD_FIRST:
+            p_do = tl.make_block_ptr(do + i_bh * T*V, (T, V), (V, 1), (i_t * BT, i_v * BV), (BT, BV), (1, 0))
+            p_v = tl.make_block_ptr(v + i_bh * T*V, (V, T), (1, V), (i_v * BV, i_t * BT), (BV, BT), (0, 1))
+        else:
+            p_do = tl.make_block_ptr(do + (bos*H + i_h) * V, (T, V), (H*V, 1), (i_t * BT, i_v * BV), (BT, BV), (1, 0))
+            p_v = tl.make_block_ptr(v + (bos*H + i_h) * V, (V, T), (1, H*V), (i_v * BV, i_t * BT), (BV, BT), (0, 1))
+        b_v = tl.load(p_v, boundary_check=(0, 1))
+        b_do = tl.load(p_do, boundary_check=(0, 1))
+        b_dA += tl.dot(b_do, b_v)
+    if HEAD_FIRST:
+        p_dA = tl.make_block_ptr(dA + i_bh * T*BT, (T, BT), (BT, 1), (i_t * BT, 0), (BT, BT), (1, 0))
+    else:
+        p_dA = tl.make_block_ptr(dA + (bos * H + i_h) * BT, (T, BT), (H*BT, 1), (i_t * BT, 0), (BT, BT), (1, 0))
+    m_s = tl.arange(0, BT)[:, None] >= tl.arange(0, BT)[None, :]
+    b_dA = tl.where(m_s, b_dA * scale, 0.)
+    tl.store(p_dA, b_dA.to(p_dA.dtype.element_ty), boundary_check=(0, 1))
+@triton.heuristics({
+    'USE_OFFSETS': lambda args: args['offsets'] is not None
+})
+@triton.autotune(
+    configs=[
+        triton.Config({'BK': BK, 'BV': BV}, num_warps=num_warps)
+        for BK in BK_LIST
+        for BV in BV_LIST
+        for num_warps in [2, 4, 8]
+    ],
+    key=['BT'],
+)
+@triton.jit(do_not_specialize=['T'])
+def chunk_gla_bwd_kernel_dv(
+    k,
+    g,
+    A,
+    do,
+    dh,
+    dv,
+    offsets,
+    indices,
+    T,
+    H: tl.constexpr,
+    K: tl.constexpr,
+    V: tl.constexpr,
+    BT: tl.constexpr,
+    BK: tl.constexpr,
+    BV: tl.constexpr,
+    USE_OFFSETS: tl.constexpr,
+    HEAD_FIRST: tl.constexpr
+):
+    i_v, i_t, i_bh = tl.program_id(0), tl.program_id(1), tl.program_id(2)
+    i_b, i_h = i_bh // H, i_bh % H
+    if USE_OFFSETS:
+        i_tg = i_t
+        i_n, i_t = tl.load(indices + i_t * 2).to(tl.int32), tl.load(indices + i_t * 2 + 1).to(tl.int32)
+        bos, eos = tl.load(offsets + i_n).to(tl.int32), tl.load(offsets + i_n + 1).to(tl.int32)
+        T = eos - bos
+        NT = tl.cdiv(T, BT)
+    else:
+        NT = tl.cdiv(T, BT)
+        i_tg = i_b * NT + i_t
+        bos, eos = i_b * T, i_b * T + T
+    if HEAD_FIRST:
+        p_A = tl.make_block_ptr(A + i_bh * T * BT, (BT, T), (1, BT), (0, i_t * BT), (BT, BT), (0, 1))
+        p_do = tl.make_block_ptr(do + i_bh * T*V, (T, V), (V, 1), (i_t * BT, i_v * BV), (BT, BV), (1, 0))
+        p_dv = tl.make_block_ptr(dv + i_bh * T*V, (T, V), (V, 1), (i_t * BT, i_v * BV), (BT, BV), (1, 0))
+    else:
+        p_A = tl.make_block_ptr(A + (bos * H + i_h) * BT, (BT, T), (1, H*BT), (0, i_t * BT), (BT, BT), (0, 1))
+        p_do = tl.make_block_ptr(do + (bos * H + i_h) * V, (T, V), (H*V, 1), (i_t * BT, i_v * BV), (BT, BV), (1, 0))
+        p_dv = tl.make_block_ptr(dv + (bos * H + i_h) * V, (T, V), (H*V, 1), (i_t * BT, i_v * BV), (BT, BV), (1, 0))
+    b_A = tl.load(p_A, boundary_check=(0, 1))
+    b_A = tl.where(tl.arange(0, BT)[:, None] <= tl.arange(0, BT)[None, :], b_A, 0.)
+    b_do = tl.load(p_do, boundary_check=(0, 1))
+    # (SY 09/17) important to disallow tf32 here to maintain a good precision.
+    b_dv = tl.dot(b_A, b_do.to(b_A.dtype), allow_tf32=False)
+    for i_k in range(tl.cdiv(K, BK)):
+        o_k = i_k * BK + tl.arange(0, BK)
+        m_k = o_k < K
+        if HEAD_FIRST:
+            p_k = tl.make_block_ptr(k + i_bh * T*K, (T, K), (K, 1), (i_t * BT, i_k * BK), (BT, BK), (1, 0))
+            p_gk = tl.make_block_ptr(g + i_bh * T*K, (T, K), (K, 1), (i_t * BT, i_k * BK), (BT, BK), (1, 0))
+            p_gn = tl.max_contiguous(tl.multiple_of(g + i_bh * T*K + min(i_t * BT + BT, T) * K - K + o_k, BK), BK)
+            p_dh = tl.make_block_ptr(dh + (i_bh * NT + i_t) * K*V, (K, V), (V, 1), (i_k * BK, i_v * BV), (BK, BV), (1, 0))
+        else:
+            p_k = tl.make_block_ptr(k + (bos * H + i_h) * K, (T, K), (H*K, 1), (i_t * BT, i_k * BK), (BT, BK), (1, 0))
+            p_gk = tl.make_block_ptr(g + (bos * H + i_h) * K, (T, K), (H*K, 1), (i_t * BT, i_k * BK), (BT, BK), (1, 0))
+            p_gn = g + (bos + min(i_t * BT + BT, T) - 1)*H*K + i_h * K + o_k
+            p_dh = tl.make_block_ptr(dh + (i_tg * H + i_h) * K*V, (K, V), (V, 1), (i_k * BK, i_v * BV), (BK, BV), (1, 0))
+        b_k = tl.load(p_k, boundary_check=(0, 1))
+        b_gk = tl.load(p_gk, boundary_check=(0, 1))
+        b_gn = exp(tl.load(p_gn, mask=m_k, other=0)[None, :] - b_gk)
+        b_k = (b_k * b_gn).to(b_k.dtype)
+        b_dh = tl.load(p_dh, boundary_check=(0, 1))
+        # [BT, BV]
+        # (SY 09/17) it is ok to have bf16 interchunk gradient contribution here
+        b_dv += tl.dot(b_k, b_dh.to(b_k.dtype))
+    tl.store(p_dv, b_dv.to(p_dv.dtype.element_ty), boundary_check=(0, 1))
+@triton.heuristics({
+    'USE_OFFSETS': lambda args: args['offsets'] is not None
+})
+@triton.autotune(
+    configs=[
+        triton.Config({'BK': BK, 'BV': BV}, num_warps=num_warps)
+        for BK in BK_LIST
+        for BV in BV_LIST
+        for num_warps in [2, 4, 8]
+    ],
+    key=['BT'],
+)
+@triton.jit(do_not_specialize=['T'])
+def chunk_gla_bwd_kernel_inter(
+    q,
+    k,
+    v,
+    h,
+    g,
+    do,
+    dh,
+    dq,
+    dk,
+    dq2,
+    dk2,
+    dg,
+    offsets,
+    indices,
+    scale,
+    T,
+    H: tl.constexpr,
+    K: tl.constexpr,
+    V: tl.constexpr,
+    BT: tl.constexpr,
+    BK: tl.constexpr,
+    BV: tl.constexpr,
+    USE_OFFSETS: tl.constexpr,
+    HEAD_FIRST: tl.constexpr
+):
+    i_k, i_t, i_bh = tl.program_id(0), tl.program_id(1), tl.program_id(2)
+    i_b, i_h = i_bh // H, i_bh % H
+    if USE_OFFSETS:
+        i_tg = i_t
+        i_n, i_t = tl.load(indices + i_t * 2).to(tl.int32), tl.load(indices + i_t * 2 + 1).to(tl.int32)
+        bos, eos = tl.load(offsets + i_n).to(tl.int32), tl.load(offsets + i_n + 1).to(tl.int32)
+        T = eos - bos
+        NT = tl.cdiv(T, BT)
+    else:
+        NT = tl.cdiv(T, BT)
+        i_tg = i_b * NT + i_t
+        bos, eos = i_b * T, i_b * T + T
+    o_k = i_k * BK + tl.arange(0, BK)
+    m_k = o_k < K
+    if HEAD_FIRST:
+        p_gk = tl.make_block_ptr(g + i_bh * T*K, (T, K), (K, 1), (i_t * BT, i_k * BK), (BT, BK), (1, 0))
+        p_gn = tl.max_contiguous(tl.multiple_of(g + i_bh * T*K + (min(T, i_t * BT + BT)-1) * K + o_k, BK), BK)
+    else:
+        p_gk = tl.make_block_ptr(g + (bos*H+i_h)*K, (T, K), (H*K, 1), (i_t * BT, i_k * BK), (BT, BK), (1, 0))
+        p_gn = g + (bos + min(T, i_t * BT + BT)-1) * H*K + i_h * K + o_k
+    b_gn = tl.load(p_gn, mask=m_k, other=0)
+    b_dq = tl.zeros([BT, BK], dtype=tl.float32)
+    b_dk = tl.zeros([BT, BK], dtype=tl.float32)
+    b_dgk = tl.zeros([BK,], dtype=tl.float32)
+    for i_v in range(tl.cdiv(V, BV)):
+        if HEAD_FIRST:
+            p_v = tl.make_block_ptr(v + i_bh * T*V, (T, V), (V, 1), (i_t * BT, i_v * BV), (BT, BV), (1, 0))
+            p_do = tl.make_block_ptr(do + i_bh * T*V, (T, V), (V, 1), (i_t * BT, i_v * BV), (BT, BV), (1, 0))
+            p_h = tl.make_block_ptr(h + i_bh * NT*K*V + i_t * K*V, (V, K), (1, V), (i_v * BV, i_k * BK), (BV, BK), (0, 1))
+            p_dh = tl.make_block_ptr(dh + i_bh * NT*K*V + i_t * K*V, (V, K), (1, V), (i_v * BV, i_k * BK), (BV, BK), (0, 1))
+        else:
+            p_v = tl.make_block_ptr(v + (bos*H + i_h) * V, (T, V), (H*V, 1), (i_t * BT, i_v * BV), (BT, BV), (1, 0))
+            p_do = tl.make_block_ptr(do + (bos*H + i_h) * V, (T, V), (H*V, 1), (i_t * BT, i_v * BV), (BT, BV), (1, 0))
+            p_h = tl.make_block_ptr(h + (i_tg * H + i_h) * K*V, (V, K), (1, V), (i_v * BV, i_k * BK), (BV, BK), (0, 1))
+            p_dh = tl.make_block_ptr(dh + (i_tg * H + i_h) * K*V, (V, K), (1, V), (i_v * BV, i_k * BK), (BV, BK), (0, 1))
+        # [BT, BV]
+        b_v = tl.load(p_v, boundary_check=(0, 1))
+        b_do = tl.load(p_do, boundary_check=(0, 1))
+        # [BV, BK]
+        b_h = tl.load(p_h, boundary_check=(0, 1))
+        b_dh = tl.load(p_dh, boundary_check=(0, 1))
+        # [BK]
+        b_dgk += tl.sum(b_h * b_dh, axis=0)
+        # [BT, BK]
+        b_dq += tl.dot(b_do, b_h.to(b_do.dtype))
+        b_dk += tl.dot(b_v, b_dh.to(b_v.dtype))
+    b_dgk *= exp(b_gn)
+    b_dq *= scale
+    b_gk = tl.load(p_gk, boundary_check=(0, 1))
+    b_dq = b_dq * exp(b_gk)
+    b_dk = b_dk * exp(b_gn[None, :] - b_gk)
+    if HEAD_FIRST:
+        p_q = tl.make_block_ptr(q + i_bh * T*K, (T, K), (K, 1), (i_t * BT, i_k * BK), (BT, BK), (1, 0))
+        p_k = tl.make_block_ptr(k + i_bh * T*K, (T, K), (K, 1), (i_t * BT, i_k * BK), (BT, BK), (1, 0))
+        p_dq = tl.make_block_ptr(dq + i_bh * T*K, (T, K), (K, 1), (i_t * BT, i_k * BK), (BT, BK), (1, 0))
+        p_dk = tl.make_block_ptr(dk + i_bh * T*K, (T, K), (K, 1), (i_t * BT, i_k * BK), (BT, BK), (1, 0))
+    else:
+        p_q = tl.make_block_ptr(q + (bos*H+i_h)*K, (T, K), (H*K, 1), (i_t * BT, i_k * BK), (BT, BK), (1, 0))
+        p_k = tl.make_block_ptr(k + (bos*H+i_h)*K, (T, K), (H*K, 1), (i_t * BT, i_k * BK), (BT, BK), (1, 0))
+        p_dq = tl.make_block_ptr(dq + (bos*H+i_h)*K, (T, K), (H*K, 1), (i_t * BT, i_k * BK), (BT, BK), (1, 0))
+        p_dk = tl.make_block_ptr(dk + (bos*H+i_h)*K, (T, K), (H*K, 1), (i_t * BT, i_k * BK), (BT, BK), (1, 0))
+    b_q = tl.load(p_q, boundary_check=(0, 1))
+    b_k = tl.load(p_k, boundary_check=(0, 1))
+    b_dgk += tl.sum(b_dk * b_k, axis=0)
+    b_dq += tl.load(p_dq, boundary_check=(0, 1))
+    b_dk += tl.load(p_dk, boundary_check=(0, 1))
+    b_dg = b_q * b_dq - b_k * b_dk
+    # tl.debug_barrier()
+    b_dg = b_dg - tl.cumsum(b_dg, axis=0) + tl.sum(b_dg, axis=0)[None, :] + b_dgk[None, :]
+    # Buggy due to strange triton compiler issue.
+    # m_s = tl.where(tl.arange(0, BT)[:, None] <= tl.arange(0, BT)[None, :], 1., 0.)
+    # b_dg = tl.dot(m_s, b_dg, allow_tf32=False) + b_dgk[None, :]
+    if HEAD_FIRST:
+        p_dq = tl.make_block_ptr(dq2 + i_bh * T*K, (T, K), (K, 1), (i_t * BT, i_k * BK), (BT, BK), (1, 0))
+        p_dk = tl.make_block_ptr(dk2 + i_bh * T*K, (T, K), (K, 1), (i_t * BT, i_k * BK), (BT, BK), (1, 0))
+        p_dg = tl.make_block_ptr(dg + i_bh * T*K, (T, K), (K, 1), (i_t * BT, i_k * BK), (BT, BK), (1, 0))
+    else:
+        p_dq = tl.make_block_ptr(dq2 + (bos * H + i_h) * K, (T, K), (H*K, 1), (i_t * BT, i_k * BK), (BT, BK), (1, 0))
+        p_dk = tl.make_block_ptr(dk2 + (bos * H + i_h) * K, (T, K), (H*K, 1), (i_t * BT, i_k * BK), (BT, BK), (1, 0))
+        p_dg = tl.make_block_ptr(dg + (bos * H + i_h) * K, (T, K), (H*K, 1), (i_t * BT, i_k * BK), (BT, BK), (1, 0))
+    tl.store(p_dq, b_dq.to(p_dq.dtype.element_ty), boundary_check=(0, 1))
+    tl.store(p_dk, b_dk.to(p_dk.dtype.element_ty), boundary_check=(0, 1))
+    tl.store(p_dg, b_dg.to(p_dg.dtype.element_ty), boundary_check=(0, 1))
+def chunk_gla_fwd_intra_gk(
+    q: torch.Tensor,
+    k: torch.Tensor,
+    g: torch.Tensor,
+    scale: float,
+    offsets: Optional[torch.LongTensor] = None,
+    indices: Optional[torch.LongTensor] = None,
+    head_first: bool = True,
+    chunk_size: int = 64
+):
+    if head_first:
+        B, H, T, K = k.shape
+    else:
+        B, T, H, K = k.shape
+    BT = min(chunk_size, max(16, triton.next_power_of_2(T)))
+    NT = triton.cdiv(T, BT) if offsets is None else len(indices)
+    BC = min(16, BT)
+    NC = triton.cdiv(BT, BC)
+    A = q.new_empty(B, *((H, T) if head_first else (T, H)), BT, dtype=torch.float)
+    grid = (NT, NC * NC, B * H)
+    chunk_gla_fwd_A_kernel_intra_sub_inter[grid](
+        q,
+        k,
+        g,
+        A,
+        offsets,
+        indices,
+        scale,
+        T=T,
+        H=H,
+        K=K,
+        BT=BT,
+        BC=BC,
+        NC=NC,
+        HEAD_FIRST=head_first
+    )
+    grid = (NT, NC, B * H)
+    # load the entire [BC, K] blocks into SRAM at once
+    if K <= 256:
+        BK = triton.next_power_of_2(K)
+        chunk_gla_fwd_A_kernel_intra_sub_intra[grid](
+            q,
+            k,
+            g,
+            A,
+            offsets,
+            indices,
+            scale,
+            T=T,
+            H=H,
+            K=K,
+            BT=BT,
+            BC=BC,
+            BK=BK,
+            HEAD_FIRST=head_first
+        )
+    # split then merge
+    else:
+        BK = min(128, triton.next_power_of_2(K))
+        NK = triton.cdiv(K, BK)
+        A_intra = q.new_empty(NK, B, *((H, T) if head_first else (T, H)), BC, dtype=torch.float)
+        grid = (NK, NT * NC, B * H)
+        chunk_gla_fwd_A_kernel_intra_sub_intra_split[grid](
+            q,
+            k,
+            g,
+            A_intra,
+            offsets,
+            indices,
+            scale,
+            T=T,
+            B=B,
+            H=H,
+            K=K,
+            BT=BT,
+            BC=BC,
+            BK=BK,
+            NC=NC,
+            HEAD_FIRST=head_first
+        )
+        grid = (NT, NC, B * H)
+        chunk_gla_fwd_A_kernel_intra_sub_intra_merge[grid](
+            A_intra,
+            A,
+            offsets,
+            indices,
+            T=T,
+            B=B,
+            H=H,
+            BT=BT,
+            BC=BC,
+            NK=NK,
+            HEAD_FIRST=head_first
+        )
+    return A
+def chunk_gla_fwd_o_gk(
+    q: torch.Tensor,
+    v: torch.Tensor,
+    g: torch.Tensor,
+    A: torch.Tensor,
+    h: torch.Tensor,
+    scale: float,
+    offsets: Optional[torch.LongTensor] = None,
+    indices: Optional[torch.LongTensor] = None,
+    head_first: bool = True,
+    chunk_size: int = 64
+):
+    if head_first:
+        B, H, T, K, V = *q.shape, v.shape[-1]
+    else:
+        B, T, H, K, V = *q.shape, v.shape[-1]
+    BT = min(chunk_size, max(16, triton.next_power_of_2(T)))
+    NT = triton.cdiv(T, BT) if offsets is None else len(indices)
+    o = torch.empty_like(v)
+    def grid(meta): return (triton.cdiv(V, meta['BV']), NT, B * H)
+    chunk_gla_fwd_kernel_o[grid](
+        q,
+        v,
+        g,
+        h,
+        o,
+        A,
+        offsets,
+        indices,
+        scale,
+        T=T,
+        H=H,
+        K=K,
+        V=V,
+        BT=BT,
+        HEAD_FIRST=head_first
+    )
+    return o
+def chunk_gla_bwd_dA(
+    v: torch.Tensor,
+    do: torch.Tensor,
+    scale: float,
+    offsets: Optional[torch.LongTensor] = None,
+    indices: Optional[torch.LongTensor] = None,
+    head_first: bool = True,
+    chunk_size: int = 64
+):
+    if head_first:
+        B, H, T, V = v.shape
+    else:
+        B, T, H, V = v.shape
+    BT = min(chunk_size, max(16, triton.next_power_of_2(T)))
+    NT = triton.cdiv(T, BT) if offsets is None else len(indices)
+    BV = min(64, triton.next_power_of_2(V))
+    dA = v.new_empty(B, *((H, T) if head_first else (T, H)), BT, dtype=torch.float)
+    grid = (NT, B * H)
+    chunk_gla_bwd_kernel_dA[grid](
+        v,
+        do,
+        dA,
+        offsets,
+        indices,
+        scale,
+        T=T,
+        H=H,
+        V=V,
+        BT=BT,
+        BV=BV,
+        HEAD_FIRST=head_first
+    )
+    return dA
+def chunk_gla_bwd_dv(
+    k: torch.Tensor,
+    g: torch.Tensor,
+    A: torch.Tensor,
+    do: torch.Tensor,
+    dh: torch.Tensor,
+    offsets: Optional[torch.LongTensor] = None,
+    indices: Optional[torch.LongTensor] = None,
+    head_first: bool = True,
+    chunk_size: int = 64
+):
+    if head_first:
+        B, H, T, K, V = *k.shape, do.shape[-1]
+    else:
+        B, T, H, K, V = *k.shape, do.shape[-1]
+    BT = min(chunk_size, max(16, triton.next_power_of_2(T)))
+    NT = triton.cdiv(T, BT) if offsets is None else len(indices)
+    dv = torch.empty_like(do)
+    def grid(meta): return (triton.cdiv(V, meta['BV']), NT, B * H)
+    chunk_gla_bwd_kernel_dv[grid](
+        k,
+        g,
+        A,
+        do,
+        dh,
+        dv,
+        offsets,
+        indices,
+        T=T,
+        H=H,
+        K=K,
+        V=V,
+        BT=BT,
+        HEAD_FIRST=head_first
+    )
+    return dv
+def chunk_gla_bwd_dqk_intra(
+    q: torch.Tensor,
+    k: torch.Tensor,
+    g: torch.Tensor,
+    dA: torch.Tensor,
+    offsets: Optional[torch.LongTensor] = None,
+    indices: Optional[torch.LongTensor] = None,
+    head_first: bool = True,
+    chunk_size: int = 64
+):
+    if head_first:
+        B, H, T, K = q.shape
+    else:
+        B, T, H, K = q.shape
+    BT = min(chunk_size, max(16, triton.next_power_of_2(T)))
+    BC = min(16, BT)
+    BK = min(64, triton.next_power_of_2(K))
+    NT = triton.cdiv(T, BT) if offsets is None else len(indices)
+    NC = triton.cdiv(BT, BC)
+    NK = triton.cdiv(K, BK)
+    dq = torch.empty_like(q, dtype=torch.float)
+    dk = torch.empty_like(k, dtype=torch.float)
+    grid = (NK, NT * NC, B * H)
+    chunk_gla_bwd_kernel_intra[grid](
+        q,
+        k,
+        g,
+        dA,
+        dq,
+        dk,
+        offsets,
+        indices,
+        T=T,
+        H=H,
+        K=K,
+        BT=BT,
+        BC=BC,
+        BK=BK,
+        NC=NC,
+        HEAD_FIRST=head_first
+    )
+    return dq, dk
+def chunk_gla_bwd_dqkg(
+    q: torch.Tensor,
+    k: torch.Tensor,
+    v: torch.Tensor,
+    h: torch.Tensor,
+    g: torch.Tensor,
+    do: torch.Tensor,
+    dh: torch.Tensor,
+    dq: torch.Tensor,
+    dk: torch.Tensor,
+    scale: float,
+    offsets: Optional[torch.LongTensor] = None,
+    indices: Optional[torch.LongTensor] = None,
+    head_first: bool = True,
+    chunk_size: int = 64
+):
+    if head_first:
+        B, H, T, K, V = *k.shape, v.shape[-1]
+    else:
+        B, T, H, K, V = *k.shape, v.shape[-1]
+    BT = min(chunk_size, max(16, triton.next_power_of_2(T)))
+    NT = triton.cdiv(T, BT) if offsets is None else len(indices)
+    dg = torch.empty_like(g)
+    # work around triton compiler bugs.
+    dq2 = torch.empty_like(dq)
+    dk2 = torch.empty_like(dk)
+    def grid(meta): return (triton.cdiv(K, meta['BK']), NT, B * H)
+    chunk_gla_bwd_kernel_inter[grid](
+        q,
+        k,
+        v,
+        h,
+        g,
+        do,
+        dh,
+        dq,
+        dk,
+        dq2,
+        dk2,
+        dg,
+        offsets,
+        indices,
+        scale,
+        T=T,
+        H=H,
+        K=K,
+        V=V,
+        BT=BT,
+        HEAD_FIRST=head_first
+    )
+    return dq2, dk2, dg
+def chunk_gla_fwd(
+    q: torch.Tensor,
+    k: torch.Tensor,
+    v: torch.Tensor,
+    g: torch.Tensor,
+    g_cumsum: Optional[torch.Tensor],
+    scale: float,
+    initial_state: torch.Tensor,
+    output_final_state: bool,
+    offsets: Optional[torch.LongTensor] = None,
+    indices: Optional[torch.LongTensor] = None,
+    head_first: bool = True,
+    chunk_size: int = 64
+) -> Tuple[torch.Tensor, torch.Tensor, torch.Tensor]:
+    T = q.shape[2] if head_first else q.shape[1]
+    BT = min(chunk_size, max(16, triton.next_power_of_2(T)))
+    if g_cumsum is None:
+        g_cumsum = chunk_local_cumsum(g, BT, offsets=offsets, indices=indices, head_first=head_first)
+    h, ht = chunk_fwd_h(
+        k=k,
+        v=v,
+        g=None,
+        gk=g_cumsum,
+        gv=None,
+        h0=initial_state,
+        output_final_state=output_final_state,
+        states_in_fp32=False,
+        offsets=offsets,
+        head_first=head_first,
+        chunk_size=BT
+    )
+    # the intra A is kept in fp32
+    # the computation has very marginal effect on the entire throughput
+    A = chunk_gla_fwd_intra_gk(
+        q=q,
+        k=k,
+        g=g_cumsum,
+        scale=scale,
+        offsets=offsets,
+        indices=indices,
+        head_first=head_first,
+        chunk_size=BT
+    )
+    o = chunk_gla_fwd_o_gk(
+        q=q,
+        v=v,
+        g=g_cumsum,
+        A=A,
+        h=h,
+        scale=scale,
+        offsets=offsets,
+        indices=indices,
+        head_first=head_first,
+        chunk_size=BT
+    )
+    return g_cumsum, A, h, ht, o
+def chunk_gla_bwd(
+    q: torch.Tensor,
+    k: torch.Tensor,
+    v: torch.Tensor,
+    g: torch.Tensor,
+    g_cumsum: Optional[torch.Tensor],
+    scale: float,
+    initial_state: torch.Tensor,
+    h: torch.Tensor,
+    A: torch.Tensor,
+    do: torch.Tensor,
+    dht: torch.Tensor,
+    offsets: Optional[torch.LongTensor] = None,
+    indices: Optional[torch.LongTensor] = None,
+    head_first: bool = True,
+    chunk_size: int = 64
+):
+    T = q.shape[2] if head_first else q.shape[1]
+    BT = min(chunk_size, max(16, triton.next_power_of_2(T)))
+    if g_cumsum is None:
+        g_cumsum = chunk_local_cumsum(g, BT, offsets=offsets, indices=indices, head_first=head_first)
+    if h is None:
+        h, _ = chunk_fwd_h(
+            k=k,
+            v=v,
+            g=None,
+            gk=g_cumsum,
+            gv=None,
+            h0=initial_state,
+            output_final_state=False,
+            offsets=offsets,
+            head_first=head_first,
+            chunk_size=BT,
+            states_in_fp32=True
+        )
+    dh, dh0 = chunk_bwd_dh(
+        q=q,
+        k=k,
+        v=v,
+        g=None,
+        gk=g_cumsum,
+        gv=None,
+        do=do,
+        h0=initial_state,
+        dht=dht,
+        scale=scale,
+        offsets=offsets,
+        head_first=head_first,
+        chunk_size=BT,
+        states_in_fp32=True
+    )
+    dv = chunk_gla_bwd_dv(
+        k=k,
+        g=g_cumsum,
+        A=A,
+        do=do,
+        dh=dh,
+        offsets=offsets,
+        indices=indices,
+        head_first=head_first,
+        chunk_size=BT
+    )
+    # dq dk in fp32
+    dA = chunk_gla_bwd_dA(
+        v=v,
+        do=do,
+        scale=scale,
+        offsets=offsets,
+        indices=indices,
+        head_first=head_first,
+        chunk_size=BT
+    )
+    dq, dk = chunk_gla_bwd_dqk_intra(
+        q=q,
+        k=k,
+        g=g_cumsum,
+        dA=dA,
+        offsets=offsets,
+        indices=indices,
+        head_first=head_first,
+        chunk_size=BT
+    )
+    dq, dk, dg = chunk_gla_bwd_dqkg(
+        q=q,
+        k=k,
+        v=v,
+        h=h,
+        g=g_cumsum,
+        do=do,
+        dh=dh,
+        dq=dq,
+        dk=dk,
+        scale=scale,
+        offsets=offsets,
+        indices=indices,
+        head_first=head_first,
+        chunk_size=BT
+    )
+    return dq, dk, dv, dg, dh0
+class ChunkGLAFunction(torch.autograd.Function):
+    @staticmethod
+    @input_guard
+    def forward(
+        ctx,
+        q,
+        k,
+        v,
+        g,
+        scale,
+        initial_state,
+        output_final_state,
+        offsets,
+        head_first
+    ):
+        T = q.shape[2] if head_first else q.shape[1]
+        chunk_size = min(64, max(16, triton.next_power_of_2(T)))
+        # 2-d indices denoting the offsets of chunks in each sequence
+        # for example, if the passed `offsets` is [0, 100, 356] and `chunk_size` is 64,
+        # then there are 2 and 4 chunks in the 1st and 2nd sequences respectively, and `indices` will be
+        # [[0, 0], [0, 1], [1, 0], [1, 1], [1, 2], [1, 3]]
+        indices = prepare_chunk_indices(offsets, chunk_size) if offsets is not None else None
+        g_cumsum, A, h, ht, o = chunk_gla_fwd(
+            q=q,
+            k=k,
+            v=v,
+            g=g,
+            g_cumsum=None,
+            scale=scale,
+            initial_state=initial_state,
+            output_final_state=output_final_state,
+            offsets=offsets,
+            indices=indices,
+            head_first=head_first,
+            chunk_size=chunk_size
+        )
+        # recompute g_cumsum in bwd pass
+        if g.dtype != torch.float:
+            g_cumsum = None
+        else:
+            g = None
+        ctx.save_for_backward(q, k, v, g, g_cumsum, initial_state, A)
+        ctx.chunk_size = chunk_size
+        ctx.scale = scale
+        ctx.offsets = offsets
+        ctx.indices = indices
+        ctx.head_first = head_first
+        return o, ht
+    @staticmethod
+    @input_guard
+    def backward(ctx, do, dht):
+        q, k, v, g, g_cumsum, initial_state, A = ctx.saved_tensors
+        chunk_size, scale, offsets, indices, head_first = ctx.chunk_size, ctx.scale, ctx.offsets, ctx.indices, ctx.head_first
+        dq, dk, dv, dg, dh0 = chunk_gla_bwd(
+            q=q,
+            k=k,
+            v=v,
+            g=g,
+            g_cumsum=g_cumsum,
+            scale=scale,
+            h=None,
+            A=A,
+            initial_state=initial_state,
+            do=do,
+            dht=dht,
+            offsets=offsets,
+            indices=indices,
+            head_first=head_first,
+            chunk_size=chunk_size
+        )
+        return dq.to(q), dk.to(k), dv.to(v), dg, None, dh0, None, None, None
+@torch.compiler.disable
+def chunk_gla(
+    q: torch.Tensor,
+    k: torch.Tensor,
+    v: torch.Tensor,
+    g: torch.Tensor,
+    scale: Optional[int] = None,
+    initial_state: torch.Tensor = None,
+    output_final_state: bool = False,
+    cu_seqlens: Optional[torch.LongTensor] = None,
+    head_first: bool = True
+) -> Tuple[torch.Tensor, torch.Tensor]:
+    r"""
+    Args:
+        q (torch.Tensor):
+            queries of shape `[B, H, T, K]` if `head_first=True` else `[B, T, H, K]`.
+        k (torch.Tensor):
+            keys of shape `[B, H, T, K]` if `head_first=True` else `[B, T, H, K]`.
+        v (torch.Tensor):
+            values of shape `[B, H, T, V]` if `head_first=True` else `[B, T, H, V]`.
+        g (torch.Tensor):
+            Forget gates of shape `[B, H, T, K]` if `head_first=True` else `[B, T, H, K]` applied to keys.
+        scale (Optional[int]):
+            Scale factor for the attention scores.
+            If not provided, it will default to `1 / sqrt(K)`. Default: `None`.
+        initial_state (Optional[torch.Tensor]):
+            Initial state of shape `[N, H, K, V]` for `N` input sequences.
+            For equal-length input sequences, `N` equals the batch size `B`.
+            Default: `None`.
+        output_final_state (Optional[bool]):
+            Whether to output the final state of shape `[N, H, K, V]`. Default: `False`.
+        cu_seqlens (torch.LongTensor):
+            Cumulative sequence lengths of shape `[N+1]` used for variable-length training,
+            consistent with the FlashAttention API.
+        head_first (Optional[bool]):
+            Whether the inputs are in the head-first format, which is not supported for variable-length inputs.
+            Default: `True`.
+    Returns:
+        o (torch.Tensor):
+            Outputs of shape `[B, H, T, V]` if `head_first=True` else `[B, T, H, V]`.
+        final_state (torch.Tensor):
+            Final state of shape `[N, H, K, V]` if `output_final_state=True` else `None`.
+    Examples::
+        >>> import torch
+        >>> import torch.nn.functional as F
+        >>> from einops import rearrange
+        >>> from fla.ops.gla import chunk_gla
+        # inputs with equal lengths
+        >>> B, T, H, K, V = 4, 2048, 4, 512, 512
+        >>> q = torch.randn(B, T, H, K, device='cuda')
+        >>> k = torch.randn(B, T, H, K, device='cuda')
+        >>> v = torch.randn(B, T, H, V, device='cuda')
+        >>> g = F.logsigmoid(torch.randn(B, T, H, K, device='cuda'))
+        >>> h0 = torch.randn(B, H, K, V, device='cuda')
+        >>> o, ht = chunk_gla(q, k, v, g,
+                              initial_state=h0,
+                              output_final_state=True,
+                              head_first=False)
+        # for variable-length inputs, the batch size `B` is expected to be 1 and `cu_seqlens` is required
+        >>> q, k, v, g = map(lambda x: rearrange(x, 'b t h d -> 1 (b t) h d'), (q, k, v, g))
+        # for a batch with 4 sequences, `cu_seqlens` with 5 start/end positions are expected
+        >>> cu_seqlens = q.new_tensor([0, 2048, 4096, 6144, 8192], dtype=torch.long)
+        >>> o_var, ht_var = chunk_gla(q, k, v, g,
+                                      initial_state=h0,
+                                      output_final_state=True,
+                                      cu_seqlens=cu_seqlens,
+                                      head_first=False)
+        >>> assert o.allclose(o_var.view(o.shape))
+        >>> assert ht.allclose(ht_var)
+    """
+    if cu_seqlens is not None:
+        if q.shape[0] != 1:
+            raise ValueError(f"The batch size is expected to be 1 rather than {q.shape[0]} when using `cu_seqlens`."
+                             f"Please flatten variable-length inputs before processing.")
+        if head_first:
+            raise RuntimeError("Sequences with variable lengths are not supported for head-first mode")
+        if initial_state is not None and initial_state.shape[0] != len(cu_seqlens) - 1:
+            raise ValueError(f"The number of initial states is expected to be equal to the number of input sequences, "
+                             f"i.e., {len(cu_seqlens) - 1} rather than {initial_state.shape[0]}.")
+    if scale is None:
+        scale = q.shape[-1] ** -0.5
+    o, final_state = ChunkGLAFunction.apply(q, k, v, g, scale, initial_state, output_final_state, cu_seqlens, head_first)
+    return o, final_state

fla/ops/gsa/__pycache__/chunk.cpython-312.pyc ADDED Viewed

Binary file (69.4 kB). View file

fla/ops/hgrn/__pycache__/__init__.cpython-312.pyc ADDED Viewed

Binary file (288 Bytes). View file

fla/ops/hgrn/__pycache__/fused_recurrent.cpython-312.pyc ADDED Viewed

Binary file (14.3 kB). View file

fla/ops/lightning_attn/chunk.py ADDED Viewed

	@@ -0,0 +1,74 @@

+# -*- coding: utf-8 -*-
+# Copyright (c) 2023-2025, Songlin Yang, Yu Zhang
+from typing import Optional, Tuple
+import torch
+from fla.ops.simple_gla.chunk import chunk_simple_gla
+@torch.compiler.disable
+def chunk_lightning_attn(
+    q: torch.Tensor,
+    k: torch.Tensor,
+    v: torch.Tensor,
+    layer_idx: int,
+    num_layers: int,
+    scale: Optional[float] = None,
+    initial_state: Optional[torch.Tensor] = None,
+    output_final_state: bool = False,
+    cu_seqlens: Optional[torch.LongTensor] = None,
+    head_first: bool = True
+) -> Tuple[torch.Tensor, torch.Tensor]:
+    r"""
+    Args:
+        q (torch.Tensor):
+            queries of shape `[B, H, T, K]` if `head_first=True` else `[B, T, H, K]`.
+        k (torch.Tensor):
+            keys of shape `[B, H, T, K]` if `head_first=True` else `[B, T, H, K]`.
+        v (torch.Tensor):
+            values of shape `[B, H, T, V]` if `head_first=True` else `[B, T, H, V]`.
+        layer_idx (int):
+            The index of the current layer.
+        num_layers (int):
+            The total number of layers. Both `layer_idx` and `num_layers` are used to compute the decay factor.
+        scale (Optional[int]):
+            Scale factor for the attention scores.
+            If not provided, it will default to `1 / sqrt(K)`. Default: `None`.
+        initial_state (Optional[torch.Tensor]):
+            Initial state of shape `[N, H, K, V]` for `N` input sequences.
+            For equal-length input sequences, `N` equals the batch size `B`.
+            Default: `None`.
+        output_final_state (Optional[bool]):
+            Whether to output the final state of shape `[N, H, K, V]`. Default: `False`.
+        cu_seqlens (torch.LongTensor):
+            Cumulative sequence lengths of shape `[N+1]` used for variable-length training,
+            consistent with the FlashAttention API.
+        head_first (Optional[bool]):
+            Whether the inputs are in the head-first format, which is not supported for variable-length inputs.
+            Default: `True`.
+    Returns:
+        o (torch.Tensor):
+            Outputs of shape `[B, H, T, V]` if `head_first=True` else `[B, T, H, V]`.
+        final_state (torch.Tensor):
+            Final state of shape `[N, H, K, V]` if `output_final_state=True` else `None`.
+    """
+    H = q.shape[1] if head_first else q.shape[2]
+    s = -(8 / H * (1 - layer_idx / num_layers)) * q.new_tensor(range(H), dtype=torch.float)
+    if head_first:
+        g = s[None, :, None].expand(q.shape[0], q.shape[1], q.shape[2]).contiguous()
+    else:
+        g = s[None, None, :].expand(q.shape[0], q.shape[1], q.shape[2]).contiguous()
+    return chunk_simple_gla(
+        q=q,
+        k=k,
+        v=v,
+        scale=scale,
+        g=g,
+        initial_state=initial_state,
+        output_final_state=output_final_state,
+        head_first=head_first,
+        cu_seqlens=cu_seqlens
+    )