kernels-community
/

flash-attn2

Kernels

Model card Files Files and versions

xet

Community

drbh commited on Mar 31

Commit

d6cc1b0

1 Parent(s): b833fce

feat improve readme and library code

Browse files

Files changed (2) hide show

README.md +80 -0
torch-ext/flash_attn/__init__.py +343 -16

README.md ADDED Viewed

	@@ -0,0 +1,80 @@

+# Flash Attention
+Flash Attention is a fast and memory-efficient implementation of the attention mechanism, designed to work with large models and long sequences. This is a Hugging Face compliant kernel build of Flash Attention.
+Original code here [https://github.com/Dao-AILab/flash-attention](https://github.com/Dao-AILab/flash-attention).
+```python
+# /// script
+# dependencies = ["numpy", "torch", "kernels"]
+# ///
+import torch
+from kernels import get_kernel
+# Setup
+torch.manual_seed(42)
+flash_attn = get_kernel("kernels-community/flash-attn")
+device = torch.device("cuda")
+# Show available functions
+print("Flash Attention functions:", [i for i in dir(flash_attn) if i.startswith("mha")])
+# 1. Standard attention
+print("\n1. Standard attention:")
+B, S, H, D = 2, 5, 4, 8  # batch, seq_len, heads, head_dim
+q = k = v = torch.randn(B, S, H, D, device=device, dtype=torch.float16)
+out = flash_attn.mha_fwd(q=q, k=k, v=v, is_causal=False)[0]
+print(f"Output: {out.shape}")
+# 2. Variable length sequences
+print("\n2. Variable length sequences:")
+q_var = torch.randn(10, H, D, device=device, dtype=torch.float16)  # total_q=10
+k_var = v_var = torch.randn(12, H, D, device=device, dtype=torch.float16)  # total_k=12
+# For 3 sequences with lengths [3,4,3] for q and [4,5,3] for k
+cu_q = torch.tensor([0, 3, 7, 10], device=device, dtype=torch.int32)
+cu_k = torch.tensor([0, 4, 9, 12], device=device, dtype=torch.int32)
+out_var = flash_attn.mha_varlen_fwd(
+    q=q_var,
+    k=k_var,
+    v=v_var,
+    cu_seqlens_q=cu_q,
+    cu_seqlens_k=cu_k,
+    max_seqlen_q=4,
+    max_seqlen_k=5,
+)[0]
+print(f"Output: {out_var.shape}")
+# 3. KV-cache for autoregressive generation
+print("\n3. KV-cache:")
+cache_len, new_len = 10, 2
+kcache = vcache = torch.randn(B, cache_len, H, D, device=device, dtype=torch.float16)
+q_new = k_new = v_new = torch.randn(
+    B, new_len, H, D, device=device, dtype=torch.float16
+)
+seqlens = torch.full((B,), cache_len + new_len, device=device, dtype=torch.int32)
+out_kv = flash_attn.mha_fwd_kvcache(
+    q=q_new,
+    kcache=kcache,
+    vcache=vcache,
+    k=k_new,
+    v=v_new,
+    seqlens_k=seqlens,
+    is_causal=True,
+)[0]
+print(f"Output: {out_kv.shape}")
+```
+expected output
+```txt
+Fetching 3 files: 100%|█████████████████████████████████████████████████████| 3/3 [00:00<00:00, 16384.00it/s]
+Flash Attention functions: ['mha_bwd', 'mha_fwd', 'mha_fwd_kvcache', 'mha_varlen_bwd', 'mha_varlen_fwd']
+1. Standard attention:
+Output: torch.Size([2, 5, 4, 8])
+2. Variable length sequences:
+Output: torch.Size([10, 4, 8])
+3. KV-cache:
+Output: torch.Size([2, 2, 4, 8])
+```

torch-ext/flash_attn/__init__.py CHANGED Viewed

@@ -1,25 +1,45 @@
-from typing import Optional
 import torch
 from ._ops import ops
 def mha_fwd(
     q: torch.Tensor,
     k: torch.Tensor,
     v: torch.Tensor,
-    out: torch.Tensor,
-    alibi_slopes: torch.Tensor,
-    p_dropout: float,
-    softmax_scale: float,
-    is_causal: bool,
-    window_size_left: int,
-    window_size_right: int,
-    softcap: float,
-    return_softmax: bool,
-    gen: Optional[torch.Generator],
-) -> torch.Tensor:
-    ops.mha_fwd(
         q,
         k,
         v,
@@ -34,4 +54,311 @@ def mha_fwd(
         return_softmax,
         gen,
     )
-    return out

+from typing import Optional, List
 import torch
 from ._ops import ops
 def mha_fwd(
     q: torch.Tensor,
     k: torch.Tensor,
     v: torch.Tensor,
+    out: Optional[torch.Tensor] = None,
+    alibi_slopes: Optional[torch.Tensor] = None,
+    p_dropout: float = 0.0,
+    softmax_scale: float = 1.0,
+    is_causal: bool = False,
+    window_size_left: int = -1,
+    window_size_right: int = -1,
+    softcap: float = 0.0,
+    return_softmax: bool = False,
+    gen: Optional[torch.Generator] = None,
+) -> List[torch.Tensor]:
+    """
+    Forward pass for multi-head attention.
+    Args:
+        q: Query tensor of shape [batch_size, seqlen_q, num_heads, head_size]
+        k: Key tensor of shape [batch_size, seqlen_k, num_heads_k, head_size]
+        v: Value tensor of shape [batch_size, seqlen_k, num_heads_k, head_size]
+        out: Optional output tensor, same shape as q
+        alibi_slopes: Optional ALiBi slopes tensor of shape [num_heads] or [batch_size, num_heads]
+        p_dropout: Dropout probability
+        softmax_scale: Scale factor for softmax
+        is_causal: Whether to use causal attention
+        window_size_left: Window size for left context (-1 for unlimited)
+        window_size_right: Window size for right context (-1 for unlimited)
+        softcap: Soft cap for attention weights
+        return_softmax: Whether to return softmax weights
+        gen: Optional random number generator
+    Returns:
+        List of tensors: [output, softmax_lse, (softmax if return_softmax)]
+    """
+    return ops.mha_fwd(
         q,
         k,
         v,
         return_softmax,
         gen,
     )
+def mha_varlen_fwd(
+    q: torch.Tensor,
+    k: torch.Tensor,
+    v: torch.Tensor,
+    cu_seqlens_q: torch.Tensor,
+    cu_seqlens_k: torch.Tensor,
+    out: Optional[torch.Tensor] = None,
+    seqused_k: Optional[torch.Tensor] = None,
+    leftpad_k: Optional[torch.Tensor] = None,
+    block_table: Optional[torch.Tensor] = None,
+    alibi_slopes: Optional[torch.Tensor] = None,
+    max_seqlen_q: int = 0,
+    max_seqlen_k: int = 0,
+    p_dropout: float = 0.0,
+    softmax_scale: float = 1.0,
+    zero_tensors: bool = False,
+    is_causal: bool = False,
+    window_size_left: int = -1,
+    window_size_right: int = -1,
+    softcap: float = 0.0,
+    return_softmax: bool = False,
+    gen: Optional[torch.Generator] = None,
+) -> List[torch.Tensor]:
+    """
+    Forward pass for multi-head attention with variable sequence lengths.
+    Args:
+        q: Query tensor of shape [total_q, num_heads, head_size]
+        k: Key tensor of shape [total_k, num_heads_k, head_size] or [num_blocks, page_block_size, num_heads_k, head_size]
+        v: Value tensor of shape [total_k, num_heads_k, head_size] or [num_blocks, page_block_size, num_heads_k, head_size]
+        cu_seqlens_q: Cumulative sequence lengths for queries of shape [batch_size+1]
+        cu_seqlens_k: Cumulative sequence lengths for keys of shape [batch_size+1]
+        out: Optional output tensor of shape [total_q, num_heads, head_size]
+        seqused_k: Optional tensor specifying how many keys to use per batch element [batch_size]
+        leftpad_k: Optional left padding for keys of shape [batch_size]
+        block_table: Optional block table of shape [batch_size, max_num_blocks_per_seq]
+        alibi_slopes: Optional ALiBi slopes tensor of shape [num_heads] or [batch_size, num_heads]
+        max_seqlen_q: Maximum sequence length for queries
+        max_seqlen_k: Maximum sequence length for keys
+        p_dropout: Dropout probability
+        softmax_scale: Scale factor for softmax
+        zero_tensors: Whether to zero tensors before computation
+        is_causal: Whether to use causal attention
+        window_size_left: Window size for left context (-1 for unlimited)
+        window_size_right: Window size for right context (-1 for unlimited)
+        softcap: Soft cap for attention weights
+        return_softmax: Whether to return softmax weights
+        gen: Optional random number generator
+    Returns:
+        List of tensors: [output, softmax_lse, (softmax if return_softmax)]
+    """
+    return ops.mha_varlen_fwd(
+        q,
+        k,
+        v,
+        out,
+        cu_seqlens_q,
+        cu_seqlens_k,
+        seqused_k,
+        leftpad_k,
+        block_table,
+        alibi_slopes,
+        max_seqlen_q,
+        max_seqlen_k,
+        p_dropout,
+        softmax_scale,
+        zero_tensors,
+        is_causal,
+        window_size_left,
+        window_size_right,
+        softcap,
+        return_softmax,
+        gen,
+    )
+def mha_bwd(
+    dout: torch.Tensor,
+    q: torch.Tensor,
+    k: torch.Tensor,
+    v: torch.Tensor,
+    out: torch.Tensor,
+    softmax_lse: torch.Tensor,
+    dq: Optional[torch.Tensor] = None,
+    dk: Optional[torch.Tensor] = None,
+    dv: Optional[torch.Tensor] = None,
+    alibi_slopes: Optional[torch.Tensor] = None,
+    p_dropout: float = 0.0,
+    softmax_scale: float = 1.0,
+    is_causal: bool = False,
+    window_size_left: int = -1,
+    window_size_right: int = -1,
+    softcap: float = 0.0,
+    deterministic: bool = False,
+    gen: Optional[torch.Generator] = None,
+    rng_state: Optional[torch.Tensor] = None,
+) -> List[torch.Tensor]:
+    """
+    Backward pass for multi-head attention.
+    Args:
+        dout: Gradient tensor of shape [batch_size, seqlen_q, num_heads, head_size]
+        q: Query tensor of shape [batch_size, seqlen_q, num_heads, head_size]
+        k: Key tensor of shape [batch_size, seqlen_k, num_heads_k, head_size]
+        v: Value tensor of shape [batch_size, seqlen_k, num_heads_k, head_size]
+        out: Output tensor from forward pass of shape [batch_size, seqlen_q, num_heads, head_size]
+        softmax_lse: Log-sum-exp values from forward pass of shape [batch_size, num_heads, seqlen_q]
+        dq: Optional gradient tensor for queries, same shape as q
+        dk: Optional gradient tensor for keys, same shape as k
+        dv: Optional gradient tensor for values, same shape as v
+        alibi_slopes: Optional ALiBi slopes tensor of shape [num_heads] or [batch_size, num_heads]
+        p_dropout: Dropout probability
+        softmax_scale: Scale factor for softmax
+        is_causal: Whether to use causal attention
+        window_size_left: Window size for left context (-1 for unlimited)
+        window_size_right: Window size for right context (-1 for unlimited)
+        softcap: Soft cap for attention weights
+        deterministic: Whether to use deterministic algorithms
+        gen: Optional random number generator
+        rng_state: Optional RNG state from forward pass
+    Returns:
+        List of tensors: [dq, dk, dv]
+    """
+    return ops.mha_bwd(
+        dout,
+        q,
+        k,
+        v,
+        out,
+        softmax_lse,
+        dq,
+        dk,
+        dv,
+        alibi_slopes,
+        p_dropout,
+        softmax_scale,
+        is_causal,
+        window_size_left,
+        window_size_right,
+        softcap,
+        deterministic,
+        gen,
+        rng_state,
+    )
+def mha_varlen_bwd(
+    dout: torch.Tensor,
+    q: torch.Tensor,
+    k: torch.Tensor,
+    v: torch.Tensor,
+    out: torch.Tensor,
+    softmax_lse: torch.Tensor,
+    cu_seqlens_q: torch.Tensor,
+    cu_seqlens_k: torch.Tensor,
+    dq: Optional[torch.Tensor] = None,
+    dk: Optional[torch.Tensor] = None,
+    dv: Optional[torch.Tensor] = None,
+    alibi_slopes: Optional[torch.Tensor] = None,
+    max_seqlen_q: int = 0,
+    max_seqlen_k: int = 0,
+    p_dropout: float = 0.0,
+    softmax_scale: float = 1.0,
+    zero_tensors: bool = False,
+    is_causal: bool = False,
+    window_size_left: int = -1,
+    window_size_right: int = -1,
+    softcap: float = 0.0,
+    deterministic: bool = False,
+    gen: Optional[torch.Generator] = None,
+    rng_state: Optional[torch.Tensor] = None,
+) -> List[torch.Tensor]:
+    """
+    Backward pass for multi-head attention with variable sequence lengths.
+    Args:
+        dout: Gradient tensor of shape [batch_size, seqlen_q, num_heads, head_size]
+        q: Query tensor of shape [batch_size, seqlen_q, num_heads, head_size]
+        k: Key tensor of shape [batch_size, seqlen_k, num_heads_k, head_size]
+        v: Value tensor of shape [batch_size, seqlen_k, num_heads_k, head_size]
+        out: Output tensor from forward pass of shape [batch_size, seqlen_q, num_heads, head_size]
+        softmax_lse: Log-sum-exp values from forward pass of shape [batch_size, num_heads, seqlen_q]
+        cu_seqlens_q: Cumulative sequence lengths for queries of shape [batch_size+1]
+        cu_seqlens_k: Cumulative sequence lengths for keys of shape [batch_size+1]
+        dq: Optional gradient tensor for queries, same shape as q
+        dk: Optional gradient tensor for keys, same shape as k
+        dv: Optional gradient tensor for values, same shape as v
+        alibi_slopes: Optional ALiBi slopes tensor of shape [num_heads] or [batch_size, num_heads]
+        max_seqlen_q: Maximum sequence length for queries
+        max_seqlen_k: Maximum sequence length for keys
+        p_dropout: Dropout probability
+        softmax_scale: Scale factor for softmax
+        zero_tensors: Whether to zero tensors before computation
+        is_causal: Whether to use causal attention
+        window_size_left: Window size for left context (-1 for unlimited)
+        window_size_right: Window size for right context (-1 for unlimited)
+        softcap: Soft cap for attention weights
+        deterministic: Whether to use deterministic algorithms
+        gen: Optional random number generator
+        rng_state: Optional RNG state from forward pass
+    Returns:
+        List of tensors: [dq, dk, dv]
+    """
+    return ops.mha_varlen_bwd(
+        dout,
+        q,
+        k,
+        v,
+        out,
+        softmax_lse,
+        dq,
+        dk,
+        dv,
+        cu_seqlens_q,
+        cu_seqlens_k,
+        alibi_slopes,
+        max_seqlen_q,
+        max_seqlen_k,
+        p_dropout,
+        softmax_scale,
+        zero_tensors,
+        is_causal,
+        window_size_left,
+        window_size_right,
+        softcap,
+        deterministic,
+        gen,
+        rng_state,
+    )
+def mha_fwd_kvcache(
+    q: torch.Tensor,
+    kcache: torch.Tensor,
+    vcache: torch.Tensor,
+    k: Optional[torch.Tensor] = None,
+    v: Optional[torch.Tensor] = None,
+    seqlens_k: Optional[torch.Tensor] = None,
+    rotary_cos: Optional[torch.Tensor] = None,
+    rotary_sin: Optional[torch.Tensor] = None,
+    cache_batch_idx: Optional[torch.Tensor] = None,
+    leftpad_k: Optional[torch.Tensor] = None,
+    block_table: Optional[torch.Tensor] = None,
+    alibi_slopes: Optional[torch.Tensor] = None,
+    out: Optional[torch.Tensor] = None,
+    softmax_scale: float = 1.0,
+    is_causal: bool = False,
+    window_size_left: int = -1,
+    window_size_right: int = -1,
+    softcap: float = 0.0,
+    is_rotary_interleaved: bool = False,
+    num_splits: int = 1,
+) -> List[torch.Tensor]:
+    """
+    Forward pass for multi-head attention with KV cache.
+    Args:
+        q: Query tensor of shape [batch_size, seqlen_q, num_heads, head_size]
+        kcache: Key cache tensor of shape [batch_size_c, seqlen_k, num_heads_k, head_size] or [num_blocks, page_block_size, num_heads_k, head_size]
+        vcache: Value cache tensor of shape [batch_size_c, seqlen_k, num_heads_k, head_size] or [num_blocks, page_block_size, num_heads_k, head_size]
+        k: Optional new keys tensor of shape [batch_size, seqlen_knew, num_heads_k, head_size]
+        v: Optional new values tensor of shape [batch_size, seqlen_knew, num_heads_k, head_size]
+        seqlens_k: Optional sequence lengths for keys of shape [batch_size]
+        rotary_cos: Optional rotary cosine tensor of shape [seqlen_ro, rotary_dim/2]
+        rotary_sin: Optional rotary sine tensor of shape [seqlen_ro, rotary_dim/2]
+        cache_batch_idx: Optional indices to index into the KV cache
+        leftpad_k: Optional left padding for keys of shape [batch_size]
+        block_table: Optional block table of shape [batch_size, max_num_blocks_per_seq]
+        alibi_slopes: Optional ALiBi slopes tensor of shape [num_heads] or [batch_size, num_heads]
+        out: Optional output tensor, same shape as q
+        softmax_scale: Scale factor for softmax
+        is_causal: Whether to use causal attention
+        window_size_left: Window size for left context (-1 for unlimited)
+        window_size_right: Window size for right context (-1 for unlimited)
+        softcap: Soft cap for attention weights
+        is_rotary_interleaved: Whether rotary embeddings are interleaved
+        num_splits: Number of splits for computation
+    Returns:
+        List of tensors: [output, softmax_lse]
+    """
+    return ops.mha_fwd_kvcache(
+        q,
+        kcache,
+        vcache,
+        k,
+        v,
+        seqlens_k,
+        rotary_cos,
+        rotary_sin,
+        cache_batch_idx,
+        leftpad_k,
+        block_table,
+        alibi_slopes,
+        out,
+        softmax_scale,
+        is_causal,
+        window_size_left,
+        window_size_right,
+        softcap,
+        is_rotary_interleaved,
+        num_splits,
+    )

feat improve readme and library code

🎉 Free Image Generator Now Available!