update to v1.1

Browse files

Files changed (11) hide show

config.json +1 -1
model-00001-of-00008.safetensors +1 -1
model-00002-of-00008.safetensors +1 -1
model-00003-of-00008.safetensors +1 -1
model-00004-of-00008.safetensors +1 -1
model-00005-of-00008.safetensors +1 -1
model-00006-of-00008.safetensors +1 -1
model-00007-of-00008.safetensors +1 -1
model-00008-of-00008.safetensors +1 -1
modeling_cogvlm.py +73 -24
util.py +0 -483

config.json CHANGED Viewed

@@ -1,5 +1,5 @@
 {
-  "_name_or_path": "cogvlm-grounding-generalist",
   "architectures": [
     "CogVLMForCausalLM"
   ],

 {
+  "_name_or_path": "cogvlm-grounding-generalist-v1-1",
   "architectures": [
     "CogVLMForCausalLM"
   ],

model-00001-of-00008.safetensors CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:ab245c9b171545099d652eefcd863aa20f95b7ec2c18dea754ffa661ddcdeebd
 size 4938885184

 version https://git-lfs.github.com/spec/v1
+oid sha256:0ff07f55a4068d8d553593122343209b30b895760fcd83924cf9001c09c683c2
 size 4938885184

model-00002-of-00008.safetensors CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:c4c34bd99e8317404ef58c253a01648c120b356b3b8f95933da46dd02fbb73ba
 size 4947290688

 version https://git-lfs.github.com/spec/v1
+oid sha256:6ab2d1ce61f81a53be36e9af77f122e74aca0ff875b58006436effa50884c005
 size 4947290688

model-00003-of-00008.safetensors CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:0affb830c00b94b9fc5ecfa41d7ae0d62fa42c73300df2537cd7c4496f947014
 size 4947307592

 version https://git-lfs.github.com/spec/v1
+oid sha256:48edfcbc0ee4a397ff8dc5e972e02f0299ba779d5929367c9caae0cc751cf892
 size 4947307592

model-00004-of-00008.safetensors CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:ed1d44e236c4af9263f8a32406769d4e8dcf2fefad205a17c2c2191d948c0e07
 size 4991331080

 version https://git-lfs.github.com/spec/v1
+oid sha256:ed00da067fca6bdbc1b65f0c1482d716e0ee4ee648170f0588dc6cb0d23588cb
 size 4991331080

model-00005-of-00008.safetensors CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:9dc0499401747dd885bd9e212cec57296fad9b0c2d59c8c1984063f9c9d22eeb
 size 4991331088

 version https://git-lfs.github.com/spec/v1
+oid sha256:b4a6c6a257805bd07721ebcb5884b1048248d732aa979cd4375761183b806d46
 size 4991331088

model-00006-of-00008.safetensors CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:7196fa96d96992a65ae041f2a2921afef149003e059bec884520d386c814ce0a
 size 4970162920

 version https://git-lfs.github.com/spec/v1
+oid sha256:37afe090f7597fa338287fb12f501cb74828005bb0f7a0d51fa6c410711de7ee
 size 4970162920

model-00007-of-00008.safetensors CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:340afeec8059f2d6fb2b890041611a448bb54440ae185d8a5433bf39e2711d9b
 size 4960543792

 version https://git-lfs.github.com/spec/v1
+oid sha256:63a91f88f413ec65b0345f30e394cdfcf9a24fcd820825ecbcd35de78b469fa9
 size 4960543792

model-00008-of-00008.safetensors CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:9122674b65c1b3c9c8d498ff684591c106081a906f0c85e9fcb8e1a2cc0a270e
 size 532677104

 version https://git-lfs.github.com/spec/v1
+oid sha256:f97723a9e977d4416d2a70739c56aa8f8f50e55b86ddc71d7a4c9262e5167bbd
 size 532677104

modeling_cogvlm.py CHANGED Viewed

@@ -5,6 +5,7 @@ from typing import TYPE_CHECKING, Optional, Tuple, List, Union, Literal, Dict, A
 import math
 import torch
 from torch import nn
 from torch.nn import CrossEntropyLoss
 from torchvision import transforms
 from einops import rearrange
@@ -15,7 +16,6 @@ from transformers.activations import ACT2FN
 from transformers.modeling_outputs import BaseModelOutputWithPast, CausalLMOutputWithPast
 from .configuration_cogvlm import CogVLMConfig
-from .util import FastRotaryEmbedding
 from .visual import EVA2CLIPModel
 if TYPE_CHECKING:
@@ -144,6 +144,57 @@ def attention_fn(
         return context_layer
 class VisionExpertAttention(nn.Module):
     def __init__(self, config):
         super().__init__()
@@ -153,8 +204,7 @@ class VisionExpertAttention(nn.Module):
         self.head_dim = self.hidden_size // self.num_heads
         self.max_position_embeddings = config.max_position_embeddings
-        # self.rotary_emb = RotaryEmbedding(self.hidden_size // self.num_heads)
-        self.rotary_emb = FastRotaryEmbedding(dim=self.head_dim, pos_idx_in_fp32=False)
         self.vision_expert_query_key_value = nn.Linear(self.hidden_size, self.hidden_size * 3, bias=False)
         self.vision_expert_dense = nn.Linear(self.hidden_size, self.hidden_size, bias=False)
         self.language_expert_query_key_value = nn.Linear(self.hidden_size, self.hidden_size * 3, bias=False)
@@ -193,8 +243,8 @@ class VisionExpertAttention(nn.Module):
         kv_seq_len = key_states.shape[-2]
         if past_key_value is not None:
             kv_seq_len += past_key_value[0].shape[-2]
-        query_states, key_states = self.rotary_emb(query_states, key_states, position_ids=position_ids, max_seqlen=position_ids.max() + 1)
         if past_key_value is not None:
             key_states = torch.cat([past_key_value[0], key_states], dim=2)
@@ -278,7 +328,7 @@ class CogVLMPreTrainedModel(PreTrainedModel):
     config_class = CogVLMConfig
     base_model_prefix = "model"
     supports_gradient_checkpointing = False
-    _no_split_modules = ["CogVLMDecoderLayer"]
     _skip_keys_device_placement = "past_key_values"
     def _init_weights(self, module):
@@ -538,25 +588,23 @@ class CogVLMModel(CogVLMPreTrainedModel):
         return combined_attention_mask
-def chat_history_to_prompt(history, query):
-    prompt = " [INST] "
-    for i, (old_query, response) in enumerate(history):
-        prompt += old_query + " [/INST] " + response + " [INST] "
-    prompt += query + " [/INST] "
-    return prompt
-def base_history_to_prompt(history, query):
-    prompt = query
     return prompt
-_history_to_prompt = {
-    "base": base_history_to_prompt,
-    "chat": chat_history_to_prompt
-}
 class CogVLMForCausalLM(CogVLMPreTrainedModel):
     _auto_class = "AutoModelForCausalLM"
@@ -708,7 +756,8 @@ class CogVLMForCausalLM(CogVLMPreTrainedModel):
         # update token_type_ids with last value
         if "token_type_ids" in model_kwargs:
             token_type_ids = model_kwargs["token_type_ids"]
-            new_token_type_ids = torch.ones(size=(token_type_ids.shape[0], 1), dtype=token_type_ids.dtype, device=token_type_ids.device) * LANGUAGE_TOKEN_TYPE
             model_kwargs["token_type_ids"] = torch.cat([token_type_ids, new_token_type_ids], dim=-1)
         if not is_encoder_decoder:
@@ -744,14 +793,14 @@ class CogVLMForCausalLM(CogVLMPreTrainedModel):
             query: str,
             history: Optional[List[Tuple[str, str]]] = None,
             images: Optional[List["PIL.Image"]] = None,
-            template_version: Optional[Literal["base", "chat"]] = None,
     ):
         image_size: int = self.config.vision_config['image_size']
         patch_size: int = self.config.vision_config['patch_size']
         template_version = template_version or self.config.template_version
         assert images is None or len(images) <= 1, f"not support multi images by now."
         history = history or []
-        text = _history_to_prompt[template_version](history, query)
         input_ids = [tokenizer.bos_token_id]
         token_type_ids = [LANGUAGE_TOKEN_TYPE]

 import math
 import torch
 from torch import nn
+from torch.nn import functional as F
 from torch.nn import CrossEntropyLoss
 from torchvision import transforms
 from einops import rearrange
 from transformers.modeling_outputs import BaseModelOutputWithPast, CausalLMOutputWithPast
 from .configuration_cogvlm import CogVLMConfig
 from .visual import EVA2CLIPModel
 if TYPE_CHECKING:
         return context_layer
+class RotaryEmbedding(torch.nn.Module):
+    def __init__(self, dim, max_position_embeddings=2048, base=10000, device=None):
+        super().__init__()
+        self.dim = dim
+        self.max_position_embeddings = max_position_embeddings
+        self.base = base
+        inv_freq = self._compute_inv_freq(device)
+        self.register_buffer("inv_freq", inv_freq)
+        self.max_seq_len_cached = 0
+    def _compute_inv_freq(self, device=None):
+        return 1.0 / (
+                self.base
+                ** (torch.arange(0, self.dim, 2, device=device) / self.dim)
+        )
+    def _set_cos_sin_cache(self, seq_len, device, dtype):
+        self.max_seq_len_cached = seq_len
+        t = torch.arange(self.max_seq_len_cached, device=device, dtype=self.inv_freq.dtype)
+        freqs = torch.einsum("i,j->ij", t, self.inv_freq)
+        # Different from paper, but it uses a different permutation in order to obtain the same calculation
+        emb = torch.cat((freqs, freqs), dim=-1)
+        self.register_buffer("cos_cached", emb.cos()[:, None, :].to(dtype), persistent=False)
+        self.register_buffer("sin_cached", emb.sin()[:, None, :].to(dtype), persistent=False)
+    def forward(self, x, seq_len):
+        # x: [bs, num_attention_heads, seq_len, head_size]
+        if seq_len > self.max_seq_len_cached:
+            self._set_cos_sin_cache(seq_len=seq_len, device=x.device, dtype=x.dtype)
+        return (
+            self.cos_cached[:seq_len, ...].to(dtype=x.dtype),
+            self.sin_cached[:seq_len, ...].to(dtype=x.dtype),
+        )
+def rotate_half(x):
+    x1, x2 = x[..., :x.shape[-1] // 2], x[..., x.shape[-1] // 2:]
+    return torch.cat((-x2, x1), dim=x1.ndim - 1)
+def apply_rotary_pos_emb_index_bhs(q, k, cos, sin, position_id):
+    # batch_size, num_head, seq_len, hidden_size
+    cos, sin = F.embedding(position_id, cos.squeeze(1)).unsqueeze(1), \
+        F.embedding(position_id, sin.squeeze(1)).unsqueeze(1)
+    q, k = (q * cos) + (rotate_half(q) * sin), (k * cos) + (rotate_half(k) * sin)
+    return q, k
 class VisionExpertAttention(nn.Module):
     def __init__(self, config):
         super().__init__()
         self.head_dim = self.hidden_size // self.num_heads
         self.max_position_embeddings = config.max_position_embeddings
+        self.rotary_emb = RotaryEmbedding(self.head_dim)
         self.vision_expert_query_key_value = nn.Linear(self.hidden_size, self.hidden_size * 3, bias=False)
         self.vision_expert_dense = nn.Linear(self.hidden_size, self.hidden_size, bias=False)
         self.language_expert_query_key_value = nn.Linear(self.hidden_size, self.hidden_size * 3, bias=False)
         kv_seq_len = key_states.shape[-2]
         if past_key_value is not None:
             kv_seq_len += past_key_value[0].shape[-2]
+        cos, sin = self.rotary_emb(value_states, seq_len=position_ids.max() + 1)
+        query_states, key_states = apply_rotary_pos_emb_index_bhs(query_states, key_states, cos, sin, position_ids)
         if past_key_value is not None:
             key_states = torch.cat([past_key_value[0], key_states], dim=2)
     config_class = CogVLMConfig
     base_model_prefix = "model"
     supports_gradient_checkpointing = False
+    _no_split_modules = ["CogVLMDecoderLayer", "TransformerLayer"]
     _skip_keys_device_placement = "past_key_values"
     def _init_weights(self, module):
         return combined_attention_mask
+def _history_to_prompt(signal_type, history, query):
+    if signal_type == 'base':
+        return query
+    elif signal_type == 'vqa':
+        answer_format = 'Short answer:'
+    elif signal_type == 'chat':
+        answer_format = 'Answer:'
+    else:
+        assert False, f"Unknown signal type {signal_type}"
+    prompt = ''
+    for i, (old_query, response) in enumerate(history):
+        prompt += 'Question: ' + old_query + " {} ".format(answer_format) + response + "\n"
+    prompt += 'Question: {} {}'.format(query, answer_format)
     return prompt
 class CogVLMForCausalLM(CogVLMPreTrainedModel):
     _auto_class = "AutoModelForCausalLM"
         # update token_type_ids with last value
         if "token_type_ids" in model_kwargs:
             token_type_ids = model_kwargs["token_type_ids"]
+            new_token_type_ids = torch.ones(size=(token_type_ids.shape[0], 1), dtype=token_type_ids.dtype,
+                                            device=token_type_ids.device) * LANGUAGE_TOKEN_TYPE
             model_kwargs["token_type_ids"] = torch.cat([token_type_ids, new_token_type_ids], dim=-1)
         if not is_encoder_decoder:
             query: str,
             history: Optional[List[Tuple[str, str]]] = None,
             images: Optional[List["PIL.Image"]] = None,
+            template_version: Optional[Literal["base", "chat", "vqa"]] = None,
     ):
         image_size: int = self.config.vision_config['image_size']
         patch_size: int = self.config.vision_config['patch_size']
         template_version = template_version or self.config.template_version
         assert images is None or len(images) <= 1, f"not support multi images by now."
         history = history or []
+        text = _history_to_prompt(template_version, history, query)
         input_ids = [tokenizer.bos_token_id]
         token_type_ids = [LANGUAGE_TOKEN_TYPE]

util.py DELETED Viewed

@@ -1,483 +0,0 @@
-from typing import Optional, Tuple, Union
-import torch
-from einops import rearrange, repeat
-import torch.nn.functional as F
-import triton
-import triton.language as tl
-# @triton.autotune(
-#     configs=[
-#         triton.Config({"BLOCK_M": 2}),
-#         triton.Config({"BLOCK_M": 4}),
-#         triton.Config({"BLOCK_M": 8}),
-#         triton.Config({"BLOCK_M": 16}),
-#     ],
-#     key=["CACHE_KEY_SEQLEN", "BLOCK_K", "INTERLEAVED"],
-# )
-@triton.jit
-def rotary_kernel(
-        OUT,  # Pointers to matrices
-        X,
-        COS,
-        SIN,
-        CU_SEQLENS,
-        SEQLEN_OFFSETS,  # this could be int or a pointer
-        # Matrix dimensions
-        seqlen,
-        nheads,
-        rotary_dim,
-        seqlen_ro,
-        CACHE_KEY_SEQLEN,
-        # strides
-        stride_out_batch,
-        stride_out_nheads,
-        stride_out_seqlen,
-        stride_out_headdim,
-        stride_x_batch,
-        stride_x_nheads,
-        stride_x_seqlen,
-        stride_x_headdim,
-        # Meta-parameters
-        BLOCK_K: tl.constexpr,
-        IS_SEQLEN_OFFSETS_TENSOR: tl.constexpr,
-        IS_VARLEN: tl.constexpr,
-        INTERLEAVED: tl.constexpr,
-        CONJUGATE: tl.constexpr,
-        BLOCK_M: tl.constexpr,
-):
-    pid_m = tl.program_id(axis=0)
-    pid_batch = tl.program_id(axis=1)
-    pid_head = tl.program_id(axis=2)
-    rotary_dim_half = rotary_dim // 2
-    if not IS_VARLEN:
-        X = X + pid_batch * stride_x_batch + pid_head * stride_x_nheads
-        OUT = OUT + pid_batch * stride_out_batch + pid_head * stride_out_nheads
-        COS = COS + pid_batch * seqlen_ro * rotary_dim_half
-        SIN = SIN + pid_batch * seqlen_ro * rotary_dim_half
-    else:
-        start_idx = tl.load(CU_SEQLENS + pid_batch)
-        seqlen = tl.load(CU_SEQLENS + pid_batch + 1) - start_idx
-        X = X + start_idx * stride_x_seqlen + pid_head * stride_x_nheads
-        OUT = OUT + start_idx * stride_out_seqlen + pid_head * stride_out_nheads
-    if pid_m * BLOCK_M >= seqlen:
-        return
-    rm = pid_m * BLOCK_M + tl.arange(0, BLOCK_M)
-    if not IS_SEQLEN_OFFSETS_TENSOR:
-        rm_cs = rm + SEQLEN_OFFSETS
-    else:
-        rm_cs = rm + tl.load(SEQLEN_OFFSETS + pid_batch)
-    rk = tl.arange(0, BLOCK_K)
-    rk_half = tl.arange(0, BLOCK_K // 2)
-    if not INTERLEAVED:
-        # Load the 1st and 2nd halves of X, do calculation, then store to 1st and 2nd halves of OUT
-        X = X + (rm[:, None] * stride_x_seqlen + rk_half[None, :] * stride_x_headdim)
-        COS = COS + (rm_cs[:, None] * rotary_dim_half + rk_half[None, :])
-        SIN = SIN + (rm_cs[:, None] * rotary_dim_half + rk_half[None, :])
-        cos = tl.load(
-            COS, mask=(rm_cs[:, None] < seqlen_ro) & (rk_half[None, :] < rotary_dim_half), other=1.0
-        )
-        sin = tl.load(
-            SIN, mask=(rm_cs[:, None] < seqlen_ro) & (rk_half[None, :] < rotary_dim_half), other=0.0
-        )
-        x0 = tl.load(
-            X, mask=(rm[:, None] < seqlen) & (rk_half[None, :] < rotary_dim_half), other=0.0
-        )
-        x1 = tl.load(
-            X + rotary_dim_half * stride_x_headdim,
-            mask=(rm[:, None] < seqlen) & (rk_half[None, :] < rotary_dim_half),
-            other=0.0,
-        )
-        if CONJUGATE:
-            sin = -sin
-        o0 = x0 * cos - x1 * sin
-        o1 = x0 * sin + x1 * cos
-        # write back result
-        OUT = OUT + (rm[:, None] * stride_out_seqlen + rk_half[None, :] * stride_out_headdim)
-        tl.store(OUT, o0, mask=(rm[:, None] < seqlen) & (rk_half[None, :] < rotary_dim_half))
-        tl.store(
-            OUT + rotary_dim_half * stride_out_headdim,
-            o1,
-            mask=(rm[:, None] < seqlen) & (rk_half[None, :] < rotary_dim_half),
-        )
-    else:
-        # We don't want to load X[0, 2, 4, ...] and X[1, 3, 5, ...] separately since both are slow.
-        # Instead, we load x0 = X[0, 1, 2, 3, ...] and x1 = X[1, 0, 3, 2, ...].
-        # Loading x0 will be fast but x1 will be slow.
-        # Then we load cos = COS[0, 0, 1, 1, ...] and sin = SIN[0, 0, 1, 1, ...].
-        # Then we do the calculation and use tl.where to pick put the right outputs for the even
-        # and for the odd indices.
-        rk_swap = rk + ((rk + 1) % 2) * 2 - 1  # 1, 0, 3, 2, 5, 4, ...
-        rk_repeat = tl.arange(0, BLOCK_K) // 2
-        X0 = X + (rm[:, None] * stride_x_seqlen + rk[None, :] * stride_x_headdim)
-        X1 = X + (rm[:, None] * stride_x_seqlen + rk_swap[None, :] * stride_x_headdim)
-        COS = COS + (rm_cs[:, None] * rotary_dim_half + rk_repeat[None, :])
-        SIN = SIN + (rm_cs[:, None] * rotary_dim_half + rk_repeat[None, :])
-        cos = tl.load(
-            COS,
-            mask=(rm_cs[:, None] < seqlen_ro) & (rk_repeat[None, :] < rotary_dim_half),
-            other=1.0,
-        ).to(tl.float32)
-        sin = tl.load(
-            SIN,
-            mask=(rm_cs[:, None] < seqlen_ro) & (rk_repeat[None, :] < rotary_dim_half),
-            other=0.0,
-        ).to(tl.float32)
-        x0 = tl.load(X0, mask=(rm[:, None] < seqlen) & (rk[None, :] < rotary_dim), other=0.0).to(
-            tl.float32
-        )
-        x1 = tl.load(
-            X1, mask=(rm[:, None] < seqlen) & (rk_swap[None, :] < rotary_dim), other=0.0
-        ).to(tl.float32)
-        if CONJUGATE:
-            sin = -sin
-        x0_cos = x0 * cos
-        x1_sin = x1 * sin
-        out = tl.where(rk[None, :] % 2 == 0, x0_cos - x1_sin, x0_cos + x1_sin)
-        OUT = OUT + (rm[:, None] * stride_out_seqlen + rk[None, :] * stride_out_headdim)
-        tl.store(OUT, out, mask=(rm[:, None] < seqlen) & (rk[None, :] < rotary_dim))
-def apply_rotary(
-        x: torch.Tensor,
-        cos: torch.Tensor,
-        sin: torch.Tensor,
-        seqlen_offsets: Union[int, torch.Tensor] = 0,
-        cu_seqlens: Optional[torch.Tensor] = None,
-        max_seqlen: Optional[int] = None,
-        interleaved=False,
-        inplace=False,
-        conjugate=False,
-) -> torch.Tensor:
-    """
-    Arguments:
-        x: (batch, seqlen, nheads, headdim) if cu_seqlens is None
-            else (total_seqlen, nheads, headdim).
-        cos: (seqlen_ro, rotary_dim / 2)
-        sin: (seqlen_ro, rotary_dim / 2)
-        seqlen_offsets: integer or integer tensor of size (batch,)
-        cu_seqlens: (batch + 1,) or None
-        max_seqlen: int
-    Returns:
-        y: (batch, seqlen, nheads, headdim)
-    """
-    batch, nheads, seqlen, headdim = x.shape
-    batch_ro, seqlen_ro, rotary_dim = cos.shape
-    assert batch == batch_ro
-    assert sin.shape == cos.shape
-    rotary_dim *= 2
-    assert rotary_dim <= headdim, "rotary_dim must be <= headdim"
-    assert headdim <= 256, "Only support headdim <= 256"
-    assert seqlen_ro >= seqlen, "seqlen_ro must be >= seqlen"
-    assert (
-            cos.dtype == sin.dtype
-    ), f"cos and sin must have the same dtype, got {cos.dtype} and {sin.dtype}"
-    assert (
-            x.dtype == cos.dtype
-    ), f"Input and cos/sin must have the same dtype, got {x.dtype} and {cos.dtype}"
-    cos, sin = cos.contiguous(), sin.contiguous()
-    if isinstance(seqlen_offsets, torch.Tensor):
-        assert seqlen_offsets.shape == (batch,)
-        assert seqlen_offsets.dtype in [torch.int32, torch.int64]
-        seqlen_offsets = seqlen_offsets.contiguous()
-    else:
-        assert seqlen_offsets + seqlen <= seqlen_ro
-    output = torch.empty_like(x) if not inplace else x
-    if rotary_dim < headdim and not inplace:
-        output[..., rotary_dim:].copy_(x[..., rotary_dim:])
-    BLOCK_K = (
-        32
-        if rotary_dim <= 32
-        else (64 if rotary_dim <= 64 else (128 if rotary_dim <= 128 else 256))
-    )
-    grid = lambda META: (triton.cdiv(seqlen, META["BLOCK_M"]), batch, nheads)  # noqa
-    BLOCK_M = 4 if interleaved else (8 if rotary_dim <= 64 else 4)
-    # Need this, otherwise Triton tries to launch from cuda:0 and we get
-    # ValueError: Pointer argument (at 0) cannot be accessed from Triton (cpu tensor?)
-    with torch.cuda.device(x.device.index):
-        rotary_kernel[grid](
-            output,  # data ptrs
-            x,
-            cos,
-            sin,
-            cu_seqlens,
-            seqlen_offsets,
-            seqlen,  # shapes
-            nheads,
-            rotary_dim,
-            seqlen_ro,
-            seqlen // 128,  # key for triton cache (limit number of compilations)
-            output.stride(0),  # batch_strides
-            output.stride(-3),  # nheads_stride
-            output.stride(-2),  # seqlen_stride
-            output.stride(-1),  # headdim_stride
-            x.stride(0),  # batch_strides
-            x.stride(-3),  # nheads stride
-            x.stride(-2),  # seqlen stride
-            x.stride(-1),  # headdim stride
-            BLOCK_K,
-            isinstance(seqlen_offsets, torch.Tensor),
-            False,
-            interleaved,
-            conjugate,
-            BLOCK_M,
-        )
-    return output
-class ApplyRotaryEmb(torch.autograd.Function):
-    @staticmethod
-    def forward(
-            ctx,
-            x,
-            cos,
-            sin,
-            interleaved=False,
-            inplace=False,
-            seqlen_offsets: Union[int, torch.Tensor] = 0,
-            cu_seqlens: Optional[torch.Tensor] = None,
-            max_seqlen: Optional[int] = None,
-    ):
-        out = apply_rotary(
-            x,
-            cos,
-            sin,
-            seqlen_offsets=seqlen_offsets,
-            cu_seqlens=cu_seqlens,
-            max_seqlen=max_seqlen,
-            interleaved=interleaved,
-            inplace=inplace,
-        )
-        if isinstance(seqlen_offsets, int):
-            ctx.save_for_backward(cos, sin, cu_seqlens)  # Can't save int with save_for_backward
-            ctx.seqlen_offsets = seqlen_offsets
-        else:
-            ctx.save_for_backward(cos, sin, cu_seqlens, seqlen_offsets)
-            ctx.seqlen_offsets = None
-        ctx.interleaved = interleaved
-        ctx.inplace = inplace
-        ctx.max_seqlen = max_seqlen
-        return out if not inplace else x
-    @staticmethod
-    def backward(ctx, do):
-        seqlen_offsets = ctx.seqlen_offsets
-        if seqlen_offsets is None:
-            cos, sin, cu_seqlens, seqlen_offsets = ctx.saved_tensors
-        else:
-            cos, sin, cu_seqlens = ctx.saved_tensors
-        # TD [2023-09-02]: For some reason Triton (2.0.0.post1) errors with
-        # "[CUDA]: invalid device context", and cloning makes it work. Idk why. Triton 2.1.0 works.
-        if not ctx.interleaved and not ctx.inplace:
-            do = do.clone()
-        dx = apply_rotary(
-            do,
-            cos,
-            sin,
-            seqlen_offsets=seqlen_offsets,
-            cu_seqlens=cu_seqlens,
-            max_seqlen=ctx.max_seqlen,
-            interleaved=ctx.interleaved,
-            inplace=ctx.inplace,
-            conjugate=True,
-        )
-        return dx, None, None, None, None, None, None, None
-def apply_rotary_emb(
-        x,
-        cos,
-        sin,
-        interleaved=False,
-        inplace=False,
-        seqlen_offsets: Union[int, torch.Tensor] = 0,
-        cu_seqlens: Optional[torch.Tensor] = None,
-        max_seqlen: Optional[int] = None,
-):
-    """
-    Arguments:
-        x: (batch_size, seqlen, nheads, headdim) if cu_seqlens is None
-            else (total_seqlen, nheads, headdim)
-        cos, sin: (seqlen_rotary, rotary_dim / 2)
-        interleaved: if True, rotate pairs of even and odd dimensions (GPT-J style) instead
-            of 1st half and 2nd half (GPT-NeoX style).
-        inplace: if True, apply rotary embedding in-place.
-        seqlen_offsets: (batch_size,) or int. Each sequence in x is shifted by this amount.
-            Most commonly used in inference when we have KV cache.
-        cu_seqlens: (batch + 1,) or None
-        max_seqlen: int
-    Return:
-        out: (batch_size, seqlen, nheads, headdim) if cu_seqlens is None
-            else (total_seqlen, nheads, headdim)
-    rotary_dim must be <= headdim
-    Apply rotary embedding to the first rotary_dim of x.
-    """
-    return ApplyRotaryEmb.apply(
-        x, cos, sin, interleaved, inplace, seqlen_offsets, cu_seqlens, max_seqlen
-    )
-# For backward compatibility
-apply_rotary_emb_func = apply_rotary_emb
-class FastRotaryEmbedding(torch.nn.Module):
-    """
-    The rotary position embeddings from RoFormer_ (Su et. al).
-    A crucial insight from the method is that the query and keys are
-    transformed by rotation matrices which depend on the relative positions.
-    Other implementations are available in the Rotary Transformer repo_ and in
-    GPT-NeoX_, GPT-NeoX was an inspiration
-    .. _RoFormer: https://arxiv.org/abs/2104.09864
-    .. _repo: https://github.com/ZhuiyiTechnology/roformer
-    .. _GPT-NeoX: https://github.com/EleutherAI/gpt-neox
-    If scale_base is not None, this implements XPos (Sun et al., https://arxiv.org/abs/2212.10554).
-    A recommended value for scale_base is 512: https://github.com/HazyResearch/flash-attention/issues/96
-    Reference: https://github.com/sunyt32/torchscale/blob/main/torchscale/component/xpos_relative_position.py
-    """
-    def __init__(
-            self,
-            dim: int,
-            base=10000,
-            interleaved=False,
-            scale_base=None,
-            pos_idx_in_fp32=True,
-            device=None,
-    ):
-        """
-        interleaved: if True, rotate pairs of even and odd dimensions (GPT-J style) instead
-            of 1st half and 2nd half (GPT-NeoX style).
-        pos_idx_in_fp32: if True, the position indices [0.0, ..., seqlen - 1] are in fp32,
-            otherwise they might be in lower precision.
-            This option was added because previously (before 2023-07-02), when we construct
-            the position indices, we use the dtype of self.inv_freq. In most cases this would
-            be fp32, but if the model is trained in pure bf16 (not mixed precision), then
-            self.inv_freq would be bf16, and the position indices are also in bf16.
-            Because of the limited precision of bf16 (e.g. 1995.0 is rounded to 2000.0), the
-            embeddings for some positions will coincide.
-            To maintain compatibility with models previously trained in pure bf16,
-            we add this option.
-        """
-        super().__init__()
-        self.dim = dim
-        self.base = base
-        self.pos_idx_in_fp32 = pos_idx_in_fp32
-        # Generate and save the inverse frequency buffer (non trainable)
-        inv_freq = self._compute_inv_freq(device)
-        self.register_buffer("inv_freq", inv_freq)
-        self.interleaved = interleaved
-        self.scale_base = scale_base
-        scale = (
-            (torch.arange(0, dim, 2, device=device, dtype=torch.float32) + 0.4 * dim) / (1.4 * dim)
-            if scale_base is not None
-            else None
-        )
-        self.register_buffer("scale", scale, persistent=False)
-        self._seq_len_cached = 0
-        self._cos_cached = None
-        self._sin_cached = None
-        self._cos_k_cached = None
-        self._sin_k_cached = None
-        self.cos = None
-        self.sin = None
-    def _compute_inv_freq(self, device=None):
-        return 1.0 / (
-                self.base
-                ** (torch.arange(0, self.dim, 2, device=device) / self.dim)
-                # ** (torch.arange(0, self.dim, 2, device=device).float() / self.dim)
-        )
-    def _update_cos_sin_cache(self, seqlen, position_id, device=None, dtype=None):
-        if (
-                seqlen > self._seq_len_cached
-        ):
-            self._seq_len_cached = seqlen
-            # We want fp32 here, not self.inv_freq.dtype, since the model could be loaded in bf16
-            # And the output of arange can be quite large, so bf16 would lose a lot of precision.
-            # However, for compatibility reason, we add an option to use the dtype of self.inv_freq.
-            if self.pos_idx_in_fp32:
-                t = torch.arange(seqlen, device=device, dtype=torch.float32)
-                # We want fp32 here as well since inv_freq will be multiplied with t, and the output
-                # will be large. Having it in bf16 will lose a lot of precision and cause the
-                # cos & sin output to change significantly.
-                # We want to recompute self.inv_freq if it was not loaded in fp32
-                if self.inv_freq.dtype != torch.float32:
-                    inv_freq = self._compute_inv_freq(device=device)
-                else:
-                    inv_freq = self.inv_freq
-            else:
-                t = torch.arange(seqlen, device=device, dtype=self.inv_freq.dtype)
-                inv_freq = self.inv_freq
-            freqs = torch.einsum("i,j->ij", t, inv_freq)
-            if self.scale is None:
-                self._cos_cached = torch.cos(freqs).to(dtype)
-                self._sin_cached = torch.sin(freqs).to(dtype)
-            else:
-                power = (
-                                torch.arange(seqlen, dtype=self.scale.dtype, device=self.scale.device)
-                                - seqlen // 2
-                        ) / self.scale_base
-                scale = self.scale.to(device=power.device) ** rearrange(power, "s -> s 1")
-                # We want the multiplication by scale to happen in fp32
-                self._cos_cached = (torch.cos(freqs) * scale).to(dtype)
-                self._sin_cached = (torch.sin(freqs) * scale).to(dtype)
-                self._cos_k_cached = (torch.cos(freqs) / scale).to(dtype)
-                self._sin_k_cached = (torch.sin(freqs) / scale).to(dtype)
-    def forward(
-            self,
-            q: torch.Tensor,
-            k: torch.Tensor,
-            position_ids: torch.Tensor,
-            max_seqlen,
-    ) -> Tuple[torch.Tensor, torch.Tensor]:
-        """
-        q: (batch, nheads, seqlen, headdim)
-        k: (batch, nheads, seqlen, headdim)
-        position_id: (batch, seqlen)
-        max_seqlen: int
-        layer_id: int
-            only if layer_id == 0, then update cons and sin
-        Apply rotary embedding *inplace* to q k.
-        """
-        self._update_cos_sin_cache(max_seqlen, position_ids, device=q.device, dtype=q.dtype)
-        cos, sin = F.embedding(position_ids, self._cos_cached), F.embedding(position_ids, self._sin_cached)
-        q = apply_rotary_emb_func(
-            q,
-            cos,
-            sin,
-            interleaved=self.interleaved,
-            inplace=True
-        )
-        k = apply_rotary_emb_func(
-            k,
-            cos,
-            sin,
-            interleaved=self.interleaved,
-            inplace=True
-        )
-        return q, k