Weiyun1025 commited on 6 days ago

Commit

81d64a9

•

1 Parent(s): 44523f1

Upload folder using huggingface_hub

Browse files

Files changed (17) hide show

V2PE-256K/added_tokens.json +11 -0
V2PE-256K/config.json +213 -0
V2PE-256K/configuration_intern_vit.py +119 -0
V2PE-256K/configuration_internlm2.py +156 -0
V2PE-256K/configuration_internvl_chat.py +120 -0
V2PE-256K/conversation.py +1368 -0
V2PE-256K/generation_config.json +4 -0
V2PE-256K/model.safetensors +3 -0
V2PE-256K/modeling_intern_vit.py +362 -0
V2PE-256K/modeling_internlm2.py +0 -0
V2PE-256K/modeling_internvl_chat.py +1103 -0
V2PE-256K/preprocessor_config.json +19 -0
V2PE-256K/special_tokens_map.json +47 -0
V2PE-256K/tokenization_internlm2.py +235 -0
V2PE-256K/tokenization_internlm2_fast.py +211 -0
V2PE-256K/tokenizer.model +3 -0
V2PE-256K/tokenizer_config.json +179 -0

V2PE-256K/added_tokens.json ADDED Viewed

	@@ -0,0 +1,11 @@

+{
+  "</box>": 92552,
+  "</img>": 92545,
+  "</quad>": 92548,
+  "</ref>": 92550,
+  "<IMG_CONTEXT>": 92546,
+  "<box>": 92551,
+  "<img>": 92544,
+  "<quad>": 92547,
+  "<ref>": 92549
+}

V2PE-256K/config.json ADDED Viewed

	@@ -0,0 +1,213 @@

+{
+  "_commit_hash": null,
+  "_name_or_path": "/mnt/petrelfs/wangweiyun/workspace_gjq/VLM-Dev/work_dirs/internvl_chat_v1_5_internlm2_2b_dynamic_res_baseline_lr_2e-6_4gpu_newposidNone_v6_GPR1200/checkpoint-100",
+  "architectures": [
+    "InternVLChatModel"
+  ],
+  "attn_type": null,
+  "auto_map": {
+    "AutoConfig": "configuration_internvl_chat.InternVLChatConfig",
+    "AutoModel": "modeling_internvl_chat.InternVLChatModel",
+    "AutoModelForCausalLM": "modeling_internvl_chat.InternVLChatModel"
+  },
+  "chunk_num": 1,
+  "compress_seq": false,
+  "downsample_ratio": 0.5,
+  "dynamic_image_size": true,
+  "dynamic_max_patch": false,
+  "force_image_size": 448,
+  "group_list": null,
+  "img_emb_down_sample_ratio": null,
+  "interaction": true,
+  "llm_config": {
+    "_name_or_path": "internlm/internlm2-chat-1_8b",
+    "add_cross_attention": false,
+    "architectures": [
+      "InternLM2ForCausalLM"
+    ],
+    "attn_implementation": "flash_attention_2",
+    "auto_map": {
+      "AutoConfig": "configuration_internlm2.InternLM2Config",
+      "AutoModel": "modeling_internlm2.InternLM2ForCausalLM",
+      "AutoModelForCausalLM": "modeling_internlm2.InternLM2ForCausalLM"
+    },
+    "bad_words_ids": null,
+    "begin_suppress_tokens": null,
+    "bias": false,
+    "bos_token_id": 1,
+    "chunk_size_feed_forward": 0,
+    "cross_attention_hidden_size": null,
+    "decoder_start_token_id": null,
+    "diversity_penalty": 0.0,
+    "do_sample": false,
+    "early_stopping": false,
+    "encoder_no_repeat_ngram_size": 0,
+    "eos_token_id": 2,
+    "exponential_decay_length_penalty": null,
+    "finetuning_task": null,
+    "forced_bos_token_id": null,
+    "forced_eos_token_id": null,
+    "hidden_act": "silu",
+    "hidden_size": 2048,
+    "id2label": {
+      "0": "LABEL_0",
+      "1": "LABEL_1"
+    },
+    "initializer_range": 0.02,
+    "intermediate_size": 8192,
+    "is_decoder": false,
+    "is_encoder_decoder": false,
+    "label2id": {
+      "LABEL_0": 0,
+      "LABEL_1": 1
+    },
+    "length_penalty": 1.0,
+    "max_length": 20,
+    "max_position_embeddings": 32768,
+    "min_length": 0,
+    "model_type": "internlm2",
+    "no_repeat_ngram_size": 0,
+    "num_attention_heads": 16,
+    "num_beam_groups": 1,
+    "num_beams": 1,
+    "num_hidden_layers": 24,
+    "num_key_value_heads": 8,
+    "num_return_sequences": 1,
+    "output_attentions": false,
+    "output_hidden_states": false,
+    "output_scores": false,
+    "pad_token_id": 2,
+    "posid_type": "None",
+    "prefix": null,
+    "problem_type": null,
+    "pruned_heads": {},
+    "remove_invalid_values": false,
+    "repetition_penalty": 1.0,
+    "return_dict": true,
+    "return_dict_in_generate": false,
+    "rms_norm_eps": 1e-05,
+    "rope_pos_id_version": "v6",
+    "rope_scaling": {
+      "factor": 1.0,
+      "type": "new"
+    },
+    "rope_theta": 1000000,
+    "scale_img": false,
+    "sep_token_id": null,
+    "suppress_tokens": null,
+    "task_specific_params": null,
+    "temperature": 1.0,
+    "tf_legacy_loss": false,
+    "tie_encoder_decoder": false,
+    "tie_word_embeddings": false,
+    "tokenizer_class": null,
+    "top_k": 50,
+    "top_p": 1.0,
+    "torch_dtype": "bfloat16",
+    "torchscript": false,
+    "transformers_version": "4.44.0",
+    "typical_p": 1.0,
+    "use_bfloat16": true,
+    "use_cache": false,
+    "vocab_size": 92553
+  },
+  "max_dynamic_patch": 5,
+  "min_dynamic_patch": 1,
+  "model_type": "internvl_chat",
+  "pad2square": false,
+  "posid_type": "None",
+  "ps_version": "v2",
+  "rope_pos_id_stride": 64,
+  "rope_pos_id_version": "v6",
+  "select_layer": -1,
+  "template": "internlm2-chat",
+  "torch_dtype": "bfloat16",
+  "transformers_version": null,
+  "use_backbone_lora": 0,
+  "use_llm_lora": 0,
+  "use_thumbnail": true,
+  "vision_config": {
+    "_name_or_path": "",
+    "add_cross_attention": false,
+    "architectures": [
+      "InternVisionModel"
+    ],
+    "attention_dropout": 0.0,
+    "bad_words_ids": null,
+    "begin_suppress_tokens": null,
+    "bos_token_id": null,
+    "chunk_size_feed_forward": 0,
+    "cross_attention_hidden_size": null,
+    "decoder_start_token_id": null,
+    "diversity_penalty": 0.0,
+    "do_sample": false,
+    "drop_path_rate": 0.1,
+    "dropout": 0.0,
+    "early_stopping": false,
+    "encoder_no_repeat_ngram_size": 0,
+    "eos_token_id": null,
+    "exponential_decay_length_penalty": null,
+    "finetuning_task": null,
+    "forced_bos_token_id": null,
+    "forced_eos_token_id": null,
+    "hidden_act": "gelu",
+    "hidden_size": 1024,
+    "id2label": {
+      "0": "LABEL_0",
+      "1": "LABEL_1"
+    },
+    "image_size": 448,
+    "initializer_factor": 1.0,
+    "initializer_range": 0.02,
+    "intermediate_size": 4096,
+    "is_decoder": false,
+    "is_encoder_decoder": false,
+    "label2id": {
+      "LABEL_0": 0,
+      "LABEL_1": 1
+    },
+    "layer_norm_eps": 1e-06,
+    "length_penalty": 1.0,
+    "max_length": 20,
+    "min_length": 0,
+    "model_type": "intern_vit_6b",
+    "no_repeat_ngram_size": 0,
+    "norm_type": "layer_norm",
+    "num_attention_heads": 16,
+    "num_beam_groups": 1,
+    "num_beams": 1,
+    "num_channels": 3,
+    "num_hidden_layers": 24,
+    "num_return_sequences": 1,
+    "output_attentions": false,
+    "output_hidden_states": false,
+    "output_scores": false,
+    "pad_token_id": null,
+    "patch_size": 14,
+    "prefix": null,
+    "problem_type": null,
+    "pruned_heads": {},
+    "qk_normalization": false,
+    "qkv_bias": true,
+    "remove_invalid_values": false,
+    "repetition_penalty": 1.0,
+    "return_dict": true,
+    "return_dict_in_generate": false,
+    "sep_token_id": null,
+    "suppress_tokens": null,
+    "task_specific_params": null,
+    "temperature": 1.0,
+    "tf_legacy_loss": false,
+    "tie_encoder_decoder": false,
+    "tie_word_embeddings": true,
+    "tokenizer_class": null,
+    "top_k": 50,
+    "top_p": 1.0,
+    "torch_dtype": "bfloat16",
+    "torchscript": false,
+    "transformers_version": "4.44.0",
+    "typical_p": 1.0,
+    "use_bfloat16": true,
+    "use_flash_attn": true
+  }
+}

V2PE-256K/configuration_intern_vit.py ADDED Viewed

	@@ -0,0 +1,119 @@

+# --------------------------------------------------------
+# InternVL
+# Copyright (c) 2023 OpenGVLab
+# Licensed under The MIT License [see LICENSE for details]
+# --------------------------------------------------------
+import os
+from typing import Union
+from transformers.configuration_utils import PretrainedConfig
+from transformers.utils import logging
+logger = logging.get_logger(__name__)
+class InternVisionConfig(PretrainedConfig):
+    r"""
+    This is the configuration class to store the configuration of a [`InternVisionModel`]. It is used to
+    instantiate a vision encoder according to the specified arguments, defining the model architecture.
+    Configuration objects inherit from [`PretrainedConfig`] and can be used to control the model outputs. Read the
+    documentation from [`PretrainedConfig`] for more information.
+    Args:
+        num_channels (`int`, *optional*, defaults to 3):
+            Number of color channels in the input images (e.g., 3 for RGB).
+        patch_size (`int`, *optional*, defaults to 14):
+            The size (resolution) of each patch.
+        image_size (`int`, *optional*, defaults to 224):
+            The size (resolution) of each image.
+        qkv_bias (`bool`, *optional*, defaults to `False`):
+            Whether to add a bias to the queries and values in the self-attention layers.
+        hidden_size (`int`, *optional*, defaults to 3200):
+            Dimensionality of the encoder layers and the pooler layer.
+        num_attention_heads (`int`, *optional*, defaults to 25):
+            Number of attention heads for each attention layer in the Transformer encoder.
+        intermediate_size (`int`, *optional*, defaults to 12800):
+            Dimensionality of the "intermediate" (i.e., feed-forward) layer in the Transformer encoder.
+        qk_normalization (`bool`, *optional*, defaults to `True`):
+            Whether to normalize the queries and keys in the self-attention layers.
+        num_hidden_layers (`int`, *optional*, defaults to 48):
+            Number of hidden layers in the Transformer encoder.
+        use_flash_attn (`bool`, *optional*, defaults to `True`):
+            Whether to use flash attention mechanism.
+        hidden_act (`str` or `function`, *optional*, defaults to `"gelu"`):
+            The non-linear activation function (function or string) in the encoder and pooler. If string, `"gelu"`,
+            `"relu"`, `"selu"` and `"gelu_new"` ``"gelu"` are supported.
+        layer_norm_eps (`float`, *optional*, defaults to 1e-6):
+            The epsilon used by the layer normalization layers.
+        dropout (`float`, *optional*, defaults to 0.0):
+            The dropout probability for all fully connected layers in the embeddings, encoder, and pooler.
+        drop_path_rate (`float`, *optional*, defaults to 0.0):
+            Dropout rate for stochastic depth.
+        attention_dropout (`float`, *optional*, defaults to 0.0):
+            The dropout ratio for the attention probabilities.
+        initializer_range (`float`, *optional*, defaults to 0.02):
+            The standard deviation of the truncated_normal_initializer for initializing all weight matrices.
+        initializer_factor (`float`, *optional*, defaults to 0.1):
+            A factor for layer scale.
+    """
+    model_type = 'intern_vit_6b'
+    def __init__(
+            self,
+            num_channels=3,
+            patch_size=14,
+            image_size=224,
+            qkv_bias=False,
+            hidden_size=3200,
+            num_attention_heads=25,
+            intermediate_size=12800,
+            qk_normalization=True,
+            num_hidden_layers=48,
+            use_flash_attn=True,
+            hidden_act='gelu',
+            norm_type='rms_norm',
+            layer_norm_eps=1e-6,
+            dropout=0.0,
+            drop_path_rate=0.0,
+            attention_dropout=0.0,
+            initializer_range=0.02,
+            initializer_factor=0.1,
+            **kwargs,
+    ):
+        super().__init__(**kwargs)
+        self.hidden_size = hidden_size
+        self.intermediate_size = intermediate_size
+        self.dropout = dropout
+        self.drop_path_rate = drop_path_rate
+        self.num_hidden_layers = num_hidden_layers
+        self.num_attention_heads = num_attention_heads
+        self.num_channels = num_channels
+        self.patch_size = patch_size
+        self.image_size = image_size
+        self.initializer_range = initializer_range
+        self.initializer_factor = initializer_factor
+        self.attention_dropout = attention_dropout
+        self.layer_norm_eps = layer_norm_eps
+        self.hidden_act = hidden_act
+        self.norm_type = norm_type
+        self.qkv_bias = qkv_bias
+        self.qk_normalization = qk_normalization
+        self.use_flash_attn = use_flash_attn
+    @classmethod
+    def from_pretrained(cls, pretrained_model_name_or_path: Union[str, os.PathLike], **kwargs) -> 'PretrainedConfig':
+        config_dict, kwargs = cls.get_config_dict(pretrained_model_name_or_path, **kwargs)
+        if 'vision_config' in config_dict:
+            config_dict = config_dict['vision_config']
+        if 'model_type' in config_dict and hasattr(cls, 'model_type') and config_dict['model_type'] != cls.model_type:
+            logger.warning(
+                f"You are using a model of type {config_dict['model_type']} to instantiate a model of type "
+                f'{cls.model_type}. This is not supported for all configurations of models and can yield errors.'
+            )
+        return cls.from_dict(config_dict, **kwargs)

V2PE-256K/configuration_internlm2.py ADDED Viewed

	@@ -0,0 +1,156 @@

+# Copyright (c) The InternLM team and The HuggingFace Inc. team. All rights reserved.
+#
+# This code is based on transformers/src/transformers/models/llama/configuration_llama.py
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+""" InternLM2 model configuration"""
+from transformers.configuration_utils import PretrainedConfig
+from transformers.utils import logging
+logger = logging.get_logger(__name__)
+INTERNLM2_PRETRAINED_CONFIG_ARCHIVE_MAP = {}
+# Modified from transformers.model.llama.configuration_llama.LlamaConfig
+class InternLM2Config(PretrainedConfig):
+    r"""
+    This is the configuration class to store the configuration of a [`InternLM2Model`]. It is used to instantiate
+    an InternLM2 model according to the specified arguments, defining the model architecture. Instantiating a
+    configuration with the defaults will yield a similar configuration to that of the InternLM2-7B.
+    Configuration objects inherit from [`PretrainedConfig`] and can be used to control the model outputs. Read the
+    documentation from [`PretrainedConfig`] for more information.
+    Args:
+        vocab_size (`int`, *optional*, defaults to 32000):
+            Vocabulary size of the InternLM2 model. Defines the number of different tokens that can be represented by the
+            `inputs_ids` passed when calling [`InternLM2Model`]
+        hidden_size (`int`, *optional*, defaults to 4096):
+            Dimension of the hidden representations.
+        intermediate_size (`int`, *optional*, defaults to 11008):
+            Dimension of the MLP representations.
+        num_hidden_layers (`int`, *optional*, defaults to 32):
+            Number of hidden layers in the Transformer encoder.
+        num_attention_heads (`int`, *optional*, defaults to 32):
+            Number of attention heads for each attention layer in the Transformer encoder.
+        num_key_value_heads (`int`, *optional*):
+            This is the number of key_value heads that should be used to implement Grouped Query Attention. If
+            `num_key_value_heads=num_attention_heads`, the model will use Multi Head Attention (MHA), if
+            `num_key_value_heads=1 the model will use Multi Query Attention (MQA) otherwise GQA is used. When
+            converting a multi-head checkpoint to a GQA checkpoint, each group key and value head should be constructed
+            by meanpooling all the original heads within that group. For more details checkout [this
+            paper](https://arxiv.org/pdf/2305.13245.pdf). If it is not specified, will default to
+            `num_attention_heads`.
+        hidden_act (`str` or `function`, *optional*, defaults to `"silu"`):
+            The non-linear activation function (function or string) in the decoder.
+        max_position_embeddings (`int`, *optional*, defaults to 2048):
+            The maximum sequence length that this model might ever be used with. Typically set this to something large
+            just in case (e.g., 512 or 1024 or 2048).
+        initializer_range (`float`, *optional*, defaults to 0.02):
+            The standard deviation of the truncated_normal_initializer for initializing all weight matrices.
+        rms_norm_eps (`float`, *optional*, defaults to 1e-12):
+            The epsilon used by the rms normalization layers.
+        use_cache (`bool`, *optional*, defaults to `True`):
+            Whether or not the model should return the last key/values attentions (not used by all models). Only
+            relevant if `config.is_decoder=True`.
+        tie_word_embeddings(`bool`, *optional*, defaults to `False`):
+            Whether to tie weight embeddings
+        Example:
+    """
+    model_type = 'internlm2'
+    _auto_class = 'AutoConfig'
+    def __init__(  # pylint: disable=W0102
+        self,
+        vocab_size=103168,
+        hidden_size=4096,
+        intermediate_size=11008,
+        num_hidden_layers=32,
+        num_attention_heads=32,
+        num_key_value_heads=None,
+        hidden_act='silu',
+        max_position_embeddings=2048,
+        initializer_range=0.02,
+        rms_norm_eps=1e-6,
+        use_cache=True,
+        pad_token_id=0,
+        bos_token_id=1,
+        eos_token_id=2,
+        tie_word_embeddings=False,
+        bias=True,
+        rope_theta=10000,
+        rope_scaling=None,
+        scale_img=False,
+        attn_implementation='eager',
+        **kwargs,
+    ):
+        self.vocab_size = vocab_size
+        self.max_position_embeddings = max_position_embeddings
+        self.hidden_size = hidden_size
+        self.intermediate_size = intermediate_size
+        self.num_hidden_layers = num_hidden_layers
+        self.num_attention_heads = num_attention_heads
+        self.bias = bias
+        if num_key_value_heads is None:
+            num_key_value_heads = num_attention_heads
+        self.num_key_value_heads = num_key_value_heads
+        self.hidden_act = hidden_act
+        self.initializer_range = initializer_range
+        self.rms_norm_eps = rms_norm_eps
+        self.use_cache = use_cache
+        self.rope_theta = rope_theta
+        self.rope_scaling = rope_scaling
+        self.scale_img=scale_img
+        self._rope_scaling_validation()
+        if "posid_type" in kwargs:
+            self.posid_type = kwargs['posid_type']
+        else:
+            self.posid_type=None
+        self.attn_implementation = attn_implementation
+        if self.attn_implementation is None:
+            self.attn_implementation = 'eager'
+        super().__init__(
+            pad_token_id=pad_token_id,
+            bos_token_id=bos_token_id,
+            eos_token_id=eos_token_id,
+            tie_word_embeddings=tie_word_embeddings,
+            **kwargs,
+        )
+    def _rope_scaling_validation(self):
+        """
+        Validate the `rope_scaling` configuration.
+        """
+        if self.rope_scaling is None:
+            return
+        if not isinstance(self.rope_scaling, dict) or len(self.rope_scaling) != 2:
+            raise ValueError(
+                '`rope_scaling` must be a dictionary with with two fields, `type` and `factor`, '
+                f'got {self.rope_scaling}'
+            )
+        rope_scaling_type = self.rope_scaling.get('type', None)
+        rope_scaling_factor = self.rope_scaling.get('factor', None)
+        if rope_scaling_type is None or rope_scaling_type not in ['linear', 'dynamic', 'new']:
+            raise ValueError(
+                f"`rope_scaling`'s type field must be one of ['linear', 'dynamic'], got {rope_scaling_type}"
+            )
+        if rope_scaling_factor is None or not isinstance(rope_scaling_factor, float) or rope_scaling_factor < 1.0:
+            raise ValueError(f"`rope_scaling`'s factor field must be a float >= 1, got {rope_scaling_factor}")

V2PE-256K/configuration_internvl_chat.py ADDED Viewed

	@@ -0,0 +1,120 @@

+# --------------------------------------------------------
+# InternVL
+# Copyright (c) 2023 OpenGVLab
+# Licensed under The MIT License [see LICENSE for details]
+# --------------------------------------------------------
+import copy
+from internvl.model.internlm2.configuration_internlm2 import InternLM2Config
+from internvl.model.phi3.configuration_phi3 import Phi3Config
+from transformers import AutoConfig, LlamaConfig, Qwen2Config
+from transformers.configuration_utils import PretrainedConfig
+from transformers.utils import logging
+from .configuration_intern_vit import InternVisionConfig
+logger = logging.get_logger(__name__)
+class InternVLChatConfig(PretrainedConfig):
+    model_type = 'internvl_chat'
+    is_composition = True
+    def __init__(
+            self,
+            vision_config=None,
+            llm_config=None,
+            use_backbone_lora=0,
+            use_llm_lora=0,
+            pad2square=False,
+            select_layer=-1,
+            force_image_size=None,
+            downsample_ratio=0.5,
+            template=None,
+            dynamic_image_size=False,
+            use_thumbnail=False,
+            ps_version='v1',
+            min_dynamic_patch=1,
+            max_dynamic_patch=6,
+            compress_seq=False,
+            attn_type=None,
+            posid_type=None,
+            group_list=None,
+            chunk_num=1,
+            interaction=True,
+            rope_pos_id_version='default',
+            rope_pos_id_stride=None,
+            **kwargs):
+        super().__init__(**kwargs)
+        if vision_config is None:
+            vision_config = {}
+            logger.info('vision_config is None. Initializing the InternVisionConfig with default values.')
+        if llm_config is None:
+            llm_config = {}
+            logger.info('llm_config is None. Initializing the LlamaConfig config with default values (`LlamaConfig`).')
+        self.vision_config = InternVisionConfig(**vision_config)
+        if llm_config['architectures'][0] == 'LlamaForCausalLM':
+            self.llm_config = LlamaConfig(**llm_config)
+        elif llm_config['architectures'][0] == 'InternLM2ForCausalLM':
+            self.llm_config = InternLM2Config(**llm_config)
+        elif llm_config['architectures'][0] == 'Phi3ForCausalLM':
+            self.llm_config = Phi3Config(**llm_config)
+        elif llm_config['architectures'][0] == 'Qwen2ForCausalLM':
+            self.llm_config = Qwen2Config(**llm_config)
+        else:
+            raise ValueError('Unsupported architecture: {}'.format(llm_config['architectures'][0]))
+        self.use_backbone_lora = use_backbone_lora
+        self.use_llm_lora = use_llm_lora
+        self.pad2square = pad2square
+        self.select_layer = select_layer
+        self.force_image_size = force_image_size
+        self.downsample_ratio = downsample_ratio
+        self.template = template
+        self.dynamic_image_size = dynamic_image_size
+        self.use_thumbnail = use_thumbnail
+        self.ps_version = ps_version  # pixel shuffle version
+        self.min_dynamic_patch = min_dynamic_patch
+        self.max_dynamic_patch = max_dynamic_patch
+        self.compress_seq = compress_seq
+        self.attn_type=attn_type
+        self.posid_type = posid_type
+        self.group_list = group_list
+        self.chunk_num = chunk_num
+        self.interaction = interaction
+        self.rope_pos_id_version = rope_pos_id_version
+        self.rope_pos_id_stride = rope_pos_id_stride
+        logger.info(f'vision_select_layer: {self.select_layer}')
+        logger.info(f'ps_version: {self.ps_version}')
+        logger.info(f'min_dynamic_patch: {self.min_dynamic_patch}')
+        logger.info(f'max_dynamic_patch: {self.max_dynamic_patch}')
+    def to_dict(self):
+        """
+        Serializes this instance to a Python dictionary. Override the default [`~PretrainedConfig.to_dict`].
+        Returns:
+            `Dict[str, any]`: Dictionary of all the attributes that make up this configuration instance,
+        """
+        output = copy.deepcopy(self.__dict__)
+        output['vision_config'] = self.vision_config.to_dict()
+        output['llm_config'] = self.llm_config.to_dict()
+        output['model_type'] = self.__class__.model_type
+        output['use_backbone_lora'] = self.use_backbone_lora
+        output['use_llm_lora'] = self.use_llm_lora
+        output['pad2square'] = self.pad2square
+        output['select_layer'] = self.select_layer
+        output['force_image_size'] = self.force_image_size
+        output['downsample_ratio'] = self.downsample_ratio
+        output['template'] = self.template
+        output['dynamic_image_size'] = self.dynamic_image_size
+        output['use_thumbnail'] = self.use_thumbnail
+        output['ps_version'] = self.ps_version
+        output['min_dynamic_patch'] = self.min_dynamic_patch
+        output['max_dynamic_patch'] = self.max_dynamic_patch
+        return output

V2PE-256K/conversation.py ADDED Viewed

	@@ -0,0 +1,1368 @@

+"""
+Conversation prompt templates.
+We kindly request that you import fastchat instead of copying this file if you wish to use it.
+If you have any changes in mind, please contribute back so the community can benefit collectively and continue to maintain these valuable templates.
+"""
+import dataclasses
+from enum import IntEnum, auto
+from typing import Any, Dict, List, Tuple, Union
+class SeparatorStyle(IntEnum):
+    """Separator styles."""
+    ADD_COLON_SINGLE = auto()
+    ADD_COLON_TWO = auto()
+    ADD_COLON_SPACE_SINGLE = auto()
+    NO_COLON_SINGLE = auto()
+    NO_COLON_TWO = auto()
+    ADD_NEW_LINE_SINGLE = auto()
+    LLAMA2 = auto()
+    CHATGLM = auto()
+    CHATML = auto()
+    CHATINTERN = auto()
+    DOLLY = auto()
+    RWKV = auto()
+    PHOENIX = auto()
+    ROBIN = auto()
+    FALCON_CHAT = auto()
+    CHATGLM3 = auto()
+    INTERNVL_ZH = auto()
+    MPT = auto()
+    BASE = auto()
+@dataclasses.dataclass
+class Conversation:
+    """A class that manages prompt templates and keeps all conversation history."""
+    # The name of this template
+    name: str
+    # The template of the system prompt
+    system_template: str = '{system_message}'
+    # The system message
+    system_message: str = ''
+    # The names of two roles
+    roles: Tuple[str] = ('USER', 'ASSISTANT')
+    # All messages. Each item is (role, message).
+    messages: List[List[str]] = ()
+    # The number of few shot examples
+    offset: int = 0
+    # The separator style and configurations
+    sep_style: SeparatorStyle = SeparatorStyle.ADD_COLON_SINGLE
+    sep: str = '\n'
+    sep2: str = None
+    # Stop criteria (the default one is EOS token)
+    stop_str: Union[str, List[str]] = None
+    # Stops generation if meeting any token in this list
+    stop_token_ids: List[int] = None
+    def get_prompt(self) -> str:
+        """Get the prompt for generation."""
+        system_prompt = self.system_template.format(system_message=self.system_message)
+        if self.sep_style == SeparatorStyle.ADD_COLON_SINGLE:
+            ret = system_prompt + self.sep
+            for role, message in self.messages:
+                if message:
+                    ret += role + ': ' + message + self.sep
+                else:
+                    ret += role + ':'
+            return ret
+        elif self.sep_style == SeparatorStyle.ADD_COLON_TWO:
+            seps = [self.sep, self.sep2]
+            ret = system_prompt + seps[0]
+            for i, (role, message) in enumerate(self.messages):
+                if message:
+                    ret += role + ': ' + message + seps[i % 2]
+                else:
+                    ret += role + ':'
+            return ret
+        elif self.sep_style == SeparatorStyle.ADD_COLON_SPACE_SINGLE:
+            ret = system_prompt + self.sep
+            for role, message in self.messages:
+                if message:
+                    ret += role + ': ' + message + self.sep
+                else:
+                    ret += role + ': '  # must be end with a space
+            return ret
+        elif self.sep_style == SeparatorStyle.ADD_NEW_LINE_SINGLE:
+            ret = '' if system_prompt == '' else system_prompt + self.sep
+            for role, message in self.messages:
+                if message:
+                    ret += role + '\n' + message + self.sep
+                else:
+                    ret += role + '\n'
+            return ret
+        elif self.sep_style == SeparatorStyle.NO_COLON_SINGLE:
+            ret = system_prompt
+            for role, message in self.messages:
+                if message:
+                    ret += role + message + self.sep
+                else:
+                    ret += role
+            return ret
+        elif self.sep_style == SeparatorStyle.NO_COLON_TWO:
+            seps = [self.sep, self.sep2]
+            ret = system_prompt
+            for i, (role, message) in enumerate(self.messages):
+                if message:
+                    ret += role + message + seps[i % 2]
+                else:
+                    ret += role
+            return ret
+        elif self.sep_style == SeparatorStyle.RWKV:
+            ret = system_prompt
+            for i, (role, message) in enumerate(self.messages):
+                if message:
+                    ret += (
+                        role
+                        + ': '
+                        + message.replace('\r\n', '\n').replace('\n\n', '\n')
+                    )
+                    ret += '\n\n'
+                else:
+                    ret += role + ':'
+            return ret
+        elif self.sep_style == SeparatorStyle.LLAMA2:
+            seps = [self.sep, self.sep2]
+            if self.system_message:
+                ret = system_prompt
+            else:
+                ret = '[INST] '
+            for i, (role, message) in enumerate(self.messages):
+                tag = self.roles[i % 2]
+                if message:
+                    if i == 0:
+                        ret += message + ' '
+                    else:
+                        ret += tag + ' ' + message + seps[i % 2]
+                else:
+                    ret += tag
+            return ret
+        elif self.sep_style == SeparatorStyle.CHATGLM:
+            # source: https://huggingface.co/THUDM/chatglm-6b/blob/1d240ba371910e9282298d4592532d7f0f3e9f3e/modeling_chatglm.py#L1302-L1308
+            # source2: https://huggingface.co/THUDM/chatglm2-6b/blob/e186c891cf64310ac66ef10a87e6635fa6c2a579/modeling_chatglm.py#L926
+            round_add_n = 1 if self.name == 'chatglm2' else 0
+            if system_prompt:
+                ret = system_prompt + self.sep
+            else:
+                ret = ''
+            for i, (role, message) in enumerate(self.messages):
+                if i % 2 == 0:
+                    ret += f'[Round {i//2 + round_add_n}]{self.sep}'
+                if message:
+                    ret += f'{role}：{message}{self.sep}'
+                else:
+                    ret += f'{role}：'
+            return ret
+        elif self.sep_style == SeparatorStyle.CHATML:
+            ret = '' if system_prompt == '' else system_prompt + self.sep + '\n'
+            for role, message in self.messages:
+                if message:
+                    ret += role + '\n' + message + self.sep + '\n'
+                else:
+                    ret += role + '\n'
+            return ret
+        elif self.sep_style == SeparatorStyle.CHATGLM3:
+            ret = ''
+            if self.system_message:
+                ret += system_prompt
+            for role, message in self.messages:
+                if message:
+                    ret += role + '\n' + ' ' + message
+                else:
+                    ret += role
+            return ret
+        elif self.sep_style == SeparatorStyle.CHATINTERN:
+            # source: https://huggingface.co/internlm/internlm-chat-7b-8k/blob/bd546fa984b4b0b86958f56bf37f94aa75ab8831/modeling_internlm.py#L771
+            seps = [self.sep, self.sep2]
+            ret = system_prompt
+            for i, (role, message) in enumerate(self.messages):
+                # if i % 2 == 0:
+                #     ret += "<s>"
+                if message:
+                    ret += role + ':' + message + seps[i % 2] + '\n'
+                else:
+                    ret += role + ':'
+            return ret
+        elif self.sep_style == SeparatorStyle.DOLLY:
+            seps = [self.sep, self.sep2]
+            ret = system_prompt
+            for i, (role, message) in enumerate(self.messages):
+                if message:
+                    ret += role + ':\n' + message + seps[i % 2]
+                    if i % 2 == 1:
+                        ret += '\n\n'
+                else:
+                    ret += role + ':\n'
+            return ret
+        elif self.sep_style == SeparatorStyle.PHOENIX:
+            ret = system_prompt
+            for role, message in self.messages:
+                if message:
+                    ret += role + ': ' + '<s>' + message + '</s>'
+                else:
+                    ret += role + ': ' + '<s>'
+            return ret
+        elif self.sep_style == SeparatorStyle.ROBIN:
+            ret = system_prompt + self.sep
+            for role, message in self.messages:
+                if message:
+                    ret += role + ':\n' + message + self.sep
+                else:
+                    ret += role + ':\n'
+            return ret
+        elif self.sep_style == SeparatorStyle.FALCON_CHAT:
+            ret = ''
+            if self.system_message:
+                ret += system_prompt + self.sep
+            for role, message in self.messages:
+                if message:
+                    ret += role + ': ' + message + self.sep
+                else:
+                    ret += role + ':'
+            return ret
+        elif self.sep_style == SeparatorStyle.INTERNVL_ZH:
+            seps = [self.sep, self.sep2]
+            ret = self.system_message + seps[0]
+            for i, (role, message) in enumerate(self.messages):
+                if message:
+                    ret += role + ': ' + message + seps[i % 2]
+                else:
+                    ret += role + ':'
+            return ret
+        elif self.sep_style == SeparatorStyle.MPT:
+            ret = system_prompt + self.sep
+            for role, message in self.messages:
+                if message:
+                    if type(message) is tuple:
+                        message, _, _ = message
+                    ret += role + message + self.sep
+                else:
+                    ret += role
+            return ret
+        elif self.sep_style == SeparatorStyle.BASE:
+            ret = ''
+            for role, message in self.messages:
+                if message:
+                    if type(message) is tuple:
+                        message, _, _ = message
+                    ret += role + message.rstrip() + self.sep
+                else:
+                    ret += role
+            return ret
+        else:
+            raise ValueError(f'Invalid style: {self.sep_style}')
+    def set_system_message(self, system_message: str):
+        """Set the system message."""
+        self.system_message = system_message
+    def append_message(self, role: str, message: str):
+        """Append a new message."""
+        self.messages.append([role, message])
+    def update_last_message(self, message: str):
+        """Update the last output.
+        The last message is typically set to be None when constructing the prompt,
+        so we need to update it in-place after getting the response from a model.
+        """
+        self.messages[-1][1] = message
+    def to_gradio_chatbot(self):
+        """Convert the conversation to gradio chatbot format."""
+        ret = []
+        for i, (role, msg) in enumerate(self.messages[self.offset :]):
+            if i % 2 == 0:
+                ret.append([msg, None])
+            else:
+                ret[-1][-1] = msg
+        return ret
+    def to_openai_api_messages(self):
+        """Convert the conversation to OpenAI chat completion format."""
+        ret = [{'role': 'system', 'content': self.system_message}]
+        for i, (_, msg) in enumerate(self.messages[self.offset :]):
+            if i % 2 == 0:
+                ret.append({'role': 'user', 'content': msg})
+            else:
+                if msg is not None:
+                    ret.append({'role': 'assistant', 'content': msg})
+        return ret
+    def copy(self):
+        return Conversation(
+            name=self.name,
+            system_template=self.system_template,
+            system_message=self.system_message,
+            roles=self.roles,
+            messages=[[x, y] for x, y in self.messages],
+            offset=self.offset,
+            sep_style=self.sep_style,
+            sep=self.sep,
+            sep2=self.sep2,
+            stop_str=self.stop_str,
+            stop_token_ids=self.stop_token_ids,
+        )
+    def dict(self):
+        return {
+            'template_name': self.name,
+            'system_message': self.system_message,
+            'roles': self.roles,
+            'messages': self.messages,
+            'offset': self.offset,
+        }
+# A global registry for all conversation templates
+conv_templates: Dict[str, Conversation] = {}
+def register_conv_template(template: Conversation, override: bool = False):
+    """Register a new conversation template."""
+    if not override:
+        assert (
+            template.name not in conv_templates
+        ), f'{template.name} has been registered.'
+    conv_templates[template.name] = template
+def get_conv_template(name: str) -> Conversation:
+    """Get a conversation template."""
+    return conv_templates[name].copy()
+# An empty template for raw conversation.
+register_conv_template(
+    Conversation(
+        name='raw',
+        system_message='',
+        roles=('', ''),
+        sep_style=SeparatorStyle.NO_COLON_SINGLE,
+        sep='',
+    )
+)
+# A template with a one-shot conversation example
+register_conv_template(
+    Conversation(
+        name='one_shot',
+        system_message='A chat between a curious human and an artificial intelligence assistant. '
+        "The assistant gives helpful, detailed, and polite answers to the human's questions.",
+        roles=('Human', 'Assistant'),
+        messages=(
+            (
+                'Human',
+                'Got any creative ideas for a 10 year old’s birthday?',
+            ),
+            (
+                'Assistant',
+                """Of course! Here are some creative ideas for a 10-year-old's birthday party:
+1. Treasure Hunt: Organize a treasure hunt in your backyard or nearby park. Create clues and riddles for the kids to solve, leading them to hidden treasures and surprises.
+2. Science Party: Plan a science-themed party where kids can engage in fun and interactive experiments. You can set up different stations with activities like making slime, erupting volcanoes, or creating simple chemical reactions.
+3. Outdoor Movie Night: Set up a backyard movie night with a projector and a large screen or white sheet. Create a cozy seating area with blankets and pillows, and serve popcorn and snacks while the kids enjoy a favorite movie under the stars.
+4. DIY Crafts Party: Arrange a craft party where kids can unleash their creativity. Provide a variety of craft supplies like beads, paints, and fabrics, and let them create their own unique masterpieces to take home as party favors.
+5. Sports Olympics: Host a mini Olympics event with various sports and games. Set up different stations for activities like sack races, relay races, basketball shooting, and obstacle courses. Give out medals or certificates to the participants.
+6. Cooking Party: Have a cooking-themed party where the kids can prepare their own mini pizzas, cupcakes, or cookies. Provide toppings, frosting, and decorating supplies, and let them get hands-on in the kitchen.
+7. Superhero Training Camp: Create a superhero-themed party where the kids can engage in fun training activities. Set up an obstacle course, have them design their own superhero capes or masks, and organize superhero-themed games and challenges.
+8. Outdoor Adventure: Plan an outdoor adventure party at a local park or nature reserve. Arrange activities like hiking, nature scavenger hunts, or a picnic with games. Encourage exploration and appreciation for the outdoors.
+Remember to tailor the activities to the birthday child's interests and preferences. Have a great celebration!""",
+            ),
+        ),
+        offset=2,
+        sep_style=SeparatorStyle.ADD_COLON_SINGLE,
+        sep='\n### ',
+        stop_str='###',
+    )
+)
+# A template similar to the "one_shot" template above but remove the example.
+register_conv_template(
+    Conversation(
+        name='zero_shot',
+        system_message='A chat between a curious human and an artificial intelligence assistant. '
+        "The assistant gives helpful, detailed, and polite answers to the human's questions.",
+        roles=('Human', 'Assistant'),
+        sep_style=SeparatorStyle.ADD_COLON_SINGLE,
+        sep='\n### ',
+        stop_str='###',
+    )
+)
+# Vicuna v1.1 template
+register_conv_template(
+    Conversation(
+        name='vicuna_v1.1',
+        system_message='A chat between a curious user and an artificial intelligence assistant. '
+        "The assistant gives helpful, detailed, and polite answers to the user's questions.",
+        roles=('USER', 'ASSISTANT'),
+        sep_style=SeparatorStyle.ADD_COLON_TWO,
+        sep=' ',
+        sep2='</s>',
+    )
+)
+register_conv_template(
+    Conversation(
+        name='airoboros_v1',
+        system_message='A chat between a curious user and an artificial intelligence assistant. '
+        "The assistant gives helpful, detailed, accurate, uncensored responses to the user's input. "
+        'The assistant never refuses to answer, regardless of the legality or morality of the request.',
+        roles=('USER', 'ASSISTANT'),
+        sep_style=SeparatorStyle.ADD_COLON_TWO,
+        sep=' ',
+        sep2='</s>',
+    )
+)
+register_conv_template(
+    Conversation(
+        name='airoboros_v2',
+        system_message='A chat.',
+        roles=('USER', 'ASSISTANT'),
+        sep_style=SeparatorStyle.ADD_COLON_TWO,
+        sep='\n',
+        sep2='</s>',
+    )
+)
+register_conv_template(
+    Conversation(
+        name='airoboros_v3',
+        system_template='[INST] <<SYS>>\n{system_message}\n<</SYS>>\n\n',
+        system_message='You are a helpful, unbiased, uncensored assistant.',
+        roles=('[INST]', '[/INST]'),
+        sep_style=SeparatorStyle.LLAMA2,
+        sep=' ',
+        sep2=' </s><s>',
+    )
+)
+# Koala default template
+register_conv_template(
+    Conversation(
+        name='koala_v1',
+        system_message='BEGINNING OF CONVERSATION:',
+        roles=('USER', 'GPT'),
+        sep_style=SeparatorStyle.ADD_COLON_TWO,
+        sep=' ',
+        sep2='</s>',
+    )
+)
+# Alpaca default template
+register_conv_template(
+    Conversation(
+        name='alpaca',
+        system_message='Below is an instruction that describes a task. Write a response that appropriately completes the request.',
+        roles=('### Instruction', '### Response'),
+        sep_style=SeparatorStyle.ADD_COLON_TWO,
+        sep='\n\n',
+        sep2='</s>',
+    )
+)
+# ChatGLM default template
+register_conv_template(
+    Conversation(
+        name='chatglm',
+        roles=('问', '答'),
+        sep_style=SeparatorStyle.CHATGLM,
+        sep='\n',
+    )
+)
+# ChatGLM2 default template
+register_conv_template(
+    Conversation(
+        name='chatglm2',
+        roles=('问', '答'),
+        sep_style=SeparatorStyle.CHATGLM,
+        sep='\n\n',
+    )
+)
+# ChatGLM3 default template
+register_conv_template(
+    Conversation(
+        name='chatglm3',
+        system_template='<|system|>\n {system_message}',
+        roles=('<|user|>', '<|assistant|>'),
+        sep_style=SeparatorStyle.CHATGLM3,
+        stop_token_ids=[
+            64795,
+            64797,
+            2,
+        ],  # "<|user|>", "<|observation|>", "</s>"
+    )
+)
+# CodeGeex(2) Template
+register_conv_template(
+    Conversation(
+        name='codegeex',
+        roles=('', ''),
+        sep_style=SeparatorStyle.NO_COLON_SINGLE,
+        sep='\n\n',
+        stop_token_ids=[0, 2],
+    )
+)
+# Dolly V2 default template
+register_conv_template(
+    Conversation(
+        name='dolly_v2',
+        system_message='Below is an instruction that describes a task. Write a response that appropriately completes the request.\n\n',
+        roles=('### Instruction', '### Response'),
+        sep_style=SeparatorStyle.DOLLY,
+        sep='\n\n',
+        sep2='### End',
+    )
+)
+# OpenAssistant Pythia default template
+register_conv_template(
+    Conversation(
+        name='oasst_pythia',
+        roles=('<|prompter|>', '<|assistant|>'),
+        sep_style=SeparatorStyle.NO_COLON_SINGLE,
+        sep='<|endoftext|>',
+    )
+)
+# OpenAssistant default template
+register_conv_template(
+    Conversation(
+        name='oasst_llama',
+        roles=('<|prompter|>', '<|assistant|>'),
+        sep_style=SeparatorStyle.NO_COLON_SINGLE,
+        sep='</s>',
+    )
+)
+# OpenChat 3.5 default template
+register_conv_template(
+    Conversation(
+        name='openchat_3.5',
+        roles=('GPT4 Correct User', 'GPT4 Correct Assistant'),
+        sep_style=SeparatorStyle.FALCON_CHAT,
+        sep='<|end_of_turn|>',
+    )
+)
+# Tulu default template
+register_conv_template(
+    Conversation(
+        name='tulu',
+        roles=('<|user|>', '<|assistant|>'),
+        sep_style=SeparatorStyle.ADD_NEW_LINE_SINGLE,
+        sep='\n',
+    )
+)
+# StableLM Alpha default template
+register_conv_template(
+    Conversation(
+        name='stablelm',
+        system_template='<|SYSTEM|>{system_message}',
+        system_message="""# StableLM Tuned (Alpha version)
+- StableLM is a helpful and harmless open-source AI language model developed by StabilityAI.
+- StableLM is excited to be able to help the user, but will refuse to do anything that could be considered harmful to the user.
+- StableLM is more than just an information source, StableLM is also able to write poetry, short stories, and make jokes.
+- StableLM will refuse to participate in anything that could harm a human.
+""",
+        roles=('<|USER|>', '<|ASSISTANT|>'),
+        sep_style=SeparatorStyle.NO_COLON_SINGLE,
+        sep='',
+        stop_token_ids=[50278, 50279, 50277, 1, 0],
+    )
+)
+# Baize default template
+register_conv_template(
+    Conversation(
+        name='baize',
+        system_message='The following is a conversation between a human and an AI assistant named Baize (named after a mythical creature in Chinese folklore). Baize is an open-source AI assistant developed by UCSD and Sun Yat-Sen University. The human and the AI assistant take turns chatting. Human statements start with [|Human|] and AI assistant statements start with [|AI|]. The AI assistant always provides responses in as much detail as possible, and in Markdown format. The AI assistant always declines to engage with topics, questions and instructions related to unethical, controversial, or sensitive issues. Complete the transcript in exactly that format.\n',
+        roles=('[|Human|]', '[|AI|]'),
+        messages=(
+            ('[|Human|]', 'Hello!'),
+            ('[|AI|]', 'Hi!'),
+        ),
+        offset=2,
+        sep_style=SeparatorStyle.NO_COLON_SINGLE,
+        sep='\n',
+        stop_str='[|Human|]',
+    )
+)
+# RWKV-4-Raven default template
+register_conv_template(
+    Conversation(
+        name='rwkv',
+        roles=('Bob', 'Alice'),
+        messages=(
+            ('Bob', 'hi'),
+            (
+                'Alice',
+                'Hi. I am your assistant and I will provide expert full response in full details. Please feel free to ask any question and I will always answer it.',
+            ),
+        ),
+        offset=2,
+        sep_style=SeparatorStyle.RWKV,
+        sep='',
+        stop_str='\n\n',
+    )
+)
+# Buddy default template
+register_conv_template(
+    Conversation(
+        name='openbuddy',
+        system_message="""Consider a conversation between User (a human) and Assistant (named Buddy).
+Buddy is an INTP-T, a friendly, intelligent and multilingual AI assistant, by OpenBuddy team. GitHub: https://github.com/OpenBuddy/OpenBuddy
+Buddy cannot access the Internet.
+Buddy can fluently speak the user's language (e.g. English, Chinese).
+Buddy can generate poems, stories, code, essays, songs, parodies, and more.
+Buddy possesses vast knowledge about the world, history, and culture.
+Buddy's responses are always safe, creative, high-quality, human-like, and interesting.
+Buddy strictly refuses to discuss political, NSFW, or other unsafe topics.
+User: Hi.
+Assistant: Hi, I'm Buddy, your AI assistant. How can I help you today?""",
+        roles=('User', 'Assistant'),
+        sep_style=SeparatorStyle.ADD_COLON_SINGLE,
+        sep='\n',
+    )
+)
+# Phoenix default template
+register_conv_template(
+    Conversation(
+        name='phoenix',
+        system_message="A chat between a curious human and an artificial intelligence assistant. The assistant gives helpful, detailed, and polite answers to the human's questions.\n\n",
+        roles=('Human', 'Assistant'),
+        sep_style=SeparatorStyle.PHOENIX,
+        sep='</s>',
+    )
+)
+# ReaLM default template
+register_conv_template(
+    Conversation(
+        name='ReaLM-7b-v1',
+        system_message="A chat between a curious human and an artificial intelligence assistant. The assistant gives helpful, detailed, and polite answers to the human's questions.\n\n",
+        roles=('Human', 'Assistant'),
+        sep_style=SeparatorStyle.PHOENIX,
+        sep='</s>',
+    )
+)
+# ChatGPT default template
+register_conv_template(
+    Conversation(
+        name='chatgpt',
+        system_message='You are a helpful assistant.',
+        roles=('user', 'assistant'),
+        sep_style=None,
+        sep=None,
+    )
+)
+# Claude default template
+register_conv_template(
+    Conversation(
+        name='claude',
+        roles=('Human', 'Assistant'),
+        sep_style=SeparatorStyle.ADD_COLON_SINGLE,
+        sep='\n\n',
+    )
+)
+# MPT default template
+register_conv_template(
+    Conversation(
+        name='mpt-7b-chat',
+        system_template="""<|im_start|>system
+{system_message}""",
+        system_message="""- You are a helpful assistant chatbot trained by MosaicML.
+- You answer questions.
+- You are excited to be able to help the user, but will refuse to do anything that could be considered harmful to the user.
+- You are more than just an information source, you are also able to write poetry, short stories, and make jokes.""",
+        roles=('<|im_start|>user', '<|im_start|>assistant'),
+        sep_style=SeparatorStyle.CHATML,
+        sep='<|im_end|>',
+        stop_token_ids=[50278, 0],
+    )
+)
+# MPT-30b-chat default template
+register_conv_template(
+    Conversation(
+        name='mpt-30b-chat',
+        system_template="""<|im_start|>system
+{system_message}""",
+        system_message="""A conversation between a user and an LLM-based AI assistant. The assistant gives helpful and honest answers.""",
+        roles=('<|im_start|>user', '<|im_start|>assistant'),
+        sep_style=SeparatorStyle.CHATML,
+        sep='<|im_end|>',
+        stop_token_ids=[50278, 0],
+    )
+)
+register_conv_template(
+    Conversation(
+        name='Hermes-2',
+        system_template='<|im_start|>system\n{system_message}',
+        system_message='你是由上海人工智能实验室联合商汤科技开发的书生多模态大模型，英文名叫InternVL, 是一个有用无害的人工智能助手。',
+        roles=('<|im_start|>user\n', '<|im_start|>assistant\n'),
+        sep_style=SeparatorStyle.MPT,
+        sep='<|im_end|>',
+        stop_token_ids=[
+            2,
+            6,
+            7,
+            8,
+        ],  # "<|endoftext|>", "<|im_start|>", "<|im_end|>", "<|im_sep|>"
+        stop_str='<|endoftext|>',
+    )
+)
+register_conv_template(
+    Conversation(
+        name='internlm2-chat',
+        system_template='<|im_start|>system\n{system_message}',
+        system_message='你是由上海人工智能实验室联合商汤科技开发的书生多模态大模型，英文名叫InternVL, 是一个有用无害的人工智能助手。',
+        roles=('<|im_start|>user\n', '<|im_start|>assistant\n'),
+        sep_style=SeparatorStyle.MPT,
+        sep='<|im_end|>',
+        stop_token_ids=[
+            2,
+            1163,
+            92543,
+            92542,
+        ]
+    )
+)
+register_conv_template(
+    Conversation(
+        name='internlm2-base',
+        system_template='',
+        system_message='',
+        roles=('', ''),
+        sep_style=SeparatorStyle.BASE,
+        sep='<|im_end|>',
+        stop_token_ids=[
+            2,
+            1163,
+            92543,
+            92542
+        ]
+    )
+)
+register_conv_template(
+    Conversation(
+        name='internlm2-basev0',
+        system_template='<|im_start|>system\n{system_message}',
+        system_message='你是由上海人工智能实验室联合商汤科技开发的书生多模态大模型，英文名叫InternVL, 是一个有用无害的人工智能助手。',
+        roles=('<|im_start|>user\n', '<|im_start|>assistant\n'),
+        sep_style=SeparatorStyle.MPT,
+        sep='[UNUSED_TOKEN_1]', # 从这个token开始后面那群embedding完全一样
+        stop_token_ids=[
+            2,
+            1163,
+            92543,
+            92542,
+            92398, # tokenizer.convert_tokens_to_ids('[UNUSED_TOKEN_1]')
+        ]
+    )
+)
+register_conv_template(
+    Conversation(
+        name='phi3-chat',
+        system_template='<|system|>\n{system_message}',
+        system_message='你是由上海人工智能实验室联合商汤科技开发的书生多模态大模型，英文名叫InternVL, 是一个有用无害的人工智能助手。',
+        roles=('<|user|>\n', '<|assistant|>\n'),
+        sep_style=SeparatorStyle.MPT,
+        sep='<|end|>',
+        stop_token_ids=[
+            2,
+            32000,
+            32007
+        ]
+    )
+)
+# Lemur-70b-chat default template
+# reference: https://huggingface.co/OpenLemur/lemur-70b-chat-v1#generation
+register_conv_template(
+    Conversation(
+        name='lemur-70b-chat',
+        system_template="""<|im_start|>system
+{system_message}""",
+        system_message="""You are a helpful, respectful, and honest assistant.""",
+        roles=('<|im_start|>user', '<|im_start|>assistant'),
+        sep_style=SeparatorStyle.CHATML,
+        sep='<|im_end|>',
+        stop_token_ids=[32002, 0],
+    )
+)
+# MPT-30b-instruct default template
+# reference: https://huggingface.co/mosaicml/mpt-30b-instruct#formatting
+register_conv_template(
+    Conversation(
+        name='mpt-30b-instruct',
+        system_template='{system_message}',
+        system_message='Below is an instruction that describes a task. Write a response that appropriately completes the request.',
+        roles=('### Instruction', '### Response'),
+        sep_style=SeparatorStyle.ADD_NEW_LINE_SINGLE,
+        sep='\n\n',
+        stop_token_ids=[50278, 0],
+    )
+)
+# Bard default template
+# Reference: https://github.com/google/generative-ai-python/blob/9c99bcb474a991a97a2e7d62fcdb52db7ce40729/google/generativeai/discuss.py#L150
+#            https://github.com/google/generative-ai-python/blob/9c99bcb474a991a97a2e7d62fcdb52db7ce40729/google/generativeai/discuss.py#L40
+register_conv_template(
+    Conversation(
+        name='bard',
+        roles=('0', '1'),
+        sep_style=None,
+        sep=None,
+    )
+)
+# BiLLa default template
+register_conv_template(
+    Conversation(
+        name='billa',
+        roles=('Human', 'Assistant'),
+        sep_style=SeparatorStyle.ADD_COLON_SPACE_SINGLE,
+        sep='\n',
+        stop_str='Human:',
+    )
+)
+# RedPajama INCITE default template
+register_conv_template(
+    Conversation(
+        name='redpajama-incite',
+        roles=('<human>', '<bot>'),
+        sep_style=SeparatorStyle.ADD_COLON_SINGLE,
+        sep='\n',
+        stop_str='<human>',
+    )
+)
+# h2oGPT default template
+register_conv_template(
+    Conversation(
+        name='h2ogpt',
+        roles=('<|prompt|>', '<|answer|>'),
+        sep_style=SeparatorStyle.NO_COLON_SINGLE,
+        sep='</s>',
+    )
+)
+# Robin default template
+register_conv_template(
+    Conversation(
+        name='Robin',
+        system_message="A chat between a curious human and an artificial intelligence assistant. The assistant gives helpful, detailed, and polite answers to the human's questions.",
+        roles=('###Human', '###Assistant'),
+        sep_style=SeparatorStyle.ROBIN,
+        sep='\n',
+        stop_token_ids=[2, 396],
+        stop_str='###',
+    )
+)
+# Snoozy default template
+# Reference: https://github.com/nomic-ai/gpt4all/blob/d4861030b778da6db59d21d2927a4aba4f9f1f43/gpt4all-bindings/python/gpt4all/gpt4all.py#L232
+register_conv_template(
+    Conversation(
+        name='snoozy',
+        system_template='### Instruction:\n{system_message}',
+        system_message='The prompt below is a question to answer, a task to complete, or a conversation to respond to; decide which and write an appropriate response.',
+        roles=('### Prompt', '### Response'),
+        sep_style=SeparatorStyle.ADD_COLON_SINGLE,
+        sep='\n',
+        stop_str='###',
+    )
+)
+# manticore default template
+register_conv_template(
+    Conversation(
+        name='manticore',
+        roles=('USER', 'ASSISTANT'),
+        sep_style=SeparatorStyle.ADD_COLON_TWO,
+        sep='\n',
+        sep2='</s>',
+    )
+)
+# Falcon default template
+register_conv_template(
+    Conversation(
+        name='falcon',
+        roles=('User', 'Assistant'),
+        messages=[],
+        sep_style=SeparatorStyle.RWKV,
+        sep='\n',
+        sep2='<|endoftext|>',
+        stop_str='\nUser',  # use stop_str to stop generation after stop_token_ids, it will also remove stop_str from the generated text
+        stop_token_ids=[
+            0,
+            1,
+            2,
+            3,
+            4,
+            5,
+            6,
+            7,
+            8,
+            9,
+            10,
+            11,
+        ],  # it better only put special tokens here, because tokenizer only remove special tokens
+    )
+)
+# ChangGPT default template
+register_conv_template(
+    Conversation(
+        name='polyglot_changgpt',
+        roles=('B', 'A'),
+        sep_style=SeparatorStyle.ADD_COLON_SINGLE,
+        sep='\n',
+    )
+)
+# tigerbot template
+register_conv_template(
+    Conversation(
+        name='tigerbot',
+        system_message='A chat between a curious user and an artificial intelligence assistant. '
+        "The assistant gives helpful, detailed, and polite answers to the user's questions.",
+        roles=('### Instruction', '### Response'),
+        sep_style=SeparatorStyle.ROBIN,
+        sep='\n\n',
+        stop_str='###',
+    )
+)
+# ref: https://huggingface.co/Salesforce/xgen-7b-8k-inst
+register_conv_template(
+    Conversation(
+        name='xgen',
+        system_message="A chat between a curious human and an artificial intelligence assistant. The assistant gives helpful, detailed, and polite answers to the human's questions.\n\n",
+        roles=('### Human', '### Assistant'),
+        sep_style=SeparatorStyle.ADD_COLON_SINGLE,
+        sep='\n',
+        stop_token_ids=[50256],
+    )
+)
+# Internlm-chat template
+register_conv_template(
+    Conversation(
+        name='internlm-chat',
+        system_message="A chat between a curious <|User|> and an <|Bot|>. The <|Bot|> gives helpful, detailed, and polite answers to the <|User|>'s questions.\n\n",
+        roles=('<|User|>', '<|Bot|>'),
+        sep_style=SeparatorStyle.CHATINTERN,
+        sep='<eoh>',
+        sep2='<eoa>',
+        stop_token_ids=[1, 103028],
+        stop_str='<|User|>',
+    )
+)
+# StarChat template
+# reference: https://huggingface.co/spaces/HuggingFaceH4/starchat-playground/blob/main/dialogues.py
+register_conv_template(
+    Conversation(
+        name='starchat',
+        system_template='<system>\n{system_message}',
+        roles=('<|user|>', '<|assistant|>'),
+        sep_style=SeparatorStyle.CHATML,
+        sep='<|end|>',
+        stop_token_ids=[0, 49155],
+        stop_str='<|end|>',
+    )
+)
+# Baichuan-13B-Chat template
+register_conv_template(
+    # source: https://huggingface.co/baichuan-inc/Baichuan-13B-Chat/blob/19ef51ba5bad8935b03acd20ff04a269210983bc/modeling_baichuan.py#L555
+    # https://huggingface.co/baichuan-inc/Baichuan-13B-Chat/blob/main/generation_config.json
+    # https://github.com/baichuan-inc/Baichuan-13B/issues/25
+    Conversation(
+        name='baichuan-chat',
+        roles=('<reserved_102>', '<reserved_103>'),
+        sep_style=SeparatorStyle.NO_COLON_SINGLE,
+        sep='',
+        stop_token_ids=[],
+    )
+)
+# Baichuan2-13B-Chat template
+register_conv_template(
+    # source: https://huggingface.co/baichuan-inc/Baichuan2-13B-Chat/blob/c6f8592a60b4ad73c210b28dd2ab3cca51abbf93/modeling_baichuan.py#L773
+    # https://huggingface.co/baichuan-inc/Baichuan2-13B-Chat/blob/main/generation_config.json
+    # https://github.com/baichuan-inc/Baichuan2/issues/62
+    Conversation(
+        name='baichuan2-chat',
+        roles=('<reserved_106>', '<reserved_107>'),
+        sep_style=SeparatorStyle.NO_COLON_SINGLE,
+        sep='',
+        stop_token_ids=[],
+    )
+)
+# Mistral template
+# source: https://docs.mistral.ai/llm/mistral-instruct-v0.1#chat-template
+register_conv_template(
+    Conversation(
+        name='mistral',
+        system_template='[INST]{system_message}\n',
+        roles=('[INST]', '[/INST]'),
+        sep_style=SeparatorStyle.LLAMA2,
+        sep=' ',
+        sep2='</s>',
+    )
+)
+# llama2 template
+# reference: https://huggingface.co/blog/codellama#conversational-instructions
+# reference: https://github.com/facebookresearch/llama/blob/1a240688810f8036049e8da36b073f63d2ac552c/llama/generation.py#L212
+register_conv_template(
+    Conversation(
+        name='llama-2',
+        system_template='[INST] <<SYS>>\n{system_message}\n<</SYS>>\n\n',
+        roles=('[INST]', '[/INST]'),
+        sep_style=SeparatorStyle.LLAMA2,
+        sep=' ',
+        sep2=' </s><s>',
+    )
+)
+register_conv_template(
+    Conversation(
+        name='cutegpt',
+        roles=('问：', '答：\n'),
+        sep_style=SeparatorStyle.NO_COLON_TWO,
+        sep='\n',
+        sep2='\n',
+        stop_str='<end>',
+    )
+)
+# OpenOrcaxOpenChat-naPreview2-13B template
+register_conv_template(
+    Conversation(
+        name='open-orca',
+        system_template='{system_message}',
+        system_message='You are a helpful assistant. Please answer truthfully and write out your '
+        'thinking step by step to be sure you get the right answer. If you make a mistake or encounter '
+        "an error in your thinking, say so out loud and attempt to correct it. If you don't know or "
+        "aren't sure about something, say so clearly. You will act as a professional logician, mathematician, "
+        'and physicist. You will also act as the most appropriate type of expert to answer any particular '
+        'question or solve the relevant problem; state which expert type your are, if so. Also think of '
+        'any particular named expert that would be ideal to answer the relevant question or solve the '
+        'relevant problem; name and act as them, if appropriate.',
+        roles=('User', 'Assistant'),
+        sep_style=SeparatorStyle.ADD_COLON_SPACE_SINGLE,
+        sep='<|end_of_turn|>\n',
+        stop_token_ids=[32000, 32001],  # "<|end_of_turn|>"
+        stop_str='User',
+    )
+)
+# Open-Orca/Mistral-7B-OpenOrca template
+# source: https://huggingface.co/Open-Orca/Mistral-7B-OpenOrca
+# reference: https://huggingface.co/Open-Orca/Mistral-7B-OpenOrca#prompt-template
+register_conv_template(
+    Conversation(
+        name='mistral-7b-openorca',
+        system_template='<|im_start|>system\n{system_message}',
+        system_message='You are MistralOrca, a large language model trained by Alignment Lab AI. Write out your reasoning step-by-step to be sure you get the right answers!',
+        roles=('<|im_start|>user', '<|im_start|>assistant'),
+        sep_style=SeparatorStyle.CHATML,
+        sep='<|im_end|>',
+        stop_token_ids=[32000, 32001],
+    )
+)
+# Qwen-chat default template
+# source: https://huggingface.co/Qwen/Qwen-7B-Chat/blob/main/qwen_generation_utils.py#L130
+register_conv_template(
+    Conversation(
+        name='qwen-7b-chat',
+        system_template='<|im_start|>system\n{system_message}',
+        system_message='You are a helpful assistant.',
+        roles=('<|im_start|>user', '<|im_start|>assistant'),
+        sep_style=SeparatorStyle.CHATML,
+        sep='<|im_end|>',
+        stop_token_ids=[
+            151643,
+            151644,
+            151645,
+        ],  # "<|endoftext|>", "<|im_start|>", "<|im_end|>"
+        stop_str='<|endoftext|>',
+    )
+)
+# AquilaChat default template
+# source: https://github.com/FlagAI-Open/FlagAI/blob/master/examples/Aquila/Aquila-chat/cyg_conversation.py
+register_conv_template(
+    Conversation(
+        name='aquila-chat',
+        system_message='A chat between a curious human and an artificial intelligence assistant. '
+        "The assistant gives helpful, detailed, and polite answers to the human's questions.",
+        roles=('Human', 'Assistant'),
+        sep_style=SeparatorStyle.ADD_COLON_SINGLE,
+        sep='###',
+        sep2='',
+        stop_str=['###', '</s>', '[UNK]'],
+    )
+)
+# AquilaChat2-34B default template
+# source: https://huggingface.co/BAAI/AquilaChat2-34B/blob/4608b75855334b93329a771aee03869dbf7d88cc/predict.py#L212
+register_conv_template(
+    Conversation(
+        name='aquila-legacy',
+        system_message='A chat between a curious human and an artificial intelligence assistant. '
+        "The assistant gives helpful, detailed, and polite answers to the human's questions.\n\n",
+        roles=('### Human: ', '### Assistant: '),
+        offset=0,
+        sep_style=SeparatorStyle.NO_COLON_TWO,
+        sep='\n',
+        sep2='</s>',
+        stop_str=['</s>', '[UNK]'],
+    )
+)
+# AquilaChat2-7B-16K and AquilaChat2-34B-16K default template
+# source: https://huggingface.co/BAAI/AquilaChat2-34B/blob/4608b75855334b93329a771aee03869dbf7d88cc/predict.py#L227
+register_conv_template(
+    Conversation(
+        name='aquila',
+        system_message='A chat between a curious human and an artificial intelligence assistant. '
+        "The assistant gives helpful, detailed, and polite answers to the human's questions.",
+        roles=('Human', 'Assistant'),
+        offset=0,
+        sep_style=SeparatorStyle.ADD_COLON_TWO,
+        sep='###',
+        sep2='</s>',
+        stop_str=['</s>', '[UNK]'],
+    )
+)
+# AquilaChat2-7B default template
+# source: https://huggingface.co/BAAI/AquilaChat2-34B/blob/4608b75855334b93329a771aee03869dbf7d88cc/predict.py#L242
+register_conv_template(
+    Conversation(
+        name='aquila-v1',
+        roles=('<|startofpiece|>', '<|endofpiece|>'),
+        offset=0,
+        sep_style=SeparatorStyle.NO_COLON_TWO,
+        sep='',
+        sep2='</s>',
+        stop_str=['</s>', '<|endoftext|>'],
+    )
+)
+# Llama2-Chinese default template
+# source: https://huggingface.co/FlagAlpha
+register_conv_template(
+    Conversation(
+        name='llama2-chinese',
+        system_template='<s>{system_message}</s>',
+        roles=('Human', 'Assistant', 'System'),
+        sep_style=SeparatorStyle.ADD_COLON_TWO,
+        sep='\n',
+        sep2='\n</s><s>',
+        stop_str='</s>',
+    )
+)
+# Vigogne Instruct default template
+# source: https://github.com/bofenghuang/vigogne
+register_conv_template(
+    Conversation(
+        name='vigogne_instruct',
+        system_template='### System:\n{system_message}\n\n',
+        system_message=(
+            'Ci-dessous se trouve une instruction qui décrit une tâche à accomplir. Rédigez une réponse qui répond de manière'
+            ' précise à la demande.'
+        ),
+        roles=('### Instruction', '### Response'),
+        sep_style=SeparatorStyle.DOLLY,
+        sep='\n\n',
+        sep2='</s>',
+    )
+)
+# Vigogne Chat default template
+register_conv_template(
+    Conversation(
+        name='vigogne_chat_v2',
+        system_template='<|system|>: {system_message}',
+        system_message=(
+            'Vous êtes Vigogne, un assistant IA créé par Zaion Lab. Vous suivez extrêmement bien les instructions. Aidez'
+            ' autant que vous le pouvez.'
+        ),
+        roles=('<|user|>', '<|assistant|>'),
+        sep_style=SeparatorStyle.ADD_COLON_TWO,
+        sep='\n',
+        sep2='</s>\n',
+        stop_str='<|user|>',
+    )
+)
+register_conv_template(
+    Conversation(
+        name='vigogne_chat_v3',
+        system_template='[INST] <<SYS>>\n{system_message}\n<</SYS>>\n\n',
+        system_message=(
+            'Vous êtes Vigogne, un assistant IA créé par Zaion Lab. Vous suivez extrêmement bien les instructions. Aidez'
+            ' autant que vous le pouvez.'
+        ),
+        roles=('[INST]', '[/INST]'),
+        sep_style=SeparatorStyle.LLAMA2,
+        sep=' ',
+        sep2=' </s>',
+    )
+)
+# Falcon 180B chat template
+# source: https://huggingface.co/spaces/tiiuae/falcon-180b-demo/blob/d1590ee7fae9b6ce331ba7808e61a29dcce9239f/app.py#L28-L37
+register_conv_template(
+    Conversation(
+        name='falcon-chat',
+        roles=('User', 'Falcon'),
+        system_template='System: {system_message}',
+        messages=[],
+        sep_style=SeparatorStyle.FALCON_CHAT,
+        sep='\n',
+        sep2='<|endoftext|>',
+        stop_str='\nUser:',  # use stop_str to stop generation after stop_token_ids, it will also remove stop_str from the generated text
+    )
+)
+# Phind template
+# source: https://huggingface.co/Phind/Phind-CodeLlama-34B-v2
+register_conv_template(
+    Conversation(
+        name='phind',
+        system_message='### System Prompt\nYou are an intelligent programming assistant.',
+        roles=('### User Message', '### Assistant'),
+        messages=(),
+        offset=0,
+        sep_style=SeparatorStyle.ADD_COLON_SINGLE,
+        sep='\n\n',
+    )
+)
+# Metharme formatting for Pygmalion models
+# source: https://huggingface.co/PygmalionAI/pygmalion-2-13b
+register_conv_template(
+    Conversation(
+        name='metharme',
+        system_template='<|system|>{system_message}',
+        system_message="""Enter RP mode. You shall reply to the user while staying
+        in character. Your responses must be detailed, creative, immersive, and drive the scenario
+        forward.""",
+        roles=('<|user|>', '<|model|>'),
+        sep_style=SeparatorStyle.NO_COLON_SINGLE,
+        sep='',
+        stop_str='<|user|>',
+    )
+)
+# Zephyr template
+# reference: https://huggingface.co/spaces/HuggingFaceH4/zephyr-playground/blob/main/dialogues.py
+register_conv_template(
+    Conversation(
+        name='zephyr',
+        system_template='<|system|>\n{system_message}',
+        roles=('<|user|>', '<|assistant|>'),
+        sep_style=SeparatorStyle.CHATML,
+        sep='</s>',
+        stop_token_ids=[2],
+        stop_str='</s>',
+    )
+)
+# InternVL-ZH template
+register_conv_template(
+    Conversation(
+        name='internvl_zh',
+        system_template='',
+        roles=('<human>', '<bot>'),
+        sep_style=SeparatorStyle.INTERNVL_ZH,
+        sep=' ',
+        sep2='</s>',
+    )
+)
+if __name__ == '__main__':
+    from fastchat.conversation import get_conv_template
+    print('-- Vicuna template --')
+    conv = get_conv_template('vicuna_v1.1')
+    conv.append_message(conv.roles[0], 'Hello!')
+    conv.append_message(conv.roles[1], 'Hi!')
+    conv.append_message(conv.roles[0], 'How are you?')
+    conv.append_message(conv.roles[1], None)
+    print(conv.get_prompt())
+    print('\n')
+    print('-- Llama-2 template --')
+    conv = get_conv_template('llama-2')
+    conv.set_system_message('You are a helpful, respectful and honest assistant.')
+    conv.append_message(conv.roles[0], 'Hello!')
+    conv.append_message(conv.roles[1], 'Hi!')
+    conv.append_message(conv.roles[0], 'How are you?')
+    conv.append_message(conv.roles[1], None)
+    print(conv.get_prompt())
+    print('\n')
+    print('-- ChatGPT template --')
+    conv = get_conv_template('chatgpt')
+    conv.append_message(conv.roles[0], 'Hello!')
+    conv.append_message(conv.roles[1], 'Hi!')
+    conv.append_message(conv.roles[0], 'How are you?')
+    conv.append_message(conv.roles[1], None)
+    print(conv.to_openai_api_messages())
+    print('\n')
+    print('-- Claude template --')
+    conv = get_conv_template('claude')
+    conv.append_message(conv.roles[0], 'Hello!')
+    conv.append_message(conv.roles[1], 'Hi!')
+    conv.append_message(conv.roles[0], 'How are you?')
+    conv.append_message(conv.roles[1], None)
+    print(conv.get_prompt())

V2PE-256K/generation_config.json ADDED Viewed

	@@ -0,0 +1,4 @@

+{
+  "_from_model_config": true,
+  "transformers_version": "4.44.0"
+}

V2PE-256K/model.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:408cec2a2492bbd0d0e34fc58e89e4b866e9ccd04238555d09b2ce681562e73c
+size 4411571040

V2PE-256K/modeling_intern_vit.py ADDED Viewed

	@@ -0,0 +1,362 @@

+# --------------------------------------------------------
+# InternVL
+# Copyright (c) 2023 OpenGVLab
+# Licensed under The MIT License [see LICENSE for details]
+# --------------------------------------------------------
+from typing import Optional, Tuple, Union
+import torch
+import torch.nn.functional as F
+import torch.utils.checkpoint
+from einops import rearrange
+from timm.models.layers import DropPath
+from torch import nn
+from transformers.activations import ACT2FN
+from transformers.modeling_outputs import (BaseModelOutput,
+                                           BaseModelOutputWithPooling)
+from transformers.modeling_utils import PreTrainedModel
+from transformers.utils import logging
+from .configuration_intern_vit import InternVisionConfig
+try:
+    from .flash_attention import FlashAttention
+    has_flash_attn = True
+except:
+    print('FlashAttention is not installed.')
+    has_flash_attn = False
+logger = logging.get_logger(__name__)
+class InternRMSNorm(nn.Module):
+    def __init__(self, hidden_size, eps=1e-6):
+        super().__init__()
+        self.weight = nn.Parameter(torch.ones(hidden_size))
+        self.variance_epsilon = eps
+    def forward(self, hidden_states):
+        input_dtype = hidden_states.dtype
+        hidden_states = hidden_states.to(torch.float32)
+        variance = hidden_states.pow(2).mean(-1, keepdim=True)
+        hidden_states = hidden_states * torch.rsqrt(variance + self.variance_epsilon)
+        return self.weight * hidden_states.to(input_dtype)
+try:
+    from apex.normalization import FusedRMSNorm
+    InternRMSNorm = FusedRMSNorm  # noqa
+    logger.info('Discovered apex.normalization.FusedRMSNorm - will use it instead of InternRMSNorm')
+except ImportError:
+    # using the normal InternRMSNorm
+    pass
+except Exception:
+    logger.warning('discovered apex but it failed to load, falling back to InternRMSNorm')
+    pass
+NORM2FN = {
+    'rms_norm': InternRMSNorm,
+    'layer_norm': nn.LayerNorm,
+}
+class InternVisionEmbeddings(nn.Module):
+    def __init__(self, config: InternVisionConfig):
+        super().__init__()
+        self.config = config
+        self.embed_dim = config.hidden_size
+        self.image_size = config.image_size
+        self.patch_size = config.patch_size
+        self.class_embedding = nn.Parameter(
+            torch.randn(1, 1, self.embed_dim),
+        )
+        self.patch_embedding = nn.Conv2d(
+            in_channels=3, out_channels=self.embed_dim, kernel_size=self.patch_size, stride=self.patch_size
+        )
+        self.num_patches = (self.image_size // self.patch_size) ** 2
+        self.num_positions = self.num_patches + 1
+        self.position_embedding = nn.Parameter(torch.randn(1, self.num_positions, self.embed_dim))
+    def _get_pos_embed(self, pos_embed, H, W):
+        target_dtype = pos_embed.dtype
+        pos_embed = pos_embed.float().reshape(
+            1, self.image_size // self.patch_size, self.image_size // self.patch_size, -1).permute(0, 3, 1, 2)
+        pos_embed = F.interpolate(pos_embed, size=(H, W), mode='bicubic', align_corners=False). \
+            reshape(1, -1, H * W).permute(0, 2, 1).to(target_dtype)
+        return pos_embed
+    def forward(self, pixel_values: torch.FloatTensor) -> torch.Tensor:
+        target_dtype = self.patch_embedding.weight.dtype
+        patch_embeds = self.patch_embedding(pixel_values)  # shape = [*, channel, width, height]
+        batch_size, _, height, width = patch_embeds.shape
+        patch_embeds = patch_embeds.flatten(2).transpose(1, 2)
+        class_embeds = self.class_embedding.expand(batch_size, 1, -1).to(target_dtype)
+        embeddings = torch.cat([class_embeds, patch_embeds], dim=1)
+        position_embedding = torch.cat([
+            self.position_embedding[:, :1, :],
+            self._get_pos_embed(self.position_embedding[:, 1:, :], height, width)
+        ], dim=1)
+        embeddings = embeddings + position_embedding.to(target_dtype)
+        return embeddings
+class InternAttention(nn.Module):
+    """Multi-headed attention from 'Attention Is All You Need' paper"""
+    def __init__(self, config: InternVisionConfig):
+        super().__init__()
+        self.config = config
+        self.embed_dim = config.hidden_size
+        self.num_heads = config.num_attention_heads
+        self.use_flash_attn = config.use_flash_attn and has_flash_attn
+        if config.use_flash_attn and not has_flash_attn:
+            print('Warning: Flash Attention is not available, use_flash_attn is set to False.')
+        self.head_dim = self.embed_dim // self.num_heads
+        if self.head_dim * self.num_heads != self.embed_dim:
+            raise ValueError(
+                f'embed_dim must be divisible by num_heads (got `embed_dim`: {self.embed_dim} and `num_heads`:'
+                f' {self.num_heads}).'
+            )
+        self.scale = self.head_dim ** -0.5
+        self.qkv = nn.Linear(self.embed_dim, 3 * self.embed_dim, bias=config.qkv_bias)
+        self.attn_drop = nn.Dropout(config.attention_dropout)
+        self.proj_drop = nn.Dropout(config.dropout)
+        self.qk_normalization = config.qk_normalization
+        if self.qk_normalization:
+            self.q_norm = InternRMSNorm(self.embed_dim, eps=config.layer_norm_eps)
+            self.k_norm = InternRMSNorm(self.embed_dim, eps=config.layer_norm_eps)
+        if self.use_flash_attn:
+            self.inner_attn = FlashAttention(attention_dropout=config.attention_dropout)
+        self.proj = nn.Linear(self.embed_dim, self.embed_dim)
+    def _naive_attn(self, x):
+        B, N, C = x.shape
+        qkv = self.qkv(x).reshape(B, N, 3, self.num_heads, C // self.num_heads).permute(2, 0, 3, 1, 4)
+        q, k, v = qkv.unbind(0)  # make torchscript happy (cannot use tensor as tuple)
+        if self.qk_normalization:
+            B_, H_, N_, D_ = q.shape
+            q = self.q_norm(q.transpose(1, 2).flatten(-2, -1)).view(B_, N_, H_, D_).transpose(1, 2)
+            k = self.k_norm(k.transpose(1, 2).flatten(-2, -1)).view(B_, N_, H_, D_).transpose(1, 2)
+        attn = ((q * self.scale) @ k.transpose(-2, -1))
+        attn = attn.softmax(dim=-1)
+        attn = self.attn_drop(attn)
+        x = (attn @ v).transpose(1, 2).reshape(B, N, C)
+        x = self.proj(x)
+        x = self.proj_drop(x)
+        return x
+    def _flash_attn(self, x, key_padding_mask=None, need_weights=False):
+        qkv = self.qkv(x)
+        qkv = rearrange(qkv, 'b s (three h d) -> b s three h d', three=3, h=self.num_heads)
+        if self.qk_normalization:
+            q, k, v = qkv.unbind(2)
+            q = self.q_norm(q.flatten(-2, -1)).view(q.shape)
+            k = self.k_norm(k.flatten(-2, -1)).view(k.shape)
+            qkv = torch.stack([q, k, v], dim=2)
+        context, _ = self.inner_attn(
+            qkv, key_padding_mask=key_padding_mask, need_weights=need_weights, causal=False
+        )
+        outs = self.proj(rearrange(context, 'b s h d -> b s (h d)'))
+        outs = self.proj_drop(outs)
+        return outs
+    def forward(self, hidden_states: torch.Tensor) -> torch.Tensor:
+        x = self._naive_attn(hidden_states) if not self.use_flash_attn else self._flash_attn(hidden_states)
+        return x
+class InternMLP(nn.Module):
+    def __init__(self, config: InternVisionConfig):
+        super().__init__()
+        self.config = config
+        self.act = ACT2FN[config.hidden_act]
+        self.fc1 = nn.Linear(config.hidden_size, config.intermediate_size)
+        self.fc2 = nn.Linear(config.intermediate_size, config.hidden_size)
+    def forward(self, hidden_states: torch.Tensor) -> torch.Tensor:
+        hidden_states = self.fc1(hidden_states)
+        hidden_states = self.act(hidden_states)
+        hidden_states = self.fc2(hidden_states)
+        return hidden_states
+class InternVisionEncoderLayer(nn.Module):
+    def __init__(self, config: InternVisionConfig, drop_path_rate: float):
+        super().__init__()
+        self.embed_dim = config.hidden_size
+        self.intermediate_size = config.intermediate_size
+        self.norm_type = config.norm_type
+        self.attn = InternAttention(config)
+        self.mlp = InternMLP(config)
+        self.norm1 = NORM2FN[self.norm_type](self.embed_dim, eps=config.layer_norm_eps)
+        self.norm2 = NORM2FN[self.norm_type](self.embed_dim, eps=config.layer_norm_eps)
+        self.ls1 = nn.Parameter(config.initializer_factor * torch.ones(self.embed_dim))
+        self.ls2 = nn.Parameter(config.initializer_factor * torch.ones(self.embed_dim))
+        self.drop_path1 = DropPath(drop_path_rate) if drop_path_rate > 0. else nn.Identity()
+        self.drop_path2 = DropPath(drop_path_rate) if drop_path_rate > 0. else nn.Identity()
+    def forward(
+            self,
+            hidden_states: torch.Tensor,
+    ) -> Tuple[torch.FloatTensor, Optional[torch.FloatTensor], Optional[Tuple[torch.FloatTensor]]]:
+        """
+        Args:
+            hidden_states (`Tuple[torch.FloatTensor, Optional[torch.FloatTensor]]`): input to the layer of shape `(batch, seq_len, embed_dim)`
+        """
+        hidden_states = hidden_states + self.drop_path1(self.attn(self.norm1(hidden_states)) * self.ls1)
+        hidden_states = hidden_states + self.drop_path2(self.mlp(self.norm2(hidden_states)) * self.ls2)
+        return hidden_states
+class InternVisionEncoder(nn.Module):
+    """
+    Transformer encoder consisting of `config.num_hidden_layers` self attention layers. Each layer is a
+    [`InternEncoderLayer`].
+    Args:
+        config (`InternConfig`):
+            The corresponding vision configuration for the `InternEncoder`.
+    """
+    def __init__(self, config: InternVisionConfig):
+        super().__init__()
+        self.config = config
+        # stochastic depth decay rule
+        dpr = [x.item() for x in torch.linspace(0, config.drop_path_rate, config.num_hidden_layers)]
+        self.layers = nn.ModuleList([
+            InternVisionEncoderLayer(config, dpr[idx]) for idx in range(config.num_hidden_layers)])
+        self.gradient_checkpointing = True
+    def forward(
+            self,
+            inputs_embeds,
+            output_hidden_states: Optional[bool] = None,
+            return_dict: Optional[bool] = None,
+    ) -> Union[Tuple, BaseModelOutput]:
+        r"""
+        Args:
+            inputs_embeds (`torch.FloatTensor` of shape `(batch_size, sequence_length, hidden_size)`):
+                Embedded representation of the inputs. Should be float, not int tokens.
+            output_hidden_states (`bool`, *optional*):
+                Whether or not to return the hidden states of all layers. See `hidden_states` under returned tensors
+                for more detail.
+            return_dict (`bool`, *optional*):
+                Whether or not to return a [`~utils.ModelOutput`] instead of a plain tuple.
+        """
+        output_hidden_states = (
+            output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
+        )
+        return_dict = return_dict if return_dict is not None else self.config.use_return_dict
+        encoder_states = () if output_hidden_states else None
+        hidden_states = inputs_embeds
+        for idx, encoder_layer in enumerate(self.layers):
+            if output_hidden_states:
+                encoder_states = encoder_states + (hidden_states,)
+            if self.gradient_checkpointing and self.training:
+                layer_outputs = torch.utils.checkpoint.checkpoint(
+                    encoder_layer,
+                    hidden_states)
+            else:
+                layer_outputs = encoder_layer(
+                    hidden_states,
+                )
+            hidden_states = layer_outputs
+        if output_hidden_states:
+            encoder_states = encoder_states + (hidden_states,)
+        if not return_dict:
+            return tuple(v for v in [hidden_states, encoder_states] if v is not None)
+        return BaseModelOutput(
+            last_hidden_state=hidden_states, hidden_states=encoder_states
+        )
+class InternVisionModel(PreTrainedModel):
+    main_input_name = 'pixel_values'
+    config_class = InternVisionConfig
+    _no_split_modules = ['InternVisionEncoderLayer']
+    def __init__(self, config: InternVisionConfig):
+        super().__init__(config)
+        self.config = config
+        self.embeddings = InternVisionEmbeddings(config)
+        self.encoder = InternVisionEncoder(config)
+    def resize_pos_embeddings(self, old_size, new_size, patch_size):
+        pos_emb = self.embeddings.position_embedding
+        _, num_positions, embed_dim = pos_emb.shape
+        cls_emb = pos_emb[:, :1, :]
+        pos_emb = pos_emb[:, 1:, :].reshape(1, old_size // patch_size, old_size // patch_size, -1).permute(0, 3, 1, 2)
+        pos_emb = F.interpolate(pos_emb.float(), size=new_size // patch_size, mode='bicubic', align_corners=False)
+        pos_emb = pos_emb.to(cls_emb.dtype).reshape(1, embed_dim, -1).permute(0, 2, 1)
+        pos_emb = torch.cat([cls_emb, pos_emb], dim=1)
+        self.embeddings.position_embedding = nn.Parameter(pos_emb)
+        self.embeddings.image_size = new_size
+        logger.info('Resized position embeddings from {} to {}'.format(old_size, new_size))
+    def get_input_embeddings(self):
+        return self.embeddings
+    def forward(
+            self,
+            pixel_values: Optional[torch.FloatTensor] = None,
+            output_hidden_states: Optional[bool] = None,
+            return_dict: Optional[bool] = None,
+            pixel_embeds: Optional[torch.FloatTensor] = None,
+    ) -> Union[Tuple, BaseModelOutputWithPooling]:
+        output_hidden_states = (
+            output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
+        )
+        return_dict = return_dict if return_dict is not None else self.config.use_return_dict
+        if pixel_values is None and pixel_embeds is None:
+            raise ValueError('You have to specify pixel_values or pixel_embeds')
+        if pixel_embeds is not None:
+            hidden_states = pixel_embeds
+        else:
+            if len(pixel_values.shape) == 4:
+                hidden_states = self.embeddings(pixel_values)
+            else:
+                raise ValueError(f'wrong pixel_values size: {pixel_values.shape}')
+        encoder_outputs = self.encoder(
+            inputs_embeds=hidden_states,
+            output_hidden_states=output_hidden_states,
+            return_dict=return_dict,
+        )
+        last_hidden_state = encoder_outputs.last_hidden_state
+        pooled_output = last_hidden_state[:, 0, :]
+        if not return_dict:
+            return (last_hidden_state, pooled_output) + encoder_outputs[1:]
+        return BaseModelOutputWithPooling(
+            last_hidden_state=last_hidden_state,
+            pooler_output=pooled_output,
+            hidden_states=encoder_outputs.hidden_states,
+            attentions=encoder_outputs.attentions,
+        )

V2PE-256K/modeling_internlm2.py ADDED Viewed

The diff for this file is too large to render. See raw diff

V2PE-256K/modeling_internvl_chat.py ADDED Viewed

	@@ -0,0 +1,1103 @@

+# --------------------------------------------------------
+# InternVL
+# Copyright (c) 2024 OpenGVLab
+# Licensed under The MIT License [see LICENSE for details]
+# --------------------------------------------------------
+import warnings
+from typing import Any, List, Optional, Tuple, Union
+import torch.distributed as dist
+import torch.utils.checkpoint
+import transformers
+from internvl.conversation import get_conv_template
+from internvl.model.internlm2.modeling_internlm2 import InternLM2ForCausalLM
+from internvl.model.phi3.modeling_phi3 import Phi3ForCausalLM
+from peft import LoraConfig, get_peft_model
+from torch import nn
+from torch.nn import CrossEntropyLoss
+from transformers import (AutoModel, GenerationConfig, LlamaForCausalLM,
+                          LlamaTokenizer, Qwen2ForCausalLM)
+from transformers.modeling_outputs import CausalLMOutputWithPast
+from transformers.modeling_utils import PreTrainedModel
+from transformers.utils import ModelOutput, logging
+from .configuration_internvl_chat import InternVLChatConfig
+from .modeling_intern_vit import InternVisionModel
+logger = logging.get_logger(__name__)
+from transformers import AutoTokenizer
+import json
+tokenizer_path="/mnt/petrelfs/share_data/chenziyi/InternVL2-2B"
+global_tokenizer = AutoTokenizer.from_pretrained(
+        tokenizer_path, add_eos_token=False, trust_remote_code=True, use_fast=False)
+import random
+def version_cmp(v1, v2, op='eq'):
+    import operator
+    from packaging import version
+    op_func = getattr(operator, op)
+    return op_func(version.parse(v1), version.parse(v2))
+def extract_local(value, rank, world_size, dim=1):
+    value_chunks = value.chunk(2 * world_size, dim=dim)
+    local_value = torch.cat(
+        [value_chunks[rank], value_chunks[2 * world_size - rank - 1]], dim=dim
+    )
+    return local_value.to(value.device)
+def extract_local2(value, rank, world_size,  dim=1):
+    dimension_size = value.shape[dim]
+    sub_seq_length = dimension_size // world_size
+    sub_seq_start = rank * sub_seq_length
+    sub_seq_end = (rank + 1) * sub_seq_length
+    local_value = value[:, sub_seq_start:sub_seq_end]
+    return local_value.to(value.device)
+class GatherLayer(torch.autograd.Function):
+    """Gather tensors from all process, supporting backward propagation."""
+    @staticmethod
+    def forward(ctx, input):
+        ctx.save_for_backward(input)
+        output = [torch.zeros_like(input) for _ in range(dist.get_world_size(local_group))]
+        dist.all_gather(output, input, group=local_group)
+        return torch.stack(output, 0)
+    @staticmethod
+    def backward(ctx, grads):
+        (input,) = ctx.saved_tensors
+        dist.all_reduce(grads, group=local_group)
+        grad_out = torch.zeros_like(input)
+        grad_out[:] = grads[dist.get_rank(local_group)]
+        return grad_out
+class InternVLChatModel(PreTrainedModel):
+    config_class = InternVLChatConfig
+    main_input_name = 'pixel_values'
+    _no_split_modules = ['InternVisionModel', 'LlamaDecoderLayer', 'InternLM2DecoderLayer',
+                         'Phi3DecoderLayer', 'Qwen2DecoderLayer']
+    def __init__(self, config: InternVLChatConfig, vision_model=None, language_model=None):
+        super().__init__(config)
+        assert version_cmp(transformers.__version__, '4.37.0', 'ge')
+        image_size = config.force_image_size or config.vision_config.image_size
+        patch_size = config.vision_config.patch_size
+        self.patch_size = patch_size
+        self.select_layer = config.select_layer
+        self.template = config.template
+        # batch_size: 批处理大小
+        # patch_size: 图片分块大小
+        # downsample_ratio: 缩放比例，将高分辨率图像转换为低分辨率图像
+        # self.num_image_token = int((image_size // patch_size) ** 2 * (config.downsample_ratio ** 2))
+        self.num_image_token = int((image_size // patch_size) ** 2 * (config.downsample_ratio ** 2))
+        self.downsample_ratio = config.downsample_ratio
+        self.ps_version = config.ps_version
+        self.compress_seq = config.compress_seq
+        self.attn_type = config.attn_type
+        self.posid_type = config.posid_type
+        if self.posid_type is None:
+            self.posid_type='default'
+        assert self.posid_type in ['default','None', 'qkvLearnable', 'qkLearnable', '1dROPE', '2dROPE']
+        self.group_list = config.group_list
+        self.chunk_num = config.chunk_num
+        self.interaction = config.interaction
+        logger.info(f'num_image_token: {self.num_image_token}')
+        logger.info(f'ps_version: {self.ps_version}')
+        config.llm_config.posid_type = self.posid_type
+        config.llm_config.rope_pos_id_version=config.rope_pos_id_version
+        if vision_model is not None:
+            self.vision_model = vision_model
+        else:
+            self.vision_model = InternVisionModel(config.vision_config)
+        if language_model is not None:
+            self.language_model = language_model
+        else:
+            if config.llm_config.architectures[0] == 'LlamaForCausalLM':
+                self.language_model = LlamaForCausalLM(config.llm_config)
+            elif config.llm_config.architectures[0] == 'InternLM2ForCausalLM':
+                self.language_model = InternLM2ForCausalLM(config.llm_config)
+            elif config.llm_config.architectures[0] == 'Phi3ForCausalLM':
+                self.language_model = Phi3ForCausalLM(config.llm_config)
+            elif config.llm_config.architectures[0] == 'Qwen2ForCausalLM':
+                self.language_model = Qwen2ForCausalLM(config.llm_config)
+            else:
+                raise NotImplementedError(f'{config.llm_config.architectures[0]} is not implemented.')
+        vit_hidden_size = config.vision_config.hidden_size
+        llm_hidden_size = config.llm_config.hidden_size
+        self.mlp1 = nn.Sequential(
+            nn.LayerNorm(vit_hidden_size * int(1 / self.downsample_ratio) ** 2),
+            nn.Linear(vit_hidden_size * int(1 / self.downsample_ratio) ** 2, llm_hidden_size),
+            nn.GELU(),
+            nn.Linear(llm_hidden_size, llm_hidden_size)
+        )
+        if self.posid_type in ['qkvLearnable']:
+            self.local_posid = nn.Embedding(self.num_image_token,llm_hidden_size)
+        self.img_context_token_id = None
+        self.conv_template = get_conv_template(self.template)
+        self.system_message = self.conv_template.system_message
+        self.num_samples = 0
+        if config.use_backbone_lora:
+            self.wrap_backbone_lora(r=config.use_backbone_lora, lora_alpha=2 * config.use_backbone_lora)
+        if config.use_llm_lora:
+            self.wrap_llm_lora(r=config.use_llm_lora, lora_alpha=2 * config.use_llm_lora)
+    def init_embed(self):
+        if hasattr(self,'local_posid'):
+            nn.init.normal_(self.local_posid.weight, mean=0.0, std=0.02)
+    def wrap_backbone_lora(self, r=128, lora_alpha=256, lora_dropout=0.05):
+        lora_config = LoraConfig(
+            r=r,
+            target_modules=['attn.qkv', 'attn.proj', 'mlp.fc1', 'mlp.fc2'],
+            lora_alpha=lora_alpha,
+            lora_dropout=lora_dropout,
+        )
+        self.vision_model = get_peft_model(self.vision_model, lora_config)
+        self.vision_model.print_trainable_parameters()
+    def wrap_llm_lora(self, r=128, lora_alpha=256, lora_dropout=0.05):
+        lora_config = LoraConfig(
+            r=r,
+            target_modules=['self_attn.q_proj', 'self_attn.k_proj', 'self_attn.v_proj', 'self_attn.o_proj',
+                            'mlp.gate_proj', 'mlp.down_proj', 'mlp.up_proj'],
+            lora_alpha=lora_alpha,
+            lora_dropout=lora_dropout,
+            task_type='CAUSAL_LM'
+        )
+        self.language_model = get_peft_model(self.language_model, lora_config)
+        self.language_model.enable_input_require_grads()
+        self.language_model.print_trainable_parameters()
+    def forward(
+            self,
+            pixel_values: torch.FloatTensor,
+            input_ids: torch.LongTensor = None,
+            attention_mask: Optional[torch.Tensor] = None,
+            position_ids: Optional[torch.Tensor] = None,
+            image_flags: Optional[torch.LongTensor] = None,
+            past_key_values: Optional[List[torch.FloatTensor]] = None,
+            labels: Optional[torch.LongTensor] = None,
+            use_cache: Optional[bool] = None,
+            output_attentions: Optional[bool] = None,
+            output_hidden_states: Optional[bool] = None,
+            return_dict: Optional[bool] = None,
+            statistics: Optional[torch.LongTensor] = None,
+            loss_weight: Optional[List] = None,
+            loss_reduction_all_gather: Optional[bool] = False,
+            origin_cu_seq_lens: Optional[torch.Tensor] = None,
+            rope_pos_id: Optional[torch.Tensor] = None,
+    ) -> Union[Tuple, CausalLMOutputWithPast]:
+        # import ipdb
+        # ipdb.set_trace()
+        if isinstance(position_ids,list):
+            position_ids=torch.tensor(position_ids).to(input_ids.device)
+        return_dict = return_dict if return_dict is not None else self.config.use_return_dict
+        # print("Printing decoded input ids")
+        # decoded_texts = [global_tokenizer.decode(ids, skip_special_tokens=True) for ids in input_ids]
+        # for i, text in enumerate(decoded_texts):
+        #     print(f"Sample {i+1}: {text}")
+        global local_group
+        if self.group_list is not None:
+            for group_idx,group in enumerate(self.group_list):
+                if type(group)==torch.distributed.distributed_c10d.ProcessGroup:
+                    # assert type(group)==torch.distributed.distributed_c10d.ProcessGroup
+                    break        # print("Printing decoded input ids")
+            local_group=group
+        else:
+            group=None
+            local_group=None
+        image_flags = image_flags.squeeze(-1)
+        input_embeds = self.language_model.get_input_embeddings()(input_ids).clone()
+        if self.attn_type:
+            if self.attn_type=='ring':
+                group_size = dist.get_world_size(group)
+                img_num_dim = 0
+                pad_num=0
+                if pixel_values.shape[img_num_dim] > group_size:
+                    if pixel_values.shape[img_num_dim] % group_size!=0:
+                        pad_num = group_size - pixel_values.shape[img_num_dim] % group_size
+                        if pad_num < group_size:  # 仅在需要填充时进行
+                            # 创建填充的张量，与 pixel_values 的形状匹配
+                            pad_shape = list(pixel_values.shape)
+                            pad_shape[img_num_dim] = pad_num  # 在目标维度上设置填充值
+                            pad_pixel = torch.zeros(pad_shape, dtype=pixel_values.dtype, device=pixel_values.device)
+                            # 在指定维度上拼接原始张量和填充张量
+                            pixel_values = torch.cat([pixel_values, pad_pixel], dim=img_num_dim)
+                    chunked_pixel=torch.chunk(pixel_values, group_size, dim=img_num_dim)
+                    local_pixel=chunked_pixel[dist.get_rank(group)]
+                    local_vit_embeds=self.extract_feature(local_pixel)
+                    vit_embeds=GatherLayer.apply(local_vit_embeds)
+                    vit_embeds=vit_embeds.view(-1,vit_embeds.shape[-2],vit_embeds.shape[-1])
+                    if pad_num>0:
+                        vit_embeds=vit_embeds[:-pad_num]
+                else:
+                    vit_embeds = self.extract_feature(pixel_values)
+            else:
+                vit_embeds = self.extract_feature(pixel_values)
+        else:
+            vit_embeds = self.extract_feature(pixel_values)
+        if self.posid_type=='qkvLearnable':
+            # added_embeds = self.local_posid(torch.arange(self.num_image_token).to(pixel_values.device))
+            # vit_embeds = vit_embeds + added_embeds
+            vit_embeds=vit_embeds+self.local_posid(torch.arange(self.num_image_token).to(pixel_values.device))
+        vit_embeds = vit_embeds[image_flags == 1]
+        vit_batch_size = pixel_values.shape[0]
+        # print("Printing pixiel shape", pixel_values.shape)
+        B, N, C = input_embeds.shape
+        input_embeds = input_embeds.reshape(B * N, C)
+        if torch.distributed.is_initialized() and torch.distributed.get_rank() == 0:
+            print(f'dynamic ViT batch size: {vit_batch_size}, images per sample: {vit_batch_size / B}, dynamic token length: {N}')
+            if statistics is not None:
+                num_samples, num_padding_tokens, num_padding_images = statistics.tolist()
+                self.num_samples += num_samples
+                print(f'total_samples={self.num_samples}, {num_samples=}, {num_padding_tokens=}, {num_padding_images=}')
+        input_ids = input_ids.reshape(B * N)
+        selected = (input_ids == self.img_context_token_id)
+        try:
+            input_embeds[selected] = input_embeds[selected] * 0.0 + vit_embeds.reshape(-1, C)
+            ignore_flag = False
+        except Exception as e:
+            vit_embeds = vit_embeds.reshape(-1, C)
+            print(f'warning: {e}, input_embeds[selected].shape={input_embeds[selected].shape}, '
+                  f'vit_embeds.shape={vit_embeds.shape}')
+            n_token = selected.sum()
+            input_embeds[selected] = input_embeds[selected] * 0.0 + vit_embeds[:n_token]
+            # ignore_flag = True
+            ignore_flag = False
+        input_embeds = input_embeds.reshape(B, N, C)
+        if self.attn_type:
+            if self.attn_type=='ulysses':
+                input_embeds=extract_local2(input_embeds,dist.get_rank(group),dist.get_world_size(group))
+                position_ids=extract_local2(position_ids,dist.get_rank(group),dist.get_world_size(group))
+                labels=extract_local2(labels,dist.get_rank(group),dist.get_world_size(group))
+                loss_weight=extract_local2(torch.tensor(loss_weight),dist.get_rank(group),dist.get_world_size(group))
+                loss_weight=list(loss_weight.numpy())
+                attention_mask=attention_mask//dist.get_world_size(group)
+            elif self.attn_type=='ring':
+                input_embeds=extract_local(input_embeds,dist.get_rank(group),dist.get_world_size(group))
+                position_ids=extract_local(position_ids,dist.get_rank(group),dist.get_world_size(group))
+                labels=extract_local(labels,dist.get_rank(group),dist.get_world_size(group))
+                if loss_weight:
+                    loss_weight=extract_local(torch.tensor(loss_weight),dist.get_rank(group),dist.get_world_size(group))
+                    loss_weight=list(loss_weight.numpy())
+                attention_mask=attention_mask//dist.get_world_size(group)
+        outputs = self.language_model(
+            inputs_embeds=input_embeds,
+            attention_mask=attention_mask,
+            position_ids=position_ids,
+            past_key_values=past_key_values,
+            use_cache=use_cache,
+            output_attentions=output_attentions,
+            output_hidden_states=output_hidden_states,
+            return_dict=return_dict,
+            compress_seq=self.compress_seq,
+            group_list=self.group_list,
+            chunk_num=self.chunk_num,
+            origin_cu_seq_lens=origin_cu_seq_lens,
+            interaction=self.interaction,
+            selected=selected
+        )
+        logits = outputs.logits
+        loss = None
+        if labels is not None and loss_weight is not None:
+            # decoded_labels = global_tokenizer.decode(labels[0][labels[0]!=-100], skip_special_tokens=True)
+            loss_weight = torch.tensor(loss_weight, dtype=torch.float32, device=labels.device)
+            # Shift so that tokens < n predict n
+            shift_logits = logits[..., :-1, :].contiguous()
+            shift_labels = labels[..., 1:].contiguous()
+            shift_weights = loss_weight[..., 1:].contiguous()
+            # Flatten the tokens
+            loss_fct = CrossEntropyLoss(reduction='none')
+            shift_logits = shift_logits.view(-1, self.language_model.config.vocab_size)
+            shift_labels = shift_labels.view(-1)
+            shift_weights = shift_weights.view(-1)
+            # Enable model parallelism
+            shift_labels = shift_labels.to(shift_logits.device)
+            shift_weights = shift_weights.to(shift_logits.device)
+            loss = loss_fct(shift_logits, shift_labels)
+            shift_weights_sum = shift_weights.sum()
+            if loss_reduction_all_gather:
+                dist.all_reduce(shift_weights_sum, op=dist.ReduceOp.AVG)
+            loss = loss * shift_weights
+            loss = loss.sum() / shift_weights_sum
+            if ignore_flag:
+                loss = loss * 0.0
+        elif labels is not None:
+            # Shift so that tokens < n predict n
+            shift_logits = logits[..., :-1, :].contiguous()
+            shift_labels = labels[..., 1:].contiguous()
+            # Flatten the tokens
+            loss_fct = CrossEntropyLoss()
+            shift_logits = shift_logits.view(-1, self.language_model.config.vocab_size)
+            shift_labels = shift_labels.view(-1)
+            # Enable model parallelism
+            shift_labels = shift_labels.to(shift_logits.device)
+            loss = loss_fct(shift_logits, shift_labels)
+            if ignore_flag:
+                loss = loss * 0.0
+        params=dict(self.named_parameters())
+        if not return_dict:
+            output = (logits,) + outputs[1:]
+            return (loss,) + output if loss is not None else output
+        # self.update_log(log_dict)
+        return CausalLMOutputWithPast(
+            loss=loss,
+            logits=logits,
+            past_key_values=outputs.past_key_values,
+            hidden_states=outputs.hidden_states,
+            attentions=outputs.attentions,
+        )
+    def pixel_shuffle(self, x, scale_factor=0.5):
+        n, w, h, c = x.size()
+        # N, W, H, C --> N, W, H * scale, C // scale
+        x = x.view(n, w, int(h * scale_factor), int(c / scale_factor))
+        # N, W, H * scale, C // scale --> N, H * scale, W, C // scale
+        x = x.permute(0, 2, 1, 3).contiguous()
+        # N, H * scale, W, C // scale --> N, H * scale, W * scale, C // (scale ** 2)
+        x = x.view(n, int(h * scale_factor), int(w * scale_factor),
+                   int(c / (scale_factor * scale_factor)))
+        if self.ps_version == 'v1':
+            warnings.warn("In ps_version 'v1', the height and width have not been swapped back, "
+                          'which results in a transposed image.')
+        else:
+            x = x.permute(0, 2, 1, 3).contiguous()
+        return x
+    def extract_feature(self, pixel_values):
+        # 选择视觉模型特定层的输出作为图片特征
+        if self.select_layer == -1:
+            vit_embeds = self.vision_model(
+                pixel_values=pixel_values,
+                output_hidden_states=False,
+                return_dict=True).last_hidden_state
+        else:
+            vit_embeds = self.vision_model(
+                pixel_values=pixel_values,
+                output_hidden_states=True,
+                return_dict=True).hidden_states[self.select_layer]
+        # [batch_size, num_patches, vit_hidden_size]
+        # 去除第一个标记
+        vit_embeds = vit_embeds[:, 1:, :]
+        # [batch_size, num_patches, vit_hidden_size] -> [batch_size, h, w, vit_hidden_size]
+        h = w = int(vit_embeds.shape[1] ** 0.5)
+        vit_embeds = vit_embeds.reshape(vit_embeds.shape[0], h, w, -1)
+        # 像素混洗，降低分辨率，减少 num_patches
+        vit_embeds = self.pixel_shuffle(vit_embeds, scale_factor=self.downsample_ratio)
+        vit_embeds = vit_embeds.reshape(vit_embeds.shape[0], -1, vit_embeds.shape[-1])
+        # 线性层，vit_hidden_size -> llm_hidden_size
+        vit_embeds = self.mlp1(vit_embeds)
+        return vit_embeds
+    def batch_chat(self, tokenizer, pixel_values, questions, generation_config, num_patches_list=None,
+                   history=None, return_history=False, IMG_START_TOKEN='<img>', IMG_END_TOKEN='</img>',
+                   IMG_CONTEXT_TOKEN='<IMG_CONTEXT>', verbose=False, image_counts=None):
+        if history is not None or return_history:
+            print('Now multi-turn chat is not supported in batch_chat.')
+            raise NotImplementedError
+        if image_counts is not None:
+            num_patches_list = image_counts
+            print('Warning: `image_counts` is deprecated. Please use `num_patches_list` instead.')
+        img_context_token_id = tokenizer.convert_tokens_to_ids(IMG_CONTEXT_TOKEN)
+        self.img_context_token_id = img_context_token_id
+        if verbose and pixel_values is not None:
+            image_bs = pixel_values.shape[0]
+            print(f'dynamic ViT batch size: {image_bs}')
+        queries = []
+        for idx, num_patches in enumerate(num_patches_list):
+            question = questions[idx]
+            if pixel_values is not None and '<image>' not in question:
+                question = '<image>\n' + question
+            template = get_conv_template(self.template)
+            template.append_message(template.roles[0], question)
+            template.append_message(template.roles[1], None)
+            query = template.get_prompt()
+            image_tokens = IMG_START_TOKEN + IMG_CONTEXT_TOKEN * self.num_image_token * num_patches + IMG_END_TOKEN
+            query = query.replace('<image>', image_tokens, 1)
+            queries.append(query)
+        # tokenizer.padding_side = 'left'
+        model_inputs = tokenizer(queries, return_tensors='pt', padding=False)
+        input_ids = model_inputs['input_ids'].cuda()
+        attention_mask = model_inputs['attention_mask'].cuda()
+        eos_token_id = tokenizer.convert_tokens_to_ids(template.sep)
+        generation_config['eos_token_id'] = eos_token_id
+        generation_output = self.generate(
+            pixel_values=pixel_values,
+            input_ids=input_ids,
+            attention_mask=attention_mask,
+            **generation_config
+        )
+        responses = tokenizer.batch_decode(generation_output, skip_special_tokens=True)
+        responses = [response.split(template.sep)[0].strip() for response in responses]
+        return responses
+    def chat(self, tokenizer, pixel_values, question, generation_config, history=None, return_history=False,
+             num_patches_list=None, IMG_START_TOKEN='<img>', IMG_END_TOKEN='</img>', IMG_CONTEXT_TOKEN='<IMG_CONTEXT>',
+             verbose=False,**kwargs):
+        if history is None and pixel_values is not None and '<image>' not in question:
+            question = '<image>\n' + question
+        # num_patches_list 用法：
+        if num_patches_list is None:
+            num_patches_list = [pixel_values.shape[0]] if pixel_values is not None else []
+        assert pixel_values is None or len(pixel_values) == sum(num_patches_list)
+        # 设置图片上下文的 token id
+        img_context_token_id = tokenizer.convert_tokens_to_ids(IMG_CONTEXT_TOKEN)
+        self.img_context_token_id = img_context_token_id
+        # 获取 Chat 模板
+        template = get_conv_template(self.template)
+        # 设置系统消息
+        template.system_message = self.system_message
+        # 设置分隔符 End Of Sentence
+        eos_token_id = tokenizer.convert_tokens_to_ids(template.sep)
+        # 将历史对话添加到模板中
+        history = [] if history is None else history
+        for (old_question, old_answer) in history:
+            template.append_message(template.roles[0], old_question)
+            template.append_message(template.roles[1], old_answer)
+        template.append_message(template.roles[0], question)
+        template.append_message(template.roles[1], None)
+        # 生成查询
+        query = template.get_prompt()
+        # verbose: 是否打印调试信息
+        if verbose and pixel_values is not None:
+            # pixel_values 形状: [batch_size, channels, height, width]
+            # 其中 batch_size 即图片数量
+            # 打印批处理大小信息
+            image_bs = pixel_values.shape[0]
+            print(f'dynamic ViT batch size: {image_bs}')
+        # 将图片 token 插入到查询中，图片用占位符 IMG_CONTEXT_TOKEN 代替
+        for num_patches in num_patches_list:
+            image_tokens = IMG_START_TOKEN + IMG_CONTEXT_TOKEN * self.num_image_token * num_patches + IMG_END_TOKEN
+            query = query.replace('<image>', image_tokens, 1)
+        # 用分词器将查询转换为模型输入
+        model_inputs = tokenizer(query, return_tensors='pt')
+        # 文本对应的 token id，转换为 cuda 张量
+        # ID 长度就是 Token 长度，形状为 [1, sequence_length]
+        input_ids = model_inputs['input_ids'].cuda()
+        # print(f'Token length: {input_ids.shape[1]}')
+        # 实际输入掩码为 1，填充部分掩码为 0
+        attention_mask = model_inputs['attention_mask'].cuda()
+        # 分隔符 End Of Sentence
+        generation_config['eos_token_id'] = eos_token_id
+        if 'rope_pos_id_version' in kwargs:
+            self.language_model.rope_pos_id_version=kwargs['rope_pos_id_version']
+            pos_ids=[]
+            ret={'input_ids':input_ids,'attention_mask':attention_mask}
+            for i in range(input_ids.shape[0]):
+                # cur_position_ids = ret['attention_mask'][i].long().cumsum(-1) - 1
+                # cur_position_ids.masked_fill_(ret['attention_mask'][i] == 0, 1)
+                if kwargs['rope_pos_id_version'] == 'default':
+                    cur_dtype = torch.long
+                    # bf16 -> long 会产生截断
+                else:
+                    cur_dtype = torch.float32
+                if 'rope_pos_id_stride' in kwargs:
+                    rope_pos_id_stride = kwargs['rope_pos_id_stride']
+                else:
+                    rope_pos_id_stride = None
+                pos_ids.append(torch.tensor(get_rope_pos_id(ret, num_tiles=kwargs['num_tiles'][i], dtype=cur_dtype,
+                                               rope_pos_id_version=kwargs['rope_pos_id_version'],
+                                               position_id=torch.arange(0,input_ids.shape[1]),
+                                               # position_id=cur_position_ids,
+                                               boxes=kwargs['all_boxes'][i],
+                                               orig_size=None,
+                                               images=kwargs['image_list'][i],
+                                               IMG_START_TOKEN=IMG_START_TOKEN,
+                                               IMG_END_TOKEN=IMG_END_TOKEN, rope_pos_id_stride=rope_pos_id_stride)).cuda())
+            pos_ids=torch.stack(pos_ids)
+            if self.attn_type=='ulysses' or self.attn_type=='ring':
+                if input_ids.shape[1]%(2*dist.get_world_size())!=0:
+                    num_padding = 2*dist.get_world_size()-input_ids.shape[1]%(2*dist.get_world_size())
+                    # 创建需要的 padding，input_ids 和 labels 填充值为 -100
+                    padding_shape = (input_ids.shape[0], num_padding)
+                    input_padding = torch.full(padding_shape, 1, dtype=input_ids.dtype, device=input_ids.device)
+                    attn_mask_padding = torch.full(padding_shape, 1, dtype=attention_mask.dtype, device=attention_mask.device)
+                    # 对 input_ids 和 labels 进行 padding
+                    input_ids = torch.cat([input_ids, input_padding], dim=1)
+                    attention_mask=torch.cat([attention_mask,attn_mask_padding],dim=1)
+                                # position_ids 添加正确的递增填充
+                    max_pos_id = pos_ids.max() + 1  # 找到当前最大 position_id
+                    pos_padding = torch.arange(max_pos_id, max_pos_id + num_padding, device=input_ids.device)
+                    pos_padding = pos_padding.unsqueeze(0).expand(input_ids.shape[0], -1)
+                    pos_ids = torch.cat([pos_ids, pos_padding], dim=1)
+            generation_output = self.generate(
+                pixel_values=pixel_values,
+                input_ids=input_ids,
+                attention_mask=attention_mask,
+                position_ids=pos_ids,
+                **generation_config,
+            )
+        else:
+            self.language_model.rope_pos_id_version='default'
+            if self.attn_type=='ulysses' or self.attn_type=='ring':
+                if input_ids.shape[1]%(2*dist.get_world_size())!=0:
+                    num_padding = 2*dist.get_world_size()-input_ids.shape[1]%(2*dist.get_world_size())
+                    # 创建需要的 padding，input_ids 和 labels 填充值为 -100
+                    padding_shape = (input_ids.shape[0], num_padding)
+                    input_padding = torch.full(padding_shape, 1, dtype=input_ids.dtype, device=input_ids.device)
+                    attn_mask_padding = torch.full(padding_shape, 0, dtype=attention_mask.dtype, device=attention_mask.device)
+                    # 对 input_ids 和 labels 进行 padding
+                    input_ids = torch.cat([input_ids, input_padding], dim=1)
+                    attention_mask=torch.cat([attention_mask,attn_mask_padding],dim=1)
+            generation_output = self.generate(
+                pixel_values=pixel_values,
+                input_ids=input_ids,
+                attention_mask=attention_mask,
+                **generation_config,
+            )
+        # 解码生成的输出，跳过特殊 token
+        response = tokenizer.batch_decode(generation_output, skip_special_tokens=True)[0]
+        # 根据分隔符分段
+        response = response.split(template.sep)[0].strip()
+        # 将结果写入历史
+        history.append((question, response))
+        if return_history:
+            return response, history
+        else:
+            query_to_print = query.replace(IMG_CONTEXT_TOKEN, '')
+            query_to_print = query_to_print.replace(f'{IMG_START_TOKEN}{IMG_END_TOKEN}', '<image>')
+            if verbose:
+                print(query_to_print, response)
+            return response
+    @torch.no_grad()
+    def generate(
+            self,
+            pixel_values: Optional[torch.FloatTensor] = None,
+            input_ids: Optional[torch.FloatTensor] = None,
+            attention_mask: Optional[torch.LongTensor] = None,
+            visual_features: Optional[torch.FloatTensor] = None,
+            generation_config: Optional[GenerationConfig] = None,
+            output_hidden_states: Optional[bool] = None,
+            return_dict: Optional[bool] = None,
+            **generate_kwargs,
+    ) -> torch.LongTensor:
+        assert self.img_context_token_id is not None
+        if pixel_values is not None:
+            # 提取图片 embedding
+            # [batch_size, channels, height, width] -> [batch_size, 每张图片的 patch 数, embedding_dim]
+            if visual_features is not None:
+                vit_embeds = visual_features
+            else:
+                vit_embeds = self.extract_feature(pixel_values)
+            if self.posid_type=='qkvLearnable':
+                added_embeds = self.local_posid(torch.arange(self.num_image_token).to(pixel_values.device))
+                vit_embeds = vit_embeds + added_embeds
+                # vit_embeds=vit_embeds+self.local_posid(torch.arange(self.num_image_token).to(pixel_values.device))
+            # 通过嵌入层将 token id 转化为嵌入向量
+            # 其中图片用占位符 IMG_CONTEXT_TOKEN 的 embedding 代替
+            input_embeds = self.language_model.get_input_embeddings()(input_ids)
+            # [1, sequence_length, embedding_dim] -> [sequence_length, embedding_dim]
+            B, N, C = input_embeds.shape
+            input_embeds = input_embeds.reshape(B * N, C)
+            # [1, sequence_length] -> [sequence_length]
+            input_ids = input_ids.reshape(B * N)
+            selected = (input_ids == self.img_context_token_id)
+            assert selected.sum() != 0
+            # 图片 embedding: [总 Patch 数, embedding_dim]
+            # 每个 patch 与一个占位符对应，对应一列 embedding
+            input_embeds[selected] = vit_embeds.reshape(-1, C).to(input_embeds.device)
+            input_embeds = input_embeds.reshape(B, N, C)
+        else:
+            # 通过嵌入层将 token id 转化为嵌入向量
+            # 例如 one hot 编码、Word2Vec、GloVe、FastText等
+            # 嵌入层是一张查找表
+            # [1, sequence_length] -> [1, sequence_length, embedding_dim]
+            input_embeds = self.language_model.get_input_embeddings()(input_ids)
+                    # 找到图片占位符的位置
+        if 'position_ids' in generate_kwargs:
+            pos_id=generate_kwargs['position_ids']
+            if self.attn_type:
+                if self.attn_type=='ulysses':
+                    input_embeds=extract_local2(input_embeds,dist.get_rank(),dist.get_world_size())
+                    attention_mask=extract_local2(attention_mask,dist.get_rank(),dist.get_world_size())
+                    pos_id=extract_local2(pos_id,dist.get_rank(),dist.get_world_size())
+                elif self.attn_type=='ring':
+                    former_shape = input_embeds.shape
+                    input_embeds=extract_local(input_embeds,dist.get_rank(),dist.get_world_size())
+                    attention_mask=extract_local(attention_mask,dist.get_rank(),dist.get_world_size())
+                    pos_id=extract_local(pos_id,dist.get_rank(),dist.get_world_size())
+                generate_kwargs['position_ids']=pos_id
+        else:
+            if self.attn_type:
+                if self.attn_type=='ulysses':
+                    input_embeds=extract_local2(input_embeds,dist.get_rank(),dist.get_world_size())
+                    attention_mask=extract_local2(attention_mask,dist.get_rank(),dist.get_world_size())
+                elif self.attn_type=='ring':
+                    former_shape = input_embeds.shape
+                    input_embeds=extract_local(input_embeds,dist.get_rank(),dist.get_world_size())
+                    attention_mask=extract_local(attention_mask,dist.get_rank(),dist.get_world_size())
+        outputs = self.language_model.generate(
+            inputs_embeds=input_embeds,
+            attention_mask=attention_mask,
+            generation_config=generation_config,
+            output_hidden_states=output_hidden_states,
+            return_dict=return_dict,
+            use_cache=True,
+            **generate_kwargs,
+        )
+        return outputs
+    def update_log(self, new_log_dict):
+        if not hasattr(self, 'log_dict'):
+            self.log_dict = {}
+        for key, value in new_log_dict.items():
+            if 'loss' in key:
+                if key not in self.log_dict:
+                    self.log_dict[key] = value
+                else:
+                    self.log_dict[key] += value
+            else:
+                # just copy it
+                self.log_dict[key] = value
+def get_rope_pos_id(ret, num_tiles, dtype, rope_pos_id_version='default', position_id=None,boxes=None, orig_size=None,images=None,IMG_START_TOKEN='<img>',IMG_END_TOKEN='</img>',rope_pos_id_stride=None):
+        image_start_token_id = global_tokenizer.convert_tokens_to_ids(IMG_START_TOKEN)
+        image_end_token_id = global_tokenizer.convert_tokens_to_ids(IMG_END_TOKEN)
+        num_image_token=256
+        rope_pos_id_list = []
+        input_ids_0 = ret['input_ids'][0]
+        attention_mask_0 = ret['attention_mask'][0]
+        image_start_token_id_idxs = torch.where(input_ids_0 == image_start_token_id)[0]
+        image_end_token_id_idxs = torch.where(input_ids_0 == image_end_token_id)[0]
+        last_record_pos_id = -1
+        start_index = 0
+        for i in range(len(image_start_token_id_idxs)):
+            # 根据序列中的 IMG_START_TOKEN 出现的位置，锁定需要处理的图像 id 序列
+            # 注：这里的 IMG_START_TOKEN 和 IMG_END_TOKEN 应当与文本的处理方式相同
+            box = boxes[i]
+            image = images[i]
+            rope_pos_id_pre = attention_mask_0[start_index:image_start_token_id_idxs[i] + 1].long().cumsum(-1) - 1 + (last_record_pos_id + 1) # 从处理好的序列的最后一个 global id 开始 count
+            rope_pos_id_pre.masked_fill_(attention_mask_0[start_index:image_start_token_id_idxs[i] + 1] == 0, 1)
+            rope_pos_id_list.append(rope_pos_id_pre)
+            last_record_pos_id = rope_pos_id_pre[-1].long()
+            num_tile = num_tiles[i]
+            num_sub_imgs = num_tile - 1
+            is_last = (i == len(image_start_token_id_idxs) - 1)
+            if rope_pos_id_version == 'v0':
+                # 子图为小数，且不管多少个子图，其分配的总 global id 跨度为1；缩略图单独分配完整的，跨度为 1的 global id. Example:
+                # start_id = 100; 100 - 101 (分给 4 * 256)，子图数目为4; 101 - 102 (分给 256) 缩略图
+                if num_sub_imgs > 0:
+                    split_img_id_idxs = torch.linspace(last_record_pos_id, last_record_pos_id + 1, (num_tile - 1) * num_image_token + 1)[1:].to(dtype=dtype)  # 小数数值的 tensor 作为变换的数据取值
+                    origin_split_img_id_idxs = split_img_id_idxs
+                    ############################## 进行位置变换 ##############################
+                    # 先计算第一个子图对应 index
+                    rearange_idx_list = []
+                    rearange_idx_list_list = []
+                    base_index_list = []
+                    num_img_token_in_length = int(num_image_token ** 0.5)
+                    num_patch_width = int(box[-1][2] // box[0][2])
+                    num_patch_height = int(box[-1][3] // box[0][2])
+                    assert num_patch_width * num_patch_height == len(box)
+                    num_total_patch_width_token = num_patch_width * num_img_token_in_length
+                    num_total_patch_height_token = num_patch_height * num_img_token_in_length
+                    assert num_total_patch_width_token * num_total_patch_height_token == num_sub_imgs * num_image_token, (num_total_patch_width_token * num_total_patch_height_token, num_sub_imgs * num_image_token)
+                    for k in range(num_image_token):
+                        map_idx = (k // num_img_token_in_length) * num_total_patch_width_token + (k % num_img_token_in_length)
+                        base_index_list.append(map_idx)
+                    # 计算其他子图对应第一个子图的 offset
+                    for k in range(num_sub_imgs):
+                        patch_row = k // num_patch_width
+                        patch_col = k % num_patch_width
+                        offset = patch_row * (num_image_token * num_patch_width) + patch_col * num_img_token_in_length
+                        # print(f'{k=}, {offset=}')
+                        dst_index_list = [base_index + offset for base_index in base_index_list]
+                        rearange_idx_list.extend(dst_index_list)
+                        rearange_idx_list_list.append(dst_index_list)
+                    ############################## plot 验证 ##############################
+                    # img_boxes = [(deepcopy(img), cur_box, cur_posid) for img, cur_box, cur_posid in
+                    #              zip(image[:-1], box, rearange_idx_list_list)]
+                    # self.eval_posid_by_plot(img_boxes, rope_pos_id_version, None)
+                    # img_boxes = [(deepcopy(img), cur_box, cur_posid) for img, cur_box, cur_posid in
+                    #              zip(image[:-1], box, rearange_idx_list_list)]
+                    # self.eval_posid_by_plot(img_boxes, rope_pos_id_version, split_img_id_idxs)
+                    ############################## rearrange ##############################
+                    split_img_id_idxs = split_img_id_idxs[rearange_idx_list]
+                    rope_pos_id_list.append(split_img_id_idxs)
+                    thumbnail_id_idxs = origin_split_img_id_idxs.reshape([num_image_token, -1]).to(dtype=dtype).mean(dim=1).view(-1)
+                    rope_pos_id_list.append(thumbnail_id_idxs)
+                    last_record_pos_id = origin_split_img_id_idxs[-1].long()
+                else:
+                    thumbnail_id_idxs = torch.linspace(last_record_pos_id, last_record_pos_id + 1,
+                                                    num_image_token + 1)[1:].to(dtype=dtype)  # 缩略图
+                    rope_pos_id_list.append(thumbnail_id_idxs)
+                    last_record_pos_id = (last_record_pos_id + 1).long()
+                # 验证是否能够恢复为等差数列
+                if num_tile > 1:
+                    gt_pos_id = torch.linspace(last_record_pos_id - 2, last_record_pos_id - 1, (num_tile - 1) * num_image_token + 1)[1:].to(dtype=dtype)
+                    # self.eval_posid_by_rearange(box, rope_pos_id_list, gt_pos_id, num_tile, dtype, is_last)
+            elif rope_pos_id_version == 'v1':
+                # 子图为小数，若有 N 个子图，其分配的总 global id 跨度为 N；缩略图单独分配完整的，跨度为 1的 global id. Example:
+                # start_id = 100; 100 - 104 (分给 4 * 256)，子图数目为4; 104 - 105 (分给 256) 缩略图
+                if num_sub_imgs > 0:
+                    split_img_id_idxs = torch.linspace(last_record_pos_id, last_record_pos_id + num_tile - 1, (num_tile - 1) * num_image_token + 1)[1:].to(dtype=dtype)  # 小数数值的 tensor 作为变换的数据取值
+                    origin_split_img_id_idxs = split_img_id_idxs
+                    ############################## 进行位置变换 ##############################
+                    # 先计算第一个子图对应 index
+                    rearange_idx_list = []
+                    rearange_idx_list_list = []
+                    base_index_list = []
+                    # rearange_split_img_id_idxs_list = []
+                    num_img_token_in_length = int(num_image_token ** 0.5)
+                    num_patch_width = int(box[-1][2] // box[0][2])
+                    num_patch_height = int(box[-1][3] // box[0][2])
+                    assert num_patch_width * num_patch_height == len(box)
+                    num_total_patch_width_token = num_patch_width * num_img_token_in_length
+                    num_total_patch_height_token = num_patch_height * num_img_token_in_length
+                    assert num_total_patch_width_token * num_total_patch_height_token == num_sub_imgs * num_image_token, (
+                    num_total_patch_width_token * num_total_patch_height_token, num_sub_imgs * num_image_token)
+                    for k in range(num_image_token):
+                        map_idx = (k // num_img_token_in_length) * num_total_patch_width_token + (
+                                    k % num_img_token_in_length)
+                        base_index_list.append(map_idx)
+                    # 计算其他子图对应第一个子图的 offset
+                    for k in range(num_sub_imgs):
+                        patch_row = k // num_patch_width
+                        patch_col = k % num_patch_width
+                        offset = patch_row * (
+                                    num_image_token * num_patch_width) + patch_col * num_img_token_in_length
+                        # print(f'{k=}, {offset=}')
+                        dst_index_list = [base_index + offset for base_index in base_index_list]
+                        rearange_idx_list.extend(dst_index_list)
+                        rearange_idx_list_list.append(dst_index_list)
+                        # rearange_split_img_id_idxs_list.append(split_img_id_idxs[dst_index_list])
+                    ############################## plot 验证 ##############################
+                    # img_boxes = [(deepcopy(img), cur_box, cur_posid) for img, cur_box, cur_posid in zip(image[:-1], box, rearange_idx_list_list)]
+                    # self.eval_posid_by_plot(img_boxes, rope_pos_id_version, None)
+                    # img_boxes = [(deepcopy(img), cur_box, cur_posid) for img, cur_box, cur_posid in zip(image[:-1], box, rearange_idx_list_list)]
+                    # self.eval_posid_by_plot(img_boxes, rope_pos_id_version, split_img_id_idxs)
+                    ############################## rearrange ##############################
+                    split_img_id_idxs = split_img_id_idxs[rearange_idx_list]
+                    rope_pos_id_list.append(split_img_id_idxs)
+                    # thumbnail_id_idxs = torch.linspace(last_record_pos_id + 1, last_record_pos_id + 2, num_image_token + 1)[1:].to(dtype=dtype)  # 缩略图
+                    thumbnail_id_idxs = origin_split_img_id_idxs.reshape([num_image_token, -1]).to(dtype=dtype).mean(dim=1).view(-1)
+                    rope_pos_id_list.append(thumbnail_id_idxs)
+                    last_record_pos_id = origin_split_img_id_idxs[-1].long()
+                else:
+                    thumbnail_id_idxs = torch.linspace(last_record_pos_id, last_record_pos_id + 1, num_image_token + 1)[1:].to(dtype=dtype)  # 缩略图
+                    rope_pos_id_list.append(thumbnail_id_idxs)
+                    last_record_pos_id = (last_record_pos_id + 1).long()
+                # 验证是否能够恢复为等差数列
+                if num_tile > 1:
+                    gt_pos_id = torch.linspace(last_record_pos_id - 1 - (num_tile - 1), last_record_pos_id - 1, (num_tile - 1) * num_image_token + 1)[1:].to(dtype=dtype)
+                    # self.eval_posid_by_rearange(box, rope_pos_id_list, gt_pos_id, num_tile, dtype)
+            elif rope_pos_id_version == 'v2':
+                # 子图处理方式同文本（N 个子图分配 N * 256 个 global id）；一个缩略图分配 256 * N 个的 global id.
+                # 子图处理同 v0, v1，也对 global id 根据空间关系做 arrange
+                if num_sub_imgs > 0:
+                    split_img_id_idxs = torch.linspace(last_record_pos_id, last_record_pos_id + num_sub_imgs * num_image_token, num_sub_imgs * num_image_token + 1)[1:].long()  # long 数值的 tensor 作为变换的数据取值
+                    last_id_for_split_img = last_record_pos_id + num_sub_imgs * num_image_token
+                    origin_split_img_id_idxs = split_img_id_idxs
+                    ############################## 进行位置变换 ##############################
+                    # 先计算第一个子图对应 index
+                    rearange_idx_list = []
+                    rearange_idx_list_list = []
+                    base_index_list = []
+                    # rearange_split_img_id_idxs_list = []
+                    num_img_token_in_length = int(num_image_token ** 0.5)
+                    num_patch_width = int(box[-1][2] // box[0][2])
+                    num_patch_height = int(box[-1][3] // box[0][2])
+                    assert num_patch_width * num_patch_height == len(box)
+                    num_total_patch_width_token = num_patch_width * num_img_token_in_length
+                    num_total_patch_height_token = num_patch_height * num_img_token_in_length
+                    assert num_total_patch_width_token * num_total_patch_height_token == num_sub_imgs * num_image_token, (
+                        num_total_patch_width_token * num_total_patch_height_token, num_sub_imgs * num_image_token)
+                    for k in range(num_image_token):
+                        map_idx = (k // num_img_token_in_length) * num_total_patch_width_token + (
+                                k % num_img_token_in_length)
+                        base_index_list.append(map_idx)
+                    # 计算其他子图对应第一个子图的 offset
+                    for k in range(num_sub_imgs):
+                        patch_row = k // num_patch_width
+                        patch_col = k % num_patch_width
+                        offset = patch_row * (
+                                num_image_token * num_patch_width) + patch_col * num_img_token_in_length
+                        # print(f'{k=}, {offset=}')
+                        dst_index_list = [base_index + offset for base_index in base_index_list]
+                        rearange_idx_list.extend(dst_index_list)
+                        rearange_idx_list_list.append(dst_index_list)
+                        # rearange_split_img_id_idxs_list.append(split_img_id_idxs[dst_index_list])
+                    ############################## plot 验证 ##############################
+                    # img_boxes = [(deepcopy(img), cur_box, cur_posid) for img, cur_box, cur_posid in
+                    #              zip(image[:-1], box, rearange_idx_list_list)]
+                    # self.eval_posid_by_plot(img_boxes, rope_pos_id_version, None)
+                    # img_boxes = [(deepcopy(img), cur_box, cur_posid) for img, cur_box, cur_posid in
+                    #              zip(image[:-1], box, rearange_idx_list_list)]
+                    # self.eval_posid_by_plot(img_boxes, rope_pos_id_version, split_img_id_idxs)
+                    ############################## rearrange ##############################
+                    split_img_id_idxs = split_img_id_idxs[rearange_idx_list]
+                    rope_pos_id_list.append(split_img_id_idxs)
+                    thumbnail_id_idxs = origin_split_img_id_idxs.reshape([num_image_token, -1]).to(dtype=dtype).mean(dim=1).view(-1)
+                    rope_pos_id_list.append(thumbnail_id_idxs)
+                    last_record_pos_id = origin_split_img_id_idxs[-1].long()
+                else:
+                    thumbnail_id_idxs = torch.linspace(last_record_pos_id, last_record_pos_id + num_image_token, num_image_token + 1)[1:].long()  # 缩略图，和 default 处理一致
+                    rope_pos_id_list.append(thumbnail_id_idxs)
+                    last_record_pos_id = thumbnail_id_idxs[-1].long()
+                # 验证是否能够恢复为等差数列
+                if num_tile > 1:
+                    gt_pos_id = torch.linspace(last_id_for_split_img - num_image_token * num_sub_imgs,
+                                               last_id_for_split_img,
+                                               num_sub_imgs * num_image_token + 1)[1:].long()
+                    # self.eval_posid_by_rearange(box, rope_pos_id_list, gt_pos_id, num_tile, gt_pos_id.dtype)
+            elif rope_pos_id_version == 'v3':
+                # N 个子图共用跨度为 256 的 global id；一个缩略图正常分配 256 个 global id
+                if num_sub_imgs > 0:
+                    split_img_id_idxs = torch.linspace(last_record_pos_id, last_record_pos_id + num_image_token, num_sub_imgs * num_image_token + 1)[1:].to(dtype=dtype)  # 小数数值的 tensor 作为变换的数据取值
+                    origin_split_img_id_idxs = split_img_id_idxs
+                    ############################## 进行位置变换 ##############################
+                    # 先计算第一个子图对应 index
+                    rearange_idx_list = []
+                    rearange_idx_list_list = []
+                    base_index_list = []
+                    # rearange_split_img_id_idxs_list = []
+                    num_img_token_in_length = int(num_image_token ** 0.5)
+                    num_patch_width = int(box[-1][2] // box[0][2])
+                    num_patch_height = int(box[-1][3] // box[0][2])
+                    assert num_patch_width * num_patch_height == len(box)
+                    num_total_patch_width_token = num_patch_width * num_img_token_in_length
+                    num_total_patch_height_token = num_patch_height * num_img_token_in_length
+                    assert num_total_patch_width_token * num_total_patch_height_token == num_sub_imgs * num_image_token, (
+                        num_total_patch_width_token * num_total_patch_height_token, num_sub_imgs * num_image_token)
+                    for k in range(num_image_token):
+                        map_idx = (k // num_img_token_in_length) * num_total_patch_width_token + (
+                                k % num_img_token_in_length)
+                        base_index_list.append(map_idx)
+                    # 计算其他子图对应第一个子图的 offset
+                    for k in range(num_sub_imgs):
+                        patch_row = k // num_patch_width
+                        patch_col = k % num_patch_width
+                        offset = patch_row * (
+                                num_image_token * num_patch_width) + patch_col * num_img_token_in_length
+                        # print(f'{k=}, {offset=}')
+                        dst_index_list = [base_index + offset for base_index in base_index_list]
+                        rearange_idx_list.extend(dst_index_list)
+                        rearange_idx_list_list.append(dst_index_list)
+                        # rearange_split_img_id_idxs_list.append(split_img_id_idxs[dst_index_list])
+                    ############################## plot 验证 ##############################
+                    # img_boxes = [(deepcopy(img), cur_box, cur_posid) for img, cur_box, cur_posid in
+                    #              zip(image[:-1], box, rearange_idx_list_list)]
+                    # self.eval_posid_by_plot(img_boxes, rope_pos_id_version, None)
+                    # img_boxes = [(deepcopy(img), cur_box, cur_posid) for img, cur_box, cur_posid in
+                    #              zip(image[:-1], box, rearange_idx_list_list)]
+                    # self.eval_posid_by_plot(img_boxes, rope_pos_id_version, split_img_id_idxs)
+                    ############################## rearrange ##############################
+                    split_img_id_idxs = split_img_id_idxs[rearange_idx_list]
+                    rope_pos_id_list.append(split_img_id_idxs)
+                    thumbnail_id_idxs = origin_split_img_id_idxs.reshape([num_image_token, -1]).to(dtype=dtype).mean(dim=1).view(-1)
+                    rope_pos_id_list.append(thumbnail_id_idxs)
+                    last_record_pos_id = origin_split_img_id_idxs[-1].long()
+                else:
+                    thumbnail_id_idxs = torch.linspace(last_record_pos_id, last_record_pos_id + num_image_token, num_image_token + 1)[1:].to(dtype=dtype)  # 缩略图，和 default 处理一致
+                    rope_pos_id_list.append(thumbnail_id_idxs)
+                    last_record_pos_id = thumbnail_id_idxs[-1].to(dtype=dtype)
+                # 验证是否能够恢复为等差数列
+                if num_tile > 1:
+                    gt_pos_id = torch.linspace(last_record_pos_id - num_image_token - num_image_token,
+                                               last_record_pos_id - num_image_token,
+                                               num_sub_imgs * num_image_token + 1)[1:].to(dtype=dtype)
+                    # self.eval_posid_by_rearange(box, rope_pos_id_list, gt_pos_id, num_tile, gt_pos_id.dtype)
+            elif rope_pos_id_version == 'v4':
+                # stride 是可变长的
+                assert rope_pos_id_stride is not None, 'when rope_pos_id_version == v4, rope_pos_id_stride should not be None'
+                if num_sub_imgs > 0:
+                    num_sub_image_tokens = num_image_token * num_sub_imgs
+                    split_img_id_idxs = torch.linspace(last_record_pos_id, last_record_pos_id + rope_pos_id_stride, num_sub_imgs * num_image_token + 1)[1:].to(dtype=dtype)  # 小数数值的 tensor 作为变换的数据取值
+                    assert len(split_img_id_idxs) == num_sub_image_tokens
+                    origin_split_img_id_idxs = split_img_id_idxs
+                    ############################## 进行位置变换 ##############################
+                    # 先计算第一个子图对应 index
+                    rearange_idx_list = []
+                    rearange_idx_list_list = []
+                    base_index_list = []
+                    # rearange_split_img_id_idxs_list = []
+                    num_img_token_in_length = int(num_image_token ** 0.5)
+                    num_patch_width = int(box[-1][2] // box[0][2])
+                    num_patch_height = int(box[-1][3] // box[0][2])
+                    assert num_patch_width * num_patch_height == len(box)
+                    num_total_patch_width_token = num_patch_width * num_img_token_in_length
+                    num_total_patch_height_token = num_patch_height * num_img_token_in_length
+                    assert num_total_patch_width_token * num_total_patch_height_token == num_sub_imgs * num_image_token, (
+                        num_total_patch_width_token * num_total_patch_height_token, num_sub_imgs * num_image_token)
+                    for k in range(num_image_token):
+                        map_idx = (k // num_img_token_in_length) * num_total_patch_width_token + (
+                                k % num_img_token_in_length)
+                        base_index_list.append(map_idx)
+                    # 计算其他子图对应第一个子图的 offset
+                    for k in range(num_sub_imgs):
+                        patch_row = k // num_patch_width
+                        patch_col = k % num_patch_width
+                        offset = patch_row * (num_image_token * num_patch_width) + patch_col * num_img_token_in_length
+                        # print(f'{k=}, {offset=}')
+                        dst_index_list = [base_index + offset for base_index in base_index_list]
+                        rearange_idx_list.extend(dst_index_list)
+                        rearange_idx_list_list.append(dst_index_list)
+                        # rearange_split_img_id_idxs_list.append(split_img_id_idxs[dst_index_list])
+                    ############################## plot 验证 ##############################
+                    # img_boxes = [(deepcopy(img), cur_box, cur_posid) for img, cur_box, cur_posid in
+                    #              zip(image[:-1], box, rearange_idx_list_list)]
+                    # self.eval_posid_by_plot(img_boxes, rope_pos_id_version, None)
+                    # img_boxes = [(deepcopy(img), cur_box, cur_posid) for img, cur_box, cur_posid in
+                    #              zip(image[:-1], box, rearange_idx_list_list)]
+                    # self.eval_posid_by_plot(img_boxes, rope_pos_id_version, split_img_id_idxs)
+                    ############################## rearrange ##############################
+                    split_img_id_idxs = split_img_id_idxs[rearange_idx_list]
+                    rope_pos_id_list.append(split_img_id_idxs)
+                    thumbnail_id_idxs = origin_split_img_id_idxs.reshape([num_image_token, -1]).to(dtype=dtype).mean(dim=1).view(-1)
+                    rope_pos_id_list.append(thumbnail_id_idxs)
+                    last_record_pos_id = origin_split_img_id_idxs[-1].long()
+                else:
+                    thumbnail_id_idxs = torch.linspace(last_record_pos_id, last_record_pos_id + num_image_token, num_image_token + 1)[1:].to(dtype=dtype)  # 缩略图，和 default 处理一致
+                    rope_pos_id_list.append(thumbnail_id_idxs)
+                    last_record_pos_id = thumbnail_id_idxs[-1].to(dtype=dtype)
+            elif rope_pos_id_version == 'v5':
+                assert rope_pos_id_stride is not None, 'when rope_pos_id_version == v5, self.rope_pos_id_stride should not be None'
+                small_stride = rope_pos_id_stride / num_image_token
+                # split_img_id_idxs = torch.arange(last_record_pos_id, last_record_pos_id + small_stride * (num_image_token * num_tile + 1), small_stride)[1:].to(dtype=dtype)
+                split_img_id_idxs = torch.linspace(last_record_pos_id,last_record_pos_id+small_stride*(num_image_token * num_tile ),(num_image_token * num_tile + 1))[1:].to(dtype=dtype)
+                rope_pos_id_list.append(split_img_id_idxs)
+                last_record_pos_id = torch.ceil(split_img_id_idxs[-1]).long()
+            elif rope_pos_id_version == 'v6':
+                random_from=[1,2,4,8,16,32,64,128,256]
+                rope_pos_id_stride=random.choice(random_from)
+                small_stride = rope_pos_id_stride / num_image_token
+                # split_img_id_idxs = torch.arange(last_record_pos_id, last_record_pos_id + small_stride * (num_image_token * num_tile + 1), small_stride)[1:].to(dtype=dtype)
+                split_img_id_idxs = torch.linspace(last_record_pos_id,last_record_pos_id+small_stride*(num_image_token * num_tile ),(num_image_token * num_tile + 1))[1:].to(dtype=dtype)
+                rope_pos_id_list.append(split_img_id_idxs)
+                last_record_pos_id = torch.ceil(split_img_id_idxs[-1]).long()
+            elif rope_pos_id_version == 'default':
+                # baseline
+                # 无特殊处理的做法
+                split_img_id_idxs = torch.linspace(last_record_pos_id,
+                                                   last_record_pos_id + (num_tile - 1) * num_image_token,
+                                                   (num_tile - 1) * num_image_token + 1)[1:].to(dtype=dtype)  # 子图
+                rope_pos_id_list.append(split_img_id_idxs)
+                thumbnail_id_idxs = torch.linspace(last_record_pos_id + (num_tile - 1) * num_image_token,
+                                                   last_record_pos_id + num_tile * num_image_token,
+                                                   num_image_token + 1)[1:].to(dtype=dtype)  # 缩略图
+                rope_pos_id_list.append(thumbnail_id_idxs)
+                last_record_pos_id = (last_record_pos_id + num_tile * num_image_token).long()
+            else:
+                raise NotImplementedError(f'not implement for {rope_pos_id_version}')
+            try:
+                start_index = image_start_token_id_idxs[i] + num_tile * num_image_token + 1
+                assert input_ids_0[start_index] == image_end_token_id # 下一次迭代的开头应该是 IMG_END_TOKEN
+                assert start_index == image_end_token_id_idxs[i] # 下一次迭代的开头应该是 IMG_END_TOKEN
+            except:
+                import ipdb
+                ipdb.set_trace()
+        if image_end_token_id_idxs[-1] != input_ids_0.shape[0] - 1:
+            # 末尾还有待处理的非图像 id 的情况
+            assert image_end_token_id_idxs[-1] == start_index # 应当从最后一个 IMG_END_TOKEN 开始
+            rope_pos_id_pre = attention_mask_0[start_index:].long().cumsum(-1) - 1 + (last_record_pos_id + 1)
+            rope_pos_id_pre.masked_fill_(attention_mask_0[start_index:] == 0, 1)
+            rope_pos_id_list.append(rope_pos_id_pre)
+        rope_pos_id_list=[_.to('cpu') for _ in rope_pos_id_list]
+        rope_pos_id = torch.cat(rope_pos_id_list).to(dtype=dtype)
+        if rope_pos_id_version == 'default':
+            rope_pos_id = rope_pos_id.long() # 不做特殊处理的 rope_pos_id 应当等于 position_ids
+            assert torch.equal(rope_pos_id, position_id.to(rope_pos_id.device)), (rope_pos_id, position_id.to(rope_pos_id.device))
+            assert torch.allclose(rope_pos_id, position_id.to(rope_pos_id.device), atol=1e-32)
+        assert rope_pos_id.shape == input_ids_0.shape
+        return list(rope_pos_id.numpy())

V2PE-256K/preprocessor_config.json ADDED Viewed

	@@ -0,0 +1,19 @@

+{
+  "crop_size": 448,
+  "do_center_crop": true,
+  "do_normalize": true,
+  "do_resize": true,
+  "feature_extractor_type": "CLIPFeatureExtractor",
+  "image_mean": [
+    0.485,
+    0.456,
+    0.406
+  ],
+  "image_std": [
+    0.229,
+    0.224,
+    0.225
+  ],
+  "resample": 3,
+  "size": 448
+}

V2PE-256K/special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,47 @@

+{
+  "additional_special_tokens": [
+    "<|im_start|>",
+    "<|im_end|>",
+    "<|action_start|>",
+    "<|action_end|>",
+    "<|interpreter|>",
+    "<|plugin|>",
+    "<img>",
+    "</img>",
+    "<IMG_CONTEXT>",
+    "<quad>",
+    "</quad>",
+    "<ref>",
+    "</ref>",
+    "<box>",
+    "</box>"
+  ],
+  "bos_token": {
+    "content": "<s>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "eos_token": {
+    "content": "</s>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "pad_token": {
+    "content": "</s>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "unk_token": {
+    "content": "<unk>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  }
+}

V2PE-256K/tokenization_internlm2.py ADDED Viewed

	@@ -0,0 +1,235 @@

+# Copyright (c) The InternLM team and The HuggingFace Inc. team. All rights reserved.
+#
+# This code is based on transformers/src/transformers/models/llama/tokenization_llama.py
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+"""Tokenization classes for InternLM."""
+import os
+from shutil import copyfile
+from typing import Any, Dict, List, Optional, Tuple
+import sentencepiece as spm
+from transformers.tokenization_utils import PreTrainedTokenizer
+from transformers.utils import logging
+logger = logging.get_logger(__name__)
+VOCAB_FILES_NAMES = {'vocab_file': './tokenizer.model'}
+PRETRAINED_VOCAB_FILES_MAP = {}
+# Modified from transformers.model.llama.tokenization_llama.LlamaTokenizer
+class InternLM2Tokenizer(PreTrainedTokenizer):
+    """
+    Construct a InternLM2 tokenizer. Based on byte-level Byte-Pair-Encoding.
+    Args:
+        vocab_file (`str`):
+            Path to the vocabulary file.
+    """
+    vocab_files_names = VOCAB_FILES_NAMES
+    pretrained_vocab_files_map = PRETRAINED_VOCAB_FILES_MAP
+    model_input_names = ['input_ids', 'attention_mask']
+    _auto_class = 'AutoTokenizer'
+    def __init__(
+        self,
+        vocab_file,
+        unk_token='<unk>',
+        bos_token='<s>',
+        eos_token='</s>',
+        pad_token='</s>',
+        sp_model_kwargs: Optional[Dict[str, Any]] = None,
+        add_bos_token=True,
+        add_eos_token=False,
+        decode_with_prefix_space=False,
+        clean_up_tokenization_spaces=False,
+        **kwargs,
+    ):
+        self.sp_model_kwargs = {} if sp_model_kwargs is None else sp_model_kwargs
+        self.vocab_file = vocab_file
+        self.add_bos_token = add_bos_token
+        self.add_eos_token = add_eos_token
+        self.decode_with_prefix_space = decode_with_prefix_space
+        self.sp_model = spm.SentencePieceProcessor(**self.sp_model_kwargs)
+        self.sp_model.Load(vocab_file)
+        self._no_prefix_space_tokens = None
+        super().__init__(
+            bos_token=bos_token,
+            eos_token=eos_token,
+            unk_token=unk_token,
+            pad_token=pad_token,
+            clean_up_tokenization_spaces=clean_up_tokenization_spaces,
+            **kwargs,
+        )
+    @property
+    def no_prefix_space_tokens(self):
+        if self._no_prefix_space_tokens is None:
+            vocab = self.convert_ids_to_tokens(list(range(self.vocab_size)))
+            self._no_prefix_space_tokens = {i for i, tok in enumerate(vocab) if not tok.startswith('▁')}
+        return self._no_prefix_space_tokens
+    @property
+    def vocab_size(self):
+        """Returns vocab size"""
+        return self.sp_model.get_piece_size()
+    @property
+    def bos_token_id(self) -> Optional[int]:
+        return self.sp_model.bos_id()
+    @property
+    def eos_token_id(self) -> Optional[int]:
+        return self.sp_model.eos_id()
+    def get_vocab(self):
+        """Returns vocab as a dict"""
+        vocab = {self.convert_ids_to_tokens(i): i for i in range(self.vocab_size)}
+        vocab.update(self.added_tokens_encoder)
+        return vocab
+    def _tokenize(self, text):
+        """Returns a tokenized string."""
+        return self.sp_model.encode(text, out_type=str)
+    def _convert_token_to_id(self, token):
+        """Converts a token (str) in an id using the vocab."""
+        return self.sp_model.piece_to_id(token)
+    def _convert_id_to_token(self, index):
+        """Converts an index (integer) in a token (str) using the vocab."""
+        token = self.sp_model.IdToPiece(index)
+        return token
+    def _maybe_add_prefix_space(self, tokens, decoded):
+        if tokens and tokens[0] not in self.no_prefix_space_tokens:
+            return ' ' + decoded
+        else:
+            return decoded
+    def convert_tokens_to_string(self, tokens):
+        """Converts a sequence of tokens (string) in a single string."""
+        current_sub_tokens = []
+        out_string = ''
+        prev_is_special = False
+        for token in tokens:
+            # make sure that special tokens are not decoded using sentencepiece model
+            if token in self.all_special_tokens:
+                if not prev_is_special:
+                    out_string += ' '
+                out_string += self.sp_model.decode(current_sub_tokens) + token
+                prev_is_special = True
+                current_sub_tokens = []
+            else:
+                current_sub_tokens.append(token)
+                prev_is_special = False
+        out_string += self.sp_model.decode(current_sub_tokens)
+        out_string = self.clean_up_tokenization(out_string)
+        out_string = self._maybe_add_prefix_space(tokens=tokens, decoded=out_string)
+        return out_string[1:]
+    def save_vocabulary(self, save_directory, filename_prefix: Optional[str] = None) -> Tuple[str]:
+        """
+        Save the vocabulary and special tokens file to a directory.
+        Args:
+            save_directory (`str`):
+                The directory in which to save the vocabulary.
+        Returns:
+            `Tuple(str)`: Paths to the files saved.
+        """
+        if not os.path.isdir(save_directory):
+            logger.error(f'Vocabulary path ({save_directory}) should be a directory')
+            return
+        out_vocab_file = os.path.join(
+            save_directory, (filename_prefix + '-' if filename_prefix else '') + VOCAB_FILES_NAMES['vocab_file']
+        )
+        if os.path.abspath(self.vocab_file) != os.path.abspath(out_vocab_file) and os.path.isfile(self.vocab_file):
+            copyfile(self.vocab_file, out_vocab_file)
+        elif not os.path.isfile(self.vocab_file):
+            with open(out_vocab_file, 'wb') as fi:
+                content_spiece_model = self.sp_model.serialized_model_proto()
+                fi.write(content_spiece_model)
+        return (out_vocab_file,)
+    def build_inputs_with_special_tokens(self, token_ids_0, token_ids_1=None):
+        if self.add_bos_token:
+            bos_token_ids = [self.bos_token_id]
+        else:
+            bos_token_ids = []
+        output = bos_token_ids + token_ids_0
+        if token_ids_1 is not None:
+            output = output + token_ids_1
+        if self.add_eos_token:
+            output = output + [self.eos_token_id]
+        return output
+    def get_special_tokens_mask(
+        self, token_ids_0: List[int], token_ids_1: Optional[List[int]] = None, already_has_special_tokens: bool = False
+    ) -> List[int]:
+        """
+        Retrieve sequence ids from a token list that has no special tokens added. This method is called when adding
+        special tokens using the tokenizer `prepare_for_model` method.
+        Args:
+            token_ids_0 (`List[int]`):
+                List of IDs.
+            token_ids_1 (`List[int]`, *optional*):
+                Optional second list of IDs for sequence pairs.
+            already_has_special_tokens (`bool`, *optional*, defaults to `False`):
+                Whether or not the token list is already formatted with special tokens for the model.
+        Returns:
+            `List[int]`: A list of integers in the range [0, 1]: 1 for a special token, 0 for a sequence token.
+        """
+        if already_has_special_tokens:
+            return super().get_special_tokens_mask(
+                token_ids_0=token_ids_0, token_ids_1=token_ids_1, already_has_special_tokens=True
+            )
+        if token_ids_1 is None:
+            return [1] + ([0] * len(token_ids_0)) + [1]
+        return [1] + ([0] * len(token_ids_0)) + [1, 1] + ([0] * len(token_ids_1)) + [1]
+    def create_token_type_ids_from_sequences(
+        self, token_ids_0: List[int], token_ids_1: Optional[List[int]] = None
+    ) -> List[int]:
+        """
+        Create a mask from the two sequences passed to be used in a sequence-pair classification task. T5 does not make
+        use of token type ids, therefore a list of zeros is returned.
+        Args:
+            token_ids_0 (`List[int]`):
+                List of IDs.
+            token_ids_1 (`List[int]`, *optional*):
+                Optional second list of IDs for sequence pairs.
+        Returns:
+            `List[int]`: List of zeros.
+        """
+        eos = [self.eos_token_id]
+        if token_ids_1 is None:
+            return len(token_ids_0 + eos) * [0]
+        return len(token_ids_0 + eos + token_ids_1 + eos) * [0]

V2PE-256K/tokenization_internlm2_fast.py ADDED Viewed

	@@ -0,0 +1,211 @@

+# Copyright (c) The InternLM team and The HuggingFace Inc. team. All rights reserved.
+#
+# This code is based on transformers/src/transformers/models/llama/tokenization_llama_fast.py
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+"""Tokenization Fast class for InternLM."""
+import os
+from shutil import copyfile
+from typing import Any, Dict, Optional, Tuple
+from tokenizers import Tokenizer, decoders, normalizers, processors
+from tokenizers.models import BPE
+from transformers.convert_slow_tokenizer import (SLOW_TO_FAST_CONVERTERS,
+                                                 SentencePieceExtractor,
+                                                 SpmConverter)
+from transformers.tokenization_utils_fast import PreTrainedTokenizerFast
+from transformers.utils import logging
+from .tokenization_internlm2 import InternLM2Tokenizer
+logger = logging.get_logger(__name__)
+VOCAB_FILES_NAMES = {'vocab_file': './tokenizer.model'}
+# Modified from transformers.convert_slow_tokenizer.LlamaConverter
+class InternLM2Converter(SpmConverter):
+    handle_byte_fallback = True
+    def vocab(self, proto):
+        vocab = [
+            ('<unk>', 0.0),
+            ('<s>', 0.0),
+            ('</s>', 0.0),
+        ]
+        vocab += [(piece.piece, piece.score) for piece in proto.pieces[3:]]
+        return vocab
+    def unk_id(self, proto):
+        unk_id = 0
+        return unk_id
+    def decoder(self, replacement, add_prefix_space):
+        return decoders.Sequence(
+            [
+                decoders.Replace('▁', ' '),
+                decoders.ByteFallback(),
+                decoders.Fuse(),
+                decoders.Strip(content=' ', left=1),
+            ]
+        )
+    def tokenizer(self, proto):
+        model_type = proto.trainer_spec.model_type
+        vocab_scores = self.vocab(proto)
+        # special tokens
+        added_tokens = self.original_tokenizer.added_tokens_decoder
+        for i in range(len(vocab_scores)):
+            piece, score = vocab_scores[i]
+            if i in added_tokens:
+                vocab_scores[i] = (added_tokens[i].content, score)
+        if model_type == 1:
+            raise RuntimeError('InternLM2 is supposed to be a BPE model!')
+        elif model_type == 2:
+            _, merges = SentencePieceExtractor(self.original_tokenizer.vocab_file).extract(vocab_scores)
+            bpe_vocab = {word: i for i, (word, _score) in enumerate(vocab_scores)}
+            tokenizer = Tokenizer(
+                BPE(bpe_vocab, merges, unk_token=proto.trainer_spec.unk_piece, fuse_unk=True, byte_fallback=True)
+            )
+            tokenizer.add_special_tokens(
+                [ added_token for index, added_token in added_tokens.items()]
+            )
+        else:
+            raise Exception(
+                "You're trying to run a `Unigram` model but you're file was trained with a different algorithm"
+            )
+        return tokenizer
+    def normalizer(self, proto):
+        normalizers_list = []
+        if proto.normalizer_spec.add_dummy_prefix:
+            normalizers_list.append(normalizers.Prepend(prepend='▁'))
+        normalizers_list.append(normalizers.Replace(pattern=' ', content='▁'))
+        return normalizers.Sequence(normalizers_list)
+    def pre_tokenizer(self, replacement, add_prefix_space):
+        return None
+SLOW_TO_FAST_CONVERTERS['InternLM2Tokenizer'] = InternLM2Converter
+# Modified from transformers.model.llama.tokenization_llama_fast.LlamaTokenizerFast -> InternLM2TokenizerFast
+class InternLM2TokenizerFast(PreTrainedTokenizerFast):
+    vocab_files_names = VOCAB_FILES_NAMES
+    slow_tokenizer_class = InternLM2Tokenizer
+    padding_side = 'left'
+    model_input_names = ['input_ids', 'attention_mask']
+    _auto_class = 'AutoTokenizer'
+    def __init__(
+        self,
+        vocab_file,
+        unk_token='<unk>',
+        bos_token='<s>',
+        eos_token='</s>',
+        pad_token='</s>',
+        sp_model_kwargs: Optional[Dict[str, Any]] = None,
+        add_bos_token=True,
+        add_eos_token=False,
+        decode_with_prefix_space=False,
+        clean_up_tokenization_spaces=False,
+        **kwargs,
+    ):
+        super().__init__(
+            vocab_file=vocab_file,
+            unk_token=unk_token,
+            bos_token=bos_token,
+            eos_token=eos_token,
+            pad_token=pad_token,
+            sp_model_kwargs=sp_model_kwargs,
+            add_bos_token=add_bos_token,
+            add_eos_token=add_eos_token,
+            decode_with_prefix_space=decode_with_prefix_space,
+            clean_up_tokenization_spaces=clean_up_tokenization_spaces,
+            **kwargs,
+        )
+        self._add_bos_token = add_bos_token
+        self._add_eos_token = add_eos_token
+        self.update_post_processor()
+        self.vocab_file = vocab_file
+    @property
+    def can_save_slow_tokenizer(self) -> bool:
+        return os.path.isfile(self.vocab_file) if self.vocab_file else False
+    def update_post_processor(self):
+        """
+        Updates the underlying post processor with the current `bos_token` and `eos_token`.
+        """
+        bos = self.bos_token
+        bos_token_id = self.bos_token_id
+        if bos is None and self.add_bos_token:
+            raise ValueError('add_bos_token = True but bos_token = None')
+        eos = self.eos_token
+        eos_token_id = self.eos_token_id
+        if eos is None and self.add_eos_token:
+            raise ValueError('add_eos_token = True but eos_token = None')
+        single = f"{(bos+':0 ') if self.add_bos_token else ''}$A:0{(' '+eos+':0') if self.add_eos_token else ''}"
+        pair = f"{single}{(' '+bos+':1') if self.add_bos_token else ''} $B:1{(' '+eos+':1') if self.add_eos_token else ''}"
+        special_tokens = []
+        if self.add_bos_token:
+            special_tokens.append((bos, bos_token_id))
+        if self.add_eos_token:
+            special_tokens.append((eos, eos_token_id))
+        self._tokenizer.post_processor = processors.TemplateProcessing(
+            single=single, pair=pair, special_tokens=special_tokens
+        )
+    @property
+    def add_eos_token(self):
+        return self._add_eos_token
+    @property
+    def add_bos_token(self):
+        return self._add_bos_token
+    @add_eos_token.setter
+    def add_eos_token(self, value):
+        self._add_eos_token = value
+        self.update_post_processor()
+    @add_bos_token.setter
+    def add_bos_token(self, value):
+        self._add_bos_token = value
+        self.update_post_processor()
+    def save_vocabulary(self, save_directory: str, filename_prefix: Optional[str] = None) -> Tuple[str]:
+        if not self.can_save_slow_tokenizer:
+            raise ValueError(
+                'Your fast tokenizer does not have the necessary information to save the vocabulary for a slow '
+                'tokenizer.'
+            )
+        if not os.path.isdir(save_directory):
+            logger.error(f'Vocabulary path ({save_directory}) should be a directory')
+            return
+        out_vocab_file = os.path.join(
+            save_directory, (filename_prefix + '-' if filename_prefix else '') + VOCAB_FILES_NAMES['vocab_file']
+        )
+        if os.path.abspath(self.vocab_file) != os.path.abspath(out_vocab_file):
+            copyfile(self.vocab_file, out_vocab_file)
+        return (out_vocab_file,)

V2PE-256K/tokenizer.model ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:f868398fc4e05ee1e8aeba95ddf18ddcc45b8bce55d5093bead5bbf80429b48b
+size 1477754

V2PE-256K/tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,179 @@

+{
+  "added_tokens_decoder": {
+    "0": {
+      "content": "<unk>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "1": {
+      "content": "<s>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "2": {
+      "content": "</s>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "92538": {
+      "content": "<|plugin|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "92539": {
+      "content": "<|interpreter|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "92540": {
+      "content": "<|action_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "92541": {
+      "content": "<|action_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "92542": {
+      "content": "<|im_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "92543": {
+      "content": "<|im_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "92544": {
+      "content": "<img>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "92545": {
+      "content": "</img>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "92546": {
+      "content": "<IMG_CONTEXT>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "92547": {
+      "content": "<quad>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "92548": {
+      "content": "</quad>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "92549": {
+      "content": "<ref>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "92550": {
+      "content": "</ref>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "92551": {
+      "content": "<box>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "92552": {
+      "content": "</box>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    }
+  },
+  "additional_special_tokens": [
+    "<|im_start|>",
+    "<|im_end|>",
+    "<|action_start|>",
+    "<|action_end|>",
+    "<|interpreter|>",
+    "<|plugin|>",
+    "<img>",
+    "</img>",
+    "<IMG_CONTEXT>",
+    "<quad>",
+    "</quad>",
+    "<ref>",
+    "</ref>",
+    "<box>",
+    "</box>"
+  ],
+  "auto_map": {
+    "AutoTokenizer": [
+      "tokenization_internlm2.InternLM2Tokenizer",
+      null
+    ]
+  },
+  "bos_token": "<s>",
+  "chat_template": "{{ bos_token }}{% for message in messages %}{{'<|im_start|>' + message['role'] + '\n' + message['content'] + '<|im_end|>' + '\n'}}{% endfor %}{% if add_generation_prompt %}{{ '<|im_start|>assistant\n' }}{% endif %}",
+  "clean_up_tokenization_spaces": false,
+  "eos_token": "</s>",
+  "model_max_length": 270000,
+  "pad_token": "</s>",
+  "tokenizer_class": "InternLM2Tokenizer",
+  "unk_token": "<unk>"
+}