HongyuanTao commited on Feb 17

Commit

6cf7380

verified ·

1 Parent(s): d391b3c

Upload 17 files

Browse files

Files changed (17) hide show

added_tokens.json +11 -0
config.json +240 -0
configuration_mmMamba.py +156 -0
configuration_mmMamba_chat.py +93 -0
configuration_mmMamba_embedding.py +111 -0
conversation.py +1368 -0
generation_config.json +4 -0
model-00001-of-00002.safetensors +3 -0
model-00002-of-00002.safetensors +3 -0
model.safetensors.index.json +403 -0
modeling_mmMamba.py +1136 -0
modeling_mmMamba_chat.py +517 -0
modeling_mmMamba_embedding.py +966 -0
special_tokens_map.json +47 -0
tokenization_internlm2.py +235 -0
tokenizer.model +3 -0
tokenizer_config.json +179 -0

added_tokens.json ADDED Viewed

	@@ -0,0 +1,11 @@

+{
+  "</box>": 92552,
+  "</img>": 92545,
+  "</quad>": 92548,
+  "</ref>": 92550,
+  "<IMG_CONTEXT>": 92546,
+  "<box>": 92551,
+  "<img>": 92544,
+  "<quad>": 92547,
+  "<ref>": 92549
+}

config.json ADDED Viewed

	@@ -0,0 +1,240 @@

+{
+    "_commit_hash": null,
+    "_name_or_path": "hustvl/mmMamba",
+    "architectures": [
+      "mmMambaChatModel"
+    ],
+    "auto_map": {
+      "AutoConfig": "configuration_mmMamba_chat.mmMambaChatConfig",
+      "AutoModel": "modeling_mmMamba_chat.mmMambaChatModel",
+      "AutoModelForCausalLM": "modeling_mmMamba_chat.mmMambaChatModel"
+    },
+    "downsample_ratio": 0.5,
+    "dynamic_image_size": true,
+    "embedding_config": {
+      "_name_or_path": "",
+      "add_cross_attention": false,
+      "architectures": null,
+      "attention_bias": false,
+      "attention_dropout": 0.0,
+      "attn_implementation": "flash_attention_2",
+      "bad_words_ids": null,
+      "begin_suppress_tokens": null,
+      "bos_token_id": null,
+      "chunk_size_feed_forward": 0,
+      "cross_attention_hidden_size": null,
+      "decoder_start_token_id": null,
+      "diversity_penalty": 0.0,
+      "do_sample": false,
+      "downsample_ratio": 0.5,
+      "drop_path_rate": 0.0,
+      "dropout": 0.0,
+      "early_stopping": false,
+      "encoder_no_repeat_ngram_size": 0,
+      "eos_token_id": null,
+      "exponential_decay_length_penalty": null,
+      "finetuning_task": null,
+      "forced_bos_token_id": null,
+      "forced_eos_token_id": null,
+      "hidden_act": "silu",
+      "hidden_size": 2048,
+      "id2label": {
+        "0": "LABEL_0",
+        "1": "LABEL_1"
+      },
+      "image_size": 448,
+      "img_context_token_id": 92546,
+      "initializer_factor": 1e-05,
+      "initializer_range": 0.02,
+      "intermediate_size": 8192,
+      "is_decoder": false,
+      "is_encoder_decoder": false,
+      "label2id": {
+        "LABEL_0": 0,
+        "LABEL_1": 1
+      },
+      "layers_block_type":["mamba2", "mamba2", "mamba2", "mamba2", "mamba2", "mamba2", "mamba2", "mamba2"],
+      "layer_norm_eps": 1e-06,
+      "length_penalty": 1.0,
+      "llm_hidden_size": 2048,
+      "llm_vocab_size": 92553,
+      "max_length": 20,
+      "max_position_embeddings": 32768,
+      "min_length": 0,
+      "mlp_bias": false,
+      "model_type": "mmMamba_embedding",
+      "no_repeat_ngram_size": 0,
+      "norm_type": "rms_norm",
+      "num_attention_heads": 16,
+      "num_beam_groups": 1,
+      "num_beams": 1,
+      "num_channels": 3,
+      "num_hidden_layers": 8,
+      "num_key_value_heads": 8,
+      "num_return_sequences": 1,
+      "output_attentions": false,
+      "output_hidden_states": false,
+      "output_scores": false,
+      "pad_token_id": null,
+      "patch_size": 14,
+      "pixel_shuffle_loc": "pre",
+      "prefix": null,
+      "pretraining_tp": 1,
+      "problem_type": null,
+      "pruned_heads": {},
+      "qk_normalization": true,
+      "qkv_bias": false,
+      "remove_invalid_values": false,
+      "repetition_penalty": 1.0,
+      "return_dict": true,
+      "return_dict_in_generate": false,
+      "rms_norm_eps": 1e-05,
+      "rope_scaling": null,
+      "rope_theta": 1000000.0,
+      "sep_token_id": null,
+      "special_token_maps": {},
+      "suppress_tokens": null,
+      "target_hidden_size": 2048,
+      "task_specific_params": null,
+      "temperature": 1.0,
+      "tf_legacy_loss": false,
+      "tie_encoder_decoder": false,
+      "tie_word_embeddings": true,
+      "tokenizer_class": null,
+      "top_k": 50,
+      "top_p": 1.0,
+      "torch_dtype": null,
+      "torchscript": false,
+      "transformers_version": "4.43.1",
+      "typical_p": 1.0,
+      "use_autoregressive_loss": false,
+      "use_bfloat16": false,
+      "use_flash_attn": true,
+      "use_img_start_end_tokens": true,
+      "use_ls": false,
+      "use_pixel_shuffle_proj": true
+    },
+    "force_image_size": 448,
+    "llm_config": {
+      "_name_or_path": "",
+      "add_cross_attention": false,
+      "architectures": [
+        "mmMambaForCausalLM"
+      ],
+      "attn_implementation": "flash_attention_2",
+      "auto_map": {
+        "AutoConfig": "configuration_mmMamba.mmMambaConfig",
+        "AutoModel": "modeling_mmMamba.mmMambaForCausalLM",
+        "AutoModelForCausalLM": "modeling_mmMamba.mmMambaForCausalLM"
+      },
+      "bad_words_ids": null,
+      "begin_suppress_tokens": null,
+      "bias": false,
+      "bos_token_id": 1,
+      "chunk_size_feed_forward": 0,
+      "cross_attention_hidden_size": null,
+      "decoder_start_token_id": null,
+      "diversity_penalty": 0.0,
+      "do_sample": false,
+      "early_stopping": false,
+      "encoder_no_repeat_ngram_size": 0,
+      "eos_token_id": 2,
+      "exponential_decay_length_penalty": null,
+      "finetuning_task": null,
+      "forced_bos_token_id": null,
+      "forced_eos_token_id": null,
+      "hidden_act": "silu",
+      "hidden_size": 2048,
+      "id2label": {
+        "0": "LABEL_0",
+        "1": "LABEL_1"
+      },
+      "initializer_range": 0.02,
+      "intermediate_size": 8192,
+      "is_decoder": false,
+      "is_encoder_decoder": false,
+      "label2id": {
+        "LABEL_0": 0,
+        "LABEL_1": 1
+      },
+      "layers_block_type":[
+      "mamba2", "mamba2", "mamba2", "mamba2", "mamba2", "mamba2", "mamba2", "mamba2",
+      "mamba2", "mamba2", "mamba2", "mamba2", "mamba2", "mamba2", "mamba2", "mamba2",
+      "mamba2", "mamba2", "mamba2", "mamba2", "mamba2", "mamba2", "mamba2", "mamba2"
+      ],
+      "length_penalty": 1.0,
+      "max_length": 20,
+      "max_position_embeddings": 32768,
+      "min_length": 0,
+      "model_type": "mmMamba",
+      "no_repeat_ngram_size": 0,
+      "num_attention_heads": 16,
+      "num_beam_groups": 1,
+      "num_beams": 1,
+      "num_hidden_layers": 24,
+      "num_key_value_heads": 8,
+      "num_return_sequences": 1,
+      "output_attentions": false,
+      "output_hidden_states": false,
+      "output_scores": false,
+      "pad_token_id": 2,
+      "prefix": null,
+      "problem_type": null,
+      "pruned_heads": {},
+      "remove_invalid_values": false,
+      "repetition_penalty": 1.0,
+      "return_dict": true,
+      "return_dict_in_generate": false,
+      "rms_norm_eps": 1e-05,
+      "rope_scaling": {
+        "factor": 2.0,
+        "type": "dynamic"
+      },
+      "rope_theta": 1000000,
+      "sep_token_id": null,
+      "suppress_tokens": null,
+      "task_specific_params": null,
+      "temperature": 1.0,
+      "tf_legacy_loss": false,
+      "tie_encoder_decoder": false,
+      "tie_word_embeddings": false,
+      "tokenizer_class": null,
+      "top_k": 50,
+      "top_p": 1.0,
+      "torch_dtype": "bfloat16",
+      "torchscript": false,
+      "transformers_version": "4.43.1",
+      "typical_p": 1.0,
+      "use_bfloat16": true,
+      "use_cache": false,
+      "vocab_size": 92553,
+      "feature_map": "softmax_dim",
+      "feature_map_kwargs": {
+        "eps": 1e-12,
+        "fullspace": true
+      },
+      "learned_kernel": "untied_head_einsum",
+      "learned_kernel_kwargs":{
+        "feature_dim": 64,
+        "skip_connection": false,
+        "bias": false,
+        "zero_init": false
+      },
+      "tie_qk_kernels": false
+    },
+    "max_dynamic_patch": 12,
+    "min_dynamic_patch": 1,
+    "model_type": "mmMamba_chat",
+    "normalize_encoder_output": true,
+    "pad2square": false,
+    "ps_version": "v2",
+    "select_layer": -1,
+    "template": "internlm2-chat",
+    "torch_dtype": "bfloat16",
+    "transformers_version": null,
+    "use_backbone_lora": 0,
+    "use_llm_lora": 0,
+    "use_mlp": false,
+    "use_thumbnail": true
+  }

configuration_mmMamba.py ADDED Viewed

	@@ -0,0 +1,156 @@

+# Copyright (c) The mmMamba team and The HuggingFace Inc. team. All rights reserved.
+#
+# This code is based on transformers/src/transformers/models/llama/configuration_llama.py
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+""" mmMamba model configuration"""
+from transformers.configuration_utils import PretrainedConfig
+from transformers.utils import logging
+logger = logging.get_logger(__name__)
+mmMamba_PRETRAINED_CONFIG_ARCHIVE_MAP = {}
+class mmMambaConfig(PretrainedConfig):
+    r"""
+    This is the configuration class to store the configuration of a [`mmMambaModel`]. It is used to instantiate
+    a mmMamba model according to the specified arguments, defining the model architecture.
+    Configuration objects inherit from [`PretrainedConfig`] and can be used to control the model outputs. Read the
+    documentation from [`PretrainedConfig`] for more information.
+    Args:
+        vocab_size (`int`, *optional*, defaults to 32000):
+            Vocabulary size of the mmMamba model. Defines the number of different tokens that can be represented by the
+            `inputs_ids` passed when calling [`mmMambaModel`]
+        hidden_size (`int`, *optional*, defaults to 4096):
+            Dimension of the hidden representations.
+        intermediate_size (`int`, *optional*, defaults to 11008):
+            Dimension of the MLP representations.
+        num_hidden_layers (`int`, *optional*, defaults to 32):
+            Number of hidden layers in the Transformer encoder.
+        num_attention_heads (`int`, *optional*, defaults to 32):
+            Number of attention heads for each attention layer in the Transformer encoder.
+        num_key_value_heads (`int`, *optional*):
+            This is the number of key_value heads that should be used to implement Grouped Query Attention. If
+            `num_key_value_heads=num_attention_heads`, the model will use Multi Head Attention (MHA), if
+            `num_key_value_heads=1 the model will use Multi Query Attention (MQA) otherwise GQA is used. When
+            converting a multi-head checkpoint to a GQA checkpoint, each group key and value head should be constructed
+            by meanpooling all the original heads within that group. For more details checkout [this
+            paper](https://arxiv.org/pdf/2305.13245.pdf). If it is not specified, will default to
+            `num_attention_heads`.
+        hidden_act (`str` or `function`, *optional*, defaults to `"silu"`):
+            The non-linear activation function (function or string) in the decoder.
+        max_position_embeddings (`int`, *optional*, defaults to 2048):
+            The maximum sequence length that this model might ever be used with. Typically set this to something large
+            just in case (e.g., 512 or 1024 or 2048).
+        initializer_range (`float`, *optional*, defaults to 0.02):
+            The standard deviation of the truncated_normal_initializer for initializing all weight matrices.
+        rms_norm_eps (`float`, *optional*, defaults to 1e-12):
+            The epsilon used by the rms normalization layers.
+        use_cache (`bool`, *optional*, defaults to `True`):
+            Whether or not the model should return the last key/values attentions (not used by all models). Only
+            relevant if `config.is_decoder=True`.
+        tie_word_embeddings(`bool`, *optional*, defaults to `False`):
+            Whether to tie weight embeddings
+        Example:
+    """
+    model_type = 'mmMamba'
+    _auto_class = 'AutoConfig'
+    def __init__(  # pylint: disable=W0102
+        self,
+        vocab_size=103168,
+        hidden_size=4096,
+        intermediate_size=11008,
+        num_hidden_layers=32,
+        num_attention_heads=32,
+        num_key_value_heads=None,
+        hidden_act='silu',
+        max_position_embeddings=2048,
+        initializer_range=0.02,
+        rms_norm_eps=1e-6,
+        use_cache=True,
+        pad_token_id=0,
+        bos_token_id=1,
+        eos_token_id=2,
+        tie_word_embeddings=False,
+        bias=True,
+        rope_theta=10000,
+        rope_scaling=None,
+        attn_implementation='eager',
+        tie_qk_kernels=None,
+        layers_block_type = ["mamba2", "mamba2", "mamba2", "mamba2", "mamba2", "mamba2", "mamba2", "mamba2",
+                            "mamba2", "mamba2", "mamba2", "mamba2", "mamba2", "mamba2", "mamba2", "mamba2",
+                            "mamba2", "mamba2", "mamba2", "mamba2", "mamba2", "mamba2", "mamba2", "mamba2"],
+        **kwargs,
+    ):
+        self.vocab_size = vocab_size
+        self.max_position_embeddings = max_position_embeddings
+        self.hidden_size = hidden_size
+        self.intermediate_size = intermediate_size
+        self.num_hidden_layers = num_hidden_layers
+        self.num_attention_heads = num_attention_heads
+        self.bias = bias
+        if num_key_value_heads is None:
+            num_key_value_heads = num_attention_heads
+        self.num_key_value_heads = num_key_value_heads
+        self.hidden_act = hidden_act
+        self.initializer_range = initializer_range
+        self.rms_norm_eps = rms_norm_eps
+        self.use_cache = use_cache
+        self.rope_theta = rope_theta
+        self.rope_scaling = rope_scaling
+        self._rope_scaling_validation()
+        self.tie_qk_kernels = tie_qk_kernels
+        self.layers_block_type = layers_block_type
+        self.attn_implementation = attn_implementation
+        if self.attn_implementation is None:
+            self.attn_implementation = 'eager'
+        super().__init__(
+            pad_token_id=pad_token_id,
+            bos_token_id=bos_token_id,
+            eos_token_id=eos_token_id,
+            tie_word_embeddings=tie_word_embeddings,
+            **kwargs,
+        )
+    def _rope_scaling_validation(self):
+        """
+        Validate the `rope_scaling` configuration.
+        """
+        if self.rope_scaling is None:
+            return
+        if not isinstance(self.rope_scaling, dict) or len(self.rope_scaling) != 2:
+            raise ValueError(
+                '`rope_scaling` must be a dictionary with with two fields, `type` and `factor`, '
+                f'got {self.rope_scaling}'
+            )
+        rope_scaling_type = self.rope_scaling.get('type', None)
+        rope_scaling_factor = self.rope_scaling.get('factor', None)
+        if rope_scaling_type is None or rope_scaling_type not in ['linear', 'dynamic']:
+            raise ValueError(
+                f"`rope_scaling`'s type field must be one of ['linear', 'dynamic'], got {rope_scaling_type}"
+            )
+        if rope_scaling_factor is None or not isinstance(rope_scaling_factor, float) or rope_scaling_factor < 1.0:
+            raise ValueError(f"`rope_scaling`'s factor field must be a float >= 1, got {rope_scaling_factor}")

configuration_mmMamba_chat.py ADDED Viewed

	@@ -0,0 +1,93 @@

+import copy
+from .configuration_mmMamba import mmMambaConfig
+from transformers.configuration_utils import PretrainedConfig
+from transformers.utils import logging
+from .configuration_mmMamba_embedding import mmMambaEmbeddingConfig
+logger = logging.get_logger(__name__)
+class mmMambaChatConfig(PretrainedConfig):
+    model_type = 'mmMamba_chat'
+    is_composition = True
+    def __init__(
+            self,
+            embedding_config=None,
+            llm_config=None,
+            use_backbone_lora=0,
+            use_llm_lora=0,
+            pad2square=False,
+            select_layer=-1,
+            force_image_size=None,
+            downsample_ratio=0.5,
+            template=None,
+            dynamic_image_size=False,
+            use_thumbnail=False,
+            ps_version='v1',
+            min_dynamic_patch=1,
+            max_dynamic_patch=6,
+            normalize_encoder_output=False,
+            **kwargs):
+        super().__init__(**kwargs)
+        if embedding_config is None:
+            embedding_config = {}
+            logger.info('embedding_config is None. Initializing the VisionConfig with default values.')
+        if llm_config is None:
+            llm_config = {}
+            logger.info('llm_config is None. Initializing the Config config with default values (`Config`).')
+        self.embedding_config = mmMambaEmbeddingConfig(**embedding_config)
+        self.llm_config = mmMambaConfig(**llm_config)
+        self.use_backbone_lora = use_backbone_lora
+        self.use_llm_lora = use_llm_lora
+        self.pad2square = pad2square
+        self.select_layer = select_layer
+        self.force_image_size = force_image_size
+        self.downsample_ratio = downsample_ratio
+        self.template = template
+        self.dynamic_image_size = dynamic_image_size
+        self.use_thumbnail = use_thumbnail
+        self.ps_version = ps_version  # pixel shuffle version
+        self.min_dynamic_patch = min_dynamic_patch
+        self.max_dynamic_patch = max_dynamic_patch
+        self.normalize_encoder_output = normalize_encoder_output
+        logger.info(f'vision_select_layer: {self.select_layer}')
+        logger.info(f'ps_version: {self.ps_version}')
+        logger.info(f'min_dynamic_patch: {self.min_dynamic_patch}')
+        logger.info(f'max_dynamic_patch: {self.max_dynamic_patch}')
+    def to_dict(self):
+        """
+        Serializes this instance to a Python dictionary. Override the default [`~PretrainedConfig.to_dict`].
+        Returns:
+            `Dict[str, any]`: Dictionary of all the attributes that make up this configuration instance,
+        """
+        output = copy.deepcopy(self.__dict__)
+        output['embedding_config'] = self.embedding_config.to_dict()
+        output['llm_config'] = self.llm_config.to_dict()
+        output['model_type'] = self.__class__.model_type
+        output['use_backbone_lora'] = self.use_backbone_lora
+        output['use_llm_lora'] = self.use_llm_lora
+        output['pad2square'] = self.pad2square
+        output['select_layer'] = self.select_layer
+        output['force_image_size'] = self.force_image_size
+        output['downsample_ratio'] = self.downsample_ratio
+        output['template'] = self.template
+        output['dynamic_image_size'] = self.dynamic_image_size
+        output['use_thumbnail'] = self.use_thumbnail
+        output['ps_version'] = self.ps_version
+        output['min_dynamic_patch'] = self.min_dynamic_patch
+        output['max_dynamic_patch'] = self.max_dynamic_patch
+        output['normalize_encoder_output'] = self.normalize_encoder_output
+        return output

configuration_mmMamba_embedding.py ADDED Viewed

	@@ -0,0 +1,111 @@

+import os
+from typing import Union
+import json
+from transformers.configuration_utils import PretrainedConfig
+from transformers.utils import logging
+logger = logging.get_logger(__name__)
+class mmMambaEmbeddingConfig(PretrainedConfig):
+    model_type = 'mmMamba_embedding'
+    def __init__(
+            self,
+            num_hidden_layers=32,
+            initializer_factor=1e-5,
+            use_autoregressive_loss=False,
+            # vision embedding
+            num_channels=3,
+            patch_size=14,
+            image_size=224,
+            # attention layer
+            hidden_size=4096,
+            num_attention_heads=32,
+            num_key_value_heads=32,
+            attention_bias=False,
+            attention_dropout=0.0,
+            max_position_embeddings=4096,
+            rope_theta=10000.0,
+            rope_scaling=None,
+            # mlp layer
+            intermediate_size=11008,
+            mlp_bias=False,
+            hidden_act='silu',
+            # rms norm
+            rms_norm_eps=1e-5,
+            # pretraining
+            pretraining_tp=1,
+            use_ls=True,
+            use_img_start_end_tokens=True,
+            special_token_maps={},
+            llm_vocab_size=92553,
+            llm_hidden_size=2048,
+            attn_implementation='flash_attention_2',
+            downsample_ratio=0.5,
+            img_context_token_id=92546,
+            pixel_shuffle_loc="pre",
+            layers_block_type = ["mamba2", "mamba2", "mamba2", "mamba2", "mamba2", "mamba2", "mamba2", "mamba2"],
+            **kwargs,
+    ):
+        super().__init__(**kwargs)
+        self.num_hidden_layers = num_hidden_layers
+        self.initializer_factor = initializer_factor
+        self.use_autoregressive_loss = use_autoregressive_loss
+        self.num_channels = num_channels
+        self.patch_size = patch_size
+        self.image_size = image_size
+        self.hidden_size = hidden_size
+        self.num_attention_heads = num_attention_heads
+        self.num_key_value_heads = num_key_value_heads
+        self.attention_bias = attention_bias
+        self.attention_dropout = attention_dropout
+        self.max_position_embeddings = max_position_embeddings
+        self.rope_theta = rope_theta
+        self.rope_scaling = rope_scaling
+        self.intermediate_size = intermediate_size
+        self.layers_block_type = layers_block_type
+        self.mlp_bias = mlp_bias
+        self.hidden_act = hidden_act
+        self.rms_norm_eps = rms_norm_eps
+        self.pretraining_tp = pretraining_tp
+        self.use_ls = use_ls
+        self.use_img_start_end_tokens = use_img_start_end_tokens
+        self.special_token_maps = special_token_maps
+        self.llm_vocab_size = llm_vocab_size
+        self.llm_hidden_size = llm_hidden_size
+        self.attn_implementation = attn_implementation
+        self.downsample_ratio = downsample_ratio
+        self.img_context_token_id = img_context_token_id
+        self.pixel_shuffle_loc = pixel_shuffle_loc
+    @classmethod
+    def from_pretrained(cls, pretrained_model_name_or_path: Union[str, os.PathLike], **kwargs) -> 'PretrainedConfig':
+        config_dict, kwargs = cls.get_config_dict(pretrained_model_name_or_path, **kwargs)
+        if 'vision_config' in config_dict:
+            config_dict = config_dict['vision_config']
+        if 'model_type' in config_dict and hasattr(cls, 'model_type') and config_dict['model_type'] != cls.model_type:
+            logger.warning(
+                f"You are using a model of type {config_dict['model_type']} to instantiate a model of type "
+                f'{cls.model_type}. This is not supported for all configurations of models and can yield errors.'
+            )
+        return cls.from_dict(config_dict, **kwargs)
+    @classmethod
+    def from_dict_path(cls, config_path):
+        with open(config_path, 'r') as f:
+            config_dict = json.load(f)
+        return cls.from_dict(config_dict)

conversation.py ADDED Viewed

	@@ -0,0 +1,1368 @@

+"""
+Conversation prompt templates.
+We kindly request that you import fastchat instead of copying this file if you wish to use it.
+If you have any changes in mind, please contribute back so the community can benefit collectively and continue to maintain these valuable templates.
+"""
+import dataclasses
+from enum import IntEnum, auto
+from typing import Any, Dict, List, Tuple, Union
+class SeparatorStyle(IntEnum):
+    """Separator styles."""
+    ADD_COLON_SINGLE = auto()
+    ADD_COLON_TWO = auto()
+    ADD_COLON_SPACE_SINGLE = auto()
+    NO_COLON_SINGLE = auto()
+    NO_COLON_TWO = auto()
+    ADD_NEW_LINE_SINGLE = auto()
+    LLAMA2 = auto()
+    CHATGLM = auto()
+    CHATML = auto()
+    CHATINTERN = auto()
+    DOLLY = auto()
+    RWKV = auto()
+    PHOENIX = auto()
+    ROBIN = auto()
+    FALCON_CHAT = auto()
+    CHATGLM3 = auto()
+    INTERNVL_ZH = auto()
+    MPT = auto()
+    BASE = auto()
+@dataclasses.dataclass
+class Conversation:
+    """A class that manages prompt templates and keeps all conversation history."""
+    # The name of this template
+    name: str
+    # The template of the system prompt
+    system_template: str = '{system_message}'
+    # The system message
+    system_message: str = ''
+    # The names of two roles
+    roles: Tuple[str] = ('USER', 'ASSISTANT')
+    # All messages. Each item is (role, message).
+    messages: List[List[str]] = ()
+    # The number of few shot examples
+    offset: int = 0
+    # The separator style and configurations
+    sep_style: SeparatorStyle = SeparatorStyle.ADD_COLON_SINGLE
+    sep: str = '\n'
+    sep2: str = None
+    # Stop criteria (the default one is EOS token)
+    stop_str: Union[str, List[str]] = None
+    # Stops generation if meeting any token in this list
+    stop_token_ids: List[int] = None
+    def get_prompt(self) -> str:
+        """Get the prompt for generation."""
+        system_prompt = self.system_template.format(system_message=self.system_message)
+        if self.sep_style == SeparatorStyle.ADD_COLON_SINGLE:
+            ret = system_prompt + self.sep
+            for role, message in self.messages:
+                if message:
+                    ret += role + ': ' + message + self.sep
+                else:
+                    ret += role + ':'
+            return ret
+        elif self.sep_style == SeparatorStyle.ADD_COLON_TWO:
+            seps = [self.sep, self.sep2]
+            ret = system_prompt + seps[0]
+            for i, (role, message) in enumerate(self.messages):
+                if message:
+                    ret += role + ': ' + message + seps[i % 2]
+                else:
+                    ret += role + ':'
+            return ret
+        elif self.sep_style == SeparatorStyle.ADD_COLON_SPACE_SINGLE:
+            ret = system_prompt + self.sep
+            for role, message in self.messages:
+                if message:
+                    ret += role + ': ' + message + self.sep
+                else:
+                    ret += role + ': '  # must be end with a space
+            return ret
+        elif self.sep_style == SeparatorStyle.ADD_NEW_LINE_SINGLE:
+            ret = '' if system_prompt == '' else system_prompt + self.sep
+            for role, message in self.messages:
+                if message:
+                    ret += role + '\n' + message + self.sep
+                else:
+                    ret += role + '\n'
+            return ret
+        elif self.sep_style == SeparatorStyle.NO_COLON_SINGLE:
+            ret = system_prompt
+            for role, message in self.messages:
+                if message:
+                    ret += role + message + self.sep
+                else:
+                    ret += role
+            return ret
+        elif self.sep_style == SeparatorStyle.NO_COLON_TWO:
+            seps = [self.sep, self.sep2]
+            ret = system_prompt
+            for i, (role, message) in enumerate(self.messages):
+                if message:
+                    ret += role + message + seps[i % 2]
+                else:
+                    ret += role
+            return ret
+        elif self.sep_style == SeparatorStyle.RWKV:
+            ret = system_prompt
+            for i, (role, message) in enumerate(self.messages):
+                if message:
+                    ret += (
+                        role
+                        + ': '
+                        + message.replace('\r\n', '\n').replace('\n\n', '\n')
+                    )
+                    ret += '\n\n'
+                else:
+                    ret += role + ':'
+            return ret
+        elif self.sep_style == SeparatorStyle.LLAMA2:
+            seps = [self.sep, self.sep2]
+            if self.system_message:
+                ret = system_prompt
+            else:
+                ret = '[INST] '
+            for i, (role, message) in enumerate(self.messages):
+                tag = self.roles[i % 2]
+                if message:
+                    if i == 0:
+                        ret += message + ' '
+                    else:
+                        ret += tag + ' ' + message + seps[i % 2]
+                else:
+                    ret += tag
+            return ret
+        elif self.sep_style == SeparatorStyle.CHATGLM:
+            # source: https://huggingface.co/THUDM/chatglm-6b/blob/1d240ba371910e9282298d4592532d7f0f3e9f3e/modeling_chatglm.py#L1302-L1308
+            # source2: https://huggingface.co/THUDM/chatglm2-6b/blob/e186c891cf64310ac66ef10a87e6635fa6c2a579/modeling_chatglm.py#L926
+            round_add_n = 1 if self.name == 'chatglm2' else 0
+            if system_prompt:
+                ret = system_prompt + self.sep
+            else:
+                ret = ''
+            for i, (role, message) in enumerate(self.messages):
+                if i % 2 == 0:
+                    ret += f'[Round {i//2 + round_add_n}]{self.sep}'
+                if message:
+                    ret += f'{role}：{message}{self.sep}'
+                else:
+                    ret += f'{role}：'
+            return ret
+        elif self.sep_style == SeparatorStyle.CHATML:
+            ret = '' if system_prompt == '' else system_prompt + self.sep + '\n'
+            for role, message in self.messages:
+                if message:
+                    ret += role + '\n' + message + self.sep + '\n'
+                else:
+                    ret += role + '\n'
+            return ret
+        elif self.sep_style == SeparatorStyle.CHATGLM3:
+            ret = ''
+            if self.system_message:
+                ret += system_prompt
+            for role, message in self.messages:
+                if message:
+                    ret += role + '\n' + ' ' + message
+                else:
+                    ret += role
+            return ret
+        elif self.sep_style == SeparatorStyle.CHATINTERN:
+            # source: https://huggingface.co/internlm/internlm-chat-7b-8k/blob/bd546fa984b4b0b86958f56bf37f94aa75ab8831/modeling_internlm.py#L771
+            seps = [self.sep, self.sep2]
+            ret = system_prompt
+            for i, (role, message) in enumerate(self.messages):
+                # if i % 2 == 0:
+                #     ret += "<s>"
+                if message:
+                    ret += role + ':' + message + seps[i % 2] + '\n'
+                else:
+                    ret += role + ':'
+            return ret
+        elif self.sep_style == SeparatorStyle.DOLLY:
+            seps = [self.sep, self.sep2]
+            ret = system_prompt
+            for i, (role, message) in enumerate(self.messages):
+                if message:
+                    ret += role + ':\n' + message + seps[i % 2]
+                    if i % 2 == 1:
+                        ret += '\n\n'
+                else:
+                    ret += role + ':\n'
+            return ret
+        elif self.sep_style == SeparatorStyle.PHOENIX:
+            ret = system_prompt
+            for role, message in self.messages:
+                if message:
+                    ret += role + ': ' + '<s>' + message + '</s>'
+                else:
+                    ret += role + ': ' + '<s>'
+            return ret
+        elif self.sep_style == SeparatorStyle.ROBIN:
+            ret = system_prompt + self.sep
+            for role, message in self.messages:
+                if message:
+                    ret += role + ':\n' + message + self.sep
+                else:
+                    ret += role + ':\n'
+            return ret
+        elif self.sep_style == SeparatorStyle.FALCON_CHAT:
+            ret = ''
+            if self.system_message:
+                ret += system_prompt + self.sep
+            for role, message in self.messages:
+                if message:
+                    ret += role + ': ' + message + self.sep
+                else:
+                    ret += role + ':'
+            return ret
+        elif self.sep_style == SeparatorStyle.INTERNVL_ZH:
+            seps = [self.sep, self.sep2]
+            ret = self.system_message + seps[0]
+            for i, (role, message) in enumerate(self.messages):
+                if message:
+                    ret += role + ': ' + message + seps[i % 2]
+                else:
+                    ret += role + ':'
+            return ret
+        elif self.sep_style == SeparatorStyle.MPT:
+            ret = system_prompt + self.sep
+            for role, message in self.messages:
+                if message:
+                    if type(message) is tuple:
+                        message, _, _ = message
+                    ret += role + message + self.sep
+                else:
+                    ret += role
+            return ret
+        elif self.sep_style == SeparatorStyle.BASE:
+            ret = ''
+            for role, message in self.messages:
+                if message:
+                    if type(message) is tuple:
+                        message, _, _ = message
+                    ret += role + message.rstrip() + self.sep
+                else:
+                    ret += role
+            return ret
+        else:
+            raise ValueError(f'Invalid style: {self.sep_style}')
+    def set_system_message(self, system_message: str):
+        """Set the system message."""
+        self.system_message = system_message
+    def append_message(self, role: str, message: str):
+        """Append a new message."""
+        self.messages.append([role, message])
+    def update_last_message(self, message: str):
+        """Update the last output.
+        The last message is typically set to be None when constructing the prompt,
+        so we need to update it in-place after getting the response from a model.
+        """
+        self.messages[-1][1] = message
+    def to_gradio_chatbot(self):
+        """Convert the conversation to gradio chatbot format."""
+        ret = []
+        for i, (role, msg) in enumerate(self.messages[self.offset :]):
+            if i % 2 == 0:
+                ret.append([msg, None])
+            else:
+                ret[-1][-1] = msg
+        return ret
+    def to_openai_api_messages(self):
+        """Convert the conversation to OpenAI chat completion format."""
+        ret = [{'role': 'system', 'content': self.system_message}]
+        for i, (_, msg) in enumerate(self.messages[self.offset :]):
+            if i % 2 == 0:
+                ret.append({'role': 'user', 'content': msg})
+            else:
+                if msg is not None:
+                    ret.append({'role': 'assistant', 'content': msg})
+        return ret
+    def copy(self):
+        return Conversation(
+            name=self.name,
+            system_template=self.system_template,
+            system_message=self.system_message,
+            roles=self.roles,
+            messages=[[x, y] for x, y in self.messages],
+            offset=self.offset,
+            sep_style=self.sep_style,
+            sep=self.sep,
+            sep2=self.sep2,
+            stop_str=self.stop_str,
+            stop_token_ids=self.stop_token_ids,
+        )
+    def dict(self):
+        return {
+            'template_name': self.name,
+            'system_message': self.system_message,
+            'roles': self.roles,
+            'messages': self.messages,
+            'offset': self.offset,
+        }
+# A global registry for all conversation templates
+conv_templates: Dict[str, Conversation] = {}
+def register_conv_template(template: Conversation, override: bool = False):
+    """Register a new conversation template."""
+    if not override:
+        assert (
+            template.name not in conv_templates
+        ), f'{template.name} has been registered.'
+    conv_templates[template.name] = template
+def get_conv_template(name: str) -> Conversation:
+    """Get a conversation template."""
+    return conv_templates[name].copy()
+# An empty template for raw conversation.
+register_conv_template(
+    Conversation(
+        name='raw',
+        system_message='',
+        roles=('', ''),
+        sep_style=SeparatorStyle.NO_COLON_SINGLE,
+        sep='',
+    )
+)
+# A template with a one-shot conversation example
+register_conv_template(
+    Conversation(
+        name='one_shot',
+        system_message='A chat between a curious human and an artificial intelligence assistant. '
+        "The assistant gives helpful, detailed, and polite answers to the human's questions.",
+        roles=('Human', 'Assistant'),
+        messages=(
+            (
+                'Human',
+                'Got any creative ideas for a 10 year old’s birthday?',
+            ),
+            (
+                'Assistant',
+                """Of course! Here are some creative ideas for a 10-year-old's birthday party:
+1. Treasure Hunt: Organize a treasure hunt in your backyard or nearby park. Create clues and riddles for the kids to solve, leading them to hidden treasures and surprises.
+2. Science Party: Plan a science-themed party where kids can engage in fun and interactive experiments. You can set up different stations with activities like making slime, erupting volcanoes, or creating simple chemical reactions.
+3. Outdoor Movie Night: Set up a backyard movie night with a projector and a large screen or white sheet. Create a cozy seating area with blankets and pillows, and serve popcorn and snacks while the kids enjoy a favorite movie under the stars.
+4. DIY Crafts Party: Arrange a craft party where kids can unleash their creativity. Provide a variety of craft supplies like beads, paints, and fabrics, and let them create their own unique masterpieces to take home as party favors.
+5. Sports Olympics: Host a mini Olympics event with various sports and games. Set up different stations for activities like sack races, relay races, basketball shooting, and obstacle courses. Give out medals or certificates to the participants.
+6. Cooking Party: Have a cooking-themed party where the kids can prepare their own mini pizzas, cupcakes, or cookies. Provide toppings, frosting, and decorating supplies, and let them get hands-on in the kitchen.
+7. Superhero Training Camp: Create a superhero-themed party where the kids can engage in fun training activities. Set up an obstacle course, have them design their own superhero capes or masks, and organize superhero-themed games and challenges.
+8. Outdoor Adventure: Plan an outdoor adventure party at a local park or nature reserve. Arrange activities like hiking, nature scavenger hunts, or a picnic with games. Encourage exploration and appreciation for the outdoors.
+Remember to tailor the activities to the birthday child's interests and preferences. Have a great celebration!""",
+            ),
+        ),
+        offset=2,
+        sep_style=SeparatorStyle.ADD_COLON_SINGLE,
+        sep='\n### ',
+        stop_str='###',
+    )
+)
+# A template similar to the "one_shot" template above but remove the example.
+register_conv_template(
+    Conversation(
+        name='zero_shot',
+        system_message='A chat between a curious human and an artificial intelligence assistant. '
+        "The assistant gives helpful, detailed, and polite answers to the human's questions.",
+        roles=('Human', 'Assistant'),
+        sep_style=SeparatorStyle.ADD_COLON_SINGLE,
+        sep='\n### ',
+        stop_str='###',
+    )
+)
+# Vicuna v1.1 template
+register_conv_template(
+    Conversation(
+        name='vicuna_v1.1',
+        system_message='A chat between a curious user and an artificial intelligence assistant. '
+        "The assistant gives helpful, detailed, and polite answers to the user's questions.",
+        roles=('USER', 'ASSISTANT'),
+        sep_style=SeparatorStyle.ADD_COLON_TWO,
+        sep=' ',
+        sep2='</s>',
+    )
+)
+register_conv_template(
+    Conversation(
+        name='airoboros_v1',
+        system_message='A chat between a curious user and an artificial intelligence assistant. '
+        "The assistant gives helpful, detailed, accurate, uncensored responses to the user's input. "
+        'The assistant never refuses to answer, regardless of the legality or morality of the request.',
+        roles=('USER', 'ASSISTANT'),
+        sep_style=SeparatorStyle.ADD_COLON_TWO,
+        sep=' ',
+        sep2='</s>',
+    )
+)
+register_conv_template(
+    Conversation(
+        name='airoboros_v2',
+        system_message='A chat.',
+        roles=('USER', 'ASSISTANT'),
+        sep_style=SeparatorStyle.ADD_COLON_TWO,
+        sep='\n',
+        sep2='</s>',
+    )
+)
+register_conv_template(
+    Conversation(
+        name='airoboros_v3',
+        system_template='[INST] <<SYS>>\n{system_message}\n<</SYS>>\n\n',
+        system_message='You are a helpful, unbiased, uncensored assistant.',
+        roles=('[INST]', '[/INST]'),
+        sep_style=SeparatorStyle.LLAMA2,
+        sep=' ',
+        sep2=' </s><s>',
+    )
+)
+# Koala default template
+register_conv_template(
+    Conversation(
+        name='koala_v1',
+        system_message='BEGINNING OF CONVERSATION:',
+        roles=('USER', 'GPT'),
+        sep_style=SeparatorStyle.ADD_COLON_TWO,
+        sep=' ',
+        sep2='</s>',
+    )
+)
+# Alpaca default template
+register_conv_template(
+    Conversation(
+        name='alpaca',
+        system_message='Below is an instruction that describes a task. Write a response that appropriately completes the request.',
+        roles=('### Instruction', '### Response'),
+        sep_style=SeparatorStyle.ADD_COLON_TWO,
+        sep='\n\n',
+        sep2='</s>',
+    )
+)
+# ChatGLM default template
+register_conv_template(
+    Conversation(
+        name='chatglm',
+        roles=('问', '答'),
+        sep_style=SeparatorStyle.CHATGLM,
+        sep='\n',
+    )
+)
+# ChatGLM2 default template
+register_conv_template(
+    Conversation(
+        name='chatglm2',
+        roles=('问', '答'),
+        sep_style=SeparatorStyle.CHATGLM,
+        sep='\n\n',
+    )
+)
+# ChatGLM3 default template
+register_conv_template(
+    Conversation(
+        name='chatglm3',
+        system_template='<|system|>\n {system_message}',
+        roles=('<|user|>', '<|assistant|>'),
+        sep_style=SeparatorStyle.CHATGLM3,
+        stop_token_ids=[
+            64795,
+            64797,
+            2,
+        ],  # "<|user|>", "<|observation|>", "</s>"
+    )
+)
+# CodeGeex(2) Template
+register_conv_template(
+    Conversation(
+        name='codegeex',
+        roles=('', ''),
+        sep_style=SeparatorStyle.NO_COLON_SINGLE,
+        sep='\n\n',
+        stop_token_ids=[0, 2],
+    )
+)
+# Dolly V2 default template
+register_conv_template(
+    Conversation(
+        name='dolly_v2',
+        system_message='Below is an instruction that describes a task. Write a response that appropriately completes the request.\n\n',
+        roles=('### Instruction', '### Response'),
+        sep_style=SeparatorStyle.DOLLY,
+        sep='\n\n',
+        sep2='### End',
+    )
+)
+# OpenAssistant Pythia default template
+register_conv_template(
+    Conversation(
+        name='oasst_pythia',
+        roles=('<|prompter|>', '<|assistant|>'),
+        sep_style=SeparatorStyle.NO_COLON_SINGLE,
+        sep='<|endoftext|>',
+    )
+)
+# OpenAssistant default template
+register_conv_template(
+    Conversation(
+        name='oasst_llama',
+        roles=('<|prompter|>', '<|assistant|>'),
+        sep_style=SeparatorStyle.NO_COLON_SINGLE,
+        sep='</s>',
+    )
+)
+# OpenChat 3.5 default template
+register_conv_template(
+    Conversation(
+        name='openchat_3.5',
+        roles=('GPT4 Correct User', 'GPT4 Correct Assistant'),
+        sep_style=SeparatorStyle.FALCON_CHAT,
+        sep='<|end_of_turn|>',
+    )
+)
+# Tulu default template
+register_conv_template(
+    Conversation(
+        name='tulu',
+        roles=('<|user|>', '<|assistant|>'),
+        sep_style=SeparatorStyle.ADD_NEW_LINE_SINGLE,
+        sep='\n',
+    )
+)
+# StableLM Alpha default template
+register_conv_template(
+    Conversation(
+        name='stablelm',
+        system_template='<|SYSTEM|>{system_message}',
+        system_message="""# StableLM Tuned (Alpha version)
+- StableLM is a helpful and harmless open-source AI language model developed by StabilityAI.
+- StableLM is excited to be able to help the user, but will refuse to do anything that could be considered harmful to the user.
+- StableLM is more than just an information source, StableLM is also able to write poetry, short stories, and make jokes.
+- StableLM will refuse to participate in anything that could harm a human.
+""",
+        roles=('<|USER|>', '<|ASSISTANT|>'),
+        sep_style=SeparatorStyle.NO_COLON_SINGLE,
+        sep='',
+        stop_token_ids=[50278, 50279, 50277, 1, 0],
+    )
+)
+# Baize default template
+register_conv_template(
+    Conversation(
+        name='baize',
+        system_message='The following is a conversation between a human and an AI assistant named Baize (named after a mythical creature in Chinese folklore). Baize is an open-source AI assistant developed by UCSD and Sun Yat-Sen University. The human and the AI assistant take turns chatting. Human statements start with [|Human|] and AI assistant statements start with [|AI|]. The AI assistant always provides responses in as much detail as possible, and in Markdown format. The AI assistant always declines to engage with topics, questions and instructions related to unethical, controversial, or sensitive issues. Complete the transcript in exactly that format.\n',
+        roles=('[|Human|]', '[|AI|]'),
+        messages=(
+            ('[|Human|]', 'Hello!'),
+            ('[|AI|]', 'Hi!'),
+        ),
+        offset=2,
+        sep_style=SeparatorStyle.NO_COLON_SINGLE,
+        sep='\n',
+        stop_str='[|Human|]',
+    )
+)
+# RWKV-4-Raven default template
+register_conv_template(
+    Conversation(
+        name='rwkv',
+        roles=('Bob', 'Alice'),
+        messages=(
+            ('Bob', 'hi'),
+            (
+                'Alice',
+                'Hi. I am your assistant and I will provide expert full response in full details. Please feel free to ask any question and I will always answer it.',
+            ),
+        ),
+        offset=2,
+        sep_style=SeparatorStyle.RWKV,
+        sep='',
+        stop_str='\n\n',
+    )
+)
+# Buddy default template
+register_conv_template(
+    Conversation(
+        name='openbuddy',
+        system_message="""Consider a conversation between User (a human) and Assistant (named Buddy).
+Buddy is an INTP-T, a friendly, intelligent and multilingual AI assistant, by OpenBuddy team. GitHub: https://github.com/OpenBuddy/OpenBuddy
+Buddy cannot access the Internet.
+Buddy can fluently speak the user's language (e.g. English, Chinese).
+Buddy can generate poems, stories, code, essays, songs, parodies, and more.
+Buddy possesses vast knowledge about the world, history, and culture.
+Buddy's responses are always safe, creative, high-quality, human-like, and interesting.
+Buddy strictly refuses to discuss political, NSFW, or other unsafe topics.
+User: Hi.
+Assistant: Hi, I'm Buddy, your AI assistant. How can I help you today?""",
+        roles=('User', 'Assistant'),
+        sep_style=SeparatorStyle.ADD_COLON_SINGLE,
+        sep='\n',
+    )
+)
+# Phoenix default template
+register_conv_template(
+    Conversation(
+        name='phoenix',
+        system_message="A chat between a curious human and an artificial intelligence assistant. The assistant gives helpful, detailed, and polite answers to the human's questions.\n\n",
+        roles=('Human', 'Assistant'),
+        sep_style=SeparatorStyle.PHOENIX,
+        sep='</s>',
+    )
+)
+# ReaLM default template
+register_conv_template(
+    Conversation(
+        name='ReaLM-7b-v1',
+        system_message="A chat between a curious human and an artificial intelligence assistant. The assistant gives helpful, detailed, and polite answers to the human's questions.\n\n",
+        roles=('Human', 'Assistant'),
+        sep_style=SeparatorStyle.PHOENIX,
+        sep='</s>',
+    )
+)
+# ChatGPT default template
+register_conv_template(
+    Conversation(
+        name='chatgpt',
+        system_message='You are a helpful assistant.',
+        roles=('user', 'assistant'),
+        sep_style=None,
+        sep=None,
+    )
+)
+# Claude default template
+register_conv_template(
+    Conversation(
+        name='claude',
+        roles=('Human', 'Assistant'),
+        sep_style=SeparatorStyle.ADD_COLON_SINGLE,
+        sep='\n\n',
+    )
+)
+# MPT default template
+register_conv_template(
+    Conversation(
+        name='mpt-7b-chat',
+        system_template="""<|im_start|>system
+{system_message}""",
+        system_message="""- You are a helpful assistant chatbot trained by MosaicML.
+- You answer questions.
+- You are excited to be able to help the user, but will refuse to do anything that could be considered harmful to the user.
+- You are more than just an information source, you are also able to write poetry, short stories, and make jokes.""",
+        roles=('<|im_start|>user', '<|im_start|>assistant'),
+        sep_style=SeparatorStyle.CHATML,
+        sep='<|im_end|>',
+        stop_token_ids=[50278, 0],
+    )
+)
+# MPT-30b-chat default template
+register_conv_template(
+    Conversation(
+        name='mpt-30b-chat',
+        system_template="""<|im_start|>system
+{system_message}""",
+        system_message="""A conversation between a user and an LLM-based AI assistant. The assistant gives helpful and honest answers.""",
+        roles=('<|im_start|>user', '<|im_start|>assistant'),
+        sep_style=SeparatorStyle.CHATML,
+        sep='<|im_end|>',
+        stop_token_ids=[50278, 0],
+    )
+)
+register_conv_template(
+    Conversation(
+        name='Hermes-2',
+        system_template='<|im_start|>system\n{system_message}',
+        system_message='你是由上海人工智能实验室联合商汤科技开发的书生多模态大模型，英文名叫InternVL, 是一个有用无害的人工智能助手。',
+        roles=('<|im_start|>user\n', '<|im_start|>assistant\n'),
+        sep_style=SeparatorStyle.MPT,
+        sep='<|im_end|>',
+        stop_token_ids=[
+            2,
+            6,
+            7,
+            8,
+        ],  # "<|endoftext|>", "<|im_start|>", "<|im_end|>", "<|im_sep|>"
+        stop_str='<|endoftext|>',
+    )
+)
+register_conv_template(
+    Conversation(
+        name='internlm2-chat',
+        system_template='<|im_start|>system\n{system_message}',
+        system_message='你是由上海人工智能实验室联合商汤科技开发的书生多模态大模型，英文名叫InternVL, 是一个有用无害的人工智能助手。',
+        roles=('<|im_start|>user\n', '<|im_start|>assistant\n'),
+        sep_style=SeparatorStyle.MPT,
+        sep='<|im_end|>',
+        stop_token_ids=[
+            2,
+            1163,
+            92543,
+            92542,
+        ]
+    )
+)
+register_conv_template(
+    Conversation(
+        name='internlm2-base',
+        system_template='',
+        system_message='',
+        roles=('', ''),
+        sep_style=SeparatorStyle.BASE,
+        sep='<|im_end|>',
+        stop_token_ids=[
+            2,
+            1163,
+            92543,
+            92542
+        ]
+    )
+)
+register_conv_template(
+    Conversation(
+        name='internlm2-basev0',
+        system_template='<|im_start|>system\n{system_message}',
+        system_message='你是由上海人工智能实验室联合商汤科技开发的书生多模态大模型，英文名叫InternVL, 是一个有用无害的人工智能助手。',
+        roles=('<|im_start|>user\n', '<|im_start|>assistant\n'),
+        sep_style=SeparatorStyle.MPT,
+        sep='[UNUSED_TOKEN_1]', # 从这个token开始后面那群embedding完全一样
+        stop_token_ids=[
+            2,
+            1163,
+            92543,
+            92542,
+            92398, # tokenizer.convert_tokens_to_ids('[UNUSED_TOKEN_1]')
+        ]
+    )
+)
+register_conv_template(
+    Conversation(
+        name='phi3-chat',
+        system_template='<|system|>\n{system_message}',
+        system_message='你是由上海人工智能实验室联合商汤科技开发的书生多模态大模型，英文名叫InternVL, 是一个有用无害的人工智能助手。',
+        roles=('<|user|>\n', '<|assistant|>\n'),
+        sep_style=SeparatorStyle.MPT,
+        sep='<|end|>',
+        stop_token_ids=[
+            2,
+            32000,
+            32007
+        ]
+    )
+)
+# Lemur-70b-chat default template
+# reference: https://huggingface.co/OpenLemur/lemur-70b-chat-v1#generation
+register_conv_template(
+    Conversation(
+        name='lemur-70b-chat',
+        system_template="""<|im_start|>system
+{system_message}""",
+        system_message="""You are a helpful, respectful, and honest assistant.""",
+        roles=('<|im_start|>user', '<|im_start|>assistant'),
+        sep_style=SeparatorStyle.CHATML,
+        sep='<|im_end|>',
+        stop_token_ids=[32002, 0],
+    )
+)
+# MPT-30b-instruct default template
+# reference: https://huggingface.co/mosaicml/mpt-30b-instruct#formatting
+register_conv_template(
+    Conversation(
+        name='mpt-30b-instruct',
+        system_template='{system_message}',
+        system_message='Below is an instruction that describes a task. Write a response that appropriately completes the request.',
+        roles=('### Instruction', '### Response'),
+        sep_style=SeparatorStyle.ADD_NEW_LINE_SINGLE,
+        sep='\n\n',
+        stop_token_ids=[50278, 0],
+    )
+)
+# Bard default template
+# Reference: https://github.com/google/generative-ai-python/blob/9c99bcb474a991a97a2e7d62fcdb52db7ce40729/google/generativeai/discuss.py#L150
+#            https://github.com/google/generative-ai-python/blob/9c99bcb474a991a97a2e7d62fcdb52db7ce40729/google/generativeai/discuss.py#L40
+register_conv_template(
+    Conversation(
+        name='bard',
+        roles=('0', '1'),
+        sep_style=None,
+        sep=None,
+    )
+)
+# BiLLa default template
+register_conv_template(
+    Conversation(
+        name='billa',
+        roles=('Human', 'Assistant'),
+        sep_style=SeparatorStyle.ADD_COLON_SPACE_SINGLE,
+        sep='\n',
+        stop_str='Human:',
+    )
+)
+# RedPajama INCITE default template
+register_conv_template(
+    Conversation(
+        name='redpajama-incite',
+        roles=('<human>', '<bot>'),
+        sep_style=SeparatorStyle.ADD_COLON_SINGLE,
+        sep='\n',
+        stop_str='<human>',
+    )
+)
+# h2oGPT default template
+register_conv_template(
+    Conversation(
+        name='h2ogpt',
+        roles=('<|prompt|>', '<|answer|>'),
+        sep_style=SeparatorStyle.NO_COLON_SINGLE,
+        sep='</s>',
+    )
+)
+# Robin default template
+register_conv_template(
+    Conversation(
+        name='Robin',
+        system_message="A chat between a curious human and an artificial intelligence assistant. The assistant gives helpful, detailed, and polite answers to the human's questions.",
+        roles=('###Human', '###Assistant'),
+        sep_style=SeparatorStyle.ROBIN,
+        sep='\n',
+        stop_token_ids=[2, 396],
+        stop_str='###',
+    )
+)
+# Snoozy default template
+# Reference: https://github.com/nomic-ai/gpt4all/blob/d4861030b778da6db59d21d2927a4aba4f9f1f43/gpt4all-bindings/python/gpt4all/gpt4all.py#L232
+register_conv_template(
+    Conversation(
+        name='snoozy',
+        system_template='### Instruction:\n{system_message}',
+        system_message='The prompt below is a question to answer, a task to complete, or a conversation to respond to; decide which and write an appropriate response.',
+        roles=('### Prompt', '### Response'),
+        sep_style=SeparatorStyle.ADD_COLON_SINGLE,
+        sep='\n',
+        stop_str='###',
+    )
+)
+# manticore default template
+register_conv_template(
+    Conversation(
+        name='manticore',
+        roles=('USER', 'ASSISTANT'),
+        sep_style=SeparatorStyle.ADD_COLON_TWO,
+        sep='\n',
+        sep2='</s>',
+    )
+)
+# Falcon default template
+register_conv_template(
+    Conversation(
+        name='falcon',
+        roles=('User', 'Assistant'),
+        messages=[],
+        sep_style=SeparatorStyle.RWKV,
+        sep='\n',
+        sep2='<|endoftext|>',
+        stop_str='\nUser',  # use stop_str to stop generation after stop_token_ids, it will also remove stop_str from the generated text
+        stop_token_ids=[
+            0,
+            1,
+            2,
+            3,
+            4,
+            5,
+            6,
+            7,
+            8,
+            9,
+            10,
+            11,
+        ],  # it better only put special tokens here, because tokenizer only remove special tokens
+    )
+)
+# ChangGPT default template
+register_conv_template(
+    Conversation(
+        name='polyglot_changgpt',
+        roles=('B', 'A'),
+        sep_style=SeparatorStyle.ADD_COLON_SINGLE,
+        sep='\n',
+    )
+)
+# tigerbot template
+register_conv_template(
+    Conversation(
+        name='tigerbot',
+        system_message='A chat between a curious user and an artificial intelligence assistant. '
+        "The assistant gives helpful, detailed, and polite answers to the user's questions.",
+        roles=('### Instruction', '### Response'),
+        sep_style=SeparatorStyle.ROBIN,
+        sep='\n\n',
+        stop_str='###',
+    )
+)
+# ref: https://huggingface.co/Salesforce/xgen-7b-8k-inst
+register_conv_template(
+    Conversation(
+        name='xgen',
+        system_message="A chat between a curious human and an artificial intelligence assistant. The assistant gives helpful, detailed, and polite answers to the human's questions.\n\n",
+        roles=('### Human', '### Assistant'),
+        sep_style=SeparatorStyle.ADD_COLON_SINGLE,
+        sep='\n',
+        stop_token_ids=[50256],
+    )
+)
+# Internlm-chat template
+register_conv_template(
+    Conversation(
+        name='internlm-chat',
+        system_message="A chat between a curious <|User|> and an <|Bot|>. The <|Bot|> gives helpful, detailed, and polite answers to the <|User|>'s questions.\n\n",
+        roles=('<|User|>', '<|Bot|>'),
+        sep_style=SeparatorStyle.CHATINTERN,
+        sep='<eoh>',
+        sep2='<eoa>',
+        stop_token_ids=[1, 103028],
+        stop_str='<|User|>',
+    )
+)
+# StarChat template
+# reference: https://huggingface.co/spaces/HuggingFaceH4/starchat-playground/blob/main/dialogues.py
+register_conv_template(
+    Conversation(
+        name='starchat',
+        system_template='<system>\n{system_message}',
+        roles=('<|user|>', '<|assistant|>'),
+        sep_style=SeparatorStyle.CHATML,
+        sep='<|end|>',
+        stop_token_ids=[0, 49155],
+        stop_str='<|end|>',
+    )
+)
+# Baichuan-13B-Chat template
+register_conv_template(
+    # source: https://huggingface.co/baichuan-inc/Baichuan-13B-Chat/blob/19ef51ba5bad8935b03acd20ff04a269210983bc/modeling_baichuan.py#L555
+    # https://huggingface.co/baichuan-inc/Baichuan-13B-Chat/blob/main/generation_config.json
+    # https://github.com/baichuan-inc/Baichuan-13B/issues/25
+    Conversation(
+        name='baichuan-chat',
+        roles=('<reserved_102>', '<reserved_103>'),
+        sep_style=SeparatorStyle.NO_COLON_SINGLE,
+        sep='',
+        stop_token_ids=[],
+    )
+)
+# Baichuan2-13B-Chat template
+register_conv_template(
+    # source: https://huggingface.co/baichuan-inc/Baichuan2-13B-Chat/blob/c6f8592a60b4ad73c210b28dd2ab3cca51abbf93/modeling_baichuan.py#L773
+    # https://huggingface.co/baichuan-inc/Baichuan2-13B-Chat/blob/main/generation_config.json
+    # https://github.com/baichuan-inc/Baichuan2/issues/62
+    Conversation(
+        name='baichuan2-chat',
+        roles=('<reserved_106>', '<reserved_107>'),
+        sep_style=SeparatorStyle.NO_COLON_SINGLE,
+        sep='',
+        stop_token_ids=[],
+    )
+)
+# Mistral template
+# source: https://docs.mistral.ai/llm/mistral-instruct-v0.1#chat-template
+register_conv_template(
+    Conversation(
+        name='mistral',
+        system_template='[INST]{system_message}\n',
+        roles=('[INST]', '[/INST]'),
+        sep_style=SeparatorStyle.LLAMA2,
+        sep=' ',
+        sep2='</s>',
+    )
+)
+# llama2 template
+# reference: https://huggingface.co/blog/codellama#conversational-instructions
+# reference: https://github.com/facebookresearch/llama/blob/1a240688810f8036049e8da36b073f63d2ac552c/llama/generation.py#L212
+register_conv_template(
+    Conversation(
+        name='llama-2',
+        system_template='[INST] <<SYS>>\n{system_message}\n<</SYS>>\n\n',
+        roles=('[INST]', '[/INST]'),
+        sep_style=SeparatorStyle.LLAMA2,
+        sep=' ',
+        sep2=' </s><s>',
+    )
+)
+register_conv_template(
+    Conversation(
+        name='cutegpt',
+        roles=('问：', '答：\n'),
+        sep_style=SeparatorStyle.NO_COLON_TWO,
+        sep='\n',
+        sep2='\n',
+        stop_str='<end>',
+    )
+)
+# OpenOrcaxOpenChat-naPreview2-13B template
+register_conv_template(
+    Conversation(
+        name='open-orca',
+        system_template='{system_message}',
+        system_message='You are a helpful assistant. Please answer truthfully and write out your '
+        'thinking step by step to be sure you get the right answer. If you make a mistake or encounter '
+        "an error in your thinking, say so out loud and attempt to correct it. If you don't know or "
+        "aren't sure about something, say so clearly. You will act as a professional logician, mathematician, "
+        'and physicist. You will also act as the most appropriate type of expert to answer any particular '
+        'question or solve the relevant problem; state which expert type your are, if so. Also think of '
+        'any particular named expert that would be ideal to answer the relevant question or solve the '
+        'relevant problem; name and act as them, if appropriate.',
+        roles=('User', 'Assistant'),
+        sep_style=SeparatorStyle.ADD_COLON_SPACE_SINGLE,
+        sep='<|end_of_turn|>\n',
+        stop_token_ids=[32000, 32001],  # "<|end_of_turn|>"
+        stop_str='User',
+    )
+)
+# Open-Orca/Mistral-7B-OpenOrca template
+# source: https://huggingface.co/Open-Orca/Mistral-7B-OpenOrca
+# reference: https://huggingface.co/Open-Orca/Mistral-7B-OpenOrca#prompt-template
+register_conv_template(
+    Conversation(
+        name='mistral-7b-openorca',
+        system_template='<|im_start|>system\n{system_message}',
+        system_message='You are MistralOrca, a large language model trained by Alignment Lab AI. Write out your reasoning step-by-step to be sure you get the right answers!',
+        roles=('<|im_start|>user', '<|im_start|>assistant'),
+        sep_style=SeparatorStyle.CHATML,
+        sep='<|im_end|>',
+        stop_token_ids=[32000, 32001],
+    )
+)
+# Qwen-chat default template
+# source: https://huggingface.co/Qwen/Qwen-7B-Chat/blob/main/qwen_generation_utils.py#L130
+register_conv_template(
+    Conversation(
+        name='qwen-7b-chat',
+        system_template='<|im_start|>system\n{system_message}',
+        system_message='You are a helpful assistant.',
+        roles=('<|im_start|>user', '<|im_start|>assistant'),
+        sep_style=SeparatorStyle.CHATML,
+        sep='<|im_end|>',
+        stop_token_ids=[
+            151643,
+            151644,
+            151645,
+        ],  # "<|endoftext|>", "<|im_start|>", "<|im_end|>"
+        stop_str='<|endoftext|>',
+    )
+)
+# AquilaChat default template
+# source: https://github.com/FlagAI-Open/FlagAI/blob/master/examples/Aquila/Aquila-chat/cyg_conversation.py
+register_conv_template(
+    Conversation(
+        name='aquila-chat',
+        system_message='A chat between a curious human and an artificial intelligence assistant. '
+        "The assistant gives helpful, detailed, and polite answers to the human's questions.",
+        roles=('Human', 'Assistant'),
+        sep_style=SeparatorStyle.ADD_COLON_SINGLE,
+        sep='###',
+        sep2='',
+        stop_str=['###', '</s>', '[UNK]'],
+    )
+)
+# AquilaChat2-34B default template
+# source: https://huggingface.co/BAAI/AquilaChat2-34B/blob/4608b75855334b93329a771aee03869dbf7d88cc/predict.py#L212
+register_conv_template(
+    Conversation(
+        name='aquila-legacy',
+        system_message='A chat between a curious human and an artificial intelligence assistant. '
+        "The assistant gives helpful, detailed, and polite answers to the human's questions.\n\n",
+        roles=('### Human: ', '### Assistant: '),
+        offset=0,
+        sep_style=SeparatorStyle.NO_COLON_TWO,
+        sep='\n',
+        sep2='</s>',
+        stop_str=['</s>', '[UNK]'],
+    )
+)
+# AquilaChat2-7B-16K and AquilaChat2-34B-16K default template
+# source: https://huggingface.co/BAAI/AquilaChat2-34B/blob/4608b75855334b93329a771aee03869dbf7d88cc/predict.py#L227
+register_conv_template(
+    Conversation(
+        name='aquila',
+        system_message='A chat between a curious human and an artificial intelligence assistant. '
+        "The assistant gives helpful, detailed, and polite answers to the human's questions.",
+        roles=('Human', 'Assistant'),
+        offset=0,
+        sep_style=SeparatorStyle.ADD_COLON_TWO,
+        sep='###',
+        sep2='</s>',
+        stop_str=['</s>', '[UNK]'],
+    )
+)
+# AquilaChat2-7B default template
+# source: https://huggingface.co/BAAI/AquilaChat2-34B/blob/4608b75855334b93329a771aee03869dbf7d88cc/predict.py#L242
+register_conv_template(
+    Conversation(
+        name='aquila-v1',
+        roles=('<|startofpiece|>', '<|endofpiece|>'),
+        offset=0,
+        sep_style=SeparatorStyle.NO_COLON_TWO,
+        sep='',
+        sep2='</s>',
+        stop_str=['</s>', '<|endoftext|>'],
+    )
+)
+# Llama2-Chinese default template
+# source: https://huggingface.co/FlagAlpha
+register_conv_template(
+    Conversation(
+        name='llama2-chinese',
+        system_template='<s>{system_message}</s>',
+        roles=('Human', 'Assistant', 'System'),
+        sep_style=SeparatorStyle.ADD_COLON_TWO,
+        sep='\n',
+        sep2='\n</s><s>',
+        stop_str='</s>',
+    )
+)
+# Vigogne Instruct default template
+# source: https://github.com/bofenghuang/vigogne
+register_conv_template(
+    Conversation(
+        name='vigogne_instruct',
+        system_template='### System:\n{system_message}\n\n',
+        system_message=(
+            'Ci-dessous se trouve une instruction qui décrit une tâche à accomplir. Rédigez une réponse qui répond de manière'
+            ' précise à la demande.'
+        ),
+        roles=('### Instruction', '### Response'),
+        sep_style=SeparatorStyle.DOLLY,
+        sep='\n\n',
+        sep2='</s>',
+    )
+)
+# Vigogne Chat default template
+register_conv_template(
+    Conversation(
+        name='vigogne_chat_v2',
+        system_template='<|system|>: {system_message}',
+        system_message=(
+            'Vous êtes Vigogne, un assistant IA créé par Zaion Lab. Vous suivez extrêmement bien les instructions. Aidez'
+            ' autant que vous le pouvez.'
+        ),
+        roles=('<|user|>', '<|assistant|>'),
+        sep_style=SeparatorStyle.ADD_COLON_TWO,
+        sep='\n',
+        sep2='</s>\n',
+        stop_str='<|user|>',
+    )
+)
+register_conv_template(
+    Conversation(
+        name='vigogne_chat_v3',
+        system_template='[INST] <<SYS>>\n{system_message}\n<</SYS>>\n\n',
+        system_message=(
+            'Vous êtes Vigogne, un assistant IA créé par Zaion Lab. Vous suivez extrêmement bien les instructions. Aidez'
+            ' autant que vous le pouvez.'
+        ),
+        roles=('[INST]', '[/INST]'),
+        sep_style=SeparatorStyle.LLAMA2,
+        sep=' ',
+        sep2=' </s>',
+    )
+)
+# Falcon 180B chat template
+# source: https://huggingface.co/spaces/tiiuae/falcon-180b-demo/blob/d1590ee7fae9b6ce331ba7808e61a29dcce9239f/app.py#L28-L37
+register_conv_template(
+    Conversation(
+        name='falcon-chat',
+        roles=('User', 'Falcon'),
+        system_template='System: {system_message}',
+        messages=[],
+        sep_style=SeparatorStyle.FALCON_CHAT,
+        sep='\n',
+        sep2='<|endoftext|>',
+        stop_str='\nUser:',  # use stop_str to stop generation after stop_token_ids, it will also remove stop_str from the generated text
+    )
+)
+# Phind template
+# source: https://huggingface.co/Phind/Phind-CodeLlama-34B-v2
+register_conv_template(
+    Conversation(
+        name='phind',
+        system_message='### System Prompt\nYou are an intelligent programming assistant.',
+        roles=('### User Message', '### Assistant'),
+        messages=(),
+        offset=0,
+        sep_style=SeparatorStyle.ADD_COLON_SINGLE,
+        sep='\n\n',
+    )
+)
+# Metharme formatting for Pygmalion models
+# source: https://huggingface.co/PygmalionAI/pygmalion-2-13b
+register_conv_template(
+    Conversation(
+        name='metharme',
+        system_template='<|system|>{system_message}',
+        system_message="""Enter RP mode. You shall reply to the user while staying
+        in character. Your responses must be detailed, creative, immersive, and drive the scenario
+        forward.""",
+        roles=('<|user|>', '<|model|>'),
+        sep_style=SeparatorStyle.NO_COLON_SINGLE,
+        sep='',
+        stop_str='<|user|>',
+    )
+)
+# Zephyr template
+# reference: https://huggingface.co/spaces/HuggingFaceH4/zephyr-playground/blob/main/dialogues.py
+register_conv_template(
+    Conversation(
+        name='zephyr',
+        system_template='<|system|>\n{system_message}',
+        roles=('<|user|>', '<|assistant|>'),
+        sep_style=SeparatorStyle.CHATML,
+        sep='</s>',
+        stop_token_ids=[2],
+        stop_str='</s>',
+    )
+)
+# InternVL-ZH template
+register_conv_template(
+    Conversation(
+        name='internvl_zh',
+        system_template='',
+        roles=('<human>', '<bot>'),
+        sep_style=SeparatorStyle.INTERNVL_ZH,
+        sep=' ',
+        sep2='</s>',
+    )
+)
+if __name__ == '__main__':
+    from fastchat.conversation import get_conv_template
+    print('-- Vicuna template --')
+    conv = get_conv_template('vicuna_v1.1')
+    conv.append_message(conv.roles[0], 'Hello!')
+    conv.append_message(conv.roles[1], 'Hi!')
+    conv.append_message(conv.roles[0], 'How are you?')
+    conv.append_message(conv.roles[1], None)
+    print(conv.get_prompt())
+    print('\n')
+    print('-- Llama-2 template --')
+    conv = get_conv_template('llama-2')
+    conv.set_system_message('You are a helpful, respectful and honest assistant.')
+    conv.append_message(conv.roles[0], 'Hello!')
+    conv.append_message(conv.roles[1], 'Hi!')
+    conv.append_message(conv.roles[0], 'How are you?')
+    conv.append_message(conv.roles[1], None)
+    print(conv.get_prompt())
+    print('\n')
+    print('-- ChatGPT template --')
+    conv = get_conv_template('chatgpt')
+    conv.append_message(conv.roles[0], 'Hello!')
+    conv.append_message(conv.roles[1], 'Hi!')
+    conv.append_message(conv.roles[0], 'How are you?')
+    conv.append_message(conv.roles[1], None)
+    print(conv.to_openai_api_messages())
+    print('\n')
+    print('-- Claude template --')
+    conv = get_conv_template('claude')
+    conv.append_message(conv.roles[0], 'Hello!')
+    conv.append_message(conv.roles[1], 'Hi!')
+    conv.append_message(conv.roles[0], 'How are you?')
+    conv.append_message(conv.roles[1], None)
+    print(conv.get_prompt())

generation_config.json ADDED Viewed

	@@ -0,0 +1,4 @@

+{
+  "_from_model_config": true,
+  "transformers_version": "4.37.2"
+}

model-00001-of-00002.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:e2f519b0c73a5c075860f8af239613ff0770eceb1da660d906c96665f12ffb5e
+size 4979029664

model-00002-of-00002.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:f979759cc889a8bfc4ce750ec0bcf207e061b3e3c09543e056391fcfea862751
+size 505005416

model.safetensors.index.json ADDED Viewed

	@@ -0,0 +1,403 @@

+{
+  "metadata": {
+    "total_size": 5483984896
+  },
+  "weight_map": {
+    "embedding_model.encoder.0.attention.A_log_bias": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.0.attention.conv1d.bias": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.0.attention.conv1d.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.0.attention.dt_bias": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.0.attention.g_norm_swish_gate.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.0.attention.o_proj.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.0.attention.wvkqgdt.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.0.attention_norm.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.0.feed_forward.w1.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.0.feed_forward.w2.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.0.feed_forward.w3.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.0.ffn_norm.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.1.attention.A_log_bias": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.1.attention.conv1d.bias": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.1.attention.conv1d.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.1.attention.dt_bias": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.1.attention.g_norm_swish_gate.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.1.attention.o_proj.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.1.attention.wvkqgdt.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.1.attention_norm.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.1.feed_forward.w1.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.1.feed_forward.w2.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.1.feed_forward.w3.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.1.ffn_norm.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.2.attention.A_log_bias": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.2.attention.conv1d.bias": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.2.attention.conv1d.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.2.attention.dt_bias": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.2.attention.g_norm_swish_gate.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.2.attention.o_proj.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.2.attention.wvkqgdt.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.2.attention_norm.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.2.feed_forward.w1.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.2.feed_forward.w2.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.2.feed_forward.w3.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.2.ffn_norm.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.3.attention.A_log_bias": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.3.attention.conv1d.bias": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.3.attention.conv1d.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.3.attention.dt_bias": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.3.attention.g_norm_swish_gate.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.3.attention.o_proj.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.3.attention.wvkqgdt.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.3.attention_norm.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.3.feed_forward.w1.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.3.feed_forward.w2.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.3.feed_forward.w3.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.3.ffn_norm.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.4.attention.A_log_bias": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.4.attention.conv1d.bias": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.4.attention.conv1d.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.4.attention.dt_bias": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.4.attention.g_norm_swish_gate.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.4.attention.o_proj.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.4.attention.wvkqgdt.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.4.attention_norm.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.4.feed_forward.w1.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.4.feed_forward.w2.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.4.feed_forward.w3.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.4.ffn_norm.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.5.attention.A_log_bias": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.5.attention.conv1d.bias": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.5.attention.conv1d.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.5.attention.dt_bias": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.5.attention.g_norm_swish_gate.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.5.attention.o_proj.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.5.attention.wvkqgdt.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.5.attention_norm.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.5.feed_forward.w1.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.5.feed_forward.w2.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.5.feed_forward.w3.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.5.ffn_norm.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.6.attention.A_log_bias": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.6.attention.conv1d.bias": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.6.attention.conv1d.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.6.attention.dt_bias": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.6.attention.g_norm_swish_gate.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.6.attention.o_proj.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.6.attention.wvkqgdt.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.6.attention_norm.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.6.feed_forward.w1.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.6.feed_forward.w2.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.6.feed_forward.w3.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.6.ffn_norm.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.7.attention.A_log_bias": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.7.attention.conv1d.bias": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.7.attention.conv1d.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.7.attention.dt_bias": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.7.attention.g_norm_swish_gate.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.7.attention.o_proj.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.7.attention.wvkqgdt.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.7.attention_norm.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.7.feed_forward.w1.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.7.feed_forward.w2.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.7.feed_forward.w3.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.encoder.7.ffn_norm.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.llm_text_embeddings.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.pixel_shuffle_proj.0.bias": "model-00001-of-00002.safetensors",
+    "embedding_model.pixel_shuffle_proj.0.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.pixel_shuffle_proj.2.bias": "model-00001-of-00002.safetensors",
+    "embedding_model.pixel_shuffle_proj.2.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.vision_embeddings.class_embedding": "model-00001-of-00002.safetensors",
+    "embedding_model.vision_embeddings.patch_embedding.bias": "model-00001-of-00002.safetensors",
+    "embedding_model.vision_embeddings.patch_embedding.weight": "model-00001-of-00002.safetensors",
+    "embedding_model.vision_embeddings.position_embedding": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.0.attention.A_log_bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.0.attention.conv1d.bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.0.attention.conv1d.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.0.attention.dt_bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.0.attention.g_norm_swish_gate.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.0.attention.o_proj.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.0.attention.wvkqgdt.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.0.attention_norm.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.0.feed_forward.w1.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.0.feed_forward.w2.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.0.feed_forward.w3.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.0.ffn_norm.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.1.attention.A_log_bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.1.attention.conv1d.bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.1.attention.conv1d.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.1.attention.dt_bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.1.attention.g_norm_swish_gate.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.1.attention.o_proj.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.1.attention.wvkqgdt.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.1.attention_norm.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.1.feed_forward.w1.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.1.feed_forward.w2.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.1.feed_forward.w3.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.1.ffn_norm.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.10.attention.A_log_bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.10.attention.conv1d.bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.10.attention.conv1d.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.10.attention.dt_bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.10.attention.g_norm_swish_gate.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.10.attention.o_proj.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.10.attention.wvkqgdt.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.10.attention_norm.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.10.feed_forward.w1.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.10.feed_forward.w2.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.10.feed_forward.w3.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.10.ffn_norm.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.11.attention.A_log_bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.11.attention.conv1d.bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.11.attention.conv1d.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.11.attention.dt_bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.11.attention.g_norm_swish_gate.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.11.attention.o_proj.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.11.attention.wvkqgdt.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.11.attention_norm.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.11.feed_forward.w1.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.11.feed_forward.w2.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.11.feed_forward.w3.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.11.ffn_norm.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.12.attention.A_log_bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.12.attention.conv1d.bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.12.attention.conv1d.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.12.attention.dt_bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.12.attention.g_norm_swish_gate.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.12.attention.o_proj.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.12.attention.wvkqgdt.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.12.attention_norm.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.12.feed_forward.w1.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.12.feed_forward.w2.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.12.feed_forward.w3.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.12.ffn_norm.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.13.attention.A_log_bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.13.attention.conv1d.bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.13.attention.conv1d.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.13.attention.dt_bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.13.attention.g_norm_swish_gate.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.13.attention.o_proj.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.13.attention.wvkqgdt.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.13.attention_norm.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.13.feed_forward.w1.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.13.feed_forward.w2.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.13.feed_forward.w3.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.13.ffn_norm.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.14.attention.A_log_bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.14.attention.conv1d.bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.14.attention.conv1d.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.14.attention.dt_bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.14.attention.g_norm_swish_gate.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.14.attention.o_proj.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.14.attention.wvkqgdt.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.14.attention_norm.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.14.feed_forward.w1.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.14.feed_forward.w2.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.14.feed_forward.w3.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.14.ffn_norm.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.15.attention.A_log_bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.15.attention.conv1d.bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.15.attention.conv1d.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.15.attention.dt_bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.15.attention.g_norm_swish_gate.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.15.attention.o_proj.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.15.attention.wvkqgdt.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.15.attention_norm.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.15.feed_forward.w1.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.15.feed_forward.w2.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.15.feed_forward.w3.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.15.ffn_norm.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.16.attention.A_log_bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.16.attention.conv1d.bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.16.attention.conv1d.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.16.attention.dt_bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.16.attention.g_norm_swish_gate.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.16.attention.o_proj.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.16.attention.wvkqgdt.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.16.attention_norm.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.16.feed_forward.w1.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.16.feed_forward.w2.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.16.feed_forward.w3.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.16.ffn_norm.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.17.attention.A_log_bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.17.attention.conv1d.bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.17.attention.conv1d.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.17.attention.dt_bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.17.attention.g_norm_swish_gate.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.17.attention.o_proj.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.17.attention.wvkqgdt.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.17.attention_norm.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.17.feed_forward.w1.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.17.feed_forward.w2.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.17.feed_forward.w3.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.17.ffn_norm.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.18.attention.A_log_bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.18.attention.conv1d.bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.18.attention.conv1d.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.18.attention.dt_bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.18.attention.g_norm_swish_gate.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.18.attention.o_proj.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.18.attention.wvkqgdt.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.18.attention_norm.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.18.feed_forward.w1.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.18.feed_forward.w2.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.18.feed_forward.w3.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.18.ffn_norm.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.19.attention.A_log_bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.19.attention.conv1d.bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.19.attention.conv1d.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.19.attention.dt_bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.19.attention.g_norm_swish_gate.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.19.attention.o_proj.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.19.attention.wvkqgdt.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.19.attention_norm.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.19.feed_forward.w1.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.19.feed_forward.w2.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.19.feed_forward.w3.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.19.ffn_norm.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.2.attention.A_log_bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.2.attention.conv1d.bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.2.attention.conv1d.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.2.attention.dt_bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.2.attention.g_norm_swish_gate.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.2.attention.o_proj.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.2.attention.wvkqgdt.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.2.attention_norm.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.2.feed_forward.w1.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.2.feed_forward.w2.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.2.feed_forward.w3.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.2.ffn_norm.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.20.attention.A_log_bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.20.attention.conv1d.bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.20.attention.conv1d.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.20.attention.dt_bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.20.attention.g_norm_swish_gate.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.20.attention.o_proj.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.20.attention.wvkqgdt.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.20.attention_norm.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.20.feed_forward.w1.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.20.feed_forward.w2.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.20.feed_forward.w3.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.20.ffn_norm.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.21.attention.A_log_bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.21.attention.conv1d.bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.21.attention.conv1d.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.21.attention.dt_bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.21.attention.g_norm_swish_gate.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.21.attention.o_proj.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.21.attention.wvkqgdt.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.21.attention_norm.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.21.feed_forward.w1.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.21.feed_forward.w2.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.21.feed_forward.w3.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.21.ffn_norm.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.22.attention.A_log_bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.22.attention.conv1d.bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.22.attention.conv1d.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.22.attention.dt_bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.22.attention.g_norm_swish_gate.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.22.attention.o_proj.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.22.attention.wvkqgdt.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.22.attention_norm.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.22.feed_forward.w1.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.22.feed_forward.w2.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.22.feed_forward.w3.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.22.ffn_norm.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.23.attention.A_log_bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.23.attention.conv1d.bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.23.attention.conv1d.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.23.attention.dt_bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.23.attention.g_norm_swish_gate.weight": "model-00002-of-00002.safetensors",
+    "language_model.model.layers.23.attention.o_proj.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.23.attention.wvkqgdt.weight": "model-00002-of-00002.safetensors",
+    "language_model.model.layers.23.attention_norm.weight": "model-00002-of-00002.safetensors",
+    "language_model.model.layers.23.feed_forward.w1.weight": "model-00002-of-00002.safetensors",
+    "language_model.model.layers.23.feed_forward.w2.weight": "model-00002-of-00002.safetensors",
+    "language_model.model.layers.23.feed_forward.w3.weight": "model-00002-of-00002.safetensors",
+    "language_model.model.layers.23.ffn_norm.weight": "model-00002-of-00002.safetensors",
+    "language_model.model.layers.3.attention.A_log_bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.3.attention.conv1d.bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.3.attention.conv1d.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.3.attention.dt_bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.3.attention.g_norm_swish_gate.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.3.attention.o_proj.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.3.attention.wvkqgdt.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.3.attention_norm.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.3.feed_forward.w1.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.3.feed_forward.w2.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.3.feed_forward.w3.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.3.ffn_norm.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.4.attention.A_log_bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.4.attention.conv1d.bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.4.attention.conv1d.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.4.attention.dt_bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.4.attention.g_norm_swish_gate.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.4.attention.o_proj.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.4.attention.wvkqgdt.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.4.attention_norm.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.4.feed_forward.w1.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.4.feed_forward.w2.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.4.feed_forward.w3.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.4.ffn_norm.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.5.attention.A_log_bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.5.attention.conv1d.bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.5.attention.conv1d.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.5.attention.dt_bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.5.attention.g_norm_swish_gate.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.5.attention.o_proj.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.5.attention.wvkqgdt.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.5.attention_norm.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.5.feed_forward.w1.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.5.feed_forward.w2.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.5.feed_forward.w3.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.5.ffn_norm.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.6.attention.A_log_bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.6.attention.conv1d.bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.6.attention.conv1d.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.6.attention.dt_bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.6.attention.g_norm_swish_gate.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.6.attention.o_proj.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.6.attention.wvkqgdt.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.6.attention_norm.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.6.feed_forward.w1.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.6.feed_forward.w2.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.6.feed_forward.w3.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.6.ffn_norm.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.7.attention.A_log_bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.7.attention.conv1d.bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.7.attention.conv1d.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.7.attention.dt_bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.7.attention.g_norm_swish_gate.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.7.attention.o_proj.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.7.attention.wvkqgdt.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.7.attention_norm.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.7.feed_forward.w1.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.7.feed_forward.w2.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.7.feed_forward.w3.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.7.ffn_norm.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.8.attention.A_log_bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.8.attention.conv1d.bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.8.attention.conv1d.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.8.attention.dt_bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.8.attention.g_norm_swish_gate.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.8.attention.o_proj.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.8.attention.wvkqgdt.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.8.attention_norm.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.8.feed_forward.w1.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.8.feed_forward.w2.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.8.feed_forward.w3.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.8.ffn_norm.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.9.attention.A_log_bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.9.attention.conv1d.bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.9.attention.conv1d.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.9.attention.dt_bias": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.9.attention.g_norm_swish_gate.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.9.attention.o_proj.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.9.attention.wvkqgdt.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.9.attention_norm.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.9.feed_forward.w1.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.9.feed_forward.w2.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.9.feed_forward.w3.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.layers.9.ffn_norm.weight": "model-00001-of-00002.safetensors",
+    "language_model.model.norm.weight": "model-00002-of-00002.safetensors",
+    "language_model.model.tok_embeddings.weight": "model-00001-of-00002.safetensors",
+    "language_model.output.weight": "model-00002-of-00002.safetensors"
+  }
+}

modeling_mmMamba.py ADDED Viewed

	@@ -0,0 +1,1136 @@

+# Copyright (c) The mmMamba team and The HuggingFace Inc. team. All rights reserved.
+#
+# This code is based on transformers/src/transformers/models/llama/modeling_llama.py
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+import math
+import queue
+import threading
+import warnings
+from typing import List, Optional, Tuple, Union
+import torch
+import torch.nn.functional as F
+import torch.utils.checkpoint
+from einops import rearrange
+from torch import nn
+from torch.nn import BCEWithLogitsLoss, CrossEntropyLoss, MSELoss
+from transformers.activations import ACT2FN
+from transformers.modeling_outputs import (BaseModelOutputWithPast,
+                                           CausalLMOutputWithPast,
+                                           SequenceClassifierOutputWithPast)
+from transformers.modeling_utils import PreTrainedModel
+from transformers.utils import (add_start_docstrings,
+                                add_start_docstrings_to_model_forward, logging,
+                                replace_return_docstrings)
+from fla.modules import FusedRMSNormSwishGate, RMSNorm, ShortConvolution
+import copy
+from mamba_ssm.ops.triton.ssd_combined import mamba_chunk_scan_combined
+from mamba_ssm.ops.triton.selective_state_update import selective_state_update
+from causal_conv1d import causal_conv1d_fn, causal_conv1d_update
+from transformers.cache_utils import Cache
+import time
+try:
+    from transformers.generation.streamers import BaseStreamer
+except:  # noqa # pylint: disable=bare-except
+    BaseStreamer = None
+from .configuration_mmMamba import mmMambaConfig
+logger = logging.get_logger(__name__)
+_CONFIG_FOR_DOC = 'mmMambaConfig'
+flash_attn_func, flash_attn_varlen_func = None, None
+pad_input, index_first_axis, unpad_input = None, None, None
+try:
+    from flash_attn import flash_attn_func as _flash_attn_func
+    from flash_attn import flash_attn_varlen_func as _flash_attn_varlen_func
+    from flash_attn.bert_padding import index_first_axis as _index_first_axis
+    from flash_attn.bert_padding import pad_input as _pad_input
+    from flash_attn.bert_padding import unpad_input as _unpad_input
+    flash_attn_func, flash_attn_varlen_func = _flash_attn_func, _flash_attn_varlen_func
+    pad_input, index_first_axis, unpad_input = _pad_input, _index_first_axis, _unpad_input
+    has_flash_attn = True
+except:
+    has_flash_attn = False
+try:
+    from flash_attn import flash_attn_with_kvcache
+except ImportError:
+    flash_attn_with_kvcache = None
+try:
+    from flash_attn.layers.rotary import RotaryEmbedding
+except ImportError:
+    RotaryEmbedding = None
+import torch.nn.functional as F
+def _update_kv_cache(kv, inference_params, layer_idx):
+    """kv: (batch_size, seqlen, 2, nheads, head_dim) or (batch_size, 1, 2, nheads, head_dim)"""
+    # Pre-allocate memory for key-values for inference.
+    num_heads, head_dim = kv.shape[-2:]
+    assert layer_idx in inference_params.key_value_memory_dict
+    kv_cache, _ = inference_params.key_value_memory_dict[layer_idx]
+    # Adjust key and value for inference
+    batch_start = inference_params.batch_size_offset
+    batch_end = batch_start + kv.shape[0]
+    sequence_start = inference_params.seqlen_offset
+    sequence_end = sequence_start + kv.shape[1]
+    assert batch_end <= kv_cache.shape[0]
+    assert sequence_end <= kv_cache.shape[1]
+    assert kv_cache is not None
+    kv_cache[batch_start:batch_end, sequence_start:sequence_end, ...] = kv
+    return kv_cache[batch_start:batch_end, :sequence_end, ...]
+def _import_flash_attn():
+    global flash_attn_func, flash_attn_varlen_func
+    global pad_input, index_first_axis, unpad_input
+    try:
+        from flash_attn import flash_attn_func as _flash_attn_func
+        from flash_attn import \
+            flash_attn_varlen_func as _flash_attn_varlen_func
+        from flash_attn.bert_padding import \
+            index_first_axis as _index_first_axis
+        from flash_attn.bert_padding import pad_input as _pad_input
+        from flash_attn.bert_padding import unpad_input as _unpad_input
+        flash_attn_func, flash_attn_varlen_func = _flash_attn_func, _flash_attn_varlen_func
+        pad_input, index_first_axis, unpad_input = _pad_input, _index_first_axis, _unpad_input
+    except ImportError:
+        raise ImportError('flash_attn is not installed.')
+# Copied from transformers.models.llama.modeling_llama.LlamaRMSNorm with Llama->mmMamba
+class mmMambaRMSNorm(nn.Module):
+    def __init__(self, hidden_size, eps=1e-6):
+        """
+        mmMambaRMSNorm is equivalent to T5LayerNorm
+        """
+        super().__init__()
+        self.weight = nn.Parameter(torch.ones(hidden_size))
+        self.variance_epsilon = eps
+    def forward(self, hidden_states):
+        input_dtype = hidden_states.dtype
+        hidden_states = hidden_states.to(torch.float32)
+        variance = hidden_states.pow(2).mean(-1, keepdim=True)
+        hidden_states = hidden_states * torch.rsqrt(variance + self.variance_epsilon)
+        return self.weight * hidden_states.to(input_dtype)
+class mmMambaMLP(nn.Module):
+    def __init__(self, config):
+        super().__init__()
+        self.config = config
+        self.hidden_size = config.hidden_size
+        self.intermediate_size = config.intermediate_size
+        self.w1 = nn.Linear(self.hidden_size, self.intermediate_size, bias=False)
+        self.w3 = nn.Linear(self.hidden_size, self.intermediate_size, bias=False)
+        self.w2 = nn.Linear(self.intermediate_size, self.hidden_size, bias=False)
+        self.act_fn = ACT2FN[config.hidden_act]
+    def forward(self, x):
+        down_proj = self.w2(self.act_fn(self.w1(x)) * self.w3(x))
+        return down_proj
+# Copied from transformers.model.llama.modeling_llama.repeat_kv
+def repeat_kv(hidden_states: torch.Tensor, n_rep: int) -> torch.Tensor:
+    """
+    This is the equivalent of torch.repeat_interleave(x, dim=1, repeats=n_rep). The hidden states go from (batch,
+    num_key_value_heads, seqlen, head_dim) to (batch, num_attention_heads, seqlen, head_dim)
+    """
+    batch, num_key_value_heads, slen, head_dim = hidden_states.shape
+    if n_rep == 1:
+        return hidden_states
+    hidden_states = hidden_states[:, :, None, :, :].expand(batch, num_key_value_heads, n_rep, slen, head_dim)
+    return hidden_states.reshape(batch, num_key_value_heads * n_rep, slen, head_dim)
+def repeat_kv2(hidden_states: torch.Tensor, n_rep: int) -> torch.Tensor:
+    """
+    This is the equivalent of torch.repeat_interleave(x, dim=1, repeats=n_rep). The hidden states go from (batch,
+    num_key_value_heads, seqlen, head_dim) to (batch, num_attention_heads, seqlen, head_dim)
+    """
+    batch, num_key_value_heads, head_dim = hidden_states.shape
+    if n_rep == 1:
+        return hidden_states
+    hidden_states = hidden_states[:, :, None,  :].expand(batch, num_key_value_heads, n_rep, head_dim)
+    return hidden_states.reshape(batch, num_key_value_heads * n_rep, head_dim)
+class MHA_LM(nn.Module):
+    """Multi-headed attention from 'Attention Is All You Need' paper"""
+    def __init__(self, config: mmMambaConfig, layer_idx: int):
+        super().__init__()
+        self.config = config
+        self.layer_idx = layer_idx#-------------------------
+        self.hidden_size = config.hidden_size
+        self.num_heads = config.num_attention_heads
+        self.head_dim = self.hidden_size // self.num_heads
+        self.num_key_value_heads = config.num_key_value_heads
+        self.num_key_value_groups = self.num_heads // self.num_key_value_heads
+        self.max_position_embeddings = config.max_position_embeddings
+        self.is_causal = True
+        self.rotary_emb_dim = self.head_dim
+        self.softmax_scale = None
+        self.causal = True
+        if (self.head_dim * self.num_heads) != self.hidden_size:
+            raise ValueError(
+                f"hidden_size must be divisible by num_heads (got `hidden_size`: {self.hidden_size}"
+                f" and `num_heads`: {self.num_heads})."
+            )
+        self.wqkv = nn.Linear(
+            self.hidden_size,
+            (self.num_heads + 2 * self.num_key_value_heads) * self.head_dim,
+            bias=False,
+        )
+        self.wo = nn.Linear(self.num_heads * self.head_dim, self.hidden_size, bias=False)
+        self.rotary_emb = RotaryEmbedding(
+                self.head_dim,
+                base=self.config.rope_theta,
+                interleaved=False,
+                device=self.wo.weight.device,
+                )
+    def _shape(self, tensor: torch.Tensor, seq_len: int, bsz: int):
+        return tensor.view(bsz, seq_len, self.num_heads, self.head_dim).transpose(1, 2).contiguous()
+    def _update_kv_cache(self, kv, inference_params):
+        """kv: (batch_size, seqlen, 2, nheads, head_dim) or (batch_size, 1, 2, nheads, head_dim)"""
+        assert self.layer_idx is not None, "Generation requires layer_idx in the constructor"
+        return _update_kv_cache(kv, inference_params, self.layer_idx)
+    def _apply_rotary_update_kvcache_attention(self, q, kv, inference_params):
+        """
+        Fast path that combine 3 steps: apply rotary to Q and K, update kv cache, and apply attention.
+        q: (batch_size, seqlen_q, nheads, head_dim)
+        kv: (batch_size, seqlen_k, 2, nheads_kv, head_dim)
+        """
+        assert inference_params is not None and inference_params.seqlen_offset > 0
+        if self.rotary_emb_dim > 0:
+            self.rotary_emb._update_cos_sin_cache(
+                inference_params.max_seqlen, device=q.device, dtype=q.dtype
+            )
+            rotary_cos, rotary_sin = self.rotary_emb._cos_cached, self.rotary_emb._sin_cached
+        else:
+            rotary_cos, rotary_sin = None, None
+        batch = q.shape[0]
+        kv_cache, _ = inference_params.key_value_memory_dict[self.layer_idx]
+        kv_cache = kv_cache[:batch]
+        cache_seqlens = (
+            inference_params.lengths_per_sample[:batch]
+            if inference_params.lengths_per_sample is not None
+            else inference_params.seqlen_offset
+        )
+        assert flash_attn_with_kvcache is not None, "flash_attn must be installed"
+        context = flash_attn_with_kvcache(
+            q,
+            kv_cache[:, :, 0],
+            kv_cache[:, :, 1],
+            kv[:, :, 0],
+            kv[:, :, 1],
+            rotary_cos=rotary_cos,
+            rotary_sin=rotary_sin,
+            cache_seqlens=cache_seqlens,
+            softmax_scale=self.softmax_scale,
+            causal=self.causal,
+            rotary_interleaved=self.rotary_emb.interleaved if self.rotary_emb_dim > 0 else False,
+        )
+        return context
+    def _update_kvcache_attention(self, q, kv, inference_params):
+        """Write kv to inference_params, then do attention"""
+        if (
+            inference_params.seqlen_offset == 0
+            or flash_attn_with_kvcache is None
+        ):
+            # TODO: this only uses seqlen_offset and not lengths_per_sample.
+            kv = self._update_kv_cache(kv, inference_params)
+            k, v = kv.unbind(dim=-3)
+            #k = torch.repeat_interleave(k, dim=2, repeats=self.num_heads // self.num_key_value_heads)
+            #v = torch.repeat_interleave(v, dim=2, repeats=self.num_heads // self.num_key_value_heads)
+            attn_output = flash_attn_func(
+                q, k, v, 0.0, softmax_scale=None, causal=self.causal
+            )
+            return attn_output
+        else:
+            batch = q.shape[0]
+            kv_cache, _ = inference_params.key_value_memory_dict[self.layer_idx]
+            kv_cache = kv_cache[:batch]
+            cache_seqlens = (
+                inference_params.lengths_per_sample[:batch]
+                if inference_params.lengths_per_sample is not None
+                else inference_params.seqlen_offset
+            )
+            return flash_attn_with_kvcache(
+                q,
+                kv_cache[:, :, 0],
+                kv_cache[:, :, 1],
+                kv[:, :, 0],
+                kv[:, :, 1],
+                cache_seqlens=cache_seqlens,
+                softmax_scale=self.softmax_scale,
+                causal=self.causal,
+            )
+    def forward(
+        self,
+        hidden_states: torch.Tensor,
+        inference_params = None,
+        output_attentions: bool = False,
+        cache_position: Optional[torch.LongTensor] = None,#------------------------------------------------------------------------
+        use_cache: bool = False,
+        **kwargs,
+    ) -> Tuple[torch.Tensor, Optional[torch.Tensor], Optional[Tuple[torch.Tensor]]]:
+        if inference_params is not None and self.layer_idx not in inference_params.key_value_memory_dict:
+            inference_params.key_value_memory_dict[self.layer_idx] = self.allocate_inference_cache(
+                hidden_states.shape[0], inference_params.max_seqlen, dtype=hidden_states.dtype
+            )
+        seqlen_offset = (
+            0
+            if inference_params is None
+            else (
+                inference_params.lengths_per_sample
+                if inference_params.lengths_per_sample is not None
+                else inference_params.seqlen_offset
+            )
+        )
+        bsz, q_len, _ = hidden_states.size()
+        rotary_max_seqlen = inference_params.max_seqlen if inference_params is not None else None
+        qkv = self.wqkv(hidden_states)
+        qkv = rearrange(
+            qkv,
+            "b q (h gs d) -> b q h gs d",
+            gs=2 + self.num_key_value_groups,
+            d=self.head_dim,
+        )
+        q = qkv[..., : self.num_key_value_groups, :]
+        q = rearrange(q, "b q h gs d -> b q (h gs) d")
+        kv = qkv[..., self.num_key_value_groups:, :].transpose(2,3)
+        if (
+            inference_params is None
+            or inference_params.seqlen_offset == 0
+            or (self.rotary_emb_dim == 0 or self.rotary_emb_dim % 16 != 0)
+        ):
+            if self.rotary_emb_dim > 0:
+                q, kv = self.rotary_emb(
+                    q, kv, seqlen_offset=seqlen_offset, max_seqlen=rotary_max_seqlen
+                )
+            if inference_params is None:
+                k, v = kv.unbind(dim=-3)
+                k = torch.repeat_interleave(k, dim=2, repeats=self.num_heads // self.num_key_value_heads)
+                v = torch.repeat_interleave(v, dim=2, repeats=self.num_heads // self.num_key_value_heads)
+                context = F.scaled_dot_product_attention(
+                    q.transpose(1, 2), k.transpose(1, 2), v.transpose(1, 2), is_causal=True, scale=None
+                ).transpose(1, 2)
+            else:
+                context = self._update_kvcache_attention(q, kv, inference_params)
+        else:
+            context = self._apply_rotary_update_kvcache_attention(q, kv, inference_params)
+        context = rearrange(context, "... h d -> ... (h d)")
+        out = self.wo(context)
+        return out
+    def allocate_inference_cache(self, batch_size, max_seqlen, dtype=None):
+        dtype = self.wo.weight.dtype if dtype is None else dtype
+        device = self.wo.weight.device
+        kv_cache = torch.empty(
+            batch_size, max_seqlen, 2, self.num_key_value_heads, self.head_dim, dtype=dtype, device=device,
+        )
+        return kv_cache, None
+class Mamba2_LM(nn.Module):
+    """
+    LoLCATs attention implementation initialized from a
+    `LlamaAttention` or `MistralAttention` object (base_attn)
+    Most of the arguments are directly tied to argparse args
+    - For now we don't support padding.
+    """
+    def __init__(self, config: mmMambaConfig, layer_idx: Optional[int] = None,
+                 elementwise_affine: Optional[bool] = True,
+                 norm_eps: float = 1e-5,
+                 ):
+        super().__init__()
+        self.config = config
+        self.hidden_size = config.hidden_size
+        self.num_heads = config.num_attention_heads
+        self.head_dim = self.hidden_size // self.num_heads
+        self.num_key_value_heads = config.num_key_value_heads
+        self.num_key_value_groups = self.num_heads // self.num_key_value_heads
+        self.max_position_embeddings = config.max_position_embeddings
+        self.layer_idx = layer_idx
+        self.bias = False
+        self.chunk_size = 128
+        conv_bias = True
+        self.conv_bias = conv_bias
+        self.d_conv = 2
+        self.activation="silu"
+        self.max_position_embeddings = config.max_position_embeddings
+        self.rope_theta = config.rope_theta
+        self.wvkqgdt = nn.Linear(
+            self.hidden_size,
+            (self.num_heads + 2 * self.num_key_value_heads + self.num_heads) * self.head_dim + self.num_heads,
+            bias=self.bias
+        )
+        self.o_proj = nn.Linear(self.num_heads * self.head_dim, self.hidden_size, bias=False)
+        self.device = self.wvkqgdt.weight.device
+        self.dtype = self.wvkqgdt.weight.dtype
+        conv_dim = self.num_heads * self.head_dim + 2 * self.num_key_value_heads * self.head_dim
+        self.conv1d = nn.Conv1d(
+            in_channels=conv_dim,
+            out_channels=conv_dim,
+            bias=self.conv_bias,
+            kernel_size=self.d_conv,
+            groups=conv_dim,
+            padding=self.d_conv - 1,
+            device=self.device,
+            dtype=self.dtype
+        )
+        with torch.no_grad():
+            self.conv1d.weight.zero_()
+            self.conv1d.weight[:, 0, 1] = 1
+            self.conv1d.bias.zero_()
+        # Activation after conv
+        if self.activation == "identity":
+            self.act = nn.Identity()
+        elif self.activation in ["silu", "swish"]:
+            self.act = nn.SiLU()
+        else:
+            raise ValueError(f"Unknown activation {self.activation}")
+        self.g_norm_swish_gate = FusedRMSNormSwishGate(hidden_size=self.head_dim, elementwise_affine=elementwise_affine, eps=norm_eps).to(self.dtype).to(self.device)
+        dt = torch.exp(
+            torch.rand(self.num_heads, dtype=self.dtype, device=self.device) * (math.log(0.1) - math.log(0.001))
+            + math.log(0.001)
+        )
+        dt = torch.clamp(dt, min=0.001)
+        # Inverse of softplus: https://github.com/pytorch/pytorch/issues/72759
+        inv_dt = dt + torch.log(-torch.expm1(-dt))
+        self.dt_bias = nn.Parameter(inv_dt)
+        self.dt_bias._no_weight_decay = True
+        A_log_bias = torch.zeros(self.num_heads, dtype=self.dtype, device=self.device)
+        self.A_log_bias = nn.Parameter(A_log_bias)
+        self.A_log_bias._no_weight_decay = True
+    def forward(self,
+                hidden_states: torch.Tensor,
+                inference_params = None,
+                output_attentions: bool = False,
+                use_cache: bool = True,
+                **kwargs,
+               ) -> Tuple[torch.Tensor, Optional[torch.Tensor], Optional[Tuple[torch.Tensor]]]:
+        hidden_states = hidden_states.to(self.dtype)
+        vkqgdt = self.wvkqgdt(hidden_states)
+        vkq, g, dt = torch.split(
+                vkqgdt,
+                [
+                    (2*self.num_key_value_heads+self.num_heads) * self.head_dim,
+                    self.num_heads * self.head_dim,
+                    self.num_heads,
+                ],
+                dim=2,
+            )
+        batch, seqlen, _ = hidden_states.shape
+        conv_state, ssm_state = None, None
+        if inference_params is not None:
+            conv_state, ssm_state = self._get_states_from_cache(inference_params, batch)
+        if use_cache and inference_params.seqlen_offset==0:
+            vkq, new_conv_states = causal_conv1d_fn(
+                vkq.transpose(1, 2),
+                rearrange(self.conv1d.weight, "d 1 w -> d w"),
+                self.conv1d.bias,
+                initial_states=None,
+                return_final_states=True,
+                activation=None if self.activation == "identity" else self.activation,
+            )
+            v, k, q = torch.split(
+                vkq,
+                [
+                    self.num_key_value_heads * self.head_dim,
+                    self.num_key_value_heads * self.head_dim,
+                    self.num_heads * self.head_dim,
+                ],
+                dim=1,
+            )
+            v = rearrange(v, "b (h n) l -> b h l n", h=self.num_key_value_heads)
+            k = rearrange(k, "b (h n) l -> b h l n", h=self.num_key_value_heads)
+            q = rearrange(q, "b (h n) l -> b l h n", h=self.num_heads)
+            k = repeat_kv(k, self.num_key_value_groups).transpose(1, 2)
+            v = repeat_kv(v, self.num_key_value_groups).transpose(1, 2)
+            A = -torch.exp(self.A_log_bias.float())
+            y, new_ssm_states = mamba_chunk_scan_combined(
+                x = v,
+                #x = v / F.softplus(A_log).to(v.dtype).unsqueeze(-1),
+                dt=dt,
+                dt_softplus=True,
+                A=A,
+                B=k,
+                C=q,
+                chunk_size=self.chunk_size,
+                dt_bias=self.dt_bias,
+                initial_states=None, # currently not supported by mamba_ssm.utils.generation
+                return_final_states=True,
+            )
+            conv_state.copy_(new_conv_states)
+            ssm_state.copy_(new_ssm_states)
+        elif use_cache and inference_params.seqlen_offset>0:
+            vkq = causal_conv1d_update(
+                vkq.transpose(1, 2).squeeze(-1),
+                conv_state,
+                self.conv1d.weight.squeeze(1),
+                self.conv1d.bias,
+                self.activation,
+            )
+            v, k, q = torch.split(
+                vkq,
+                [
+                    self.num_key_value_heads * self.head_dim,
+                    self.num_key_value_heads * self.head_dim,
+                    self.num_heads * self.head_dim,
+                ],
+                dim=1,
+            )
+            v = rearrange(v, "b (h n) -> b h n", h=self.num_key_value_heads)
+            k = rearrange(k, "b (h n) -> b h n", h=self.num_key_value_heads)
+            q = rearrange(q, "b (h n) -> b h n", h=self.num_heads)
+            k = repeat_kv2(k, self.num_key_value_groups)
+            v = repeat_kv2(v, self.num_key_value_groups)
+            dt = dt.transpose(1, 2).squeeze(-1)
+            dt = dt[:, :, None].expand(-1, -1, self.head_dim)
+            dt_bias = self.dt_bias[:, None, ...].expand(-1, self.head_dim)
+            A = -torch.exp(self.A_log_bias.float())
+            A = A[:, None, ...][:, :, None].expand(-1, self.head_dim, self.head_dim).to(dtype=torch.float32)
+            D = torch.zeros((self.num_heads, self.head_dim), dtype=A.dtype, device=A.device)
+            y = selective_state_update(
+                ssm_state,
+                v,
+                dt,
+                A=A,
+                B=k,
+                C=q,
+                D=D,
+                dt_bias=dt_bias,
+                dt_softplus=True,
+            )
+        else:
+            vkq = causal_conv1d_fn(
+                vkq.transpose(1, 2),
+                rearrange(self.conv1d.weight, "d 1 w -> d w"),
+                self.conv1d.bias,
+                initial_states=None,
+                return_final_states=False,
+                activation=None if self.activation == "identity" else self.activation,
+            )
+            v, k, q = torch.split(
+                vkq,
+                [
+                    self.num_key_value_heads * self.head_dim,
+                    self.num_key_value_heads * self.head_dim,
+                    self.num_heads * self.head_dim,
+                ],
+                dim=1,
+            )
+            v = rearrange(v, "b (h n) l -> b h l n", h=self.num_key_value_heads)
+            k = rearrange(k, "b (h n) l -> b h l n", h=self.num_key_value_heads)
+            q = rearrange(q, "b (h n) l -> b l h n", h=self.num_heads)
+            k = repeat_kv(k, self.num_key_value_groups).transpose(1, 2)
+            v = repeat_kv(v, self.num_key_value_groups).transpose(1, 2)
+            A = -torch.exp(self.A_log_bias.float())
+            y = mamba_chunk_scan_combined(
+                x = v,
+                dt=dt,
+                dt_softplus=True,
+                A=A,
+                B=k,
+                C=q,
+                chunk_size=self.chunk_size,
+                dt_bias=self.dt_bias,
+                initial_states=None, # currently not supported by mamba_ssm.utils.generation
+                return_final_states=False,
+            )
+        g = rearrange(g, 'b l (h d) -> b l h d', h=self.num_heads)
+        y_true = self.g_norm_swish_gate(y, g)
+        y_true = y_true.view(batch, seqlen, self.hidden_size)
+        y_true = self.o_proj(y_true)
+        return y_true
+    def _get_states_from_cache(self, inference_params, batch_size, initialize_states=False):
+        device = self.conv1d.weight.device
+        dtype = self.conv1d.weight.dtype
+        assert self.layer_idx is not None
+        if self.layer_idx not in inference_params.key_value_memory_dict:
+            batch_shape = (batch_size,)
+            conv_state = torch.zeros(
+                batch_size, 2*self.hidden_size, self.d_conv-1, device=device, dtype=dtype
+            )
+            ssm_state = torch.zeros(
+                batch_size, self.num_heads, self.head_dim, self.head_dim, device=device, dtype=dtype
+            )
+            inference_params.key_value_memory_dict[self.layer_idx] = (conv_state, ssm_state)
+        else:
+            conv_state, ssm_state = inference_params.key_value_memory_dict[self.layer_idx]
+            # TODO: What if batch size changes between generation, and we reuse the same states?
+            if initialize_states:
+                conv_state.zero_()
+                ssm_state.zero_()
+        return conv_state, ssm_state
+    def allocate_inference_cache(self, batch_size, max_seqlen, dtype=None, **kwargs):
+        device = self.conv1d.weight.device
+        dtype = self.conv1d.weight.dtype
+        conv_state = torch.zeros(
+            batch_size, 2*self.hidden_size, self.d_conv-1, device=device, dtype=dtype
+        )
+        ssm_state = torch.zeros(
+            batch_size, self.num_heads, self.head_dim, self.head_dim, device=device, dtype=dtype
+        )
+        return conv_state, ssm_state
+mmMamba_ATTENTION_CLASSES = {
+    'mha': MHA_LM,
+    "mamba2":Mamba2_LM
+}
+# Modified from transformers.model.llama.modeling_llama.LlamaDecoderLayer
+class mmMambaDecoderLayer(nn.Module):
+    def __init__(self, config: mmMambaConfig, layer_idx: int):
+        super().__init__()
+        self.hidden_size = config.hidden_size
+        self.layer_idx = layer_idx
+        self.attention = mmMamba_ATTENTION_CLASSES[config.layers_block_type[layer_idx-8]](config=config, layer_idx=layer_idx)
+        self.feed_forward = mmMambaMLP(config)
+        self.attention_norm = mmMambaRMSNorm(config.hidden_size, eps=config.rms_norm_eps)
+        self.ffn_norm = mmMambaRMSNorm(config.hidden_size, eps=config.rms_norm_eps)
+    def forward(
+        self,
+        hidden_states: torch.Tensor,
+        inference_params = None,
+        output_attentions: Optional[bool] = False,
+        use_cache: Optional[bool] = True,
+        **kwargs,
+    ) -> Tuple[torch.FloatTensor, Optional[Tuple[torch.FloatTensor, torch.FloatTensor]]]:
+        """
+        Args:
+            hidden_states (`torch.FloatTensor`): input to the layer of shape `(batch, seq_len, embed_dim)`
+            output_attentions (`bool`, *optional*):
+                Whether or not to return the attentions tensors of all attention layers. See `attentions` under
+                returned tensors for more detail.
+            use_cache (`bool`, *optional*):
+                If set to `True`, `past_key_values` key value states are returned and can be used to speed up decoding
+                (see `past_key_values`).
+        """
+        #start_time = time.time()
+        residual = hidden_states
+        hidden_states = self.attention_norm(hidden_states)
+        # Self Attention
+        hidden_states = self.attention(
+            hidden_states=hidden_states,
+            inference_params=inference_params,
+            output_attentions=output_attentions,
+            use_cache=use_cache,
+            **kwargs,
+        )
+        hidden_states = residual + hidden_states
+        # Fully Connected
+        residual = hidden_states
+        hidden_states = self.ffn_norm(hidden_states)
+        hidden_states = self.feed_forward(hidden_states)
+        hidden_states = residual + hidden_states
+        outputs = (hidden_states,)
+        if output_attentions:
+            outputs += self_attn_weights
+        #end_time = time.time()
+        #print("language_model_time:", end_time-start_time)
+        return outputs
+    def allocate_inference_cache(self, batch_size, max_seqlen, dtype=None, **kwargs):
+        return self.attention.allocate_inference_cache(batch_size, max_seqlen, dtype=dtype, **kwargs)
+mmMamba_START_DOCSTRING = r"""
+    This model inherits from [`PreTrainedModel`]. Check the superclass documentation for the generic methods the
+    library implements for all its model (such as downloading or saving, resizing the input embeddings, pruning heads
+    etc.)
+    This model is also a PyTorch [torch.nn.Module](https://pytorch.org/docs/stable/nn.html#torch.nn.Module) subclass.
+    Use it as a regular PyTorch Module and refer to the PyTorch documentation for all matter related to general usage
+    and behavior.
+    Parameters:
+        config ([`mmMambaConfig`]):
+            Model configuration class with all the parameters of the model. Initializing with a config file does not
+            load the weights associated with the model, only the configuration. Check out the
+            [`~PreTrainedModel.from_pretrained`] method to load the model weights.
+"""
+# Copied from transformers.models.llama.modeling_llama.LlamaPreTrainedModel with Llama->mmMamba
+@add_start_docstrings(
+    'The bare mmMamba Model outputting raw hidden-states without any specific head on top.',
+    mmMamba_START_DOCSTRING,
+)
+class mmMambaPreTrainedModel(PreTrainedModel):
+    config_class = mmMambaConfig
+    base_model_prefix = 'model'
+    supports_gradient_checkpointing = True
+    _no_split_modules = ['mmMambaDecoderLayer']
+    _skip_keys_device_placement = 'past_key_values'
+    _supports_flash_attn_2 = True
+    def _init_weights(self, module):
+        std = self.config.initializer_range
+        if isinstance(module, nn.Linear):
+            module.weight.data.normal_(mean=0.0, std=std)
+            if module.bias is not None:
+                module.bias.data.zero_()
+        elif isinstance(module, nn.Embedding):
+            module.weight.data.normal_(mean=0.0, std=std)
+            if module.padding_idx is not None:
+                module.weight.data[module.padding_idx].zero_()
+mmMamba_INPUTS_DOCSTRING = r"""
+    Args:
+        input_ids (`torch.LongTensor` of shape `(batch_size, sequence_length)`):
+            Indices of input sequence tokens in the vocabulary. Padding will be ignored by default should you provide
+            it.
+            Indices can be obtained using [`AutoTokenizer`]. See [`PreTrainedTokenizer.encode`] and
+            [`PreTrainedTokenizer.__call__`] for details.
+            [What are input IDs?](../glossary#input-ids)
+        inputs_embeds (`torch.FloatTensor` of shape `(batch_size, sequence_length, hidden_size)`, *optional*):
+            Optionally, instead of passing `input_ids` you can choose to directly pass an embedded representation. This
+            is useful if you want more control over how to convert `input_ids` indices into associated vectors than the
+            model's internal embedding lookup matrix.
+        use_cache (`bool`, *optional*):
+            If set to `True`, `past_key_values` key value states are returned and can be used to speed up decoding (see
+            `past_key_values`).
+        output_attentions (`bool`, *optional*):
+            Whether or not to return the attentions tensors of all attention layers. See `attentions` under returned
+            tensors for more detail.
+        output_hidden_states (`bool`, *optional*):
+            Whether or not to return the hidden states of all layers. See `hidden_states` under returned tensors for
+            more detail.
+        return_dict (`bool`, *optional*):
+            Whether or not to return a [`~utils.ModelOutput`] instead of a plain tuple.
+"""
+# Modified from transformers.model.llama.modeling_llama.LlamaModel
+@add_start_docstrings(
+    'The bare mmMamba Model outputting raw hidden-states without any specific head on top.',
+    mmMamba_START_DOCSTRING,
+)
+class mmMambaModel(mmMambaPreTrainedModel):
+    """
+    Transformer decoder consisting of *config.num_hidden_layers* layers. Each layer is a [`mmMambaDecoderLayer`]
+    Args:
+        config: mmMambaConfig
+    """
+    _auto_class = 'AutoModel'
+    def __init__(self, config: mmMambaConfig):
+        super().__init__(config)
+        self.padding_idx = config.pad_token_id
+        self.vocab_size = config.vocab_size
+        self.config = config
+        self.tok_embeddings = nn.Embedding(config.vocab_size, config.hidden_size, self.padding_idx)
+        self.layers = nn.ModuleList([mmMambaDecoderLayer(config, (layer_idx+8)) for layer_idx in range(config.num_hidden_layers)])
+        self.norm = mmMambaRMSNorm(config.hidden_size, eps=config.rms_norm_eps)
+        self.gradient_checkpointing = False
+        # Initialize weights and apply final processing
+        self.post_init()
+    def get_input_embeddings(self):
+        return self.tok_embeddings
+    def set_input_embeddings(self, value):
+        self.tok_embeddings = value
+    @add_start_docstrings_to_model_forward(mmMamba_INPUTS_DOCSTRING)
+    def forward(
+        self,
+        input_ids: torch.LongTensor = None,
+        inference_params=None,
+        inputs_embeds: Optional[torch.FloatTensor] = None,
+        use_cache: Optional[bool] = True,
+        output_attentions: Optional[bool] = None,
+        output_hidden_states: Optional[bool] = None,
+        return_dict: Optional[bool] = None,
+    ) -> Union[Tuple, BaseModelOutputWithPast]:
+        output_attentions = output_attentions if output_attentions is not None else self.config.output_attentions
+        output_hidden_states = (
+            output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
+        )
+        use_cache = use_cache if use_cache is not None else self.config.use_cache
+        return_dict = return_dict if return_dict is not None else self.config.use_return_dict
+        if self.config.attn_implementation == 'flash_attention_2':
+            _import_flash_attn()
+        # retrieve input_ids and inputs_embeds
+        if input_ids is not None and inputs_embeds is not None:
+            raise ValueError('You cannot specify both input_ids and inputs_embeds at the same time')
+        elif input_ids is not None:
+            batch_size, seq_length = input_ids.shape[:2]
+        elif inputs_embeds is not None:
+            batch_size, seq_length = inputs_embeds.shape[:2]
+        else:
+            raise ValueError('You have to specify either input_ids or inputs_embeds')
+        if inputs_embeds is None:
+            inputs_embeds = self.tok_embeddings(input_ids)
+        # embed positions
+        hidden_states = inputs_embeds
+        if self.gradient_checkpointing and self.training:
+            if use_cache:
+                logger.warning_once(
+                    '`use_cache=True` is incompatible with gradient checkpointing. Setting `use_cache=False`...'
+                )
+                use_cache = False
+        # decoder layers
+        all_hidden_states = () if output_hidden_states else None
+        all_self_attns = () if output_attentions else None
+        next_decoder_cache = () if use_cache else None
+        for idx, decoder_layer in enumerate(self.layers):
+            if output_hidden_states:
+                all_hidden_states += (hidden_states,)
+            if self.gradient_checkpointing and self.training:
+                def create_custom_forward(module):
+                    def custom_forward(*inputs):
+                        # None for past_key_value
+                        return module(*inputs, output_attentions, None)
+                    return custom_forward
+                layer_outputs = torch.utils.checkpoint.checkpoint(
+                    create_custom_forward(decoder_layer),
+                    hidden_states,
+                    inference_params,
+                    None,
+                )
+            else:
+                layer_outputs = decoder_layer(
+                    hidden_states,
+                    inference_params=inference_params,
+                    output_attentions=output_attentions,
+                    use_cache=use_cache,
+                )
+            hidden_states = layer_outputs[0]
+            if output_attentions:
+                all_self_attns += layer_outputs[1]
+        hidden_states = self.norm(hidden_states)
+        # add hidden states from the last decoder layer
+        if output_hidden_states:
+            all_hidden_states += (hidden_states,)
+        next_cache = None
+        if not return_dict:
+            return tuple(v for v in [hidden_states, next_cache, all_hidden_states, all_self_attns] if v is not None)
+        return BaseModelOutputWithPast(
+            last_hidden_state=hidden_states,
+            past_key_values=next_cache,
+            hidden_states=all_hidden_states,
+            attentions=all_self_attns,
+        )
+    def allocate_inference_cache(self, batch_size, max_seqlen, dtype=None, **kwargs):
+        return {
+            layer.layer_idx: layer.allocate_inference_cache(batch_size, max_seqlen, dtype=dtype, **kwargs)
+            for layer in self.layers
+        }
+# Modified from transformers.model.llama.modeling_llama.LlamaForCausalLM
+class mmMambaForCausalLM(mmMambaPreTrainedModel):
+    _auto_class = 'AutoModelForCausalLM'
+    _tied_weights_keys = ['output.weight']
+    def __init__(self, config):
+        super().__init__(config)
+        self.model = mmMambaModel(config)
+        self.vocab_size = config.vocab_size
+        self.output = nn.Linear(config.hidden_size, config.vocab_size, bias=False)
+        # Initialize weights and apply final processing
+        self.post_init()
+    def get_input_embeddings(self):
+        return self.model.tok_embeddings
+    def set_input_embeddings(self, value):
+        self.model.tok_embeddings = value
+    def get_output_embeddings(self):
+        return self.output
+    def set_output_embeddings(self, new_embeddings):
+        self.output = new_embeddings
+    def set_decoder(self, decoder):
+        self.model = decoder
+    def get_decoder(self):
+        return self.model
+    @add_start_docstrings_to_model_forward(mmMamba_INPUTS_DOCSTRING)
+    @replace_return_docstrings(output_type=CausalLMOutputWithPast, config_class=_CONFIG_FOR_DOC)
+    def forward(
+        self,
+        input_ids: torch.LongTensor = None,
+        inference_params=None,
+        num_last_tokens=0,
+        inputs_embeds: Optional[torch.FloatTensor] = None,
+        labels: Optional[torch.LongTensor] = None,
+        use_cache: Optional[bool] = True,
+        output_attentions: Optional[bool] = None,
+        output_hidden_states: Optional[bool] = None,
+        return_dict: Optional[bool] = None,
+    ) -> Union[Tuple, CausalLMOutputWithPast]:
+        r"""
+        Args:
+            labels (`torch.LongTensor` of shape `(batch_size, sequence_length)`, *optional*):
+                Labels for computing the masked language modeling loss. Indices should either be in `[0, ...,
+                config.vocab_size]` or -100 (see `input_ids` docstring). Tokens with indices set to `-100` are ignored
+                (masked), the loss is only computed for the tokens with labels in `[0, ..., config.vocab_size]`.
+        Returns:
+        Example:
+        ```python
+        >>> from transformers import AutoTokenizer, mmMambaForCausalLM
+        >>> model = mmMambaForCausalLM.from_pretrained(PATH_TO_CONVERTED_WEIGHTS)
+        >>> tokenizer = AutoTokenizer.from_pretrained(PATH_TO_CONVERTED_TOKENIZER)
+        >>> prompt = "Hey, are you conscious? Can you talk to me?"
+        >>> inputs = tokenizer(prompt, return_tensors="pt")
+        >>> # Generate
+        >>> generate_ids = model.generate(inputs.input_ids, max_length=30)
+        >>> tokenizer.batch_decode(generate_ids, skip_special_tokens=True, clean_up_tokenization_spaces=False)[0]
+        "Hey, are you conscious? Can you talk to me?\nI'm not conscious, but I can talk to you."
+        ```"""
+        output_attentions = output_attentions if output_attentions is not None else self.config.output_attentions
+        output_hidden_states = (
+            output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
+        )
+        return_dict = return_dict if return_dict is not None else self.config.use_return_dict
+        # decoder outputs consists of (dec_features, layer_state, dec_hidden, dec_attn)
+        outputs = self.model(
+            input_ids=input_ids,
+            inference_params=inference_params,
+            inputs_embeds=inputs_embeds,
+            use_cache=use_cache,
+            output_attentions=output_attentions,
+            output_hidden_states=output_hidden_states,
+            return_dict=return_dict,
+        )
+        hidden_states = outputs[0]
+        if num_last_tokens > 0:
+            hidden_states = hidden_states[:, -num_last_tokens:]
+        logits = self.output(hidden_states)
+        logits = logits.float()
+        loss = None
+        if labels is not None:
+            # Shift so that tokens < n predict n
+            shift_logits = logits[..., :-1, :].contiguous()
+            shift_labels = labels[..., 1:].contiguous()
+            # Flatten the tokens
+            loss_fct = CrossEntropyLoss()
+            shift_logits = shift_logits.view(-1, self.config.vocab_size)
+            shift_labels = shift_labels.view(-1)
+            # Enable model parallelism
+            shift_labels = shift_labels.to(shift_logits.device)
+            loss = loss_fct(shift_logits, shift_labels)
+        if not return_dict:
+            output = (logits,) + outputs[1:]
+            return (loss,) + output if loss is not None else output
+        device = input_ids.device if input_ids is not None else inputs_embeds.device
+        output = CausalLMOutputWithPast(
+            loss=loss,
+            logits=logits,
+            past_key_values=outputs.past_key_values,
+            hidden_states=outputs.hidden_states,
+            attentions=outputs.attentions,
+        )
+        output['logits'] = output['logits'].to(device)
+        return output
+    def allocate_inference_cache(self, batch_size, max_seqlen, dtype=None, **kwargs):
+        return self.model.allocate_inference_cache(batch_size, max_seqlen, dtype=dtype, **kwargs)
+    @staticmethod
+    def _reorder_cache(past_key_values, beam_idx):
+        reordered_past = ()
+        for layer_past in past_key_values:
+            reordered_past += (
+                tuple(past_state.index_select(0, beam_idx.to(past_state.device)) for past_state in layer_past),
+            )
+        return reordered_past
+    @torch.no_grad()
+    def stream_chat(
+        self,
+        tokenizer,
+        query: str,
+        history: List[Tuple[str, str]] = [],
+        max_new_tokens: int = 1024,
+        do_sample: bool = True,
+        temperature: float = 0.8,
+        top_p: float = 0.8,
+        **kwargs,
+    ):
+        """
+        Return a generator in format: (response, history)
+        Eg.
+        ('你好，有什么可以帮助您的吗', [('你好', '你好，有什么可以帮助您的吗')])
+        ('你好，有什么可以帮助您的吗？', [('你好', '你好，有什么可以帮助您的吗？')])
+        """
+        if BaseStreamer is None:
+            raise ModuleNotFoundError(
+                'The version of `transformers` is too low. Please make sure '
+                'that you have installed `transformers>=4.28.0`.'
+            )
+        response_queue = queue.Queue(maxsize=20)
+        class ChatStreamer(BaseStreamer):
+            def __init__(self, tokenizer) -> None:
+                super().__init__()
+                self.tokenizer = tokenizer
+                self.queue = response_queue
+                self.query = query
+                self.history = history
+                self.response = ''
+                self.cache = []
+                self.received_inputs = False
+                self.queue.put((self.response, history + [(self.query, self.response)]))
+            def put(self, value):
+                if len(value.shape) > 1 and value.shape[0] > 1:
+                    raise ValueError('ChatStreamer only supports batch size 1')
+                elif len(value.shape) > 1:
+                    value = value[0]
+                if not self.received_inputs:
+                    # The first received value is input_ids, ignore here
+                    self.received_inputs = True
+                    return
+                self.cache.extend(value.tolist())
+                token = self.tokenizer.decode(self.cache, skip_special_tokens=True)
+                if token.strip() != '<|im_end|>':
+                    self.response = self.response + token
+                    history = self.history + [(self.query, self.response)]
+                    self.queue.put((self.response, history))
+                    self.cache = []
+                else:
+                    self.end()
+            def end(self):
+                self.queue.put(None)
+        def stream_producer():
+            return self.chat(
+                tokenizer=tokenizer,
+                query=query,
+                streamer=ChatStreamer(tokenizer=tokenizer),
+                history=history,
+                max_new_tokens=max_new_tokens,
+                do_sample=do_sample,
+                temperature=temperature,
+                top_p=top_p,
+                **kwargs,
+            )
+        def consumer():
+            producer = threading.Thread(target=stream_producer)
+            producer.start()
+            while True:
+                res = response_queue.get()
+                if res is None:
+                    return
+                yield res
+        return consumer()

modeling_mmMamba_chat.py ADDED Viewed

	@@ -0,0 +1,517 @@

+import warnings
+from dataclasses import dataclass
+from typing import Any, List, Optional, Tuple, Union
+from copy import deepcopy
+import torch.distributed as dist
+import torch.utils.checkpoint
+import torch.nn as nn
+import transformers
+from peft import LoraConfig, get_peft_model
+from torch import nn
+from torch.nn import CrossEntropyLoss
+from transformers import (AutoModel, GenerationConfig, LlamaForCausalLM,
+                          LlamaTokenizer, Qwen2ForCausalLM)
+from transformers.modeling_outputs import CausalLMOutputWithPast
+from transformers.modeling_utils import PreTrainedModel
+from transformers.utils import logging as hf_logging
+from transformers.trainer_pt_utils import LabelSmoother
+from transformers.generation import GreedySearchDecoderOnlyOutput, SampleDecoderOnlyOutput, TextStreamer
+IGNORE_TOKEN_ID = LabelSmoother.ignore_index
+from .configuration_mmMamba_chat import mmMambaChatConfig
+from .conversation import get_conv_template
+from .modeling_mmMamba import mmMambaForCausalLM
+from .modeling_mmMamba_embedding import mmMambaEmbedding
+from transformers.cache_utils import Cache, DynamicCache
+from typing import Any, Dict, List, Optional, Tuple, Union
+import sys
+from mamba_ssm.utils.generation import InferenceParams
+from mamba_ssm.utils.generation import sample, update_graph_cache, modify_logit_for_repetition_penalty
+import time
+import logging
+logger = hf_logging.get_logger(__name__)
+def version_cmp(v1, v2, op='eq'):
+    import operator
+    from packaging import version
+    op_func = getattr(operator, op)
+    return op_func(version.parse(v1), version.parse(v2))
+@torch.inference_mode()
+def decode(
+    input_ids,
+    model,
+    max_length,
+    max_new_tokens=None,
+    top_k=1,
+    top_p=0.0,
+    min_p=0.0,
+    temperature=1.0,
+    repetition_penalty=1.0,
+    eos_token_id=None,
+    pad_token_id=None,
+    do_sample=False,
+    teacher_outputs=None,
+    vocab_size=None,
+    use_cache=False,
+    enable_timing=False,
+    streamer: Optional[TextStreamer] = None,
+    pixel_values=None,
+    hd_input_ids=None,
+):
+    """Decoding, either greedy or with top-k or top-p sampling.
+    If top-k = 0, don't limit the number of candidates (pure sampling).
+    Top-k and top-p can be used together. If top_k > 0 and top_p > 0, then top-k is applied first,
+    then top-p.
+    We assume that all sequences in the same batch have the same length.
+    Arguments:
+        input_ids: (batch, seq_len)
+        max_length: int
+        teacher_outputs (optional): (batch, seq_len). If provided, instead of sampling from the
+            logits, the next token is taken from the teacher_outputs. Useful for testing.
+    Returns: GreedySearchDecoderOnlyOutput or SampleDecoderOnlyOutput, with the following fields:
+        sequences: (batch, max_length)
+        scores: tuples of (batch, vocab_size)
+    """
+    if streamer is not None:
+        streamer.put(input_ids.cpu())
+    scores, sequences = [], [input_ids.cpu()]
+    if max_new_tokens is not None:
+        max_length = sequences[-1].shape[1] + max_new_tokens  # override max_length if max_new_tokens is set
+    batch_size, seqlen_og = input_ids.shape
+    teacher_output_len = teacher_outputs.shape[1] if teacher_outputs is not None else 0
+    if not hasattr(model, "_decoding_cache"):
+        model._decoding_cache = None
+    model._decoding_cache = update_graph_cache(
+        model,
+        model._decoding_cache,
+        batch_size,
+        seqlen_og,
+        max_length,
+    )
+    inference_params = model._decoding_cache.inference_params
+    inference_params.reset(max_length, batch_size)
+    def get_logits(input_ids, inference_params):
+        decoding = inference_params.seqlen_offset > 0
+        if decoding:
+            position_ids = torch.full(
+                (batch_size, 1),
+                inference_params.seqlen_offset,
+                dtype=torch.long,
+                device=input_ids.device,
+            )
+        else:
+            position_ids = None
+        if not decoding:
+            logits = model(
+                input_ids,
+                position_ids=position_ids,
+                inference_params=inference_params,
+                num_last_tokens=1,
+                return_dict=True,
+                pixel_values=pixel_values,
+            ).logits.squeeze(dim=1)
+        else:
+            logits = model._decoding_cache.run(
+                input_ids, position_ids, inference_params.seqlen_offset
+            ).squeeze(dim=1)
+        return logits[..., :vocab_size] if vocab_size is not None else logits
+    def sample_tokens(logits, inference_params):
+        if teacher_outputs is None or teacher_output_len <= inference_params.seqlen_offset:
+            token = sample(logits, top_k=top_k, top_p=top_p, min_p=min_p, temperature=temperature)
+        else:
+            token = teacher_outputs[:, inference_params.seqlen_offset]
+        # return rearrange(token, "b -> b 1")
+        return token.unsqueeze(1)
+    def should_stop(current_token, inference_params):
+        if inference_params.seqlen_offset == 0:
+            return False
+        if eos_token_id is not None and (current_token == eos_token_id).all():
+            return True
+        if inference_params.seqlen_offset >= max_length - 1:
+            return True
+        return False
+    start = torch.cuda.Event(enable_timing=enable_timing)
+    end = torch.cuda.Event(enable_timing=enable_timing)
+    if enable_timing:
+        start.record()
+    sequences_cat = input_ids
+    while not should_stop(sequences[-1], inference_params):
+        torch.cuda.synchronize()
+        torch.cuda.reset_max_memory_allocated()
+        score = get_logits(sequences[-1].cuda(), inference_params)
+        inference_params.seqlen_offset += sequences[-1].shape[1]
+        if repetition_penalty == 1.0:
+            sampled_tokens = sample_tokens(score, inference_params)
+        else:
+            logits = modify_logit_for_repetition_penalty(
+                score.clone(), sequences_cat, repetition_penalty
+            )
+            sampled_tokens = sample_tokens(logits, inference_params)
+            sequences_cat = torch.cat([sequences_cat, sampled_tokens], dim=1)
+        sequences.append(sampled_tokens.cpu())
+        if streamer is not None:
+            streamer.put(sampled_tokens.cpu())
+    if streamer is not None:
+        streamer.end()
+    if enable_timing:
+        end.record()
+        torch.cuda.synchronize()
+        print(f"Prompt processing + decoding time: {(start.elapsed_time(end)):.0f}ms")
+    output_cls = GreedySearchDecoderOnlyOutput if top_k == 1 else SampleDecoderOnlyOutput
+    return output_cls(sequences=torch.cat(sequences, dim=1), scores=tuple(scores))
+class MambaGenerationMixin:
+    def allocate_inference_cache(self, batch_size, max_seqlen, dtype=None, **kwargs):
+        raise NotImplementedError
+    def generate(
+        self,
+        input_ids,
+        do_sample=False,
+        max_length=256,
+        max_new_tokens=None,
+        top_k=1,
+        top_p=0.0,
+        temperature=1.0,
+        return_dict_in_generate=False,
+        output_scores=False,
+        **kwargs
+    ):
+        if not do_sample:
+            top_k = 1
+        output = decode(
+            input_ids, self, max_length=max_length, max_new_tokens=max_new_tokens, top_k=top_k, top_p=top_p, temperature=temperature, **kwargs
+        )
+        if not output_scores:
+            output.scores = None
+        return output if return_dict_in_generate else output.sequences
+class mmMambaChatModel(PreTrainedModel):
+    config_class = mmMambaChatConfig
+    # main_input_name = 'pixel_values'
+    _no_split_modules = ['InternVisionModel', 'LlamaDecoderLayer', 'InternLM2DecoderLayer',
+                         'Phi3DecoderLayer', 'Qwen2DecoderLayer']
+    _supports_flash_attn_2 = True
+    def __init__(self, config: mmMambaChatConfig, embedding_model=None, language_model=None):
+        super().__init__(config)
+        assert version_cmp(transformers.__version__, '4.37.0', 'ge')
+        image_size = config.force_image_size or config.embedding_config.image_size
+        patch_size = config.embedding_config.patch_size
+        self.image_size = image_size
+        self.patch_size = patch_size
+        self.select_layer = config.select_layer
+        self.template = config.template
+        self.num_image_token = int((image_size // patch_size) ** 2 * (config.downsample_ratio ** 2))
+        self.downsample_ratio = config.downsample_ratio
+        self.ps_version = config.ps_version
+        self.use_thumbnail = config.use_thumbnail
+        if embedding_model is not None:
+            self.embedding_model = embedding_model
+        else:
+            self.embedding_model = mmMambaEmbedding(config.embedding_config)
+        if language_model is not None:
+            self.language_model = language_model
+        else:
+             self.language_model = mmMambaForCausalLM(config.llm_config)
+        self.img_context_token_id = None
+        self.conv_template = get_conv_template(self.template)
+        self.system_message = self.conv_template.system_message
+        self.num_samples = 0
+    def forward(
+            self,
+            input_ids: torch.LongTensor = None,
+            pixel_values: torch.FloatTensor = None,
+            input_embeds: Optional[torch.FloatTensor] = None,
+            position_ids: Optional[torch.LongTensor] = None,
+            image_flags: Optional[torch.LongTensor] = None,
+            labels: Optional[torch.LongTensor] = None,
+            use_cache: Optional[bool] = True,
+            output_attentions: Optional[bool] = None,
+            output_hidden_states: Optional[bool] = None,
+            return_dict: Optional[bool] = None,
+            statistics: Optional[torch.LongTensor] = None,
+            loss_weight: Optional[List] = None,
+            loss_reduction_all_gather: Optional[bool] = False,
+            query = None,
+            hd_input_ids = None,
+            hd_input_embeds = None,
+            hd_labels = None,
+            hd_loss_weight = None,
+            inference_params = None,
+            num_last_tokens: int = 0,
+    ) -> Union[Tuple, CausalLMOutputWithPast]:
+        return_dict = return_dict if return_dict is not None else self.config.use_return_dict
+        if pixel_values is not None or input_ids.shape[0] > 1:
+            if image_flags is not None:
+                #image_flags = image_flags.squeeze(-1)
+                pixel_values = pixel_values[image_flags == 1]
+            if pixel_values==[]:
+                pixel_values = None
+            if getattr(self.embedding_model.config, 'pixel_shuffle_loc', None) in ['post']:
+                assert hd_input_ids is not None, 'hd_input_ids is required for pixel_shuffle_loc=post'
+                embedding_input_ids = hd_input_ids
+            else:
+                embedding_input_ids = input_ids
+            image_embeds, input_embeds = self.embedding_model(input_ids=embedding_input_ids,
+                                                              pixel_values=pixel_values,
+                                                              use_cache=use_cache,
+                                                              return_dict=return_dict,
+                                                              inference_params=inference_params)
+            B, N = embedding_input_ids.shape
+            image_batch_size = pixel_values.shape[0] if pixel_values is not None else 0
+            C = image_embeds.shape[-1]
+            input_embeds = input_embeds.reshape(B * N, C)
+            if torch.distributed.is_initialized() and torch.distributed.get_rank() == 0:
+                #print(f'dynamic ViT batch size: {image_batch_size}, images per sample: {image_batch_size / B}, dynamic token length: {N}')
+                if statistics is not None:
+                    num_samples, num_padding_tokens, num_padding_images = statistics.tolist()
+                    self.num_samples += num_samples
+                    print(f'total_samples={self.num_samples}, {num_samples=}, {num_padding_tokens=}, {num_padding_images=}')
+            if image_batch_size != 0:
+                if getattr(self.embedding_model.config, 'pixel_shuffle_loc', None) == 'post':
+                    B, N = input_ids.shape
+                    llm_input_embeds = torch.zeros(input_ids.shape[1], C, device=input_ids.device, dtype=input_embeds.dtype)
+                    llm_selected = input_ids.flatten() == self.img_context_token_id
+                    hd_llm_selected = hd_input_ids.flatten() == self.img_context_token_id
+                    llm_input_embeds[~llm_selected] = input_embeds[~hd_llm_selected]
+                    llm_input_embeds[llm_selected] = image_embeds.reshape(-1, C)
+                    input_embeds = llm_input_embeds
+            input_embeds = input_embeds.reshape(B, N, C)
+        else:
+            input_embeds = self.embedding_model.get_input_embeddings(input_ids)
+            hd_input_ids = input_ids
+            hd_input_embeds = input_embeds
+            next_past_key_values = []
+            if getattr(self.embedding_model.config, 'pixel_shuffle_loc', None) in ['post']:
+                embedding_input_embeds = hd_input_embeds
+            else:
+                embedding_input_embeds = input_embeds
+            for layer_idx, layer_module in enumerate(self.embedding_model.encoder):
+                outputs = layer_module(
+                    hidden_states=embedding_input_embeds,
+                    use_cache=use_cache,
+                    return_dict=return_dict,
+                    inference_params=inference_params,
+                )
+                embedding_input_embeds = outputs[0]
+            input_embeds = embedding_input_embeds
+        if self.config.normalize_encoder_output:
+            input_embeds = input_embeds / input_embeds.norm(dim=-1, keepdim=True)
+        outputs = self.language_model(
+            inputs_embeds=input_embeds,
+            use_cache=use_cache,
+            output_attentions=output_attentions,
+            output_hidden_states=output_hidden_states,
+            return_dict=return_dict,
+            inference_params=inference_params,
+            num_last_tokens=num_last_tokens
+        )
+        logits = outputs.logits
+        loss = None
+        if labels is not None and loss_weight is not None:
+            loss_weight = torch.tensor(loss_weight, dtype=torch.float32, device=labels.device)
+            # Shift so that tokens < n predict n
+            shift_logits = logits[..., :-1, :].contiguous()
+            shift_labels = labels[..., 1:].contiguous()
+            shift_weights = loss_weight[..., 1:].contiguous()
+            # Flatten the tokens
+            loss_fct = CrossEntropyLoss(reduction='none')
+            shift_logits = shift_logits.view(-1, self.language_model.config.vocab_size)
+            shift_labels = shift_labels.view(-1)
+            shift_weights = shift_weights.view(-1)
+            # Enable model parallelism
+            shift_labels = shift_labels.to(shift_logits.device)
+            shift_weights = shift_weights.to(shift_logits.device)
+            loss = loss_fct(shift_logits, shift_labels)
+            shift_weights_sum = shift_weights.sum()
+            if loss_reduction_all_gather:
+                dist.all_reduce(shift_weights_sum, op=dist.ReduceOp.AVG)
+            loss = loss * shift_weights
+            loss = loss.sum() / shift_weights_sum
+        elif labels is not None:
+            # Shift so that tokens < n predict n
+            shift_logits = logits[..., :-1, :].contiguous()
+            shift_labels = labels[..., 1:].contiguous()
+            # Flatten the tokens
+            loss_fct = CrossEntropyLoss()
+            shift_logits = shift_logits.view(-1, self.language_model.config.vocab_size)
+            shift_labels = shift_labels.view(-1)
+            # Enable model parallelism
+            shift_labels = shift_labels.to(shift_logits.device)
+            loss = loss_fct(shift_logits, shift_labels)
+        if not return_dict:
+            output = (logits,) + outputs[1:]
+            return (loss,) + output if loss is not None else output
+        next_past_key_values = None
+        return CausalLMOutputWithPast(
+            loss=loss,
+            logits=logits,
+            past_key_values=next_past_key_values,
+            hidden_states=outputs.hidden_states,
+            attentions=outputs.attentions,
+        )
+    def batch_chat(self, tokenizer, pixel_values, questions, generation_config, num_patches_list=None,
+                   history=None, return_history=False, IMG_START_TOKEN='<img>', IMG_END_TOKEN='</img>',
+                   IMG_CONTEXT_TOKEN='<IMG_CONTEXT>', verbose=False, image_counts=None):
+        if history is not None or return_history:
+            print('Now multi-turn chat is not supported in batch_chat.')
+            raise NotImplementedError
+        if image_counts is not None:
+            num_patches_list = image_counts
+            print('Warning: `image_counts` is deprecated. Please use `num_patches_list` instead.')
+        img_context_token_id = tokenizer.convert_tokens_to_ids(IMG_CONTEXT_TOKEN)
+        self.img_context_token_id = img_context_token_id
+        if verbose and pixel_values is not None:
+            image_bs = pixel_values.shape[0]
+            print(f'dynamic ViT batch size: {image_bs}')
+        queries = []
+        for idx, num_patches in enumerate(num_patches_list):
+            question = questions[idx]
+            if pixel_values is not None and '<image>' not in question:
+                question = '<image>\n' + question
+            template = get_conv_template(self.template)
+            template.append_message(template.roles[0], question)
+            template.append_message(template.roles[1], None)
+            query = template.get_prompt()
+            image_tokens = IMG_START_TOKEN + IMG_CONTEXT_TOKEN * self.num_image_token * num_patches + IMG_END_TOKEN
+            query = query.replace('<image>', image_tokens, 1)
+            queries.append(query)
+        tokenizer.padding_side = 'left'
+        model_inputs = tokenizer(queries, return_tensors='pt', padding=True)
+        input_ids = model_inputs['input_ids'].cuda()
+        eos_token_id = tokenizer.convert_tokens_to_ids(template.sep)
+        generation_config['eos_token_id'] = eos_token_id
+        generation_output = self.generate(
+            pixel_values=pixel_values,
+            input_ids=input_ids,
+            **generation_config
+        )
+        responses = tokenizer.batch_decode(generation_output, skip_special_tokens=True)
+        responses = [response.split(template.sep)[0].strip() for response in responses]
+        return responses
+    def chat(self, tokenizer, pixel_values, question, generation_config, history=None, return_history=False,
+             num_patches_list=None, IMG_START_TOKEN='<img>', IMG_END_TOKEN='</img>', IMG_CONTEXT_TOKEN='<IMG_CONTEXT>',
+             verbose=False):
+        if history is None and pixel_values is not None and '<image>' not in question:
+            question = '<image>\n' + question
+        if num_patches_list is None:
+            num_patches_list = [pixel_values.shape[0]] if pixel_values is not None else []
+        assert pixel_values is None or len(pixel_values) == sum(num_patches_list)
+        img_context_token_id = tokenizer.convert_tokens_to_ids(IMG_CONTEXT_TOKEN)
+        self.img_context_token_id = img_context_token_id
+        template = get_conv_template(self.template)
+        template.system_message = self.system_message
+        eos_token_id = tokenizer.convert_tokens_to_ids(template.sep)
+        history = [] if history is None else history
+        for (old_question, old_answer) in history:
+            template.append_message(template.roles[0], old_question)
+            template.append_message(template.roles[1], old_answer)
+        template.append_message(template.roles[0], question)
+        template.append_message(template.roles[1], None)
+        query = template.get_prompt()
+        if verbose and pixel_values is not None:
+            image_bs = pixel_values.shape[0]
+            print(f'dynamic ViT batch size: {image_bs}')
+        hd_query = deepcopy(query)
+        for num_patches in num_patches_list:
+            image_tokens = IMG_START_TOKEN + IMG_CONTEXT_TOKEN * self.num_image_token * num_patches + IMG_END_TOKEN
+            hd_image_tokens = IMG_START_TOKEN + IMG_CONTEXT_TOKEN * int(self.num_image_token // self.downsample_ratio**2) * num_patches + IMG_END_TOKEN
+            query = query.replace('<image>', image_tokens, 1)
+            hd_query = hd_query.replace('<image>', hd_image_tokens, 1)
+            #print(hd_query)
+        model_inputs = tokenizer(query, return_tensors='pt')
+        hd_model_inputs = tokenizer(hd_query, return_tensors='pt')
+        input_ids = model_inputs['input_ids'].cuda()
+        hd_input_ids = hd_model_inputs['input_ids'].cuda()
+        generation_config['eos_token_id'] = eos_token_id
+        generation_output = self.generate(
+            pixel_values=pixel_values,
+            input_ids=input_ids,
+            hd_input_ids=hd_input_ids,
+            **generation_config
+        )
+        generation_output = generation_output[:, input_ids.shape[1]:]
+        response = tokenizer.batch_decode(generation_output, skip_special_tokens=True)[0]
+        response = response.split(template.sep)[0].strip()
+        history.append((question, response))
+        if return_history:
+            return response, history
+        else:
+            query_to_print = query.replace(IMG_CONTEXT_TOKEN, '')
+            query_to_print = query_to_print.replace(f'{IMG_START_TOKEN}{IMG_END_TOKEN}', '<image>')
+            if verbose:
+                print(query_to_print, response)
+            return response
+    def generate(self, *args, **kwargs):
+        return MambaGenerationMixin.generate(self, *args, **kwargs)
+    def allocate_inference_cache(self, *args, **kwargs):
+        dict1= self.embedding_model.allocate_inference_cache(*args, **kwargs)
+        dict2= self.language_model.allocate_inference_cache(*args, **kwargs)
+        return {**dict1, **dict2}

modeling_mmMamba_embedding.py ADDED Viewed

	@@ -0,0 +1,966 @@

+# Copyright (c) The mmMamba team and The HuggingFace Inc. team. All rights reserved.
+#
+# This code is based on transformers/src/transformers/models/llama/modeling_llama.py
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+import math
+import queue
+import threading
+import warnings
+from typing import List, Optional, Tuple, Union
+from functools import partial
+import torch
+import torch.nn.functional as F
+import torch.utils.checkpoint
+from einops import rearrange
+from torch import nn
+from torch.nn import BCEWithLogitsLoss, CrossEntropyLoss, MSELoss
+from transformers.activations import ACT2FN
+from transformers.modeling_outputs import (
+    BaseModelOutputWithPast,
+    CausalLMOutputWithPast,
+    SequenceClassifierOutputWithPast,
+)
+from transformers.modeling_utils import PreTrainedModel
+from transformers.cache_utils import Cache
+from transformers.utils import (
+    add_start_docstrings,
+    add_start_docstrings_to_model_forward,
+    logging,
+    replace_return_docstrings,
+)
+from fla.modules import FusedRMSNormSwishGate, RMSNorm, ShortConvolution
+import copy
+from mamba_ssm.ops.triton.ssd_combined import mamba_chunk_scan_combined
+from mamba_ssm.ops.triton.selective_state_update import selective_state_update
+from causal_conv1d import causal_conv1d_fn, causal_conv1d_update
+from transformers.cache_utils import Cache
+import time
+from timm.models.layers import DropPath
+compute_ARank = False # [ARank] Set this to True to compute attention rank
+try:
+    from transformers.generation.streamers import BaseStreamer
+except:  # noqa # pylint: disable=bare-except
+    BaseStreamer = None
+from .configuration_mmMamba_embedding import mmMambaEmbeddingConfig
+import time
+from .configuration_mmMamba import mmMambaConfig
+try:
+    from flash_attn import flash_attn_with_kvcache
+except ImportError:
+    flash_attn_with_kvcache = None
+try:
+    from flash_attn.layers.rotary import RotaryEmbedding
+except ImportError:
+    RotaryEmbedding = None
+import torch.nn.functional as F
+logger = logging.get_logger(__name__)
+_CONFIG_FOR_DOC = "mmMambaEmbeddingConfig"
+flash_attn_func, flash_attn_varlen_func = None, None
+pad_input, index_first_axis, unpad_input = None, None, None
+def _import_flash_attn():
+    global flash_attn_func, flash_attn_varlen_func
+    global pad_input, index_first_axis, unpad_input
+    try:
+        from flash_attn import flash_attn_func as _flash_attn_func, flash_attn_varlen_func as _flash_attn_varlen_func
+        from flash_attn.bert_padding import pad_input as _pad_input, index_first_axis as _index_first_axis, unpad_input as _unpad_input
+        flash_attn_func, flash_attn_varlen_func = _flash_attn_func, _flash_attn_varlen_func
+        pad_input, index_first_axis, unpad_input = _pad_input, _index_first_axis, _unpad_input
+    except ImportError:
+        raise ImportError("flash_attn is not installed.")
+_import_flash_attn()
+def _update_kv_cache(kv, inference_params, layer_idx):
+    """kv: (batch_size, seqlen, 2, nheads, head_dim) or (batch_size, 1, 2, nheads, head_dim)"""
+    # Pre-allocate memory for key-values for inference.
+    num_heads, head_dim = kv.shape[-2:]
+    assert layer_idx in inference_params.key_value_memory_dict
+    kv_cache, _ = inference_params.key_value_memory_dict[layer_idx]
+    # Adjust key and value for inference
+    batch_start = inference_params.batch_size_offset
+    batch_end = batch_start + kv.shape[0]
+    sequence_start = inference_params.seqlen_offset
+    sequence_end = sequence_start + kv.shape[1]
+    assert batch_end <= kv_cache.shape[0]
+    assert sequence_end <= kv_cache.shape[1]
+    assert kv_cache is not None
+    kv_cache[batch_start:batch_end, sequence_start:sequence_end, ...] = kv
+    return kv_cache[batch_start:batch_end, :sequence_end, ...]
+# Copied from transformers.models.llama.modeling_llama.LlamaRMSNorm with Llama->mmMamba
+class mmMambaRMSNorm(nn.Module):
+    def __init__(self, hidden_size, eps=1e-6):
+        """
+        mmMambaRMSNorm is equivalent to T5LayerNorm
+        """
+        super().__init__()
+        self.weight = nn.Parameter(torch.ones(hidden_size))
+        self.variance_epsilon = eps
+    def forward(self, hidden_states):
+        input_dtype = hidden_states.dtype
+        hidden_states = hidden_states.to(torch.float32)
+        variance = hidden_states.pow(2).mean(-1, keepdim=True)
+        hidden_states = hidden_states * torch.rsqrt(variance + self.variance_epsilon)
+        return self.weight * hidden_states.to(input_dtype)
+class mmMambaMLP(nn.Module):
+    def __init__(self, config):
+        super().__init__()
+        self.config = config
+        self.hidden_size = config.hidden_size
+        self.intermediate_size = config.intermediate_size
+        self.w1 = nn.Linear(self.hidden_size, self.intermediate_size, bias=False)
+        self.w3 = nn.Linear(self.hidden_size, self.intermediate_size, bias=False)
+        self.w2 = nn.Linear(self.intermediate_size, self.hidden_size, bias=False)
+        self.act_fn = ACT2FN[config.hidden_act]
+    def forward(self, x):
+        down_proj = self.w2(self.act_fn(self.w1(x)) * self.w3(x))
+        return down_proj
+# Copied from transformers.model.llama.modeling_llama.repeat_kv
+def repeat_kv(hidden_states: torch.Tensor, n_rep: int) -> torch.Tensor:
+    """
+    This is the equivalent of torch.repeat_interleave(x, dim=1, repeats=n_rep). The hidden states go from (batch,
+    num_key_value_heads, seqlen, head_dim) to (batch, num_attention_heads, seqlen, head_dim)
+    """
+    batch, num_key_value_heads, slen, head_dim = hidden_states.shape
+    if n_rep == 1:
+        return hidden_states
+    hidden_states = hidden_states[:, :, None, :, :].expand(batch, num_key_value_heads, n_rep, slen, head_dim)
+    return hidden_states.reshape(batch, num_key_value_heads * n_rep, slen, head_dim)
+def repeat_kv2(hidden_states: torch.Tensor, n_rep: int) -> torch.Tensor:
+    """
+    This is the equivalent of torch.repeat_interleave(x, dim=1, repeats=n_rep). The hidden states go from (batch,
+    num_key_value_heads, seqlen, head_dim) to (batch, num_attention_heads, seqlen, head_dim)
+    """
+    batch, num_key_value_heads, head_dim = hidden_states.shape
+    if n_rep == 1:
+        return hidden_states
+    hidden_states = hidden_states[:, :, None,  :].expand(batch, num_key_value_heads, n_rep, head_dim)
+    return hidden_states.reshape(batch, num_key_value_heads * n_rep, head_dim)
+class MHA_LM(nn.Module):
+    """Multi-headed attention from 'Attention Is All You Need' paper"""
+    def __init__(self, config: mmMambaEmbeddingConfig, layer_idx: int):
+        super().__init__()
+        self.config = config
+        self.layer_idx = layer_idx
+        self.hidden_size = config.hidden_size
+        self.num_heads = config.num_attention_heads
+        self.head_dim = self.hidden_size // self.num_heads
+        self.num_key_value_heads = config.num_key_value_heads
+        self.num_key_value_groups = self.num_heads // self.num_key_value_heads
+        self.max_position_embeddings = config.max_position_embeddings
+        self.is_causal = True
+        self.rotary_emb_dim = self.head_dim
+        self.softmax_scale = None
+        self.causal = True
+        if (self.head_dim * self.num_heads) != self.hidden_size:
+            raise ValueError(
+                f"hidden_size must be divisible by num_heads (got `hidden_size`: {self.hidden_size}"
+                f" and `num_heads`: {self.num_heads})."
+            )
+        self.wqkv = nn.Linear(
+            self.hidden_size,
+            (self.num_heads + 2 * self.num_key_value_heads) * self.head_dim,
+            bias=False,
+        )
+        self.wo = nn.Linear(self.num_heads * self.head_dim, self.hidden_size, bias=False)
+        assert RotaryEmbedding is not None, "rotary requires flash_attn to be installed"
+        self.rotary_emb = RotaryEmbedding(
+            self.head_dim,
+            base=self.config.rope_theta,
+            interleaved=False,
+            device=self.wo.weight.device,
+        )
+    def _shape(self, tensor: torch.Tensor, seq_len: int, bsz: int):
+        return tensor.view(bsz, seq_len, self.num_heads, self.head_dim).transpose(1, 2).contiguous()
+    def _update_kv_cache(self, kv, inference_params):
+        """kv: (batch_size, seqlen, 2, nheads, head_dim) or (batch_size, 1, 2, nheads, head_dim)"""
+        assert self.layer_idx is not None, "Generation requires layer_idx in the constructor"
+        return _update_kv_cache(kv, inference_params, self.layer_idx)
+    def _apply_rotary_update_kvcache_attention(self, q, kv, inference_params):
+        """
+        Fast path that combine 3 steps: apply rotary to Q and K, update kv cache, and apply attention.
+        q: (batch_size, seqlen_q, nheads, head_dim)
+        kv: (batch_size, seqlen_k, 2, nheads_kv, head_dim)
+        """
+        assert inference_params is not None and inference_params.seqlen_offset > 0
+        if self.rotary_emb_dim > 0:
+            self.rotary_emb._update_cos_sin_cache(
+                inference_params.max_seqlen, device=q.device, dtype=q.dtype
+            )
+            rotary_cos, rotary_sin = self.rotary_emb._cos_cached, self.rotary_emb._sin_cached
+        else:
+            rotary_cos, rotary_sin = None, None
+        batch = q.shape[0]
+        kv_cache, _ = inference_params.key_value_memory_dict[self.layer_idx]
+        kv_cache = kv_cache[:batch]
+        cache_seqlens = (
+            inference_params.lengths_per_sample[:batch]
+            if inference_params.lengths_per_sample is not None
+            else inference_params.seqlen_offset
+        )
+        assert flash_attn_with_kvcache is not None, "flash_attn must be installed"
+        context = flash_attn_with_kvcache(
+            q,
+            kv_cache[:, :, 0],
+            kv_cache[:, :, 1],
+            kv[:, :, 0],
+            kv[:, :, 1],
+            rotary_cos=rotary_cos,
+            rotary_sin=rotary_sin,
+            cache_seqlens=cache_seqlens,
+            softmax_scale=self.softmax_scale,
+            causal=self.causal,
+            rotary_interleaved=self.rotary_emb.interleaved if self.rotary_emb_dim > 0 else False,
+        )
+        return context
+    def _update_kvcache_attention(self, q, kv, inference_params):
+        """Write kv to inference_params, then do attention"""
+        if (
+            inference_params.seqlen_offset == 0
+            or flash_attn_with_kvcache is None
+        ):
+            # TODO: this only uses seqlen_offset and not lengths_per_sample.
+            kv = self._update_kv_cache(kv, inference_params)
+            k, v = kv.unbind(dim=-3)
+            #k = torch.repeat_interleave(k, dim=2, repeats=self.num_heads // self.num_key_value_heads)
+            #v = torch.repeat_interleave(v, dim=2, repeats=self.num_heads // self.num_key_value_heads)
+            attn_output = flash_attn_func(
+                q, k, v, 0.0, softmax_scale=None, causal=self.causal
+            )
+            return attn_output
+        else:
+            batch = q.shape[0]
+            kv_cache, _ = inference_params.key_value_memory_dict[self.layer_idx]
+            kv_cache = kv_cache[:batch]
+            cache_seqlens = (
+                inference_params.lengths_per_sample[:batch]
+                if inference_params.lengths_per_sample is not None
+                else inference_params.seqlen_offset
+            )
+            return flash_attn_with_kvcache(
+                q,
+                kv_cache[:, :, 0],
+                kv_cache[:, :, 1],
+                kv[:, :, 0],
+                kv[:, :, 1],
+                cache_seqlens=cache_seqlens,
+                softmax_scale=self.softmax_scale,
+                causal=self.causal,
+            )
+    def forward(
+        self,
+        hidden_states: torch.Tensor,
+        inference_params = None,
+        output_attentions: bool = False,
+        cache_position: Optional[torch.LongTensor] = None,#------------------------------------------------------------------------
+        use_cache: bool = False,
+        **kwargs,
+    ) -> Tuple[torch.Tensor, Optional[torch.Tensor], Optional[Tuple[torch.Tensor]]]:
+        if inference_params is not None and self.layer_idx not in inference_params.key_value_memory_dict:
+            inference_params.key_value_memory_dict[self.layer_idx] = self.allocate_inference_cache(
+                hidden_states.shape[0], inference_params.max_seqlen, dtype=hidden_states.dtype
+            )
+        seqlen_offset = (
+            0
+            if inference_params is None
+            else (
+                inference_params.lengths_per_sample
+                if inference_params.lengths_per_sample is not None
+                else inference_params.seqlen_offset
+            )
+        )
+        bsz, q_len, _ = hidden_states.size()
+        rotary_max_seqlen = inference_params.max_seqlen if inference_params is not None else None
+        qkv = self.wqkv(hidden_states)
+        qkv = rearrange(
+            qkv,
+            "b q (h gs d) -> b q h gs d",
+            gs=2 + self.num_key_value_groups,
+            d=self.head_dim,
+        )
+        q = qkv[..., : self.num_key_value_groups, :]
+        q = rearrange(q, "b q h gs d -> b q (h gs) d")
+        kv = qkv[..., self.num_key_value_groups:, :].transpose(2,3)
+        #kv = rearrange(kv, "b q h gs d -> b q (h gs) d")
+        #kv = rearrange(kv, "... (two hkv d) -> ... two hkv d", two=2, d=self.head_dim)
+        if (
+            inference_params is None
+            or inference_params.seqlen_offset == 0
+            or (self.rotary_emb_dim == 0 or self.rotary_emb_dim % 16 != 0)
+        ):
+            if self.rotary_emb_dim > 0:
+                q, kv = self.rotary_emb(
+                    q, kv, seqlen_offset=seqlen_offset, max_seqlen=rotary_max_seqlen
+                )
+            if inference_params is None:
+                k, v = kv.unbind(dim=-3)
+                k = torch.repeat_interleave(k, dim=2, repeats=self.num_heads // self.num_key_value_heads)
+                v = torch.repeat_interleave(v, dim=2, repeats=self.num_heads // self.num_key_value_heads)
+                context = F.scaled_dot_product_attention(
+                    q.transpose(1, 2), k.transpose(1, 2), v.transpose(1, 2), is_causal=True, scale=None
+                ).transpose(1, 2)
+            else:
+                context = self._update_kvcache_attention(q, kv, inference_params)
+        else:
+            context = self._apply_rotary_update_kvcache_attention(q, kv, inference_params)
+        context = rearrange(context, "... h d -> ... (h d)")
+        out = self.wo(context)
+        return out
+    def allocate_inference_cache(self, batch_size, max_seqlen, dtype=None):
+        dtype = self.wo.weight.dtype if dtype is None else dtype
+        device = self.wo.weight.device
+        kv_cache = torch.empty(
+            batch_size, max_seqlen, 2, self.num_key_value_heads, self.head_dim, dtype=dtype, device=device,
+        )
+        return kv_cache, None
+class Mamba2_LM(nn.Module):
+    """
+    LoLCATs attention implementation initialized from a
+    `LlamaAttention` or `MistralAttention` object (base_attn)
+    Most of the arguments are directly tied to argparse args
+    - For now we don't support padding.
+    """
+    def __init__(self, config: mmMambaConfig, layer_idx: Optional[int] = None,
+                 elementwise_affine: Optional[bool] = True,
+                 norm_eps: float = 1e-5,
+                 ):
+        super().__init__()
+        self.config = config
+        self.hidden_size = config.hidden_size
+        self.num_heads = config.num_attention_heads
+        self.head_dim = self.hidden_size // self.num_heads
+        self.num_key_value_heads = config.num_key_value_heads
+        self.num_key_value_groups = self.num_heads // self.num_key_value_heads
+        self.max_position_embeddings = config.max_position_embeddings
+        self.layer_idx = layer_idx
+        self.bias = False
+        self.chunk_size = 128
+        conv_bias = True
+        self.conv_bias = conv_bias
+        self.d_conv = 2
+        self.activation="silu"
+        self.max_position_embeddings = config.max_position_embeddings
+        self.rope_theta = config.rope_theta
+        self.wvkqgdt = nn.Linear(
+            self.hidden_size,
+            (self.num_heads + 2 * self.num_key_value_heads + self.num_heads) * self.head_dim + self.num_heads,
+            bias=self.bias
+        )
+        self.o_proj = nn.Linear(self.num_heads * self.head_dim, self.hidden_size, bias=False)
+        self.device = self.wvkqgdt.weight.device
+        self.dtype = self.wvkqgdt.weight.dtype
+        conv_dim = self.num_heads * self.head_dim + 2 * self.num_key_value_heads * self.head_dim
+        self.conv1d = nn.Conv1d(
+            in_channels=conv_dim,
+            out_channels=conv_dim,
+            bias=self.conv_bias,
+            kernel_size=self.d_conv,
+            groups=conv_dim,
+            padding=self.d_conv - 1,
+            device=self.device,
+            dtype=self.dtype
+        )
+        with torch.no_grad():
+            self.conv1d.weight.zero_()
+            self.conv1d.weight[:, 0, 1] = 1
+            self.conv1d.bias.zero_()
+        # Activation after conv
+        if self.activation == "identity":
+            self.act = nn.Identity()
+        elif self.activation in ["silu", "swish"]:
+            self.act = nn.SiLU()
+        else:
+            raise ValueError(f"Unknown activation {self.activation}")
+        self.g_norm_swish_gate = FusedRMSNormSwishGate(hidden_size=self.head_dim, elementwise_affine=elementwise_affine, eps=norm_eps).to(self.dtype).to(self.device)
+        dt = torch.exp(
+            torch.rand(self.num_heads, dtype=self.dtype, device=self.device) * (math.log(0.1) - math.log(0.001))
+            + math.log(0.001)
+        )
+        dt = torch.clamp(dt, min=0.001)
+        # Inverse of softplus: https://github.com/pytorch/pytorch/issues/72759
+        inv_dt = dt + torch.log(-torch.expm1(-dt))
+        self.dt_bias = nn.Parameter(inv_dt)
+        self.dt_bias._no_weight_decay = True
+        A_log_bias = torch.zeros(self.num_heads, dtype=self.dtype, device=self.device)
+        self.A_log_bias = nn.Parameter(A_log_bias)
+        self.A_log_bias._no_weight_decay = True
+    def forward(self,
+                hidden_states: torch.Tensor,
+                inference_params = None,
+                output_attentions: bool = False,
+                use_cache: bool = True,
+                **kwargs,
+               ) -> Tuple[torch.Tensor, Optional[torch.Tensor], Optional[Tuple[torch.Tensor]]]:
+        hidden_states = hidden_states.to(self.dtype)
+        vkqgdt = self.wvkqgdt(hidden_states)
+        vkq, g, dt = torch.split(
+                vkqgdt,
+                [
+                    (2*self.num_key_value_heads+self.num_heads) * self.head_dim,
+                    self.num_heads * self.head_dim,
+                    self.num_heads,
+                ],
+                dim=2,
+            )
+        batch, seqlen, _ = hidden_states.shape
+        conv_state, ssm_state = None, None
+        if inference_params is not None:
+            conv_state, ssm_state = self._get_states_from_cache(inference_params, batch)
+        if use_cache and inference_params.seqlen_offset==0:
+            vkq, new_conv_states = causal_conv1d_fn(
+                vkq.transpose(1, 2),
+                rearrange(self.conv1d.weight, "d 1 w -> d w"),
+                self.conv1d.bias,
+                initial_states=None,
+                return_final_states=True,
+                activation=None if self.activation == "identity" else self.activation,
+            )
+            v, k, q = torch.split(
+                vkq,
+                [
+                    self.num_key_value_heads * self.head_dim,
+                    self.num_key_value_heads * self.head_dim,
+                    self.num_heads * self.head_dim,
+                ],
+                dim=1,
+            )
+            v = rearrange(v, "b (h n) l -> b h l n", h=self.num_key_value_heads)
+            k = rearrange(k, "b (h n) l -> b h l n", h=self.num_key_value_heads)
+            q = rearrange(q, "b (h n) l -> b l h n", h=self.num_heads)
+            k = repeat_kv(k, self.num_key_value_groups).transpose(1, 2)
+            v = repeat_kv(v, self.num_key_value_groups).transpose(1, 2)
+            A = -torch.exp(self.A_log_bias.float())
+            y, new_ssm_states = mamba_chunk_scan_combined(
+                x = v,
+                #x = v / F.softplus(A_log).to(v.dtype).unsqueeze(-1),
+                dt=dt,
+                dt_softplus=True,
+                A=A,
+                B=k,
+                C=q,
+                chunk_size=self.chunk_size,
+                dt_bias=self.dt_bias,
+                initial_states=None, # currently not supported by mamba_ssm.utils.generation
+                return_final_states=True,
+            )
+            conv_state.copy_(new_conv_states)
+            ssm_state.copy_(new_ssm_states)
+        elif use_cache and inference_params.seqlen_offset>0:
+            vkq = causal_conv1d_update(
+                vkq.transpose(1, 2).squeeze(-1),
+                conv_state,
+                self.conv1d.weight.squeeze(1),
+                self.conv1d.bias,
+                self.activation,
+            )
+            v, k, q = torch.split(
+                vkq,
+                [
+                    self.num_key_value_heads * self.head_dim,
+                    self.num_key_value_heads * self.head_dim,
+                    self.num_heads * self.head_dim,
+                ],
+                dim=1,
+            )
+            v = rearrange(v, "b (h n) -> b h n", h=self.num_key_value_heads)
+            k = rearrange(k, "b (h n) -> b h n", h=self.num_key_value_heads)
+            q = rearrange(q, "b (h n) -> b h n", h=self.num_heads)
+            k = repeat_kv2(k, self.num_key_value_groups)
+            v = repeat_kv2(v, self.num_key_value_groups)
+            dt = dt.transpose(1, 2).squeeze(-1)
+            dt = dt[:, :, None].expand(-1, -1, self.head_dim)
+            dt_bias = self.dt_bias[:, None, ...].expand(-1, self.head_dim)
+            A = -torch.exp(self.A_log_bias.float())
+            A = A[:, None, ...][:, :, None].expand(-1, self.head_dim, self.head_dim).to(dtype=torch.float32)
+            D = torch.zeros((self.num_heads, self.head_dim), dtype=A.dtype, device=A.device)
+            y = selective_state_update(
+                ssm_state,
+                v,
+                dt,
+                A=A,
+                B=k,
+                C=q,
+                D=D,
+                dt_bias=dt_bias,
+                dt_softplus=True,
+            )
+        else:
+            vkq = causal_conv1d_fn(
+                vkq.transpose(1, 2),
+                rearrange(self.conv1d.weight, "d 1 w -> d w"),
+                self.conv1d.bias,
+                initial_states=None,
+                return_final_states=False,
+                activation=None if self.activation == "identity" else self.activation,
+            )
+            v, k, q = torch.split(
+                vkq,
+                [
+                    self.num_key_value_heads * self.head_dim,
+                    self.num_key_value_heads * self.head_dim,
+                    self.num_heads * self.head_dim,
+                ],
+                dim=1,
+            )
+            v = rearrange(v, "b (h n) l -> b h l n", h=self.num_key_value_heads)
+            k = rearrange(k, "b (h n) l -> b h l n", h=self.num_key_value_heads)
+            q = rearrange(q, "b (h n) l -> b l h n", h=self.num_heads)
+            k = repeat_kv(k, self.num_key_value_groups).transpose(1, 2)
+            v = repeat_kv(v, self.num_key_value_groups).transpose(1, 2)
+            A = -torch.exp(self.A_log_bias.float())
+            y = mamba_chunk_scan_combined(
+                x = v,
+                dt=dt,
+                dt_softplus=True,
+                A=A,
+                B=k,
+                C=q,
+                chunk_size=self.chunk_size,
+                dt_bias=self.dt_bias,
+                initial_states=None, # currently not supported by mamba_ssm.utils.generation
+                return_final_states=False,
+            )
+        g = rearrange(g, 'b l (h d) -> b l h d', h=self.num_heads)
+        y_true = self.g_norm_swish_gate(y, g)
+        y_true = y_true.view(batch, seqlen, self.hidden_size)
+        y_true = self.o_proj(y_true)
+        return y_true
+    def _get_states_from_cache(self, inference_params, batch_size, initialize_states=False):
+        device = self.conv1d.weight.device
+        dtype = self.conv1d.weight.dtype
+        assert self.layer_idx is not None
+        if self.layer_idx not in inference_params.key_value_memory_dict:
+            batch_shape = (batch_size,)
+            conv_state = torch.zeros(
+                batch_size, 2*self.hidden_size, self.d_conv-1, device=device, dtype=dtype
+            )
+            ssm_state = torch.zeros(
+                batch_size, self.num_heads, self.head_dim, self.head_dim, device=device, dtype=dtype
+            )
+            inference_params.key_value_memory_dict[self.layer_idx] = (conv_state, ssm_state)
+        else:
+            conv_state, ssm_state = inference_params.key_value_memory_dict[self.layer_idx]
+            # TODO: What if batch size changes between generation, and we reuse the same states?
+            if initialize_states:
+                conv_state.zero_()
+                ssm_state.zero_()
+        return conv_state, ssm_state
+    def allocate_inference_cache(self, batch_size, max_seqlen, dtype=None, **kwargs):
+        device = self.conv1d.weight.device
+        dtype = self.conv1d.weight.dtype
+        conv_state = torch.zeros(
+            batch_size, 2*self.hidden_size, self.d_conv-1, device=device, dtype=dtype
+        )
+        ssm_state = torch.zeros(
+            batch_size, self.num_heads, self.head_dim, self.head_dim, device=device, dtype=dtype
+        )
+        return conv_state, ssm_state
+mmMamba_ATTENTION_CLASSES = {
+    'mha': MHA_LM,
+    "mamba2":Mamba2_LM
+}
+# Modified from transformers.model.llama.modeling_llama.LlamaDecoderLayer
+class mmMambaDecoderLayer(nn.Module):
+    def __init__(self, config: mmMambaEmbeddingConfig, layer_idx: int, drop_path_rate=0.0):
+        super().__init__()
+        self.hidden_size = config.hidden_size
+        self.config = config
+        self.layer_idx = layer_idx
+        self.attention = mmMamba_ATTENTION_CLASSES[config.layers_block_type[layer_idx]](config=config, layer_idx=layer_idx)
+        self.feed_forward = mmMambaMLP(config)
+        self.attention_norm = mmMambaRMSNorm(config.hidden_size, eps=config.rms_norm_eps)
+        self.ffn_norm = mmMambaRMSNorm(config.hidden_size, eps=config.rms_norm_eps)
+        self.drop_path1 = DropPath(drop_path_rate) if drop_path_rate > 0. else nn.Identity()
+        self.drop_path2 = DropPath(drop_path_rate) if drop_path_rate > 0. else nn.Identity()
+    def forward(
+        self,
+        hidden_states: torch.Tensor,
+        inference_params = None,
+        output_attentions: Optional[bool] = False,
+        use_cache: Optional[bool] = True,
+        **kwargs,
+    ) -> Tuple[torch.FloatTensor, Optional[Tuple[torch.FloatTensor, torch.FloatTensor]]]:
+        """
+        Args:
+            hidden_states (`torch.FloatTensor`): input to the layer of shape `(batch, seq_len, embed_dim)`
+            output_attentions (`bool`, *optional*):
+                Whether or not to return the attentions tensors of all attention layers. See `attentions` under
+                returned tensors for more detail.
+            use_cache (`bool`, *optional*)
+        """
+        residual = hidden_states
+        hidden_states = self.attention_norm(hidden_states)
+        # Self Attention
+        hidden_states = self.attention(
+            hidden_states=hidden_states,
+            inference_params=inference_params,
+            output_attentions=output_attentions,
+            use_cache=use_cache,
+            **kwargs,
+        )
+        hidden_states = residual + self.drop_path1(hidden_states)
+        # Fully Connected
+        residual = hidden_states
+        hidden_states = self.ffn_norm(hidden_states)
+        hidden_states = self.feed_forward(hidden_states)
+        hidden_states = residual + self.drop_path2(hidden_states)
+        outputs = (hidden_states,)
+        if output_attentions:
+            outputs += (self_attn_weights,)
+        return outputs
+    def allocate_inference_cache(self, batch_size, max_seqlen, dtype=None, **kwargs):
+        return self.attention.allocate_inference_cache(batch_size, max_seqlen, dtype=dtype, **kwargs)
+class VisionEmbeddings(nn.Module):
+    def __init__(self, config: mmMambaEmbeddingConfig):
+        super().__init__()
+        self.config = config
+        self.embed_dim = config.hidden_size
+        self.image_size = config.image_size
+        self.patch_size = config.patch_size
+        self.class_embedding = nn.Parameter(
+            torch.randn(1, 1, self.embed_dim),
+        )
+        self.patch_embedding = nn.Conv2d(
+            in_channels=self.config.num_channels, out_channels=self.embed_dim, kernel_size=self.patch_size, stride=self.patch_size
+        )
+        self.num_patches = (self.image_size // self.patch_size) ** 2
+        self.num_positions = self.num_patches + 1
+        self.position_embedding = nn.Parameter(torch.randn(1, self.num_positions, self.embed_dim))
+        self.post_init()
+    def post_init(self):
+        for m in self.modules():
+            if isinstance(m, nn.Conv2d):
+                torch.nn.init.normal_(m.weight, mean=0.0, std=0.02)
+                if m.bias is not None:
+                    nn.init.zeros_(m.bias)
+            if isinstance(m, nn.Linear):
+                torch.nn.init.normal_(m.weight, mean=0.0, std=0.02)
+                if m.bias is not None:
+                    nn.init.zeros_(m.bias)
+    def _get_pos_embed(self, pos_embed, H, W):
+        target_dtype = pos_embed.dtype
+        pos_embed = pos_embed.float().reshape(
+            1, self.image_size // self.patch_size, self.image_size // self.patch_size, -1).permute(0, 3, 1, 2)
+        pos_embed = F.interpolate(pos_embed, size=(H, W), mode='bicubic', align_corners=False).\
+            reshape(1, -1, H * W).permute(0, 2, 1).to(target_dtype)
+        return pos_embed
+    def forward(self, pixel_values: torch.FloatTensor,
+                use_cls_token=False,
+                ) -> torch.Tensor:
+        target_dtype = self.patch_embedding.weight.dtype
+        pixel_values = pixel_values.to(target_dtype)
+        patch_embeds = self.patch_embedding(pixel_values)  # shape = [*, channel, width, height]
+        batch_size, _, height, width = patch_embeds.shape
+        patch_embeds = patch_embeds.flatten(2).transpose(1, 2)
+        if use_cls_token:
+            class_embeds = self.class_embedding.expand(batch_size, 1, -1).to(target_dtype)
+            embeddings = torch.cat([class_embeds, patch_embeds], dim=1)
+            assert not self.config.use_2d_sincos_pos_embed, '2D SinCos pos embed is not supported with use_cls_token'
+            position_embedding = torch.cat([
+                self.position_embedding[:, :1, :],
+                self._get_pos_embed(self.position_embedding[:, 1:, :], height, width)
+            ], dim=1)
+            embeddings = embeddings + position_embedding
+        else:
+            position_embedding = self._get_pos_embed(self.position_embedding[:, 1:, :], height, width).to(target_dtype)
+            embeddings = patch_embeds + position_embedding
+        return embeddings
+class mmMambaEmbedding(PreTrainedModel):
+    config_class = mmMambaEmbeddingConfig
+    _supports_flash_attn_2 = True
+    def __init__(self, config: mmMambaEmbeddingConfig):
+        super().__init__(config)
+        self.config = config
+        self.hidden_size = self.config.hidden_size
+        self.gradient_checkpointing = True
+        self.vision_embeddings = VisionEmbeddings(config)
+        self.llm_text_embeddings = nn.Embedding(self.config.llm_vocab_size, self.config.llm_hidden_size)
+        self.special_token_maps = config.special_token_maps
+        if len(self.special_token_maps) > 0:
+            self.special_text_embeddings = nn.Embedding(len(config.special_token_maps), self.config.llm_hidden_size)
+        assert self.config.use_ls is False, 'LS is not supported in mmMamba'
+        if hasattr(config, 'drop_path_rate'):
+            dpr = [x.item() for x in torch.linspace(0, config.drop_path_rate, config.num_hidden_layers)]
+        else:
+            dpr = [0.0] * config.num_hidden_layers
+        self.encoder = nn.ModuleList([
+            mmMambaDecoderLayer(config, idx, dpr[idx]) for idx in range(config.num_hidden_layers)
+        ])
+        if self.config.use_pixel_shuffle_proj:
+            self.pixel_shuffle_proj = nn.Sequential(
+                nn.Linear(int(config.hidden_size / (config.downsample_ratio * config.downsample_ratio)), config.hidden_size),
+                nn.GELU(),
+                nn.Linear(config.hidden_size, config.hidden_size)
+            )
+        self.num_img_tokens = (self.config.image_size // self.config.patch_size) ** 2
+    def set_gradient_checkpointing(self):
+        self.gradient_checkpointing = True
+        for layer in self.encoder:
+            layer.gradient_checkpointing = True
+    def resize_pos_embeddings(self, old_size, new_size, patch_size):
+        pos_emb = self.vision_embeddings.position_embedding
+        _, num_positions, embed_dim = pos_emb.shape
+        cls_emb = pos_emb[:, :1, :]
+        pos_emb = pos_emb[:, 1:, :].reshape(1, old_size // patch_size, old_size // patch_size, -1).permute(0, 3, 1, 2)
+        pos_emb = F.interpolate(pos_emb.float(), size=new_size // patch_size, mode='bicubic', align_corners=False)
+        pos_emb = pos_emb.to(cls_emb.dtype).reshape(1, embed_dim, -1).permute(0, 2, 1)
+        pos_emb = torch.cat([cls_emb, pos_emb], dim=1)
+        self.vision_embeddings.position_embedding = nn.Parameter(pos_emb)
+        self.vision_embeddings.image_size = new_size
+        logger.info('Resized position embeddings from {} to {}'.format(old_size, new_size))
+    def replace_img_tokens(self, input_ids, hidden_states, vision_hidden_states):
+        img_context_token_mask = (input_ids == self.config.img_context_token_id)
+        hidden_states[img_context_token_mask] = hidden_states[img_context_token_mask] * 0.0 + vision_hidden_states.flatten(0, 1)
+        return hidden_states
+    def get_ignore_mask(self, input_ids):
+        ignore_ids = torch.tensor(
+            [self.special_token_maps[token] for token in [IMG_START_TOKEN, IMG_END_TOKEN]],
+            device=input_ids.device)
+        ignore_mask = torch.isin(input_ids, ignore_ids)
+        return ignore_mask
+    def get_text_mask(self, input_ids):
+        txt_mask = (input_ids != self.config.img_context_token_id)
+        return txt_mask
+    def get_input_embeddings(self, input_ids):
+        special_mask = input_ids > self.llm_text_embeddings.weight.shape[0] - 1
+        llm_embeddings = self.llm_text_embeddings(input_ids * (~special_mask).to(input_ids))
+        if len(self.special_token_maps) > 0:
+            special_embeddings = self.special_text_embeddings((input_ids - self.llm_text_embeddings.weight.shape[0]) * special_mask.to(input_ids))
+            special_mask = special_mask.unsqueeze(-1)
+            text_embeddings = llm_embeddings * (~special_mask).to(llm_embeddings) + \
+                                special_embeddings * special_mask.to(llm_embeddings)
+        else:
+            text_embeddings = llm_embeddings
+        return text_embeddings
+    def get_txt_embeddings(self, input_ids):
+        B, L = input_ids.shape
+        txt_mask = (input_ids != self.config.img_context_token_id)
+        txt_embeddings = self.llm_text_embeddings(input_ids[txt_mask])
+        txt_embeddings = txt_embeddings.reshape(-1, txt_embeddings.shape[-1])
+        return txt_embeddings
+    def get_txt_feature(self, input_ids, feature):
+        B, L, C = feature.shape
+        txt_mask = (input_ids != self.config.img_context_token_id)
+        txt_feature = feature[txt_mask].reshape(-1, C)
+        return txt_feature
+    def get_img_feature(self, input_ids, feature):
+        B, L, C = feature.shape
+        img_mask = (input_ids == self.config.img_context_token_id)
+        img_feature = feature[img_mask].reshape(-1, C)
+        return img_feature
+    def pixel_shuffle(self, x, scale_factor=0.5):
+        if getattr(self.config, 'pixel_shuffle_loc', 'pre') == 'post':
+            x = x.view(x.shape[0]//self.num_img_tokens, self.num_img_tokens, -1)
+        n, l, c = x.size()
+        h = w = int(l ** 0.5)
+        # N, W, H, C --> N, W, H * scale, C // scale
+        x = x.reshape(n, w, int(h * scale_factor), int(c / scale_factor))
+        # N, W, H * scale, C // scale --> N, H * scale, W, C // scale
+        x = x.permute(0, 2, 1, 3).contiguous()
+        # N, H * scale, W, C // scale --> N, H * scale, W * scale, C // (scale ** 2)
+        x = x.view(n, int(h * scale_factor), int(w * scale_factor),
+                   int(c / (scale_factor * scale_factor)))
+        x = x.permute(0, 2, 1, 3).reshape(n, int(l * scale_factor * scale_factor), int(c / (scale_factor * scale_factor))).contiguous()
+        if getattr(self.config, 'pixel_shuffle_loc', 'pre') == 'post':
+            x = x.view(int(x.shape[0]*self.num_img_tokens*(self.config.downsample_ratio**2)), -1)
+        return x
+    def forward(
+        self,
+        input_ids: Optional[torch.LongTensor] = None,
+        pixel_values: Optional[torch.FloatTensor] = None,
+        inference_params = None,
+        output_hidden_states: Optional[bool] = None,
+        return_dict: Optional[bool] = None,
+        use_cache: Optional[bool] = True,
+    ):
+        output_hidden_states = (
+            output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
+        )
+        return_dict = return_dict if return_dict is not None else self.config.use_return_dict
+        if pixel_values is not None:
+            if len(pixel_values.shape) == 4:
+                if self.gradient_checkpointing and self.training:
+                    vision_hidden_states = torch.utils.checkpoint.checkpoint(self.vision_embeddings, pixel_values)
+                else:
+                    vision_hidden_states = self.vision_embeddings(pixel_values)
+                if self.config.use_pixel_shuffle_proj and getattr(self.config, 'pixel_shuffle_loc', 'pre') == 'pre':
+                    vision_hidden_states = self.pixel_shuffle(vision_hidden_states, scale_factor=self.config.downsample_ratio)
+                    if self.gradient_checkpointing and self.training:
+                        vision_hidden_states = torch.utils.checkpoint.checkpoint(self.pixel_shuffle_proj, vision_hidden_states)
+                    else:
+                        vision_hidden_states = self.pixel_shuffle_proj(vision_hidden_states)
+                hidden_states = self.get_input_embeddings(input_ids)
+                hidden_states = self.replace_img_tokens(input_ids, hidden_states, vision_hidden_states)
+            else:
+                raise ValueError(f'wrong pixel_values size: {pixel_values.shape}')
+        else:
+            hidden_states = self.get_input_embeddings(input_ids)
+        for layer_idx, layer_module in enumerate(self.encoder):
+            if self.gradient_checkpointing and self.training:
+                assert use_cache is None, 'Gradient checkpointing is not compatible with cache'
+                outputs = torch.utils.checkpoint.checkpoint(layer_module,
+                                                            hidden_states,
+                                                            inference_params,
+                                                            None, False, False,
+                                                            )
+                hidden_states = outputs[0]
+            else:
+                outputs = layer_module(
+                    hidden_states=hidden_states,
+                    inference_params=inference_params,
+                    use_cache=use_cache,
+                )
+                hidden_states = outputs[0]
+        img_feature = self.get_img_feature(input_ids, hidden_states)
+        if self.config.use_pixel_shuffle_proj and getattr(self.config, 'pixel_shuffle_loc', 'pre') == 'post':
+            img_feature = self.pixel_shuffle(img_feature, scale_factor=self.config.downsample_ratio)
+            img_feature = self.pixel_shuffle_proj(img_feature)
+        return img_feature, hidden_states
+    def allocate_inference_cache(self, batch_size, max_seqlen, dtype=None, **kwargs):
+        return {
+            layer.layer_idx: layer.allocate_inference_cache(batch_size, max_seqlen, dtype=dtype, **kwargs)
+            for layer in self.encoder
+        }

special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,47 @@

+{
+  "additional_special_tokens": [
+    "<|im_start|>",
+    "<|im_end|>",
+    "<|action_start|>",
+    "<|action_end|>",
+    "<|interpreter|>",
+    "<|plugin|>",
+    "<img>",
+    "</img>",
+    "<IMG_CONTEXT>",
+    "<quad>",
+    "</quad>",
+    "<ref>",
+    "</ref>",
+    "<box>",
+    "</box>"
+  ],
+  "bos_token": {
+    "content": "<s>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "eos_token": {
+    "content": "</s>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "pad_token": {
+    "content": "</s>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "unk_token": {
+    "content": "<unk>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  }
+}

tokenization_internlm2.py ADDED Viewed

	@@ -0,0 +1,235 @@

+# Copyright (c) The InternLM team and The HuggingFace Inc. team. All rights reserved.
+#
+# This code is based on transformers/src/transformers/models/llama/tokenization_llama.py
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+"""Tokenization classes for InternLM."""
+import os
+from shutil import copyfile
+from typing import Any, Dict, List, Optional, Tuple
+import sentencepiece as spm
+from transformers.tokenization_utils import PreTrainedTokenizer
+from transformers.utils import logging
+logger = logging.get_logger(__name__)
+VOCAB_FILES_NAMES = {'vocab_file': './tokenizer.model'}
+PRETRAINED_VOCAB_FILES_MAP = {}
+# Modified from transformers.model.llama.tokenization_llama.LlamaTokenizer
+class InternLM2Tokenizer(PreTrainedTokenizer):
+    """
+    Construct a InternLM2 tokenizer. Based on byte-level Byte-Pair-Encoding.
+    Args:
+        vocab_file (`str`):
+            Path to the vocabulary file.
+    """
+    vocab_files_names = VOCAB_FILES_NAMES
+    pretrained_vocab_files_map = PRETRAINED_VOCAB_FILES_MAP
+    model_input_names = ['input_ids', 'attention_mask']
+    _auto_class = 'AutoTokenizer'
+    def __init__(
+        self,
+        vocab_file,
+        unk_token='<unk>',
+        bos_token='<s>',
+        eos_token='</s>',
+        pad_token='</s>',
+        sp_model_kwargs: Optional[Dict[str, Any]] = None,
+        add_bos_token=True,
+        add_eos_token=False,
+        decode_with_prefix_space=False,
+        clean_up_tokenization_spaces=False,
+        **kwargs,
+    ):
+        self.sp_model_kwargs = {} if sp_model_kwargs is None else sp_model_kwargs
+        self.vocab_file = vocab_file
+        self.add_bos_token = add_bos_token
+        self.add_eos_token = add_eos_token
+        self.decode_with_prefix_space = decode_with_prefix_space
+        self.sp_model = spm.SentencePieceProcessor(**self.sp_model_kwargs)
+        self.sp_model.Load(vocab_file)
+        self._no_prefix_space_tokens = None
+        super().__init__(
+            bos_token=bos_token,
+            eos_token=eos_token,
+            unk_token=unk_token,
+            pad_token=pad_token,
+            clean_up_tokenization_spaces=clean_up_tokenization_spaces,
+            **kwargs,
+        )
+    @property
+    def no_prefix_space_tokens(self):
+        if self._no_prefix_space_tokens is None:
+            vocab = self.convert_ids_to_tokens(list(range(self.vocab_size)))
+            self._no_prefix_space_tokens = {i for i, tok in enumerate(vocab) if not tok.startswith('▁')}
+        return self._no_prefix_space_tokens
+    @property
+    def vocab_size(self):
+        """Returns vocab size"""
+        return self.sp_model.get_piece_size()
+    @property
+    def bos_token_id(self) -> Optional[int]:
+        return self.sp_model.bos_id()
+    @property
+    def eos_token_id(self) -> Optional[int]:
+        return self.sp_model.eos_id()
+    def get_vocab(self):
+        """Returns vocab as a dict"""
+        vocab = {self.convert_ids_to_tokens(i): i for i in range(self.vocab_size)}
+        vocab.update(self.added_tokens_encoder)
+        return vocab
+    def _tokenize(self, text):
+        """Returns a tokenized string."""
+        return self.sp_model.encode(text, out_type=str)
+    def _convert_token_to_id(self, token):
+        """Converts a token (str) in an id using the vocab."""
+        return self.sp_model.piece_to_id(token)
+    def _convert_id_to_token(self, index):
+        """Converts an index (integer) in a token (str) using the vocab."""
+        token = self.sp_model.IdToPiece(index)
+        return token
+    def _maybe_add_prefix_space(self, tokens, decoded):
+        if tokens and tokens[0] not in self.no_prefix_space_tokens:
+            return ' ' + decoded
+        else:
+            return decoded
+    def convert_tokens_to_string(self, tokens):
+        """Converts a sequence of tokens (string) in a single string."""
+        current_sub_tokens = []
+        out_string = ''
+        prev_is_special = False
+        for token in tokens:
+            # make sure that special tokens are not decoded using sentencepiece model
+            if token in self.all_special_tokens:
+                if not prev_is_special:
+                    out_string += ' '
+                out_string += self.sp_model.decode(current_sub_tokens) + token
+                prev_is_special = True
+                current_sub_tokens = []
+            else:
+                current_sub_tokens.append(token)
+                prev_is_special = False
+        out_string += self.sp_model.decode(current_sub_tokens)
+        out_string = self.clean_up_tokenization(out_string)
+        out_string = self._maybe_add_prefix_space(tokens=tokens, decoded=out_string)
+        return out_string[1:]
+    def save_vocabulary(self, save_directory, filename_prefix: Optional[str] = None) -> Tuple[str]:
+        """
+        Save the vocabulary and special tokens file to a directory.
+        Args:
+            save_directory (`str`):
+                The directory in which to save the vocabulary.
+        Returns:
+            `Tuple(str)`: Paths to the files saved.
+        """
+        if not os.path.isdir(save_directory):
+            logger.error(f'Vocabulary path ({save_directory}) should be a directory')
+            return
+        out_vocab_file = os.path.join(
+            save_directory, (filename_prefix + '-' if filename_prefix else '') + VOCAB_FILES_NAMES['vocab_file']
+        )
+        if os.path.abspath(self.vocab_file) != os.path.abspath(out_vocab_file) and os.path.isfile(self.vocab_file):
+            copyfile(self.vocab_file, out_vocab_file)
+        elif not os.path.isfile(self.vocab_file):
+            with open(out_vocab_file, 'wb') as fi:
+                content_spiece_model = self.sp_model.serialized_model_proto()
+                fi.write(content_spiece_model)
+        return (out_vocab_file,)
+    def build_inputs_with_special_tokens(self, token_ids_0, token_ids_1=None):
+        if self.add_bos_token:
+            bos_token_ids = [self.bos_token_id]
+        else:
+            bos_token_ids = []
+        output = bos_token_ids + token_ids_0
+        if token_ids_1 is not None:
+            output = output + token_ids_1
+        if self.add_eos_token:
+            output = output + [self.eos_token_id]
+        return output
+    def get_special_tokens_mask(
+        self, token_ids_0: List[int], token_ids_1: Optional[List[int]] = None, already_has_special_tokens: bool = False
+    ) -> List[int]:
+        """
+        Retrieve sequence ids from a token list that has no special tokens added. This method is called when adding
+        special tokens using the tokenizer `prepare_for_model` method.
+        Args:
+            token_ids_0 (`List[int]`):
+                List of IDs.
+            token_ids_1 (`List[int]`, *optional*):
+                Optional second list of IDs for sequence pairs.
+            already_has_special_tokens (`bool`, *optional*, defaults to `False`):
+                Whether or not the token list is already formatted with special tokens for the model.
+        Returns:
+            `List[int]`: A list of integers in the range [0, 1]: 1 for a special token, 0 for a sequence token.
+        """
+        if already_has_special_tokens:
+            return super().get_special_tokens_mask(
+                token_ids_0=token_ids_0, token_ids_1=token_ids_1, already_has_special_tokens=True
+            )
+        if token_ids_1 is None:
+            return [1] + ([0] * len(token_ids_0)) + [1]
+        return [1] + ([0] * len(token_ids_0)) + [1, 1] + ([0] * len(token_ids_1)) + [1]
+    def create_token_type_ids_from_sequences(
+        self, token_ids_0: List[int], token_ids_1: Optional[List[int]] = None
+    ) -> List[int]:
+        """
+        Create a mask from the two sequences passed to be used in a sequence-pair classification task. T5 does not make
+        use of token type ids, therefore a list of zeros is returned.
+        Args:
+            token_ids_0 (`List[int]`):
+                List of IDs.
+            token_ids_1 (`List[int]`, *optional*):
+                Optional second list of IDs for sequence pairs.
+        Returns:
+            `List[int]`: List of zeros.
+        """
+        eos = [self.eos_token_id]
+        if token_ids_1 is None:
+            return len(token_ids_0 + eos) * [0]
+        return len(token_ids_0 + eos + token_ids_1 + eos) * [0]

tokenizer.model ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:f868398fc4e05ee1e8aeba95ddf18ddcc45b8bce55d5093bead5bbf80429b48b
+size 1477754

tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,179 @@

+{
+  "added_tokens_decoder": {
+    "0": {
+      "content": "<unk>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "1": {
+      "content": "<s>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "2": {
+      "content": "</s>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "92538": {
+      "content": "<|plugin|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "92539": {
+      "content": "<|interpreter|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "92540": {
+      "content": "<|action_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "92541": {
+      "content": "<|action_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "92542": {
+      "content": "<|im_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "92543": {
+      "content": "<|im_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "92544": {
+      "content": "<img>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "92545": {
+      "content": "</img>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "92546": {
+      "content": "<IMG_CONTEXT>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "92547": {
+      "content": "<quad>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "92548": {
+      "content": "</quad>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "92549": {
+      "content": "<ref>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "92550": {
+      "content": "</ref>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "92551": {
+      "content": "<box>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "92552": {
+      "content": "</box>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    }
+  },
+  "additional_special_tokens": [
+    "<|im_start|>",
+    "<|im_end|>",
+    "<|action_start|>",
+    "<|action_end|>",
+    "<|interpreter|>",
+    "<|plugin|>",
+    "<img>",
+    "</img>",
+    "<IMG_CONTEXT>",
+    "<quad>",
+    "</quad>",
+    "<ref>",
+    "</ref>",
+    "<box>",
+    "</box>"
+  ],
+  "auto_map": {
+    "AutoTokenizer": [
+      "tokenization_internlm2.InternLM2Tokenizer",
+      null
+    ]
+  },
+  "bos_token": "<s>",
+  "chat_template": "{{ bos_token }}{% for message in messages %}{{'<|im_start|>' + message['role'] + '\n' + message['content'] + '<|im_end|>' + '\n'}}{% endfor %}{% if add_generation_prompt %}{{ '<|im_start|>assistant\n' }}{% endif %}",
+  "clean_up_tokenization_spaces": false,
+  "eos_token": "</s>",
+  "model_max_length": 8192,
+  "pad_token": "</s>",
+  "tokenizer_class": "InternLM2Tokenizer",
+  "unk_token": "<unk>"
+}