Upload 5 files

Browse files

Files changed (5) hide show

added_tokens.json +40 -0
config.json +35 -0
configuration_mixformer_sequential.py +59 -0
generation_config.json +4 -0
merges.txt +0 -0

added_tokens.json ADDED Viewed

	@@ -0,0 +1,40 @@

+{
+  "\t\t": 50294,
+  "\t\t\t": 50293,
+  "\t\t\t\t": 50292,
+  "\t\t\t\t\t": 50291,
+  "\t\t\t\t\t\t": 50290,
+  "\t\t\t\t\t\t\t": 50289,
+  "\t\t\t\t\t\t\t\t": 50288,
+  "\t\t\t\t\t\t\t\t\t": 50287,
+  "  ": 50286,
+  "   ": 50285,
+  "    ": 50284,
+  "     ": 50283,
+  "      ": 50282,
+  "       ": 50281,
+  "        ": 50280,
+  "         ": 50279,
+  "          ": 50278,
+  "           ": 50277,
+  "            ": 50276,
+  "             ": 50275,
+  "              ": 50274,
+  "               ": 50273,
+  "                ": 50272,
+  "                 ": 50271,
+  "                  ": 50270,
+  "                   ": 50269,
+  "                    ": 50268,
+  "                     ": 50267,
+  "                      ": 50266,
+  "                       ": 50265,
+  "                        ": 50264,
+  "                         ": 50263,
+  "                          ": 50262,
+  "                           ": 50261,
+  "                            ": 50260,
+  "                             ": 50259,
+  "                              ": 50258,
+  "                               ": 50257
+}

config.json ADDED Viewed

	@@ -0,0 +1,35 @@

+{
+  "_name_or_path": "phi-1.5-half",
+  "activation_function": "gelu_new",
+  "architecture": {
+    "block_cls": "parallel",
+    "mixer": {},
+    "mlp": {
+      "mlp_cls": "mlp"
+    }
+  },
+  "architectures": [
+    "MixFormerSequentialForCausalLM"
+  ],
+  "auto_map": {
+    "AutoConfig": "configuration_mixformer_sequential.MixFormerSequentialConfig",
+    "AutoModelForCausalLM": "modeling_mixformer_sequential.MixFormerSequentialForCausalLM"
+  },
+  "embd_layer": "default",
+  "embd_pdrop": 0.0,
+  "initializer_range": 0.02,
+  "layer_norm_epsilon": 1e-05,
+  "model_type": "mixformer-sequential",
+  "n_embd": 2048,
+  "n_head": 32,
+  "n_inner": null,
+  "n_layer": 24,
+  "n_positions": 2048,
+  "phyagi_version": "0.0.4.dev",
+  "resid_pdrop": 0.0,
+  "rotary_dim": 32,
+  "tie_word_embeddings": false,
+  "torch_dtype": "float16",
+  "transformers_version": "4.32.1",
+  "vocab_size": 51200
+}

configuration_mixformer_sequential.py ADDED Viewed

	@@ -0,0 +1,59 @@

+# Copyright (c) Microsoft Corporation.
+# Licensed under the MIT license.
+import math
+from typing import Any, Dict, List, Optional, Union
+from transformers import PretrainedConfig
+class MixFormerSequentialConfig(PretrainedConfig):
+    """MixFormer (sequential for DeepSpeed) configuration."""
+    model_type = "mixformer-sequential"
+    attribute_map = {
+        "max_position_embeddings": "n_positions",
+        "hidden_size": "n_embd",
+        "num_attention_heads": "n_head",
+        "num_hidden_layers": "n_layer",
+        "input_emb_layer": "embd_layer",  # `input_emb_layer` key is for backward compatibility
+        "blocks": "architecture",  # `blocks` key is for backward compatibility
+    }
+    def __init__(
+        self,
+        vocab_size: Optional[int] = 50304,
+        n_positions: Optional[int] = 2048,
+        n_embd: Optional[int] = 1024,
+        n_layer: Optional[int] = 20,
+        n_inner: Optional[int] = None,
+        n_head: Optional[int] = 16,
+        rotary_dim: Optional[int] = 32,
+        activation_function: Optional[str] = "gelu_new",
+        embd_layer: Optional[str] = "default",
+        architecture: Union[Dict[str, Any], List[Dict[str, Any]]] = None,
+        embd_pdrop: Optional[float] = 0.0,
+        resid_pdrop: Optional[float] = 0.0,
+        layer_norm_epsilon: Optional[float] = 1e-5,
+        initializer_range: Optional[float] = 0.02,
+        tie_word_embeddings: Optional[bool] = False,
+        pad_vocab_size_multiple: Optional[int] = 64,
+        **kwargs
+    ) -> None:
+        self.vocab_size = int(math.ceil(vocab_size / pad_vocab_size_multiple) * pad_vocab_size_multiple)
+        self.n_positions = n_positions
+        self.n_embd = n_embd
+        self.n_layer = n_layer
+        self.n_inner = n_inner
+        self.n_head = n_head
+        self.rotary_dim = min(rotary_dim, n_embd // n_head)
+        self.activation_function = activation_function
+        self.embd_layer = embd_layer
+        self.architecture = architecture
+        self.embd_pdrop = embd_pdrop
+        self.resid_pdrop = resid_pdrop
+        self.layer_norm_epsilon = layer_norm_epsilon
+        self.initializer_range = initializer_range
+        super().__init__(tie_word_embeddings=tie_word_embeddings, **kwargs)

generation_config.json ADDED Viewed

	@@ -0,0 +1,4 @@

+{
+  "_from_model_config": true,
+  "transformers_version": "4.32.1"
+}

merges.txt ADDED Viewed

The diff for this file is too large to render. See raw diff