Upload 11 files

Browse files

Files changed (11) hide show

config.json +28 -12
generation_config.json +8 -4
pytorch_model-00001-of-00003.bin +2 -2
pytorch_model-00002-of-00003.bin +2 -2
pytorch_model-00003-of-00003.bin +2 -2
pytorch_model.bin.index.json +327 -407
qwen.tiktoken +0 -0
special_tokens_map.json +3 -22
tokenizer_config.json +11 -31
trainer_state.json +0 -0
training_args.bin +1 -1

config.json CHANGED Viewed

@@ -1,18 +1,26 @@
 {
   "_name_or_path": "TranscoreM",
   "architectures": [
-    "TransCoreMLlamaForCausalLM"
   ],
-  "bos_token_id": 1,
-  "eos_token_id": 2,
   "freeze_mm_mlp_adapter": false,
-  "hidden_act": "silu",
   "hidden_size": 5120,
   "image_aspect_ratio": "pad",
   "image_grid_pinpoints": null,
   "initializer_range": 0.02,
-  "intermediate_size": 13824,
-  "max_position_embeddings": 4096,
   "mm_hidden_size": 1024,
   "mm_projector_lr": null,
   "mm_projector_type": "mlp2x_gelu",
@@ -22,21 +30,29 @@
   "mm_vision_select_layer": -2,
   "mm_vision_tower": "PCIResearch/clip-vit-large-patch14-336",
   "model_type": "transcorem",
   "num_attention_heads": 40,
   "num_hidden_layers": 40,
-  "num_key_value_heads": 40,
-  "pad_token_id": 0,
-  "pretraining_tp": 1,
-  "rms_norm_eps": 1e-06,
-  "rope_scaling": null,
   "tie_word_embeddings": false,
   "torch_dtype": "bfloat16",
   "transformers_version": "4.31.0",
   "tune_entire_model": false,
   "tune_mm_mlp_adapter": false,
   "tune_vision_tower": false,
   "use_cache": true,
   "use_mm_proj": true,
   "vision_tower_lr": null,
-  "vocab_size": 40076
 }

 {
   "_name_or_path": "TranscoreM",
   "architectures": [
+    "TransCoreMQWenForCausalLM"
   ],
+  "attn_dropout_prob": 0.0,
+  "auto_map": {
+    "AutoConfig": "configuration_qwen.QWenConfig",
+    "AutoModelForCausalLM": "modeling_qwen.QWenLMHeadModel"
+  },
+  "bf16": false,
+  "emb_dropout_prob": 0.0,
+  "fp16": false,
+  "fp32": false,
   "freeze_mm_mlp_adapter": false,
   "hidden_size": 5120,
   "image_aspect_ratio": "pad",
   "image_grid_pinpoints": null,
   "initializer_range": 0.02,
+  "intermediate_size": 27392,
+  "kv_channels": 128,
+  "layer_norm_epsilon": 1e-06,
+  "max_position_embeddings": 8192,
   "mm_hidden_size": 1024,
   "mm_projector_lr": null,
   "mm_projector_type": "mlp2x_gelu",
   "mm_vision_select_layer": -2,
   "mm_vision_tower": "PCIResearch/clip-vit-large-patch14-336",
   "model_type": "transcorem",
+  "no_bias": true,
   "num_attention_heads": 40,
   "num_hidden_layers": 40,
+  "onnx_safe": null,
+  "rotary_emb_base": 10000,
+  "rotary_pct": 1.0,
+  "scale_attn_weights": true,
+  "seq_length": 2048,
+  "softmax_in_fp32": false,
   "tie_word_embeddings": false,
+  "tokenizer_class": "QWenTokenizer",
   "torch_dtype": "bfloat16",
   "transformers_version": "4.31.0",
   "tune_entire_model": false,
   "tune_mm_mlp_adapter": false,
   "tune_vision_tower": false,
   "use_cache": true,
+  "use_cache_kernel": false,
+  "use_cache_quantization": false,
+  "use_dynamic_ntk": true,
+  "use_flash_attn": "auto",
+  "use_logn_attn": true,
   "use_mm_proj": true,
   "vision_tower_lr": null,
+  "vocab_size": 152064
 }

generation_config.json CHANGED Viewed

@@ -1,7 +1,11 @@
 {
-  "_from_model_config": true,
-  "bos_token_id": 1,
-  "eos_token_id": 2,
-  "pad_token_id": 0,
   "transformers_version": "4.31.0"
 }

 {
+  "chat_format": "chatml",
+  "do_sample": true,
+  "eos_token_id": 151643,
+  "max_new_tokens": 512,
+  "max_window_size": 6144,
+  "pad_token_id": 151643,
+  "top_k": 0,
+  "top_p": 0.5,
   "transformers_version": "4.31.0"
 }

pytorch_model-00001-of-00003.bin CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:de02cbbc48e89b149e1f62c021d493334e182df83bf25088cb4a83233d78eddb
-size 9978995635

 version https://git-lfs.github.com/spec/v1
+oid sha256:c3d26e6bdb05f2a63b3f57771b1d12dc2eb46d87ef1c5ef7acad6a2503682c59
+size 9963536445

pytorch_model-00002-of-00003.bin CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:1feffa38d5faaec608395a5c26c8922e87dada3f4f60f6c0cf1e9b816bf1e870
-size 9956592151

 version https://git-lfs.github.com/spec/v1
+oid sha256:f4a334d3fc0e4d08f5f421e086e67dd1e290ea8aaa9f9dbf2087c23f6047e514
+size 9878405831

pytorch_model-00003-of-00003.bin CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:477fa092a4777ee58e25ae12a4a8020532bf40b3e082e9895a7280893e21ebad
-size 6324617089

 version https://git-lfs.github.com/spec/v1
+oid sha256:8912e2567776930ec3d1c7f534340f639432c6705c43b738353a35335fc972a6
+size 8555684085

pytorch_model.bin.index.json CHANGED Viewed

@@ -1,414 +1,334 @@
 {
   "metadata": {
-    "total_size": 26260065280
   },
   "weight_map": {
     "lm_head.weight": "pytorch_model-00003-of-00003.bin",
-    "model.embed_tokens.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.0.input_layernorm.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.0.mlp.down_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.0.mlp.gate_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.0.mlp.up_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.0.post_attention_layernorm.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.0.self_attn.k_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.0.self_attn.o_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.0.self_attn.q_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.0.self_attn.rotary_emb.inv_freq": "pytorch_model-00001-of-00003.bin",
-    "model.layers.0.self_attn.v_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.1.input_layernorm.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.1.mlp.down_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.1.mlp.gate_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.1.mlp.up_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.1.post_attention_layernorm.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.1.self_attn.k_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.1.self_attn.o_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.1.self_attn.q_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.1.self_attn.rotary_emb.inv_freq": "pytorch_model-00001-of-00003.bin",
-    "model.layers.1.self_attn.v_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.10.input_layernorm.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.10.mlp.down_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.10.mlp.gate_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.10.mlp.up_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.10.post_attention_layernorm.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.10.self_attn.k_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.10.self_attn.o_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.10.self_attn.q_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.10.self_attn.rotary_emb.inv_freq": "pytorch_model-00001-of-00003.bin",
-    "model.layers.10.self_attn.v_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.11.input_layernorm.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.11.mlp.down_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.11.mlp.gate_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.11.mlp.up_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.11.post_attention_layernorm.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.11.self_attn.k_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.11.self_attn.o_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.11.self_attn.q_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.11.self_attn.rotary_emb.inv_freq": "pytorch_model-00001-of-00003.bin",
-    "model.layers.11.self_attn.v_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.12.input_layernorm.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.12.mlp.down_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.12.mlp.gate_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.12.mlp.up_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.12.post_attention_layernorm.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.12.self_attn.k_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.12.self_attn.o_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.12.self_attn.q_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.12.self_attn.rotary_emb.inv_freq": "pytorch_model-00001-of-00003.bin",
-    "model.layers.12.self_attn.v_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.13.input_layernorm.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.13.mlp.down_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.13.mlp.gate_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.13.mlp.up_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.13.post_attention_layernorm.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.13.self_attn.k_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.13.self_attn.o_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.13.self_attn.q_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.13.self_attn.rotary_emb.inv_freq": "pytorch_model-00001-of-00003.bin",
-    "model.layers.13.self_attn.v_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.14.input_layernorm.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.14.mlp.down_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.14.mlp.gate_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.14.mlp.up_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.14.post_attention_layernorm.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.14.self_attn.k_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.14.self_attn.o_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.14.self_attn.q_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.14.self_attn.rotary_emb.inv_freq": "pytorch_model-00001-of-00003.bin",
-    "model.layers.14.self_attn.v_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.15.input_layernorm.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.15.mlp.down_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.15.mlp.gate_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.15.mlp.up_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.15.post_attention_layernorm.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.15.self_attn.k_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.15.self_attn.o_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.15.self_attn.q_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.15.self_attn.rotary_emb.inv_freq": "pytorch_model-00002-of-00003.bin",
-    "model.layers.15.self_attn.v_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.16.input_layernorm.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.16.mlp.down_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.16.mlp.gate_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.16.mlp.up_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.16.post_attention_layernorm.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.16.self_attn.k_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.16.self_attn.o_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.16.self_attn.q_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.16.self_attn.rotary_emb.inv_freq": "pytorch_model-00002-of-00003.bin",
-    "model.layers.16.self_attn.v_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.17.input_layernorm.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.17.mlp.down_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.17.mlp.gate_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.17.mlp.up_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.17.post_attention_layernorm.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.17.self_attn.k_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.17.self_attn.o_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.17.self_attn.q_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.17.self_attn.rotary_emb.inv_freq": "pytorch_model-00002-of-00003.bin",
-    "model.layers.17.self_attn.v_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.18.input_layernorm.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.18.mlp.down_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.18.mlp.gate_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.18.mlp.up_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.18.post_attention_layernorm.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.18.self_attn.k_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.18.self_attn.o_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.18.self_attn.q_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.18.self_attn.rotary_emb.inv_freq": "pytorch_model-00002-of-00003.bin",
-    "model.layers.18.self_attn.v_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.19.input_layernorm.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.19.mlp.down_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.19.mlp.gate_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.19.mlp.up_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.19.post_attention_layernorm.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.19.self_attn.k_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.19.self_attn.o_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.19.self_attn.q_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.19.self_attn.rotary_emb.inv_freq": "pytorch_model-00002-of-00003.bin",
-    "model.layers.19.self_attn.v_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.2.input_layernorm.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.2.mlp.down_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.2.mlp.gate_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.2.mlp.up_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.2.post_attention_layernorm.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.2.self_attn.k_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.2.self_attn.o_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.2.self_attn.q_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.2.self_attn.rotary_emb.inv_freq": "pytorch_model-00001-of-00003.bin",
-    "model.layers.2.self_attn.v_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.20.input_layernorm.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.20.mlp.down_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.20.mlp.gate_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.20.mlp.up_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.20.post_attention_layernorm.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.20.self_attn.k_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.20.self_attn.o_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.20.self_attn.q_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.20.self_attn.rotary_emb.inv_freq": "pytorch_model-00002-of-00003.bin",
-    "model.layers.20.self_attn.v_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.21.input_layernorm.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.21.mlp.down_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.21.mlp.gate_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.21.mlp.up_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.21.post_attention_layernorm.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.21.self_attn.k_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.21.self_attn.o_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.21.self_attn.q_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.21.self_attn.rotary_emb.inv_freq": "pytorch_model-00002-of-00003.bin",
-    "model.layers.21.self_attn.v_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.22.input_layernorm.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.22.mlp.down_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.22.mlp.gate_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.22.mlp.up_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.22.post_attention_layernorm.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.22.self_attn.k_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.22.self_attn.o_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.22.self_attn.q_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.22.self_attn.rotary_emb.inv_freq": "pytorch_model-00002-of-00003.bin",
-    "model.layers.22.self_attn.v_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.23.input_layernorm.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.23.mlp.down_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.23.mlp.gate_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.23.mlp.up_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.23.post_attention_layernorm.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.23.self_attn.k_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.23.self_attn.o_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.23.self_attn.q_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.23.self_attn.rotary_emb.inv_freq": "pytorch_model-00002-of-00003.bin",
-    "model.layers.23.self_attn.v_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.24.input_layernorm.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.24.mlp.down_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.24.mlp.gate_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.24.mlp.up_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.24.post_attention_layernorm.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.24.self_attn.k_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.24.self_attn.o_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.24.self_attn.q_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.24.self_attn.rotary_emb.inv_freq": "pytorch_model-00002-of-00003.bin",
-    "model.layers.24.self_attn.v_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.25.input_layernorm.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.25.mlp.down_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.25.mlp.gate_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.25.mlp.up_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.25.post_attention_layernorm.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.25.self_attn.k_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.25.self_attn.o_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.25.self_attn.q_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.25.self_attn.rotary_emb.inv_freq": "pytorch_model-00002-of-00003.bin",
-    "model.layers.25.self_attn.v_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.26.input_layernorm.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.26.mlp.down_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.26.mlp.gate_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.26.mlp.up_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.26.post_attention_layernorm.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.26.self_attn.k_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.26.self_attn.o_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.26.self_attn.q_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.26.self_attn.rotary_emb.inv_freq": "pytorch_model-00002-of-00003.bin",
-    "model.layers.26.self_attn.v_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.27.input_layernorm.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.27.mlp.down_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.27.mlp.gate_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.27.mlp.up_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.27.post_attention_layernorm.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.27.self_attn.k_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.27.self_attn.o_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.27.self_attn.q_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.27.self_attn.rotary_emb.inv_freq": "pytorch_model-00002-of-00003.bin",
-    "model.layers.27.self_attn.v_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.28.input_layernorm.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.28.mlp.down_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.28.mlp.gate_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.28.mlp.up_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.28.post_attention_layernorm.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.28.self_attn.k_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.28.self_attn.o_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.28.self_attn.q_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.28.self_attn.rotary_emb.inv_freq": "pytorch_model-00002-of-00003.bin",
-    "model.layers.28.self_attn.v_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.29.input_layernorm.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.29.mlp.down_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.29.mlp.gate_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.29.mlp.up_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.29.post_attention_layernorm.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.29.self_attn.k_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.29.self_attn.o_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.29.self_attn.q_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.29.self_attn.rotary_emb.inv_freq": "pytorch_model-00002-of-00003.bin",
-    "model.layers.29.self_attn.v_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.3.input_layernorm.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.3.mlp.down_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.3.mlp.gate_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.3.mlp.up_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.3.post_attention_layernorm.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.3.self_attn.k_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.3.self_attn.o_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.3.self_attn.q_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.3.self_attn.rotary_emb.inv_freq": "pytorch_model-00001-of-00003.bin",
-    "model.layers.3.self_attn.v_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.30.input_layernorm.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.30.mlp.down_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.30.mlp.gate_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.30.mlp.up_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.30.post_attention_layernorm.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.30.self_attn.k_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.30.self_attn.o_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.30.self_attn.q_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.30.self_attn.rotary_emb.inv_freq": "pytorch_model-00002-of-00003.bin",
-    "model.layers.30.self_attn.v_proj.weight": "pytorch_model-00002-of-00003.bin",
-    "model.layers.31.input_layernorm.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.31.mlp.down_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.31.mlp.gate_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.31.mlp.up_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.31.post_attention_layernorm.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.31.self_attn.k_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.31.self_attn.o_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.31.self_attn.q_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.31.self_attn.rotary_emb.inv_freq": "pytorch_model-00003-of-00003.bin",
-    "model.layers.31.self_attn.v_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.32.input_layernorm.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.32.mlp.down_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.32.mlp.gate_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.32.mlp.up_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.32.post_attention_layernorm.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.32.self_attn.k_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.32.self_attn.o_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.32.self_attn.q_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.32.self_attn.rotary_emb.inv_freq": "pytorch_model-00003-of-00003.bin",
-    "model.layers.32.self_attn.v_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.33.input_layernorm.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.33.mlp.down_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.33.mlp.gate_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.33.mlp.up_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.33.post_attention_layernorm.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.33.self_attn.k_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.33.self_attn.o_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.33.self_attn.q_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.33.self_attn.rotary_emb.inv_freq": "pytorch_model-00003-of-00003.bin",
-    "model.layers.33.self_attn.v_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.34.input_layernorm.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.34.mlp.down_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.34.mlp.gate_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.34.mlp.up_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.34.post_attention_layernorm.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.34.self_attn.k_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.34.self_attn.o_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.34.self_attn.q_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.34.self_attn.rotary_emb.inv_freq": "pytorch_model-00003-of-00003.bin",
-    "model.layers.34.self_attn.v_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.35.input_layernorm.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.35.mlp.down_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.35.mlp.gate_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.35.mlp.up_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.35.post_attention_layernorm.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.35.self_attn.k_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.35.self_attn.o_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.35.self_attn.q_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.35.self_attn.rotary_emb.inv_freq": "pytorch_model-00003-of-00003.bin",
-    "model.layers.35.self_attn.v_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.36.input_layernorm.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.36.mlp.down_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.36.mlp.gate_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.36.mlp.up_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.36.post_attention_layernorm.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.36.self_attn.k_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.36.self_attn.o_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.36.self_attn.q_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.36.self_attn.rotary_emb.inv_freq": "pytorch_model-00003-of-00003.bin",
-    "model.layers.36.self_attn.v_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.37.input_layernorm.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.37.mlp.down_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.37.mlp.gate_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.37.mlp.up_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.37.post_attention_layernorm.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.37.self_attn.k_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.37.self_attn.o_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.37.self_attn.q_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.37.self_attn.rotary_emb.inv_freq": "pytorch_model-00003-of-00003.bin",
-    "model.layers.37.self_attn.v_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.38.input_layernorm.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.38.mlp.down_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.38.mlp.gate_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.38.mlp.up_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.38.post_attention_layernorm.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.38.self_attn.k_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.38.self_attn.o_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.38.self_attn.q_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.38.self_attn.rotary_emb.inv_freq": "pytorch_model-00003-of-00003.bin",
-    "model.layers.38.self_attn.v_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.39.input_layernorm.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.39.mlp.down_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.39.mlp.gate_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.39.mlp.up_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.39.post_attention_layernorm.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.39.self_attn.k_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.39.self_attn.o_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.39.self_attn.q_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.39.self_attn.rotary_emb.inv_freq": "pytorch_model-00003-of-00003.bin",
-    "model.layers.39.self_attn.v_proj.weight": "pytorch_model-00003-of-00003.bin",
-    "model.layers.4.input_layernorm.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.4.mlp.down_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.4.mlp.gate_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.4.mlp.up_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.4.post_attention_layernorm.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.4.self_attn.k_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.4.self_attn.o_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.4.self_attn.q_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.4.self_attn.rotary_emb.inv_freq": "pytorch_model-00001-of-00003.bin",
-    "model.layers.4.self_attn.v_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.5.input_layernorm.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.5.mlp.down_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.5.mlp.gate_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.5.mlp.up_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.5.post_attention_layernorm.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.5.self_attn.k_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.5.self_attn.o_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.5.self_attn.q_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.5.self_attn.rotary_emb.inv_freq": "pytorch_model-00001-of-00003.bin",
-    "model.layers.5.self_attn.v_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.6.input_layernorm.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.6.mlp.down_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.6.mlp.gate_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.6.mlp.up_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.6.post_attention_layernorm.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.6.self_attn.k_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.6.self_attn.o_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.6.self_attn.q_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.6.self_attn.rotary_emb.inv_freq": "pytorch_model-00001-of-00003.bin",
-    "model.layers.6.self_attn.v_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.7.input_layernorm.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.7.mlp.down_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.7.mlp.gate_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.7.mlp.up_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.7.post_attention_layernorm.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.7.self_attn.k_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.7.self_attn.o_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.7.self_attn.q_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.7.self_attn.rotary_emb.inv_freq": "pytorch_model-00001-of-00003.bin",
-    "model.layers.7.self_attn.v_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.8.input_layernorm.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.8.mlp.down_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.8.mlp.gate_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.8.mlp.up_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.8.post_attention_layernorm.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.8.self_attn.k_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.8.self_attn.o_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.8.self_attn.q_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.8.self_attn.rotary_emb.inv_freq": "pytorch_model-00001-of-00003.bin",
-    "model.layers.8.self_attn.v_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.9.input_layernorm.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.9.mlp.down_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.9.mlp.gate_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.9.mlp.up_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.9.post_attention_layernorm.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.9.self_attn.k_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.9.self_attn.o_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.9.self_attn.q_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.layers.9.self_attn.rotary_emb.inv_freq": "pytorch_model-00001-of-00003.bin",
-    "model.layers.9.self_attn.v_proj.weight": "pytorch_model-00001-of-00003.bin",
-    "model.mm_projector.0.bias": "pytorch_model-00003-of-00003.bin",
-    "model.mm_projector.0.weight": "pytorch_model-00003-of-00003.bin",
-    "model.mm_projector.2.bias": "pytorch_model-00003-of-00003.bin",
-    "model.mm_projector.2.weight": "pytorch_model-00003-of-00003.bin",
-    "model.norm.weight": "pytorch_model-00003-of-00003.bin"
   }
 }

 {
   "metadata": {
+    "total_size": 28397516800
   },
   "weight_map": {
     "lm_head.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.0.attn.c_attn.bias": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.0.attn.c_attn.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.0.attn.c_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.0.ln_1.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.0.ln_2.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.0.mlp.c_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.0.mlp.w1.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.0.mlp.w2.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.1.attn.c_attn.bias": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.1.attn.c_attn.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.1.attn.c_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.1.ln_1.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.1.ln_2.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.1.mlp.c_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.1.mlp.w1.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.1.mlp.w2.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.10.attn.c_attn.bias": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.10.attn.c_attn.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.10.attn.c_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.10.ln_1.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.10.ln_2.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.10.mlp.c_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.10.mlp.w1.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.10.mlp.w2.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.11.attn.c_attn.bias": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.11.attn.c_attn.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.11.attn.c_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.11.ln_1.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.11.ln_2.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.11.mlp.c_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.11.mlp.w1.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.11.mlp.w2.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.12.attn.c_attn.bias": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.12.attn.c_attn.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.12.attn.c_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.12.ln_1.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.12.ln_2.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.12.mlp.c_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.12.mlp.w1.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.12.mlp.w2.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.13.attn.c_attn.bias": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.13.attn.c_attn.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.13.attn.c_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.13.ln_1.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.13.ln_2.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.13.mlp.c_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.13.mlp.w1.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.13.mlp.w2.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.14.attn.c_attn.bias": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.14.attn.c_attn.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.14.attn.c_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.14.ln_1.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.14.ln_2.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.14.mlp.c_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.14.mlp.w1.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.14.mlp.w2.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.15.attn.c_attn.bias": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.15.attn.c_attn.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.15.attn.c_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.15.ln_1.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.15.ln_2.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.15.mlp.c_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.15.mlp.w1.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.15.mlp.w2.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.16.attn.c_attn.bias": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.16.attn.c_attn.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.16.attn.c_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.16.ln_1.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.16.ln_2.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.16.mlp.c_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.16.mlp.w1.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.16.mlp.w2.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.17.attn.c_attn.bias": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.17.attn.c_attn.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.17.attn.c_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.17.ln_1.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.17.ln_2.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.17.mlp.c_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.17.mlp.w1.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.17.mlp.w2.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.18.attn.c_attn.bias": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.18.attn.c_attn.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.18.attn.c_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.18.ln_1.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.18.ln_2.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.18.mlp.c_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.18.mlp.w1.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.18.mlp.w2.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.19.attn.c_attn.bias": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.19.attn.c_attn.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.19.attn.c_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.19.ln_1.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.19.ln_2.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.19.mlp.c_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.19.mlp.w1.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.19.mlp.w2.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.2.attn.c_attn.bias": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.2.attn.c_attn.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.2.attn.c_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.2.ln_1.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.2.ln_2.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.2.mlp.c_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.2.mlp.w1.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.2.mlp.w2.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.20.attn.c_attn.bias": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.20.attn.c_attn.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.20.attn.c_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.20.ln_1.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.20.ln_2.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.20.mlp.c_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.20.mlp.w1.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.20.mlp.w2.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.21.attn.c_attn.bias": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.21.attn.c_attn.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.21.attn.c_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.21.ln_1.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.21.ln_2.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.21.mlp.c_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.21.mlp.w1.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.21.mlp.w2.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.22.attn.c_attn.bias": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.22.attn.c_attn.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.22.attn.c_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.22.ln_1.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.22.ln_2.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.22.mlp.c_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.22.mlp.w1.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.22.mlp.w2.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.23.attn.c_attn.bias": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.23.attn.c_attn.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.23.attn.c_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.23.ln_1.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.23.ln_2.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.23.mlp.c_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.23.mlp.w1.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.23.mlp.w2.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.24.attn.c_attn.bias": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.24.attn.c_attn.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.24.attn.c_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.24.ln_1.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.24.ln_2.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.24.mlp.c_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.24.mlp.w1.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.24.mlp.w2.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.25.attn.c_attn.bias": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.25.attn.c_attn.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.25.attn.c_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.25.ln_1.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.25.ln_2.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.25.mlp.c_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.25.mlp.w1.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.25.mlp.w2.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.26.attn.c_attn.bias": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.26.attn.c_attn.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.26.attn.c_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.26.ln_1.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.26.ln_2.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.26.mlp.c_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.26.mlp.w1.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.26.mlp.w2.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.27.attn.c_attn.bias": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.27.attn.c_attn.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.27.attn.c_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.27.ln_1.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.27.ln_2.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.27.mlp.c_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.27.mlp.w1.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.27.mlp.w2.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.28.attn.c_attn.bias": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.28.attn.c_attn.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.28.attn.c_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.28.ln_1.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.28.ln_2.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.28.mlp.c_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.28.mlp.w1.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.28.mlp.w2.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.29.attn.c_attn.bias": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.29.attn.c_attn.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.29.attn.c_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.29.ln_1.weight": "pytorch_model-00002-of-00003.bin",
+    "transformer.h.29.ln_2.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.29.mlp.c_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.29.mlp.w1.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.29.mlp.w2.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.3.attn.c_attn.bias": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.3.attn.c_attn.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.3.attn.c_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.3.ln_1.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.3.ln_2.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.3.mlp.c_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.3.mlp.w1.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.3.mlp.w2.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.30.attn.c_attn.bias": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.30.attn.c_attn.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.30.attn.c_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.30.ln_1.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.30.ln_2.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.30.mlp.c_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.30.mlp.w1.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.30.mlp.w2.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.31.attn.c_attn.bias": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.31.attn.c_attn.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.31.attn.c_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.31.ln_1.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.31.ln_2.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.31.mlp.c_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.31.mlp.w1.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.31.mlp.w2.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.32.attn.c_attn.bias": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.32.attn.c_attn.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.32.attn.c_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.32.ln_1.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.32.ln_2.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.32.mlp.c_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.32.mlp.w1.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.32.mlp.w2.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.33.attn.c_attn.bias": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.33.attn.c_attn.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.33.attn.c_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.33.ln_1.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.33.ln_2.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.33.mlp.c_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.33.mlp.w1.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.33.mlp.w2.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.34.attn.c_attn.bias": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.34.attn.c_attn.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.34.attn.c_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.34.ln_1.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.34.ln_2.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.34.mlp.c_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.34.mlp.w1.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.34.mlp.w2.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.35.attn.c_attn.bias": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.35.attn.c_attn.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.35.attn.c_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.35.ln_1.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.35.ln_2.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.35.mlp.c_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.35.mlp.w1.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.35.mlp.w2.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.36.attn.c_attn.bias": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.36.attn.c_attn.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.36.attn.c_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.36.ln_1.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.36.ln_2.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.36.mlp.c_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.36.mlp.w1.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.36.mlp.w2.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.37.attn.c_attn.bias": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.37.attn.c_attn.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.37.attn.c_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.37.ln_1.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.37.ln_2.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.37.mlp.c_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.37.mlp.w1.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.37.mlp.w2.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.38.attn.c_attn.bias": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.38.attn.c_attn.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.38.attn.c_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.38.ln_1.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.38.ln_2.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.38.mlp.c_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.38.mlp.w1.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.38.mlp.w2.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.39.attn.c_attn.bias": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.39.attn.c_attn.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.39.attn.c_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.39.ln_1.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.39.ln_2.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.39.mlp.c_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.39.mlp.w1.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.39.mlp.w2.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.h.4.attn.c_attn.bias": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.4.attn.c_attn.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.4.attn.c_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.4.ln_1.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.4.ln_2.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.4.mlp.c_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.4.mlp.w1.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.4.mlp.w2.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.5.attn.c_attn.bias": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.5.attn.c_attn.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.5.attn.c_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.5.ln_1.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.5.ln_2.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.5.mlp.c_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.5.mlp.w1.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.5.mlp.w2.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.6.attn.c_attn.bias": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.6.attn.c_attn.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.6.attn.c_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.6.ln_1.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.6.ln_2.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.6.mlp.c_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.6.mlp.w1.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.6.mlp.w2.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.7.attn.c_attn.bias": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.7.attn.c_attn.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.7.attn.c_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.7.ln_1.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.7.ln_2.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.7.mlp.c_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.7.mlp.w1.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.7.mlp.w2.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.8.attn.c_attn.bias": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.8.attn.c_attn.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.8.attn.c_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.8.ln_1.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.8.ln_2.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.8.mlp.c_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.8.mlp.w1.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.8.mlp.w2.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.9.attn.c_attn.bias": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.9.attn.c_attn.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.9.attn.c_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.9.ln_1.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.9.ln_2.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.9.mlp.c_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.9.mlp.w1.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.h.9.mlp.w2.weight": "pytorch_model-00001-of-00003.bin",
+    "transformer.ln_f.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.mm_projector.0.bias": "pytorch_model-00003-of-00003.bin",
+    "transformer.mm_projector.0.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.mm_projector.2.bias": "pytorch_model-00003-of-00003.bin",
+    "transformer.mm_projector.2.weight": "pytorch_model-00003-of-00003.bin",
+    "transformer.wte.weight": "pytorch_model-00001-of-00003.bin"
   }
 }

qwen.tiktoken ADDED Viewed

The diff for this file is too large to render. See raw diff

special_tokens_map.json CHANGED Viewed

@@ -1,24 +1,5 @@
 {
-  "bos_token": {
-    "content": "<s>",
-    "lstrip": false,
-    "normalized": true,
-    "rstrip": false,
-    "single_word": false
-  },
-  "eos_token": {
-    "content": "</s>",
-    "lstrip": false,
-    "normalized": true,
-    "rstrip": false,
-    "single_word": false
-  },
-  "pad_token": "<unk>",
-  "unk_token": {
-    "content": "<unk>",
-    "lstrip": false,
-    "normalized": true,
-    "rstrip": false,
-    "single_word": false
-  }
 }

 {
+  "bos_token": "<|im_start|>",
+  "eos_token": "<|im_end|>",
+  "pad_token": "<|endoftext|>"
 }

tokenizer_config.json CHANGED Viewed

@@ -1,36 +1,16 @@
 {
-  "add_bos_token": true,
-  "add_eos_token": false,
-  "bos_token": {
-    "__type": "AddedToken",
-    "content": "<s>",
-    "lstrip": false,
-    "normalized": true,
-    "rstrip": false,
-    "single_word": false
   },
-  "clean_up_tokenization_spaces": false,
-  "eos_token": {
-    "__type": "AddedToken",
-    "content": "</s>",
-    "lstrip": false,
-    "normalized": true,
-    "rstrip": false,
-    "single_word": false
-  },
-  "legacy": true,
   "model_max_length": 2048,
-  "pad_token": null,
   "padding_side": "right",
-  "sp_model_kwargs": {},
-  "tokenizer_class": "LlamaTokenizer",
-  "unk_token": {
-    "__type": "AddedToken",
-    "content": "<unk>",
-    "lstrip": false,
-    "normalized": true,
-    "rstrip": false,
-    "single_word": false
-  },
-  "use_fast": true
 }

 {
+  "auto_map": {
+    "AutoTokenizer": [
+      "tokenization_qwen.QWenTokenizer",
+      null
+    ]
   },
+  "bos_token": "<|im_start|>",
+  "clean_up_tokenization_spaces": true,
+  "eos_token": "<|im_end|>",
   "model_max_length": 2048,
+  "pad_token": "<|endoftext|>",
   "padding_side": "right",
+  "tokenizer_class": "QWenTokenizer",
+  "use_fast": false
 }

trainer_state.json CHANGED Viewed

The diff for this file is too large to render. See raw diff

training_args.bin CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:2d0e6b0bde37a458a6ed069e0ae44ee7c96a376d4fd79093e2cb4b5c393918aa
 size 6139

 version https://git-lfs.github.com/spec/v1
+oid sha256:b2b180ca7b1e5234bc38c8afe17d9cd88753628f5ba99e8ed5b0b3d694bd8924
 size 6139