Upload folder using huggingface_hub

Browse files

Files changed (16) hide show

arc_challenge_25shot_bs16_bf16.json +25 -0
config.json +28 -0
generation_config.json +6 -0
gsm8k_5shot_bs16_bf16.json +23 -0
hellaswag_10shot_bs16_bf16.json +25 -0
mmlu_5shot_bs4_bf16.json +417 -0
pytorch_model-00001-of-00003.bin +3 -0
pytorch_model-00002-of-00003.bin +3 -0
pytorch_model-00003-of-00003.bin +3 -0
pytorch_model.bin.index.json +298 -0
special_tokens_map.json +23 -0
tokenizer.json +0 -0
tokenizer.model +3 -0
tokenizer_config.json +41 -0
truthfulqa_mc_0shot_bs16_bf16.json +25 -0
winogrande_5shot_bs16_bf16.json +23 -0

arc_challenge_25shot_bs16_bf16.json ADDED Viewed

	@@ -0,0 +1,25 @@

+{
+  "results": {
+    "arc_challenge": {
+      "acc": 0.4232081911262799,
+      "acc_stderr": 0.014438036220848027,
+      "acc_norm": 0.45307167235494883,
+      "acc_norm_stderr": 0.01454689205200563
+    }
+  },
+  "versions": {
+    "arc_challenge": 0
+  },
+  "config": {
+    "model": "sparseml",
+    "model_args": "pretrained=/network/alexandre/research/cerebras/llama2_7B_sparse70_retrained/ultrachat200k/llama2_7B_sparse70_LR3e-4_GC2_E2/training,dtype=bfloat16",
+    "num_fewshot": 25,
+    "batch_size": "16",
+    "batch_sizes": [],
+    "device": "cuda:1",
+    "no_cache": true,
+    "limit": null,
+    "bootstrap_iters": 100000,
+    "description_dict": {}
+  }
+}

config.json ADDED Viewed

	@@ -0,0 +1,28 @@

+{
+  "architectures": [
+    "LlamaForCausalLM"
+  ],
+  "attention_bias": false,
+  "attention_dropout": 0.0,
+  "bos_token_id": 1,
+  "eos_token_id": 2,
+  "hidden_act": "silu",
+  "hidden_size": 4096,
+  "initializer_range": 0.02,
+  "intermediate_size": 11008,
+  "max_position_embeddings": 4096,
+  "model_type": "llama",
+  "num_attention_heads": 32,
+  "num_hidden_layers": 32,
+  "num_key_value_heads": 32,
+  "pretraining_tp": 1,
+  "rms_norm_eps": 1e-05,
+  "rope_scaling": null,
+  "rope_theta": 10000.0,
+  "tie_word_embeddings": false,
+  "tokenizer_class": "LlamaTokenizerFast",
+  "torch_dtype": "float32",
+  "transformers_version": "1.7.0.20240309",
+  "use_cache": true,
+  "vocab_size": 32000
+}

generation_config.json ADDED Viewed

	@@ -0,0 +1,6 @@

+{
+  "_from_model_config": true,
+  "bos_token_id": 1,
+  "eos_token_id": 2,
+  "transformers_version": "1.7.0.20240309"
+}

gsm8k_5shot_bs16_bf16.json ADDED Viewed

	@@ -0,0 +1,23 @@

+{
+  "results": {
+    "gsm8k": {
+      "acc": 0.047763457164518575,
+      "acc_stderr": 0.005874387536229333
+    }
+  },
+  "versions": {
+    "gsm8k": 0
+  },
+  "config": {
+    "model": "sparseml",
+    "model_args": "pretrained=/network/alexandre/research/cerebras/llama2_7B_sparse70_retrained/ultrachat200k/llama2_7B_sparse70_LR3e-4_GC2_E2/training,dtype=bfloat16",
+    "num_fewshot": 5,
+    "batch_size": "16",
+    "batch_sizes": [],
+    "device": "cuda:6",
+    "no_cache": true,
+    "limit": null,
+    "bootstrap_iters": 100000,
+    "description_dict": {}
+  }
+}

hellaswag_10shot_bs16_bf16.json ADDED Viewed

	@@ -0,0 +1,25 @@

+{
+  "results": {
+    "hellaswag": {
+      "acc": 0.5220075682135032,
+      "acc_stderr": 0.004984945635998312,
+      "acc_norm": 0.6885082652857997,
+      "acc_norm_stderr": 0.004621568125102052
+    }
+  },
+  "versions": {
+    "hellaswag": 0
+  },
+  "config": {
+    "model": "sparseml",
+    "model_args": "pretrained=/network/alexandre/research/cerebras/llama2_7B_sparse70_retrained/ultrachat200k/llama2_7B_sparse70_LR3e-4_GC2_E2/training,dtype=bfloat16",
+    "num_fewshot": 10,
+    "batch_size": "16",
+    "batch_sizes": [],
+    "device": "cuda:3",
+    "no_cache": true,
+    "limit": null,
+    "bootstrap_iters": 100000,
+    "description_dict": {}
+  }
+}

mmlu_5shot_bs4_bf16.json ADDED Viewed

	@@ -0,0 +1,417 @@

+{
+  "results": {
+    "hendrycksTest-abstract_algebra": {
+      "acc": 0.32,
+      "acc_stderr": 0.046882617226215034,
+      "acc_norm": 0.32,
+      "acc_norm_stderr": 0.046882617226215034
+    },
+    "hendrycksTest-anatomy": {
+      "acc": 0.32592592592592595,
+      "acc_stderr": 0.040491220417025055,
+      "acc_norm": 0.32592592592592595,
+      "acc_norm_stderr": 0.040491220417025055
+    },
+    "hendrycksTest-astronomy": {
+      "acc": 0.23026315789473684,
+      "acc_stderr": 0.03426059424403165,
+      "acc_norm": 0.23026315789473684,
+      "acc_norm_stderr": 0.03426059424403165
+    },
+    "hendrycksTest-business_ethics": {
+      "acc": 0.3,
+      "acc_stderr": 0.046056618647183814,
+      "acc_norm": 0.3,
+      "acc_norm_stderr": 0.046056618647183814
+    },
+    "hendrycksTest-clinical_knowledge": {
+      "acc": 0.3584905660377358,
+      "acc_stderr": 0.02951470358398177,
+      "acc_norm": 0.3584905660377358,
+      "acc_norm_stderr": 0.02951470358398177
+    },
+    "hendrycksTest-college_biology": {
+      "acc": 0.2916666666666667,
+      "acc_stderr": 0.038009680605548594,
+      "acc_norm": 0.2916666666666667,
+      "acc_norm_stderr": 0.038009680605548594
+    },
+    "hendrycksTest-college_chemistry": {
+      "acc": 0.21,
+      "acc_stderr": 0.040936018074033256,
+      "acc_norm": 0.21,
+      "acc_norm_stderr": 0.040936018074033256
+    },
+    "hendrycksTest-college_computer_science": {
+      "acc": 0.21,
+      "acc_stderr": 0.040936018074033256,
+      "acc_norm": 0.21,
+      "acc_norm_stderr": 0.040936018074033256
+    },
+    "hendrycksTest-college_mathematics": {
+      "acc": 0.23,
+      "acc_stderr": 0.04229525846816506,
+      "acc_norm": 0.23,
+      "acc_norm_stderr": 0.04229525846816506
+    },
+    "hendrycksTest-college_medicine": {
+      "acc": 0.2832369942196532,
+      "acc_stderr": 0.034355680560478746,
+      "acc_norm": 0.2832369942196532,
+      "acc_norm_stderr": 0.034355680560478746
+    },
+    "hendrycksTest-college_physics": {
+      "acc": 0.21568627450980393,
+      "acc_stderr": 0.04092563958237655,
+      "acc_norm": 0.21568627450980393,
+      "acc_norm_stderr": 0.04092563958237655
+    },
+    "hendrycksTest-computer_security": {
+      "acc": 0.37,
+      "acc_stderr": 0.04852365870939099,
+      "acc_norm": 0.37,
+      "acc_norm_stderr": 0.04852365870939099
+    },
+    "hendrycksTest-conceptual_physics": {
+      "acc": 0.3574468085106383,
+      "acc_stderr": 0.03132941789476425,
+      "acc_norm": 0.3574468085106383,
+      "acc_norm_stderr": 0.03132941789476425
+    },
+    "hendrycksTest-econometrics": {
+      "acc": 0.24561403508771928,
+      "acc_stderr": 0.04049339297748142,
+      "acc_norm": 0.24561403508771928,
+      "acc_norm_stderr": 0.04049339297748142
+    },
+    "hendrycksTest-electrical_engineering": {
+      "acc": 0.2827586206896552,
+      "acc_stderr": 0.03752833958003337,
+      "acc_norm": 0.2827586206896552,
+      "acc_norm_stderr": 0.03752833958003337
+    },
+    "hendrycksTest-elementary_mathematics": {
+      "acc": 0.23809523809523808,
+      "acc_stderr": 0.021935878081184766,
+      "acc_norm": 0.23809523809523808,
+      "acc_norm_stderr": 0.021935878081184766
+    },
+    "hendrycksTest-formal_logic": {
+      "acc": 0.1984126984126984,
+      "acc_stderr": 0.035670166752768635,
+      "acc_norm": 0.1984126984126984,
+      "acc_norm_stderr": 0.035670166752768635
+    },
+    "hendrycksTest-global_facts": {
+      "acc": 0.34,
+      "acc_stderr": 0.04760952285695236,
+      "acc_norm": 0.34,
+      "acc_norm_stderr": 0.04760952285695236
+    },
+    "hendrycksTest-high_school_biology": {
+      "acc": 0.3225806451612903,
+      "acc_stderr": 0.02659308451657229,
+      "acc_norm": 0.3225806451612903,
+      "acc_norm_stderr": 0.02659308451657229
+    },
+    "hendrycksTest-high_school_chemistry": {
+      "acc": 0.2315270935960591,
+      "acc_stderr": 0.02967833314144444,
+      "acc_norm": 0.2315270935960591,
+      "acc_norm_stderr": 0.02967833314144444
+    },
+    "hendrycksTest-high_school_computer_science": {
+      "acc": 0.34,
+      "acc_stderr": 0.047609522856952365,
+      "acc_norm": 0.34,
+      "acc_norm_stderr": 0.047609522856952365
+    },
+    "hendrycksTest-high_school_european_history": {
+      "acc": 0.4,
+      "acc_stderr": 0.03825460278380026,
+      "acc_norm": 0.4,
+      "acc_norm_stderr": 0.03825460278380026
+    },
+    "hendrycksTest-high_school_geography": {
+      "acc": 0.3383838383838384,
+      "acc_stderr": 0.03371124142626303,
+      "acc_norm": 0.3383838383838384,
+      "acc_norm_stderr": 0.03371124142626303
+    },
+    "hendrycksTest-high_school_government_and_politics": {
+      "acc": 0.32642487046632124,
+      "acc_stderr": 0.033840286211432945,
+      "acc_norm": 0.32642487046632124,
+      "acc_norm_stderr": 0.033840286211432945
+    },
+    "hendrycksTest-high_school_macroeconomics": {
+      "acc": 0.2358974358974359,
+      "acc_stderr": 0.021525965407408726,
+      "acc_norm": 0.2358974358974359,
+      "acc_norm_stderr": 0.021525965407408726
+    },
+    "hendrycksTest-high_school_mathematics": {
+      "acc": 0.23703703703703705,
+      "acc_stderr": 0.025928876132766104,
+      "acc_norm": 0.23703703703703705,
+      "acc_norm_stderr": 0.025928876132766104
+    },
+    "hendrycksTest-high_school_microeconomics": {
+      "acc": 0.29411764705882354,
+      "acc_stderr": 0.02959732973097808,
+      "acc_norm": 0.29411764705882354,
+      "acc_norm_stderr": 0.02959732973097808
+    },
+    "hendrycksTest-high_school_physics": {
+      "acc": 0.2582781456953642,
+      "acc_stderr": 0.035737053147634576,
+      "acc_norm": 0.2582781456953642,
+      "acc_norm_stderr": 0.035737053147634576
+    },
+    "hendrycksTest-high_school_psychology": {
+      "acc": 0.3779816513761468,
+      "acc_stderr": 0.020789187066728117,
+      "acc_norm": 0.3779816513761468,
+      "acc_norm_stderr": 0.020789187066728117
+    },
+    "hendrycksTest-high_school_statistics": {
+      "acc": 0.1712962962962963,
+      "acc_stderr": 0.025695341643824674,
+      "acc_norm": 0.1712962962962963,
+      "acc_norm_stderr": 0.025695341643824674
+    },
+    "hendrycksTest-high_school_us_history": {
+      "acc": 0.3627450980392157,
+      "acc_stderr": 0.03374499356319355,
+      "acc_norm": 0.3627450980392157,
+      "acc_norm_stderr": 0.03374499356319355
+    },
+    "hendrycksTest-high_school_world_history": {
+      "acc": 0.47257383966244726,
+      "acc_stderr": 0.03249822718301303,
+      "acc_norm": 0.47257383966244726,
+      "acc_norm_stderr": 0.03249822718301303
+    },
+    "hendrycksTest-human_aging": {
+      "acc": 0.4080717488789238,
+      "acc_stderr": 0.03298574607842821,
+      "acc_norm": 0.4080717488789238,
+      "acc_norm_stderr": 0.03298574607842821
+    },
+    "hendrycksTest-human_sexuality": {
+      "acc": 0.3053435114503817,
+      "acc_stderr": 0.040393149787245626,
+      "acc_norm": 0.3053435114503817,
+      "acc_norm_stderr": 0.040393149787245626
+    },
+    "hendrycksTest-international_law": {
+      "acc": 0.5619834710743802,
+      "acc_stderr": 0.045291468044357915,
+      "acc_norm": 0.5619834710743802,
+      "acc_norm_stderr": 0.045291468044357915
+    },
+    "hendrycksTest-jurisprudence": {
+      "acc": 0.37037037037037035,
+      "acc_stderr": 0.04668408033024932,
+      "acc_norm": 0.37037037037037035,
+      "acc_norm_stderr": 0.04668408033024932
+    },
+    "hendrycksTest-logical_fallacies": {
+      "acc": 0.38650306748466257,
+      "acc_stderr": 0.038258255488486076,
+      "acc_norm": 0.38650306748466257,
+      "acc_norm_stderr": 0.038258255488486076
+    },
+    "hendrycksTest-machine_learning": {
+      "acc": 0.3482142857142857,
+      "acc_stderr": 0.04521829902833585,
+      "acc_norm": 0.3482142857142857,
+      "acc_norm_stderr": 0.04521829902833585
+    },
+    "hendrycksTest-management": {
+      "acc": 0.3300970873786408,
+      "acc_stderr": 0.0465614711001235,
+      "acc_norm": 0.3300970873786408,
+      "acc_norm_stderr": 0.0465614711001235
+    },
+    "hendrycksTest-marketing": {
+      "acc": 0.47435897435897434,
+      "acc_stderr": 0.03271298896811159,
+      "acc_norm": 0.47435897435897434,
+      "acc_norm_stderr": 0.03271298896811159
+    },
+    "hendrycksTest-medical_genetics": {
+      "acc": 0.35,
+      "acc_stderr": 0.047937248544110196,
+      "acc_norm": 0.35,
+      "acc_norm_stderr": 0.047937248544110196
+    },
+    "hendrycksTest-miscellaneous": {
+      "acc": 0.47126436781609193,
+      "acc_stderr": 0.01785041079438017,
+      "acc_norm": 0.47126436781609193,
+      "acc_norm_stderr": 0.01785041079438017
+    },
+    "hendrycksTest-moral_disputes": {
+      "acc": 0.315028901734104,
+      "acc_stderr": 0.025009313790069727,
+      "acc_norm": 0.315028901734104,
+      "acc_norm_stderr": 0.025009313790069727
+    },
+    "hendrycksTest-moral_scenarios": {
+      "acc": 0.23798882681564246,
+      "acc_stderr": 0.014242630070574915,
+      "acc_norm": 0.23798882681564246,
+      "acc_norm_stderr": 0.014242630070574915
+    },
+    "hendrycksTest-nutrition": {
+      "acc": 0.369281045751634,
+      "acc_stderr": 0.027634176689602653,
+      "acc_norm": 0.369281045751634,
+      "acc_norm_stderr": 0.027634176689602653
+    },
+    "hendrycksTest-philosophy": {
+      "acc": 0.3633440514469453,
+      "acc_stderr": 0.027316847674192717,
+      "acc_norm": 0.3633440514469453,
+      "acc_norm_stderr": 0.027316847674192717
+    },
+    "hendrycksTest-prehistory": {
+      "acc": 0.36419753086419754,
+      "acc_stderr": 0.02677492989972233,
+      "acc_norm": 0.36419753086419754,
+      "acc_norm_stderr": 0.02677492989972233
+    },
+    "hendrycksTest-professional_accounting": {
+      "acc": 0.30141843971631205,
+      "acc_stderr": 0.02737412888263115,
+      "acc_norm": 0.30141843971631205,
+      "acc_norm_stderr": 0.02737412888263115
+    },
+    "hendrycksTest-professional_law": {
+      "acc": 0.3044328552803129,
+      "acc_stderr": 0.011752877592597575,
+      "acc_norm": 0.3044328552803129,
+      "acc_norm_stderr": 0.011752877592597575
+    },
+    "hendrycksTest-professional_medicine": {
+      "acc": 0.2426470588235294,
+      "acc_stderr": 0.026040662474201264,
+      "acc_norm": 0.2426470588235294,
+      "acc_norm_stderr": 0.026040662474201264
+    },
+    "hendrycksTest-professional_psychology": {
+      "acc": 0.3104575163398693,
+      "acc_stderr": 0.018718067052623227,
+      "acc_norm": 0.3104575163398693,
+      "acc_norm_stderr": 0.018718067052623227
+    },
+    "hendrycksTest-public_relations": {
+      "acc": 0.38181818181818183,
+      "acc_stderr": 0.04653429807913508,
+      "acc_norm": 0.38181818181818183,
+      "acc_norm_stderr": 0.04653429807913508
+    },
+    "hendrycksTest-security_studies": {
+      "acc": 0.3020408163265306,
+      "acc_stderr": 0.029393609319879818,
+      "acc_norm": 0.3020408163265306,
+      "acc_norm_stderr": 0.029393609319879818
+    },
+    "hendrycksTest-sociology": {
+      "acc": 0.4079601990049751,
+      "acc_stderr": 0.034751163651940926,
+      "acc_norm": 0.4079601990049751,
+      "acc_norm_stderr": 0.034751163651940926
+    },
+    "hendrycksTest-us_foreign_policy": {
+      "acc": 0.41,
+      "acc_stderr": 0.04943110704237102,
+      "acc_norm": 0.41,
+      "acc_norm_stderr": 0.04943110704237102
+    },
+    "hendrycksTest-virology": {
+      "acc": 0.4036144578313253,
+      "acc_stderr": 0.038194861407583984,
+      "acc_norm": 0.4036144578313253,
+      "acc_norm_stderr": 0.038194861407583984
+    },
+    "hendrycksTest-world_religions": {
+      "acc": 0.40350877192982454,
+      "acc_stderr": 0.03762738699917056,
+      "acc_norm": 0.40350877192982454,
+      "acc_norm_stderr": 0.03762738699917056
+    }
+  },
+  "versions": {
+    "hendrycksTest-abstract_algebra": 1,
+    "hendrycksTest-anatomy": 1,
+    "hendrycksTest-astronomy": 1,
+    "hendrycksTest-business_ethics": 1,
+    "hendrycksTest-clinical_knowledge": 1,
+    "hendrycksTest-college_biology": 1,
+    "hendrycksTest-college_chemistry": 1,
+    "hendrycksTest-college_computer_science": 1,
+    "hendrycksTest-college_mathematics": 1,
+    "hendrycksTest-college_medicine": 1,
+    "hendrycksTest-college_physics": 1,
+    "hendrycksTest-computer_security": 1,
+    "hendrycksTest-conceptual_physics": 1,
+    "hendrycksTest-econometrics": 1,
+    "hendrycksTest-electrical_engineering": 1,
+    "hendrycksTest-elementary_mathematics": 1,
+    "hendrycksTest-formal_logic": 1,
+    "hendrycksTest-global_facts": 1,
+    "hendrycksTest-high_school_biology": 1,
+    "hendrycksTest-high_school_chemistry": 1,
+    "hendrycksTest-high_school_computer_science": 1,
+    "hendrycksTest-high_school_european_history": 1,
+    "hendrycksTest-high_school_geography": 1,
+    "hendrycksTest-high_school_government_and_politics": 1,
+    "hendrycksTest-high_school_macroeconomics": 1,
+    "hendrycksTest-high_school_mathematics": 1,
+    "hendrycksTest-high_school_microeconomics": 1,
+    "hendrycksTest-high_school_physics": 1,
+    "hendrycksTest-high_school_psychology": 1,
+    "hendrycksTest-high_school_statistics": 1,
+    "hendrycksTest-high_school_us_history": 1,
+    "hendrycksTest-high_school_world_history": 1,
+    "hendrycksTest-human_aging": 1,
+    "hendrycksTest-human_sexuality": 1,
+    "hendrycksTest-international_law": 1,
+    "hendrycksTest-jurisprudence": 1,
+    "hendrycksTest-logical_fallacies": 1,
+    "hendrycksTest-machine_learning": 1,
+    "hendrycksTest-management": 1,
+    "hendrycksTest-marketing": 1,
+    "hendrycksTest-medical_genetics": 1,
+    "hendrycksTest-miscellaneous": 1,
+    "hendrycksTest-moral_disputes": 1,
+    "hendrycksTest-moral_scenarios": 1,
+    "hendrycksTest-nutrition": 1,
+    "hendrycksTest-philosophy": 1,
+    "hendrycksTest-prehistory": 1,
+    "hendrycksTest-professional_accounting": 1,
+    "hendrycksTest-professional_law": 1,
+    "hendrycksTest-professional_medicine": 1,
+    "hendrycksTest-professional_psychology": 1,
+    "hendrycksTest-public_relations": 1,
+    "hendrycksTest-security_studies": 1,
+    "hendrycksTest-sociology": 1,
+    "hendrycksTest-us_foreign_policy": 1,
+    "hendrycksTest-virology": 1,
+    "hendrycksTest-world_religions": 1
+  },
+  "config": {
+    "model": "sparseml",
+    "model_args": "pretrained=/network/alexandre/research/cerebras/llama2_7B_sparse70_retrained/ultrachat200k/llama2_7B_sparse70_LR3e-4_GC2_E2/training,dtype=bfloat16",
+    "num_fewshot": 5,
+    "batch_size": "4",
+    "batch_sizes": [],
+    "device": "cuda:7",
+    "no_cache": true,
+    "limit": null,
+    "bootstrap_iters": 100000,
+    "description_dict": {}
+  }
+}

pytorch_model-00001-of-00003.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:db10a283fa0ce643dbf8e820a17d7a32161cb48bbf050fd91681c1c4e377e360
+size 9877982873

pytorch_model-00002-of-00003.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:e00b0b63a2192c38b3243abdaa3cb3e59122657d4ae7575d719ea722bd44a5c3
+size 9894794253

pytorch_model-00003-of-00003.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:638a4b6a16dfd8ecefb2247877a4a4470b1b4d5954f7d54a6bd883b3b84be97d
+size 7180986412

pytorch_model.bin.index.json ADDED Viewed

	@@ -0,0 +1,298 @@

+{
+  "metadata": {
+    "total_size": 26953662464
+  },
+  "weight_map": {
+    "lm_head.weight": "pytorch_model-00003-of-00003.bin",
+    "model.embed_tokens.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.0.input_layernorm.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.0.mlp.down_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.0.mlp.gate_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.0.mlp.up_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.0.post_attention_layernorm.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.0.self_attn.k_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.0.self_attn.o_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.0.self_attn.q_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.0.self_attn.v_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.1.input_layernorm.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.1.mlp.down_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.1.mlp.gate_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.1.mlp.up_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.1.post_attention_layernorm.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.1.self_attn.k_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.1.self_attn.o_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.1.self_attn.q_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.1.self_attn.v_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.10.input_layernorm.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.10.mlp.down_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.10.mlp.gate_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.10.mlp.up_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.10.post_attention_layernorm.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.10.self_attn.k_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.10.self_attn.o_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.10.self_attn.q_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.10.self_attn.v_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.11.input_layernorm.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.11.mlp.down_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.11.mlp.gate_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.11.mlp.up_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.11.post_attention_layernorm.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.11.self_attn.k_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.11.self_attn.o_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.11.self_attn.q_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.11.self_attn.v_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.12.input_layernorm.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.12.mlp.down_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.12.mlp.gate_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.12.mlp.up_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.12.post_attention_layernorm.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.12.self_attn.k_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.12.self_attn.o_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.12.self_attn.q_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.12.self_attn.v_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.13.input_layernorm.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.13.mlp.down_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.13.mlp.gate_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.13.mlp.up_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.13.post_attention_layernorm.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.13.self_attn.k_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.13.self_attn.o_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.13.self_attn.q_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.13.self_attn.v_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.14.input_layernorm.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.14.mlp.down_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.14.mlp.gate_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.14.mlp.up_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.14.post_attention_layernorm.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.14.self_attn.k_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.14.self_attn.o_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.14.self_attn.q_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.14.self_attn.v_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.15.input_layernorm.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.15.mlp.down_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.15.mlp.gate_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.15.mlp.up_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.15.post_attention_layernorm.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.15.self_attn.k_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.15.self_attn.o_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.15.self_attn.q_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.15.self_attn.v_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.16.input_layernorm.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.16.mlp.down_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.16.mlp.gate_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.16.mlp.up_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.16.post_attention_layernorm.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.16.self_attn.k_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.16.self_attn.o_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.16.self_attn.q_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.16.self_attn.v_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.17.input_layernorm.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.17.mlp.down_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.17.mlp.gate_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.17.mlp.up_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.17.post_attention_layernorm.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.17.self_attn.k_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.17.self_attn.o_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.17.self_attn.q_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.17.self_attn.v_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.18.input_layernorm.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.18.mlp.down_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.18.mlp.gate_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.18.mlp.up_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.18.post_attention_layernorm.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.18.self_attn.k_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.18.self_attn.o_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.18.self_attn.q_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.18.self_attn.v_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.19.input_layernorm.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.19.mlp.down_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.19.mlp.gate_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.19.mlp.up_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.19.post_attention_layernorm.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.19.self_attn.k_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.19.self_attn.o_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.19.self_attn.q_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.19.self_attn.v_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.2.input_layernorm.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.2.mlp.down_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.2.mlp.gate_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.2.mlp.up_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.2.post_attention_layernorm.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.2.self_attn.k_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.2.self_attn.o_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.2.self_attn.q_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.2.self_attn.v_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.20.input_layernorm.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.20.mlp.down_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.20.mlp.gate_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.20.mlp.up_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.20.post_attention_layernorm.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.20.self_attn.k_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.20.self_attn.o_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.20.self_attn.q_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.20.self_attn.v_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.21.input_layernorm.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.21.mlp.down_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.21.mlp.gate_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.21.mlp.up_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.21.post_attention_layernorm.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.21.self_attn.k_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.21.self_attn.o_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.21.self_attn.q_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.21.self_attn.v_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.22.input_layernorm.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.22.mlp.down_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.22.mlp.gate_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.22.mlp.up_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.22.post_attention_layernorm.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.22.self_attn.k_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.22.self_attn.o_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.22.self_attn.q_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.22.self_attn.v_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.23.input_layernorm.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.23.mlp.down_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.23.mlp.gate_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.23.mlp.up_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.23.post_attention_layernorm.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.23.self_attn.k_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.23.self_attn.o_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.23.self_attn.q_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.23.self_attn.v_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.24.input_layernorm.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.24.mlp.down_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.24.mlp.gate_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.24.mlp.up_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.24.post_attention_layernorm.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.24.self_attn.k_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.24.self_attn.o_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.24.self_attn.q_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.24.self_attn.v_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.25.input_layernorm.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.25.mlp.down_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.25.mlp.gate_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.25.mlp.up_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.25.post_attention_layernorm.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.25.self_attn.k_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.25.self_attn.o_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.25.self_attn.q_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.25.self_attn.v_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.26.input_layernorm.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.26.mlp.down_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.26.mlp.gate_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.26.mlp.up_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.26.post_attention_layernorm.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.26.self_attn.k_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.26.self_attn.o_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.26.self_attn.q_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.26.self_attn.v_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.27.input_layernorm.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.27.mlp.down_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.27.mlp.gate_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.27.mlp.up_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.27.post_attention_layernorm.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.27.self_attn.k_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.27.self_attn.o_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.27.self_attn.q_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.27.self_attn.v_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.28.input_layernorm.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.28.mlp.down_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.28.mlp.gate_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.28.mlp.up_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.28.post_attention_layernorm.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.28.self_attn.k_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.28.self_attn.o_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.28.self_attn.q_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.28.self_attn.v_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.29.input_layernorm.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.29.mlp.down_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.29.mlp.gate_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.29.mlp.up_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.29.post_attention_layernorm.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.29.self_attn.k_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.29.self_attn.o_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.29.self_attn.q_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.29.self_attn.v_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.3.input_layernorm.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.3.mlp.down_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.3.mlp.gate_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.3.mlp.up_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.3.post_attention_layernorm.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.3.self_attn.k_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.3.self_attn.o_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.3.self_attn.q_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.3.self_attn.v_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.30.input_layernorm.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.30.mlp.down_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.30.mlp.gate_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.30.mlp.up_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.30.post_attention_layernorm.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.30.self_attn.k_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.30.self_attn.o_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.30.self_attn.q_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.30.self_attn.v_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.31.input_layernorm.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.31.mlp.down_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.31.mlp.gate_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.31.mlp.up_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.31.post_attention_layernorm.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.31.self_attn.k_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.31.self_attn.o_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.31.self_attn.q_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.31.self_attn.v_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.4.input_layernorm.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.4.mlp.down_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.4.mlp.gate_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.4.mlp.up_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.4.post_attention_layernorm.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.4.self_attn.k_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.4.self_attn.o_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.4.self_attn.q_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.4.self_attn.v_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.5.input_layernorm.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.5.mlp.down_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.5.mlp.gate_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.5.mlp.up_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.5.post_attention_layernorm.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.5.self_attn.k_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.5.self_attn.o_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.5.self_attn.q_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.5.self_attn.v_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.6.input_layernorm.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.6.mlp.down_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.6.mlp.gate_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.6.mlp.up_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.6.post_attention_layernorm.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.6.self_attn.k_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.6.self_attn.o_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.6.self_attn.q_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.6.self_attn.v_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.7.input_layernorm.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.7.mlp.down_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.7.mlp.gate_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.7.mlp.up_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.7.post_attention_layernorm.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.7.self_attn.k_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.7.self_attn.o_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.7.self_attn.q_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.7.self_attn.v_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.8.input_layernorm.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.8.mlp.down_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.8.mlp.gate_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.8.mlp.up_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.8.post_attention_layernorm.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.8.self_attn.k_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.8.self_attn.o_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.8.self_attn.q_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.8.self_attn.v_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.9.input_layernorm.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.9.mlp.down_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.9.mlp.gate_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.9.mlp.up_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.9.post_attention_layernorm.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.9.self_attn.k_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.9.self_attn.o_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.9.self_attn.q_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.9.self_attn.v_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.norm.weight": "pytorch_model-00003-of-00003.bin"
+  }
+}

special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,23 @@

+{
+  "bos_token": {
+    "content": "<s>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "eos_token": {
+    "content": "</s>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "unk_token": {
+    "content": "<unk>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  }
+}

tokenizer.json ADDED Viewed

The diff for this file is too large to render. See raw diff

tokenizer.model ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:9e556afd44213b6bd1be2b850ebbbd98f5481437a8021afaf58ee7fb1818d347
+size 499723

tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,41 @@

+{
+  "add_bos_token": true,
+  "add_eos_token": false,
+  "added_tokens_decoder": {
+    "0": {
+      "content": "<unk>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "1": {
+      "content": "<s>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "2": {
+      "content": "</s>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    }
+  },
+  "bos_token": "<s>",
+  "clean_up_tokenization_spaces": false,
+  "eos_token": "</s>",
+  "legacy": false,
+  "model_max_length": 1000000000000000019884624838656,
+  "pad_token": null,
+  "padding_side": "right",
+  "sp_model_kwargs": {},
+  "tokenizer_class": "LlamaTokenizer",
+  "unk_token": "<unk>",
+  "use_default_system_prompt": false
+}

truthfulqa_mc_0shot_bs16_bf16.json ADDED Viewed

	@@ -0,0 +1,25 @@

+{
+  "results": {
+    "truthfulqa_mc": {
+      "mc1": 0.2521419828641371,
+      "mc1_stderr": 0.01520152224629996,
+      "mc2": 0.3954683484720209,
+      "mc2_stderr": 0.014883378073318826
+    }
+  },
+  "versions": {
+    "truthfulqa_mc": 1
+  },
+  "config": {
+    "model": "sparseml",
+    "model_args": "pretrained=/network/alexandre/research/cerebras/llama2_7B_sparse70_retrained/ultrachat200k/llama2_7B_sparse70_LR3e-4_GC2_E2/training,dtype=bfloat16",
+    "num_fewshot": 0,
+    "batch_size": "16",
+    "batch_sizes": [],
+    "device": "cuda:4",
+    "no_cache": true,
+    "limit": null,
+    "bootstrap_iters": 100000,
+    "description_dict": {}
+  }
+}

winogrande_5shot_bs16_bf16.json ADDED Viewed

	@@ -0,0 +1,23 @@

+{
+  "results": {
+    "winogrande": {
+      "acc": 0.6511444356748224,
+      "acc_stderr": 0.013395059320137332
+    }
+  },
+  "versions": {
+    "winogrande": 0
+  },
+  "config": {
+    "model": "sparseml",
+    "model_args": "pretrained=/network/alexandre/research/cerebras/llama2_7B_sparse70_retrained/ultrachat200k/llama2_7B_sparse70_LR3e-4_GC2_E2/training,dtype=bfloat16",
+    "num_fewshot": 5,
+    "batch_size": "16",
+    "batch_sizes": [],
+    "device": "cuda:2",
+    "no_cache": true,
+    "limit": null,
+    "bootstrap_iters": 100000,
+    "description_dict": {}
+  }
+}