diff --git a/added_tokens.json b/added_tokens.json
new file mode 100644
index 0000000000000000000000000000000000000000..56f8f607add57bb7f9c2a5cb5c57865e1fcf2b31
--- /dev/null
+++ b/added_tokens.json
@@ -0,0 +1,9 @@
+{
+ "<|im_end|>": 122753,
+ "<|im_start|>": 122757,
+ "<|tool_call|>": 122756,
+ "▁": 122758,
+ "▁": 122755,
+ "▁": 122754,
+ "▁": 122759
+}
diff --git a/mlc-chat-config.json b/mlc-chat-config.json
new file mode 100644
index 0000000000000000000000000000000000000000..7c9a9e100c2892b9093d0f478b721aee01644ba0
--- /dev/null
+++ b/mlc-chat-config.json
@@ -0,0 +1,86 @@
+{
+ "version": "0.1.0",
+ "model_type": "minicpm",
+ "quantization": "q4f16_2",
+ "model_config": {
+ "vocab_size": 122760,
+ "hidden_size": 2304,
+ "num_hidden_layers": 40,
+ "num_attention_heads": 36,
+ "num_key_value_heads": 36,
+ "hidden_act": "silu",
+ "rms_norm_eps": 1e-05,
+ "intermediate_size": 5760,
+ "scale_emb": 12,
+ "scale_depth": 1.4,
+ "dim_model_base": 256,
+ "use_cache": true,
+ "bos_token_id": 1,
+ "eos_token_id": 2,
+ "tie_word_embeddings": false,
+ "rope_theta": 1000000.0,
+ "context_window_size": 65536,
+ "prefill_chunk_size": 128,
+ "tensor_parallel_shards": 1,
+ "head_dim": 64,
+ "max_batch_size": 80,
+ "num_experts_per_tok": 0,
+ "num_experts": 0
+ },
+ "vocab_size": 122760,
+ "context_window_size": 65536,
+ "sliding_window_size": -1,
+ "prefill_chunk_size": 128,
+ "attention_sink_size": -1,
+ "tensor_parallel_shards": 1,
+ "pipeline_parallel_stages": 1,
+ "temperature": 1.0,
+ "presence_penalty": 0.0,
+ "frequency_penalty": 0.0,
+ "repetition_penalty": 1.0,
+ "top_p": 1.0,
+ "tokenizer_files": [
+ "tokenizer.model",
+ "tokenizer.json",
+ "added_tokens.json",
+ "tokenizer_config.json"
+ ],
+ "tokenizer_info": {
+ "token_postproc_method": "byte_fallback",
+ "prepend_space_in_encode": true,
+ "strip_space_in_decode": true
+ },
+ "conv_template": {
+ "name": "LM",
+ "system_template": "{system_message}",
+ "system_message": "",
+ "system_prefix_token_ids": [
+ 1
+ ],
+ "add_role_after_system_message": true,
+ "roles": {
+ "user": "",
+ "assistant": ""
+ },
+ "role_templates": {
+ "user": "{user_message}",
+ "assistant": "{assistant_message}",
+ "tool": "{tool_message}"
+ },
+ "messages": [],
+ "seps": [
+ ""
+ ],
+ "role_content_sep": "",
+ "role_empty_sep": "",
+ "stop_str": [],
+ "stop_token_ids": [
+ 2
+ ],
+ "function_string": "",
+ "use_function_calling": false
+ },
+ "pad_token_id": 0,
+ "bos_token_id": 1,
+ "eos_token_id": 2
+}
\ No newline at end of file
diff --git a/ndarray-cache.json b/ndarray-cache.json
new file mode 100644
index 0000000000000000000000000000000000000000..c24b5e58eeeaee79c32079aae3d4384d4431e85d
--- /dev/null
+++ b/ndarray-cache.json
@@ -0,0 +1,4753 @@
+{
+ "metadata": {
+ "ParamSize": 403,
+ "ParamBytes": 2505282048.0,
+ "BitsPerParam": 6.663568862935973
+ },
+ "records": [
+ {
+ "dataPath": "params_shard_0.bin",
+ "format": "raw-shard",
+ "nbytes": 565678080,
+ "records": [
+ {
+ "name": "model.embed_tokens.weight",
+ "shape": [
+ 122760,
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 565678080,
+ "byteOffset": 0
+ }
+ ],
+ "md5sum": "f6da78c822dc36f9e8b28d2bd90b2a00"
+ },
+ {
+ "dataPath": "params_shard_1.bin",
+ "format": "raw-shard",
+ "nbytes": 565678080,
+ "records": [
+ {
+ "name": "lm_head.weight",
+ "shape": [
+ 122760,
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 565678080,
+ "byteOffset": 0
+ }
+ ],
+ "md5sum": "0967948a9206a7d43b836daf5b26f6fe"
+ },
+ {
+ "dataPath": "params_shard_2.bin",
+ "format": "raw-shard",
+ "nbytes": 33523200,
+ "records": [
+ {
+ "name": "model.norm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 0
+ },
+ {
+ "name": "model.layers.0.input_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 4608
+ },
+ {
+ "name": "model.layers.0.self_attn.wqkv_pack.q_weight",
+ "shape": [
+ 6912,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 7962624,
+ "byteOffset": 9216
+ },
+ {
+ "name": "model.layers.0.self_attn.wqkv_pack.q_scale",
+ "shape": [
+ 6912,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 995328,
+ "byteOffset": 7971840
+ },
+ {
+ "name": "model.layers.0.self_attn.o_proj.q_weight",
+ "shape": [
+ 2304,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 2654208,
+ "byteOffset": 8967168
+ },
+ {
+ "name": "model.layers.0.self_attn.o_proj.q_scale",
+ "shape": [
+ 2304,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 331776,
+ "byteOffset": 11621376
+ },
+ {
+ "name": "model.layers.0.post_attention_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 11953152
+ },
+ {
+ "name": "model.layers.0.mlp.gate_up_proj.q_weight",
+ "shape": [
+ 11520,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 13271040,
+ "byteOffset": 11957760
+ },
+ {
+ "name": "model.layers.0.mlp.gate_up_proj.q_scale",
+ "shape": [
+ 11520,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 1658880,
+ "byteOffset": 25228800
+ },
+ {
+ "name": "model.layers.0.mlp.down_proj.q_weight",
+ "shape": [
+ 2304,
+ 720
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 6635520,
+ "byteOffset": 26887680
+ }
+ ],
+ "md5sum": "0f16d37af3c0c6651607ca530e325a36"
+ },
+ {
+ "dataPath": "params_shard_3.bin",
+ "format": "raw-shard",
+ "nbytes": 27712512,
+ "records": [
+ {
+ "name": "model.layers.0.mlp.down_proj.q_scale",
+ "shape": [
+ 2304,
+ 180
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 829440,
+ "byteOffset": 0
+ },
+ {
+ "name": "model.layers.1.input_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 829440
+ },
+ {
+ "name": "model.layers.1.self_attn.wqkv_pack.q_weight",
+ "shape": [
+ 6912,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 7962624,
+ "byteOffset": 834048
+ },
+ {
+ "name": "model.layers.1.self_attn.wqkv_pack.q_scale",
+ "shape": [
+ 6912,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 995328,
+ "byteOffset": 8796672
+ },
+ {
+ "name": "model.layers.1.self_attn.o_proj.q_weight",
+ "shape": [
+ 2304,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 2654208,
+ "byteOffset": 9792000
+ },
+ {
+ "name": "model.layers.1.self_attn.o_proj.q_scale",
+ "shape": [
+ 2304,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 331776,
+ "byteOffset": 12446208
+ },
+ {
+ "name": "model.layers.1.post_attention_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 12777984
+ },
+ {
+ "name": "model.layers.1.mlp.gate_up_proj.q_weight",
+ "shape": [
+ 11520,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 13271040,
+ "byteOffset": 12782592
+ },
+ {
+ "name": "model.layers.1.mlp.gate_up_proj.q_scale",
+ "shape": [
+ 11520,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 1658880,
+ "byteOffset": 26053632
+ }
+ ],
+ "md5sum": "d7cec04a244363558b50eb37aa85cc5d"
+ },
+ {
+ "dataPath": "params_shard_4.bin",
+ "format": "raw-shard",
+ "nbytes": 32689152,
+ "records": [
+ {
+ "name": "model.layers.1.mlp.down_proj.q_weight",
+ "shape": [
+ 2304,
+ 720
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 6635520,
+ "byteOffset": 0
+ },
+ {
+ "name": "model.layers.1.mlp.down_proj.q_scale",
+ "shape": [
+ 2304,
+ 180
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 829440,
+ "byteOffset": 6635520
+ },
+ {
+ "name": "model.layers.2.input_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 7464960
+ },
+ {
+ "name": "model.layers.2.self_attn.wqkv_pack.q_weight",
+ "shape": [
+ 6912,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 7962624,
+ "byteOffset": 7469568
+ },
+ {
+ "name": "model.layers.2.self_attn.wqkv_pack.q_scale",
+ "shape": [
+ 6912,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 995328,
+ "byteOffset": 15432192
+ },
+ {
+ "name": "model.layers.2.self_attn.o_proj.q_weight",
+ "shape": [
+ 2304,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 2654208,
+ "byteOffset": 16427520
+ },
+ {
+ "name": "model.layers.2.self_attn.o_proj.q_scale",
+ "shape": [
+ 2304,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 331776,
+ "byteOffset": 19081728
+ },
+ {
+ "name": "model.layers.2.post_attention_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 19413504
+ },
+ {
+ "name": "model.layers.2.mlp.gate_up_proj.q_weight",
+ "shape": [
+ 11520,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 13271040,
+ "byteOffset": 19418112
+ }
+ ],
+ "md5sum": "879a9c794de918d5497f3334dc71f0eb"
+ },
+ {
+ "dataPath": "params_shard_5.bin",
+ "format": "raw-shard",
+ "nbytes": 21076992,
+ "records": [
+ {
+ "name": "model.layers.2.mlp.gate_up_proj.q_scale",
+ "shape": [
+ 11520,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 1658880,
+ "byteOffset": 0
+ },
+ {
+ "name": "model.layers.2.mlp.down_proj.q_weight",
+ "shape": [
+ 2304,
+ 720
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 6635520,
+ "byteOffset": 1658880
+ },
+ {
+ "name": "model.layers.2.mlp.down_proj.q_scale",
+ "shape": [
+ 2304,
+ 180
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 829440,
+ "byteOffset": 8294400
+ },
+ {
+ "name": "model.layers.3.input_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 9123840
+ },
+ {
+ "name": "model.layers.3.self_attn.wqkv_pack.q_weight",
+ "shape": [
+ 6912,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 7962624,
+ "byteOffset": 9128448
+ },
+ {
+ "name": "model.layers.3.self_attn.wqkv_pack.q_scale",
+ "shape": [
+ 6912,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 995328,
+ "byteOffset": 17091072
+ },
+ {
+ "name": "model.layers.3.self_attn.o_proj.q_weight",
+ "shape": [
+ 2304,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 2654208,
+ "byteOffset": 18086400
+ },
+ {
+ "name": "model.layers.3.self_attn.o_proj.q_scale",
+ "shape": [
+ 2304,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 331776,
+ "byteOffset": 20740608
+ },
+ {
+ "name": "model.layers.3.post_attention_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 21072384
+ }
+ ],
+ "md5sum": "c401b065b8a70be7d2fee17667abd312"
+ },
+ {
+ "dataPath": "params_shard_6.bin",
+ "format": "raw-shard",
+ "nbytes": 31357440,
+ "records": [
+ {
+ "name": "model.layers.3.mlp.gate_up_proj.q_weight",
+ "shape": [
+ 11520,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 13271040,
+ "byteOffset": 0
+ },
+ {
+ "name": "model.layers.3.mlp.gate_up_proj.q_scale",
+ "shape": [
+ 11520,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 1658880,
+ "byteOffset": 13271040
+ },
+ {
+ "name": "model.layers.3.mlp.down_proj.q_weight",
+ "shape": [
+ 2304,
+ 720
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 6635520,
+ "byteOffset": 14929920
+ },
+ {
+ "name": "model.layers.3.mlp.down_proj.q_scale",
+ "shape": [
+ 2304,
+ 180
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 829440,
+ "byteOffset": 21565440
+ },
+ {
+ "name": "model.layers.4.input_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 22394880
+ },
+ {
+ "name": "model.layers.4.self_attn.wqkv_pack.q_weight",
+ "shape": [
+ 6912,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 7962624,
+ "byteOffset": 22399488
+ },
+ {
+ "name": "model.layers.4.self_attn.wqkv_pack.q_scale",
+ "shape": [
+ 6912,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 995328,
+ "byteOffset": 30362112
+ }
+ ],
+ "md5sum": "620394d400b5ec4dbab227b493f16e16"
+ },
+ {
+ "dataPath": "params_shard_7.bin",
+ "format": "raw-shard",
+ "nbytes": 33352704,
+ "records": [
+ {
+ "name": "model.layers.4.self_attn.o_proj.q_weight",
+ "shape": [
+ 2304,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 2654208,
+ "byteOffset": 0
+ },
+ {
+ "name": "model.layers.4.self_attn.o_proj.q_scale",
+ "shape": [
+ 2304,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 331776,
+ "byteOffset": 2654208
+ },
+ {
+ "name": "model.layers.4.post_attention_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 2985984
+ },
+ {
+ "name": "model.layers.4.mlp.gate_up_proj.q_weight",
+ "shape": [
+ 11520,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 13271040,
+ "byteOffset": 2990592
+ },
+ {
+ "name": "model.layers.4.mlp.gate_up_proj.q_scale",
+ "shape": [
+ 11520,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 1658880,
+ "byteOffset": 16261632
+ },
+ {
+ "name": "model.layers.4.mlp.down_proj.q_weight",
+ "shape": [
+ 2304,
+ 720
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 6635520,
+ "byteOffset": 17920512
+ },
+ {
+ "name": "model.layers.4.mlp.down_proj.q_scale",
+ "shape": [
+ 2304,
+ 180
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 829440,
+ "byteOffset": 24556032
+ },
+ {
+ "name": "model.layers.5.input_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 25385472
+ },
+ {
+ "name": "model.layers.5.self_attn.wqkv_pack.q_weight",
+ "shape": [
+ 6912,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 7962624,
+ "byteOffset": 25390080
+ }
+ ],
+ "md5sum": "7ca34268bce24b3495a8fec175fe310e"
+ },
+ {
+ "dataPath": "params_shard_8.bin",
+ "format": "raw-shard",
+ "nbytes": 26385408,
+ "records": [
+ {
+ "name": "model.layers.5.self_attn.wqkv_pack.q_scale",
+ "shape": [
+ 6912,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 995328,
+ "byteOffset": 0
+ },
+ {
+ "name": "model.layers.5.self_attn.o_proj.q_weight",
+ "shape": [
+ 2304,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 2654208,
+ "byteOffset": 995328
+ },
+ {
+ "name": "model.layers.5.self_attn.o_proj.q_scale",
+ "shape": [
+ 2304,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 331776,
+ "byteOffset": 3649536
+ },
+ {
+ "name": "model.layers.5.post_attention_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 3981312
+ },
+ {
+ "name": "model.layers.5.mlp.gate_up_proj.q_weight",
+ "shape": [
+ 11520,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 13271040,
+ "byteOffset": 3985920
+ },
+ {
+ "name": "model.layers.5.mlp.gate_up_proj.q_scale",
+ "shape": [
+ 11520,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 1658880,
+ "byteOffset": 17256960
+ },
+ {
+ "name": "model.layers.5.mlp.down_proj.q_weight",
+ "shape": [
+ 2304,
+ 720
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 6635520,
+ "byteOffset": 18915840
+ },
+ {
+ "name": "model.layers.5.mlp.down_proj.q_scale",
+ "shape": [
+ 2304,
+ 180
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 829440,
+ "byteOffset": 25551360
+ },
+ {
+ "name": "model.layers.6.input_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 26380800
+ }
+ ],
+ "md5sum": "7e8ee8186a577b7b01e6fd14e4656658"
+ },
+ {
+ "dataPath": "params_shard_9.bin",
+ "format": "raw-shard",
+ "nbytes": 33513984,
+ "records": [
+ {
+ "name": "model.layers.6.self_attn.wqkv_pack.q_weight",
+ "shape": [
+ 6912,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 7962624,
+ "byteOffset": 0
+ },
+ {
+ "name": "model.layers.6.self_attn.wqkv_pack.q_scale",
+ "shape": [
+ 6912,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 995328,
+ "byteOffset": 7962624
+ },
+ {
+ "name": "model.layers.6.self_attn.o_proj.q_weight",
+ "shape": [
+ 2304,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 2654208,
+ "byteOffset": 8957952
+ },
+ {
+ "name": "model.layers.6.self_attn.o_proj.q_scale",
+ "shape": [
+ 2304,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 331776,
+ "byteOffset": 11612160
+ },
+ {
+ "name": "model.layers.6.post_attention_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 11943936
+ },
+ {
+ "name": "model.layers.6.mlp.gate_up_proj.q_weight",
+ "shape": [
+ 11520,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 13271040,
+ "byteOffset": 11948544
+ },
+ {
+ "name": "model.layers.6.mlp.gate_up_proj.q_scale",
+ "shape": [
+ 11520,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 1658880,
+ "byteOffset": 25219584
+ },
+ {
+ "name": "model.layers.6.mlp.down_proj.q_weight",
+ "shape": [
+ 2304,
+ 720
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 6635520,
+ "byteOffset": 26878464
+ }
+ ],
+ "md5sum": "978b7fe0335164b651251df97b3b88aa"
+ },
+ {
+ "dataPath": "params_shard_10.bin",
+ "format": "raw-shard",
+ "nbytes": 27712512,
+ "records": [
+ {
+ "name": "model.layers.6.mlp.down_proj.q_scale",
+ "shape": [
+ 2304,
+ 180
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 829440,
+ "byteOffset": 0
+ },
+ {
+ "name": "model.layers.7.input_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 829440
+ },
+ {
+ "name": "model.layers.7.self_attn.wqkv_pack.q_weight",
+ "shape": [
+ 6912,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 7962624,
+ "byteOffset": 834048
+ },
+ {
+ "name": "model.layers.7.self_attn.wqkv_pack.q_scale",
+ "shape": [
+ 6912,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 995328,
+ "byteOffset": 8796672
+ },
+ {
+ "name": "model.layers.7.self_attn.o_proj.q_weight",
+ "shape": [
+ 2304,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 2654208,
+ "byteOffset": 9792000
+ },
+ {
+ "name": "model.layers.7.self_attn.o_proj.q_scale",
+ "shape": [
+ 2304,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 331776,
+ "byteOffset": 12446208
+ },
+ {
+ "name": "model.layers.7.post_attention_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 12777984
+ },
+ {
+ "name": "model.layers.7.mlp.gate_up_proj.q_weight",
+ "shape": [
+ 11520,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 13271040,
+ "byteOffset": 12782592
+ },
+ {
+ "name": "model.layers.7.mlp.gate_up_proj.q_scale",
+ "shape": [
+ 11520,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 1658880,
+ "byteOffset": 26053632
+ }
+ ],
+ "md5sum": "05bbef4d9fae05ce6947b57eb7a92d56"
+ },
+ {
+ "dataPath": "params_shard_11.bin",
+ "format": "raw-shard",
+ "nbytes": 32689152,
+ "records": [
+ {
+ "name": "model.layers.7.mlp.down_proj.q_weight",
+ "shape": [
+ 2304,
+ 720
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 6635520,
+ "byteOffset": 0
+ },
+ {
+ "name": "model.layers.7.mlp.down_proj.q_scale",
+ "shape": [
+ 2304,
+ 180
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 829440,
+ "byteOffset": 6635520
+ },
+ {
+ "name": "model.layers.8.input_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 7464960
+ },
+ {
+ "name": "model.layers.8.self_attn.wqkv_pack.q_weight",
+ "shape": [
+ 6912,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 7962624,
+ "byteOffset": 7469568
+ },
+ {
+ "name": "model.layers.8.self_attn.wqkv_pack.q_scale",
+ "shape": [
+ 6912,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 995328,
+ "byteOffset": 15432192
+ },
+ {
+ "name": "model.layers.8.self_attn.o_proj.q_weight",
+ "shape": [
+ 2304,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 2654208,
+ "byteOffset": 16427520
+ },
+ {
+ "name": "model.layers.8.self_attn.o_proj.q_scale",
+ "shape": [
+ 2304,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 331776,
+ "byteOffset": 19081728
+ },
+ {
+ "name": "model.layers.8.post_attention_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 19413504
+ },
+ {
+ "name": "model.layers.8.mlp.gate_up_proj.q_weight",
+ "shape": [
+ 11520,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 13271040,
+ "byteOffset": 19418112
+ }
+ ],
+ "md5sum": "0f9c838793a16d8bfc26169ba224e908"
+ },
+ {
+ "dataPath": "params_shard_12.bin",
+ "format": "raw-shard",
+ "nbytes": 21076992,
+ "records": [
+ {
+ "name": "model.layers.8.mlp.gate_up_proj.q_scale",
+ "shape": [
+ 11520,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 1658880,
+ "byteOffset": 0
+ },
+ {
+ "name": "model.layers.8.mlp.down_proj.q_weight",
+ "shape": [
+ 2304,
+ 720
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 6635520,
+ "byteOffset": 1658880
+ },
+ {
+ "name": "model.layers.8.mlp.down_proj.q_scale",
+ "shape": [
+ 2304,
+ 180
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 829440,
+ "byteOffset": 8294400
+ },
+ {
+ "name": "model.layers.9.input_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 9123840
+ },
+ {
+ "name": "model.layers.9.self_attn.wqkv_pack.q_weight",
+ "shape": [
+ 6912,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 7962624,
+ "byteOffset": 9128448
+ },
+ {
+ "name": "model.layers.9.self_attn.wqkv_pack.q_scale",
+ "shape": [
+ 6912,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 995328,
+ "byteOffset": 17091072
+ },
+ {
+ "name": "model.layers.9.self_attn.o_proj.q_weight",
+ "shape": [
+ 2304,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 2654208,
+ "byteOffset": 18086400
+ },
+ {
+ "name": "model.layers.9.self_attn.o_proj.q_scale",
+ "shape": [
+ 2304,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 331776,
+ "byteOffset": 20740608
+ },
+ {
+ "name": "model.layers.9.post_attention_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 21072384
+ }
+ ],
+ "md5sum": "c6c55951336b9b0c7caa1d55aab3f1b8"
+ },
+ {
+ "dataPath": "params_shard_13.bin",
+ "format": "raw-shard",
+ "nbytes": 31357440,
+ "records": [
+ {
+ "name": "model.layers.9.mlp.gate_up_proj.q_weight",
+ "shape": [
+ 11520,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 13271040,
+ "byteOffset": 0
+ },
+ {
+ "name": "model.layers.9.mlp.gate_up_proj.q_scale",
+ "shape": [
+ 11520,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 1658880,
+ "byteOffset": 13271040
+ },
+ {
+ "name": "model.layers.9.mlp.down_proj.q_weight",
+ "shape": [
+ 2304,
+ 720
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 6635520,
+ "byteOffset": 14929920
+ },
+ {
+ "name": "model.layers.9.mlp.down_proj.q_scale",
+ "shape": [
+ 2304,
+ 180
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 829440,
+ "byteOffset": 21565440
+ },
+ {
+ "name": "model.layers.10.input_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 22394880
+ },
+ {
+ "name": "model.layers.10.self_attn.wqkv_pack.q_weight",
+ "shape": [
+ 6912,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 7962624,
+ "byteOffset": 22399488
+ },
+ {
+ "name": "model.layers.10.self_attn.wqkv_pack.q_scale",
+ "shape": [
+ 6912,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 995328,
+ "byteOffset": 30362112
+ }
+ ],
+ "md5sum": "b46505794047dc23a39bf9dcae22835d"
+ },
+ {
+ "dataPath": "params_shard_14.bin",
+ "format": "raw-shard",
+ "nbytes": 33352704,
+ "records": [
+ {
+ "name": "model.layers.10.self_attn.o_proj.q_weight",
+ "shape": [
+ 2304,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 2654208,
+ "byteOffset": 0
+ },
+ {
+ "name": "model.layers.10.self_attn.o_proj.q_scale",
+ "shape": [
+ 2304,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 331776,
+ "byteOffset": 2654208
+ },
+ {
+ "name": "model.layers.10.post_attention_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 2985984
+ },
+ {
+ "name": "model.layers.10.mlp.gate_up_proj.q_weight",
+ "shape": [
+ 11520,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 13271040,
+ "byteOffset": 2990592
+ },
+ {
+ "name": "model.layers.10.mlp.gate_up_proj.q_scale",
+ "shape": [
+ 11520,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 1658880,
+ "byteOffset": 16261632
+ },
+ {
+ "name": "model.layers.10.mlp.down_proj.q_weight",
+ "shape": [
+ 2304,
+ 720
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 6635520,
+ "byteOffset": 17920512
+ },
+ {
+ "name": "model.layers.10.mlp.down_proj.q_scale",
+ "shape": [
+ 2304,
+ 180
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 829440,
+ "byteOffset": 24556032
+ },
+ {
+ "name": "model.layers.11.input_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 25385472
+ },
+ {
+ "name": "model.layers.11.self_attn.wqkv_pack.q_weight",
+ "shape": [
+ 6912,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 7962624,
+ "byteOffset": 25390080
+ }
+ ],
+ "md5sum": "d16fd1e450ef28ab4691a43e4c964919"
+ },
+ {
+ "dataPath": "params_shard_15.bin",
+ "format": "raw-shard",
+ "nbytes": 26385408,
+ "records": [
+ {
+ "name": "model.layers.11.self_attn.wqkv_pack.q_scale",
+ "shape": [
+ 6912,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 995328,
+ "byteOffset": 0
+ },
+ {
+ "name": "model.layers.11.self_attn.o_proj.q_weight",
+ "shape": [
+ 2304,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 2654208,
+ "byteOffset": 995328
+ },
+ {
+ "name": "model.layers.11.self_attn.o_proj.q_scale",
+ "shape": [
+ 2304,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 331776,
+ "byteOffset": 3649536
+ },
+ {
+ "name": "model.layers.11.post_attention_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 3981312
+ },
+ {
+ "name": "model.layers.11.mlp.gate_up_proj.q_weight",
+ "shape": [
+ 11520,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 13271040,
+ "byteOffset": 3985920
+ },
+ {
+ "name": "model.layers.11.mlp.gate_up_proj.q_scale",
+ "shape": [
+ 11520,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 1658880,
+ "byteOffset": 17256960
+ },
+ {
+ "name": "model.layers.11.mlp.down_proj.q_weight",
+ "shape": [
+ 2304,
+ 720
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 6635520,
+ "byteOffset": 18915840
+ },
+ {
+ "name": "model.layers.11.mlp.down_proj.q_scale",
+ "shape": [
+ 2304,
+ 180
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 829440,
+ "byteOffset": 25551360
+ },
+ {
+ "name": "model.layers.12.input_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 26380800
+ }
+ ],
+ "md5sum": "c97ed39429fb1be3dc3df1376c3db45f"
+ },
+ {
+ "dataPath": "params_shard_16.bin",
+ "format": "raw-shard",
+ "nbytes": 33513984,
+ "records": [
+ {
+ "name": "model.layers.12.self_attn.wqkv_pack.q_weight",
+ "shape": [
+ 6912,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 7962624,
+ "byteOffset": 0
+ },
+ {
+ "name": "model.layers.12.self_attn.wqkv_pack.q_scale",
+ "shape": [
+ 6912,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 995328,
+ "byteOffset": 7962624
+ },
+ {
+ "name": "model.layers.12.self_attn.o_proj.q_weight",
+ "shape": [
+ 2304,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 2654208,
+ "byteOffset": 8957952
+ },
+ {
+ "name": "model.layers.12.self_attn.o_proj.q_scale",
+ "shape": [
+ 2304,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 331776,
+ "byteOffset": 11612160
+ },
+ {
+ "name": "model.layers.12.post_attention_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 11943936
+ },
+ {
+ "name": "model.layers.12.mlp.gate_up_proj.q_weight",
+ "shape": [
+ 11520,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 13271040,
+ "byteOffset": 11948544
+ },
+ {
+ "name": "model.layers.12.mlp.gate_up_proj.q_scale",
+ "shape": [
+ 11520,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 1658880,
+ "byteOffset": 25219584
+ },
+ {
+ "name": "model.layers.12.mlp.down_proj.q_weight",
+ "shape": [
+ 2304,
+ 720
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 6635520,
+ "byteOffset": 26878464
+ }
+ ],
+ "md5sum": "72241fdaf3b53ad4bb90aad323027d05"
+ },
+ {
+ "dataPath": "params_shard_17.bin",
+ "format": "raw-shard",
+ "nbytes": 27712512,
+ "records": [
+ {
+ "name": "model.layers.12.mlp.down_proj.q_scale",
+ "shape": [
+ 2304,
+ 180
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 829440,
+ "byteOffset": 0
+ },
+ {
+ "name": "model.layers.13.input_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 829440
+ },
+ {
+ "name": "model.layers.13.self_attn.wqkv_pack.q_weight",
+ "shape": [
+ 6912,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 7962624,
+ "byteOffset": 834048
+ },
+ {
+ "name": "model.layers.13.self_attn.wqkv_pack.q_scale",
+ "shape": [
+ 6912,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 995328,
+ "byteOffset": 8796672
+ },
+ {
+ "name": "model.layers.13.self_attn.o_proj.q_weight",
+ "shape": [
+ 2304,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 2654208,
+ "byteOffset": 9792000
+ },
+ {
+ "name": "model.layers.13.self_attn.o_proj.q_scale",
+ "shape": [
+ 2304,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 331776,
+ "byteOffset": 12446208
+ },
+ {
+ "name": "model.layers.13.post_attention_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 12777984
+ },
+ {
+ "name": "model.layers.13.mlp.gate_up_proj.q_weight",
+ "shape": [
+ 11520,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 13271040,
+ "byteOffset": 12782592
+ },
+ {
+ "name": "model.layers.13.mlp.gate_up_proj.q_scale",
+ "shape": [
+ 11520,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 1658880,
+ "byteOffset": 26053632
+ }
+ ],
+ "md5sum": "7fd9a18405493607ddd4fe028afa184f"
+ },
+ {
+ "dataPath": "params_shard_18.bin",
+ "format": "raw-shard",
+ "nbytes": 32689152,
+ "records": [
+ {
+ "name": "model.layers.13.mlp.down_proj.q_weight",
+ "shape": [
+ 2304,
+ 720
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 6635520,
+ "byteOffset": 0
+ },
+ {
+ "name": "model.layers.13.mlp.down_proj.q_scale",
+ "shape": [
+ 2304,
+ 180
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 829440,
+ "byteOffset": 6635520
+ },
+ {
+ "name": "model.layers.14.input_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 7464960
+ },
+ {
+ "name": "model.layers.14.self_attn.wqkv_pack.q_weight",
+ "shape": [
+ 6912,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 7962624,
+ "byteOffset": 7469568
+ },
+ {
+ "name": "model.layers.14.self_attn.wqkv_pack.q_scale",
+ "shape": [
+ 6912,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 995328,
+ "byteOffset": 15432192
+ },
+ {
+ "name": "model.layers.14.self_attn.o_proj.q_weight",
+ "shape": [
+ 2304,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 2654208,
+ "byteOffset": 16427520
+ },
+ {
+ "name": "model.layers.14.self_attn.o_proj.q_scale",
+ "shape": [
+ 2304,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 331776,
+ "byteOffset": 19081728
+ },
+ {
+ "name": "model.layers.14.post_attention_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 19413504
+ },
+ {
+ "name": "model.layers.14.mlp.gate_up_proj.q_weight",
+ "shape": [
+ 11520,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 13271040,
+ "byteOffset": 19418112
+ }
+ ],
+ "md5sum": "59e8a7fea884bf13f4093ca0c9a0c8a4"
+ },
+ {
+ "dataPath": "params_shard_19.bin",
+ "format": "raw-shard",
+ "nbytes": 21076992,
+ "records": [
+ {
+ "name": "model.layers.14.mlp.gate_up_proj.q_scale",
+ "shape": [
+ 11520,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 1658880,
+ "byteOffset": 0
+ },
+ {
+ "name": "model.layers.14.mlp.down_proj.q_weight",
+ "shape": [
+ 2304,
+ 720
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 6635520,
+ "byteOffset": 1658880
+ },
+ {
+ "name": "model.layers.14.mlp.down_proj.q_scale",
+ "shape": [
+ 2304,
+ 180
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 829440,
+ "byteOffset": 8294400
+ },
+ {
+ "name": "model.layers.15.input_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 9123840
+ },
+ {
+ "name": "model.layers.15.self_attn.wqkv_pack.q_weight",
+ "shape": [
+ 6912,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 7962624,
+ "byteOffset": 9128448
+ },
+ {
+ "name": "model.layers.15.self_attn.wqkv_pack.q_scale",
+ "shape": [
+ 6912,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 995328,
+ "byteOffset": 17091072
+ },
+ {
+ "name": "model.layers.15.self_attn.o_proj.q_weight",
+ "shape": [
+ 2304,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 2654208,
+ "byteOffset": 18086400
+ },
+ {
+ "name": "model.layers.15.self_attn.o_proj.q_scale",
+ "shape": [
+ 2304,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 331776,
+ "byteOffset": 20740608
+ },
+ {
+ "name": "model.layers.15.post_attention_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 21072384
+ }
+ ],
+ "md5sum": "159ccad5e226afa240dc728d9156cbcb"
+ },
+ {
+ "dataPath": "params_shard_20.bin",
+ "format": "raw-shard",
+ "nbytes": 31357440,
+ "records": [
+ {
+ "name": "model.layers.15.mlp.gate_up_proj.q_weight",
+ "shape": [
+ 11520,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 13271040,
+ "byteOffset": 0
+ },
+ {
+ "name": "model.layers.15.mlp.gate_up_proj.q_scale",
+ "shape": [
+ 11520,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 1658880,
+ "byteOffset": 13271040
+ },
+ {
+ "name": "model.layers.15.mlp.down_proj.q_weight",
+ "shape": [
+ 2304,
+ 720
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 6635520,
+ "byteOffset": 14929920
+ },
+ {
+ "name": "model.layers.15.mlp.down_proj.q_scale",
+ "shape": [
+ 2304,
+ 180
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 829440,
+ "byteOffset": 21565440
+ },
+ {
+ "name": "model.layers.16.input_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 22394880
+ },
+ {
+ "name": "model.layers.16.self_attn.wqkv_pack.q_weight",
+ "shape": [
+ 6912,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 7962624,
+ "byteOffset": 22399488
+ },
+ {
+ "name": "model.layers.16.self_attn.wqkv_pack.q_scale",
+ "shape": [
+ 6912,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 995328,
+ "byteOffset": 30362112
+ }
+ ],
+ "md5sum": "dfe9788128a7e4b5ea6b2f122b32e743"
+ },
+ {
+ "dataPath": "params_shard_21.bin",
+ "format": "raw-shard",
+ "nbytes": 33352704,
+ "records": [
+ {
+ "name": "model.layers.16.self_attn.o_proj.q_weight",
+ "shape": [
+ 2304,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 2654208,
+ "byteOffset": 0
+ },
+ {
+ "name": "model.layers.16.self_attn.o_proj.q_scale",
+ "shape": [
+ 2304,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 331776,
+ "byteOffset": 2654208
+ },
+ {
+ "name": "model.layers.16.post_attention_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 2985984
+ },
+ {
+ "name": "model.layers.16.mlp.gate_up_proj.q_weight",
+ "shape": [
+ 11520,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 13271040,
+ "byteOffset": 2990592
+ },
+ {
+ "name": "model.layers.16.mlp.gate_up_proj.q_scale",
+ "shape": [
+ 11520,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 1658880,
+ "byteOffset": 16261632
+ },
+ {
+ "name": "model.layers.16.mlp.down_proj.q_weight",
+ "shape": [
+ 2304,
+ 720
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 6635520,
+ "byteOffset": 17920512
+ },
+ {
+ "name": "model.layers.16.mlp.down_proj.q_scale",
+ "shape": [
+ 2304,
+ 180
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 829440,
+ "byteOffset": 24556032
+ },
+ {
+ "name": "model.layers.17.input_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 25385472
+ },
+ {
+ "name": "model.layers.17.self_attn.wqkv_pack.q_weight",
+ "shape": [
+ 6912,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 7962624,
+ "byteOffset": 25390080
+ }
+ ],
+ "md5sum": "8024b1af5238fe280ed39f18a1cc1231"
+ },
+ {
+ "dataPath": "params_shard_22.bin",
+ "format": "raw-shard",
+ "nbytes": 26385408,
+ "records": [
+ {
+ "name": "model.layers.17.self_attn.wqkv_pack.q_scale",
+ "shape": [
+ 6912,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 995328,
+ "byteOffset": 0
+ },
+ {
+ "name": "model.layers.17.self_attn.o_proj.q_weight",
+ "shape": [
+ 2304,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 2654208,
+ "byteOffset": 995328
+ },
+ {
+ "name": "model.layers.17.self_attn.o_proj.q_scale",
+ "shape": [
+ 2304,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 331776,
+ "byteOffset": 3649536
+ },
+ {
+ "name": "model.layers.17.post_attention_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 3981312
+ },
+ {
+ "name": "model.layers.17.mlp.gate_up_proj.q_weight",
+ "shape": [
+ 11520,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 13271040,
+ "byteOffset": 3985920
+ },
+ {
+ "name": "model.layers.17.mlp.gate_up_proj.q_scale",
+ "shape": [
+ 11520,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 1658880,
+ "byteOffset": 17256960
+ },
+ {
+ "name": "model.layers.17.mlp.down_proj.q_weight",
+ "shape": [
+ 2304,
+ 720
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 6635520,
+ "byteOffset": 18915840
+ },
+ {
+ "name": "model.layers.17.mlp.down_proj.q_scale",
+ "shape": [
+ 2304,
+ 180
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 829440,
+ "byteOffset": 25551360
+ },
+ {
+ "name": "model.layers.18.input_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 26380800
+ }
+ ],
+ "md5sum": "005fc256810375dc0213cb1c04209f74"
+ },
+ {
+ "dataPath": "params_shard_23.bin",
+ "format": "raw-shard",
+ "nbytes": 33513984,
+ "records": [
+ {
+ "name": "model.layers.18.self_attn.wqkv_pack.q_weight",
+ "shape": [
+ 6912,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 7962624,
+ "byteOffset": 0
+ },
+ {
+ "name": "model.layers.18.self_attn.wqkv_pack.q_scale",
+ "shape": [
+ 6912,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 995328,
+ "byteOffset": 7962624
+ },
+ {
+ "name": "model.layers.18.self_attn.o_proj.q_weight",
+ "shape": [
+ 2304,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 2654208,
+ "byteOffset": 8957952
+ },
+ {
+ "name": "model.layers.18.self_attn.o_proj.q_scale",
+ "shape": [
+ 2304,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 331776,
+ "byteOffset": 11612160
+ },
+ {
+ "name": "model.layers.18.post_attention_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 11943936
+ },
+ {
+ "name": "model.layers.18.mlp.gate_up_proj.q_weight",
+ "shape": [
+ 11520,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 13271040,
+ "byteOffset": 11948544
+ },
+ {
+ "name": "model.layers.18.mlp.gate_up_proj.q_scale",
+ "shape": [
+ 11520,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 1658880,
+ "byteOffset": 25219584
+ },
+ {
+ "name": "model.layers.18.mlp.down_proj.q_weight",
+ "shape": [
+ 2304,
+ 720
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 6635520,
+ "byteOffset": 26878464
+ }
+ ],
+ "md5sum": "40e3fa2cd43966f109041829e5d06c28"
+ },
+ {
+ "dataPath": "params_shard_24.bin",
+ "format": "raw-shard",
+ "nbytes": 27712512,
+ "records": [
+ {
+ "name": "model.layers.18.mlp.down_proj.q_scale",
+ "shape": [
+ 2304,
+ 180
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 829440,
+ "byteOffset": 0
+ },
+ {
+ "name": "model.layers.19.input_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 829440
+ },
+ {
+ "name": "model.layers.19.self_attn.wqkv_pack.q_weight",
+ "shape": [
+ 6912,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 7962624,
+ "byteOffset": 834048
+ },
+ {
+ "name": "model.layers.19.self_attn.wqkv_pack.q_scale",
+ "shape": [
+ 6912,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 995328,
+ "byteOffset": 8796672
+ },
+ {
+ "name": "model.layers.19.self_attn.o_proj.q_weight",
+ "shape": [
+ 2304,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 2654208,
+ "byteOffset": 9792000
+ },
+ {
+ "name": "model.layers.19.self_attn.o_proj.q_scale",
+ "shape": [
+ 2304,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 331776,
+ "byteOffset": 12446208
+ },
+ {
+ "name": "model.layers.19.post_attention_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 12777984
+ },
+ {
+ "name": "model.layers.19.mlp.gate_up_proj.q_weight",
+ "shape": [
+ 11520,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 13271040,
+ "byteOffset": 12782592
+ },
+ {
+ "name": "model.layers.19.mlp.gate_up_proj.q_scale",
+ "shape": [
+ 11520,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 1658880,
+ "byteOffset": 26053632
+ }
+ ],
+ "md5sum": "b12141385af881b0068122f002ad25ba"
+ },
+ {
+ "dataPath": "params_shard_25.bin",
+ "format": "raw-shard",
+ "nbytes": 32689152,
+ "records": [
+ {
+ "name": "model.layers.19.mlp.down_proj.q_weight",
+ "shape": [
+ 2304,
+ 720
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 6635520,
+ "byteOffset": 0
+ },
+ {
+ "name": "model.layers.19.mlp.down_proj.q_scale",
+ "shape": [
+ 2304,
+ 180
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 829440,
+ "byteOffset": 6635520
+ },
+ {
+ "name": "model.layers.20.input_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 7464960
+ },
+ {
+ "name": "model.layers.20.self_attn.wqkv_pack.q_weight",
+ "shape": [
+ 6912,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 7962624,
+ "byteOffset": 7469568
+ },
+ {
+ "name": "model.layers.20.self_attn.wqkv_pack.q_scale",
+ "shape": [
+ 6912,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 995328,
+ "byteOffset": 15432192
+ },
+ {
+ "name": "model.layers.20.self_attn.o_proj.q_weight",
+ "shape": [
+ 2304,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 2654208,
+ "byteOffset": 16427520
+ },
+ {
+ "name": "model.layers.20.self_attn.o_proj.q_scale",
+ "shape": [
+ 2304,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 331776,
+ "byteOffset": 19081728
+ },
+ {
+ "name": "model.layers.20.post_attention_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 19413504
+ },
+ {
+ "name": "model.layers.20.mlp.gate_up_proj.q_weight",
+ "shape": [
+ 11520,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 13271040,
+ "byteOffset": 19418112
+ }
+ ],
+ "md5sum": "2918eb06d55105230e7f021f8cdf2628"
+ },
+ {
+ "dataPath": "params_shard_26.bin",
+ "format": "raw-shard",
+ "nbytes": 21076992,
+ "records": [
+ {
+ "name": "model.layers.20.mlp.gate_up_proj.q_scale",
+ "shape": [
+ 11520,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 1658880,
+ "byteOffset": 0
+ },
+ {
+ "name": "model.layers.20.mlp.down_proj.q_weight",
+ "shape": [
+ 2304,
+ 720
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 6635520,
+ "byteOffset": 1658880
+ },
+ {
+ "name": "model.layers.20.mlp.down_proj.q_scale",
+ "shape": [
+ 2304,
+ 180
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 829440,
+ "byteOffset": 8294400
+ },
+ {
+ "name": "model.layers.21.input_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 9123840
+ },
+ {
+ "name": "model.layers.21.self_attn.wqkv_pack.q_weight",
+ "shape": [
+ 6912,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 7962624,
+ "byteOffset": 9128448
+ },
+ {
+ "name": "model.layers.21.self_attn.wqkv_pack.q_scale",
+ "shape": [
+ 6912,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 995328,
+ "byteOffset": 17091072
+ },
+ {
+ "name": "model.layers.21.self_attn.o_proj.q_weight",
+ "shape": [
+ 2304,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 2654208,
+ "byteOffset": 18086400
+ },
+ {
+ "name": "model.layers.21.self_attn.o_proj.q_scale",
+ "shape": [
+ 2304,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 331776,
+ "byteOffset": 20740608
+ },
+ {
+ "name": "model.layers.21.post_attention_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 21072384
+ }
+ ],
+ "md5sum": "f8a02dbafd20623c4522fc3a1b8c6c56"
+ },
+ {
+ "dataPath": "params_shard_27.bin",
+ "format": "raw-shard",
+ "nbytes": 31357440,
+ "records": [
+ {
+ "name": "model.layers.21.mlp.gate_up_proj.q_weight",
+ "shape": [
+ 11520,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 13271040,
+ "byteOffset": 0
+ },
+ {
+ "name": "model.layers.21.mlp.gate_up_proj.q_scale",
+ "shape": [
+ 11520,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 1658880,
+ "byteOffset": 13271040
+ },
+ {
+ "name": "model.layers.21.mlp.down_proj.q_weight",
+ "shape": [
+ 2304,
+ 720
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 6635520,
+ "byteOffset": 14929920
+ },
+ {
+ "name": "model.layers.21.mlp.down_proj.q_scale",
+ "shape": [
+ 2304,
+ 180
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 829440,
+ "byteOffset": 21565440
+ },
+ {
+ "name": "model.layers.22.input_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 22394880
+ },
+ {
+ "name": "model.layers.22.self_attn.wqkv_pack.q_weight",
+ "shape": [
+ 6912,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 7962624,
+ "byteOffset": 22399488
+ },
+ {
+ "name": "model.layers.22.self_attn.wqkv_pack.q_scale",
+ "shape": [
+ 6912,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 995328,
+ "byteOffset": 30362112
+ }
+ ],
+ "md5sum": "0b6733da06a01725b01e108da5f599ce"
+ },
+ {
+ "dataPath": "params_shard_28.bin",
+ "format": "raw-shard",
+ "nbytes": 33352704,
+ "records": [
+ {
+ "name": "model.layers.22.self_attn.o_proj.q_weight",
+ "shape": [
+ 2304,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 2654208,
+ "byteOffset": 0
+ },
+ {
+ "name": "model.layers.22.self_attn.o_proj.q_scale",
+ "shape": [
+ 2304,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 331776,
+ "byteOffset": 2654208
+ },
+ {
+ "name": "model.layers.22.post_attention_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 2985984
+ },
+ {
+ "name": "model.layers.22.mlp.gate_up_proj.q_weight",
+ "shape": [
+ 11520,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 13271040,
+ "byteOffset": 2990592
+ },
+ {
+ "name": "model.layers.22.mlp.gate_up_proj.q_scale",
+ "shape": [
+ 11520,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 1658880,
+ "byteOffset": 16261632
+ },
+ {
+ "name": "model.layers.22.mlp.down_proj.q_weight",
+ "shape": [
+ 2304,
+ 720
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 6635520,
+ "byteOffset": 17920512
+ },
+ {
+ "name": "model.layers.22.mlp.down_proj.q_scale",
+ "shape": [
+ 2304,
+ 180
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 829440,
+ "byteOffset": 24556032
+ },
+ {
+ "name": "model.layers.23.input_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 25385472
+ },
+ {
+ "name": "model.layers.23.self_attn.wqkv_pack.q_weight",
+ "shape": [
+ 6912,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 7962624,
+ "byteOffset": 25390080
+ }
+ ],
+ "md5sum": "fd2875a385436a1a892bcc392eb6beba"
+ },
+ {
+ "dataPath": "params_shard_29.bin",
+ "format": "raw-shard",
+ "nbytes": 26385408,
+ "records": [
+ {
+ "name": "model.layers.23.self_attn.wqkv_pack.q_scale",
+ "shape": [
+ 6912,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 995328,
+ "byteOffset": 0
+ },
+ {
+ "name": "model.layers.23.self_attn.o_proj.q_weight",
+ "shape": [
+ 2304,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 2654208,
+ "byteOffset": 995328
+ },
+ {
+ "name": "model.layers.23.self_attn.o_proj.q_scale",
+ "shape": [
+ 2304,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 331776,
+ "byteOffset": 3649536
+ },
+ {
+ "name": "model.layers.23.post_attention_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 3981312
+ },
+ {
+ "name": "model.layers.23.mlp.gate_up_proj.q_weight",
+ "shape": [
+ 11520,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 13271040,
+ "byteOffset": 3985920
+ },
+ {
+ "name": "model.layers.23.mlp.gate_up_proj.q_scale",
+ "shape": [
+ 11520,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 1658880,
+ "byteOffset": 17256960
+ },
+ {
+ "name": "model.layers.23.mlp.down_proj.q_weight",
+ "shape": [
+ 2304,
+ 720
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 6635520,
+ "byteOffset": 18915840
+ },
+ {
+ "name": "model.layers.23.mlp.down_proj.q_scale",
+ "shape": [
+ 2304,
+ 180
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 829440,
+ "byteOffset": 25551360
+ },
+ {
+ "name": "model.layers.24.input_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 26380800
+ }
+ ],
+ "md5sum": "79060029b5316087f486339e530d3a4a"
+ },
+ {
+ "dataPath": "params_shard_30.bin",
+ "format": "raw-shard",
+ "nbytes": 33513984,
+ "records": [
+ {
+ "name": "model.layers.24.self_attn.wqkv_pack.q_weight",
+ "shape": [
+ 6912,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 7962624,
+ "byteOffset": 0
+ },
+ {
+ "name": "model.layers.24.self_attn.wqkv_pack.q_scale",
+ "shape": [
+ 6912,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 995328,
+ "byteOffset": 7962624
+ },
+ {
+ "name": "model.layers.24.self_attn.o_proj.q_weight",
+ "shape": [
+ 2304,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 2654208,
+ "byteOffset": 8957952
+ },
+ {
+ "name": "model.layers.24.self_attn.o_proj.q_scale",
+ "shape": [
+ 2304,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 331776,
+ "byteOffset": 11612160
+ },
+ {
+ "name": "model.layers.24.post_attention_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 11943936
+ },
+ {
+ "name": "model.layers.24.mlp.gate_up_proj.q_weight",
+ "shape": [
+ 11520,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 13271040,
+ "byteOffset": 11948544
+ },
+ {
+ "name": "model.layers.24.mlp.gate_up_proj.q_scale",
+ "shape": [
+ 11520,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 1658880,
+ "byteOffset": 25219584
+ },
+ {
+ "name": "model.layers.24.mlp.down_proj.q_weight",
+ "shape": [
+ 2304,
+ 720
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 6635520,
+ "byteOffset": 26878464
+ }
+ ],
+ "md5sum": "f838cda80b72da618c26e62272ef5e74"
+ },
+ {
+ "dataPath": "params_shard_31.bin",
+ "format": "raw-shard",
+ "nbytes": 27712512,
+ "records": [
+ {
+ "name": "model.layers.24.mlp.down_proj.q_scale",
+ "shape": [
+ 2304,
+ 180
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 829440,
+ "byteOffset": 0
+ },
+ {
+ "name": "model.layers.25.input_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 829440
+ },
+ {
+ "name": "model.layers.25.self_attn.wqkv_pack.q_weight",
+ "shape": [
+ 6912,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 7962624,
+ "byteOffset": 834048
+ },
+ {
+ "name": "model.layers.25.self_attn.wqkv_pack.q_scale",
+ "shape": [
+ 6912,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 995328,
+ "byteOffset": 8796672
+ },
+ {
+ "name": "model.layers.25.self_attn.o_proj.q_weight",
+ "shape": [
+ 2304,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 2654208,
+ "byteOffset": 9792000
+ },
+ {
+ "name": "model.layers.25.self_attn.o_proj.q_scale",
+ "shape": [
+ 2304,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 331776,
+ "byteOffset": 12446208
+ },
+ {
+ "name": "model.layers.25.post_attention_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 12777984
+ },
+ {
+ "name": "model.layers.25.mlp.gate_up_proj.q_weight",
+ "shape": [
+ 11520,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 13271040,
+ "byteOffset": 12782592
+ },
+ {
+ "name": "model.layers.25.mlp.gate_up_proj.q_scale",
+ "shape": [
+ 11520,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 1658880,
+ "byteOffset": 26053632
+ }
+ ],
+ "md5sum": "81002e391eb69dd901892ecf3c5501e1"
+ },
+ {
+ "dataPath": "params_shard_32.bin",
+ "format": "raw-shard",
+ "nbytes": 32689152,
+ "records": [
+ {
+ "name": "model.layers.25.mlp.down_proj.q_weight",
+ "shape": [
+ 2304,
+ 720
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 6635520,
+ "byteOffset": 0
+ },
+ {
+ "name": "model.layers.25.mlp.down_proj.q_scale",
+ "shape": [
+ 2304,
+ 180
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 829440,
+ "byteOffset": 6635520
+ },
+ {
+ "name": "model.layers.26.input_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 7464960
+ },
+ {
+ "name": "model.layers.26.self_attn.wqkv_pack.q_weight",
+ "shape": [
+ 6912,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 7962624,
+ "byteOffset": 7469568
+ },
+ {
+ "name": "model.layers.26.self_attn.wqkv_pack.q_scale",
+ "shape": [
+ 6912,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 995328,
+ "byteOffset": 15432192
+ },
+ {
+ "name": "model.layers.26.self_attn.o_proj.q_weight",
+ "shape": [
+ 2304,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 2654208,
+ "byteOffset": 16427520
+ },
+ {
+ "name": "model.layers.26.self_attn.o_proj.q_scale",
+ "shape": [
+ 2304,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 331776,
+ "byteOffset": 19081728
+ },
+ {
+ "name": "model.layers.26.post_attention_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 19413504
+ },
+ {
+ "name": "model.layers.26.mlp.gate_up_proj.q_weight",
+ "shape": [
+ 11520,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 13271040,
+ "byteOffset": 19418112
+ }
+ ],
+ "md5sum": "b3c454f47ced4a62df37bc0d4c01122c"
+ },
+ {
+ "dataPath": "params_shard_33.bin",
+ "format": "raw-shard",
+ "nbytes": 21076992,
+ "records": [
+ {
+ "name": "model.layers.26.mlp.gate_up_proj.q_scale",
+ "shape": [
+ 11520,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 1658880,
+ "byteOffset": 0
+ },
+ {
+ "name": "model.layers.26.mlp.down_proj.q_weight",
+ "shape": [
+ 2304,
+ 720
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 6635520,
+ "byteOffset": 1658880
+ },
+ {
+ "name": "model.layers.26.mlp.down_proj.q_scale",
+ "shape": [
+ 2304,
+ 180
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 829440,
+ "byteOffset": 8294400
+ },
+ {
+ "name": "model.layers.27.input_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 9123840
+ },
+ {
+ "name": "model.layers.27.self_attn.wqkv_pack.q_weight",
+ "shape": [
+ 6912,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 7962624,
+ "byteOffset": 9128448
+ },
+ {
+ "name": "model.layers.27.self_attn.wqkv_pack.q_scale",
+ "shape": [
+ 6912,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 995328,
+ "byteOffset": 17091072
+ },
+ {
+ "name": "model.layers.27.self_attn.o_proj.q_weight",
+ "shape": [
+ 2304,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 2654208,
+ "byteOffset": 18086400
+ },
+ {
+ "name": "model.layers.27.self_attn.o_proj.q_scale",
+ "shape": [
+ 2304,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 331776,
+ "byteOffset": 20740608
+ },
+ {
+ "name": "model.layers.27.post_attention_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 21072384
+ }
+ ],
+ "md5sum": "c5101542425646b176f7030b6f67cc5e"
+ },
+ {
+ "dataPath": "params_shard_34.bin",
+ "format": "raw-shard",
+ "nbytes": 31357440,
+ "records": [
+ {
+ "name": "model.layers.27.mlp.gate_up_proj.q_weight",
+ "shape": [
+ 11520,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 13271040,
+ "byteOffset": 0
+ },
+ {
+ "name": "model.layers.27.mlp.gate_up_proj.q_scale",
+ "shape": [
+ 11520,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 1658880,
+ "byteOffset": 13271040
+ },
+ {
+ "name": "model.layers.27.mlp.down_proj.q_weight",
+ "shape": [
+ 2304,
+ 720
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 6635520,
+ "byteOffset": 14929920
+ },
+ {
+ "name": "model.layers.27.mlp.down_proj.q_scale",
+ "shape": [
+ 2304,
+ 180
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 829440,
+ "byteOffset": 21565440
+ },
+ {
+ "name": "model.layers.28.input_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 22394880
+ },
+ {
+ "name": "model.layers.28.self_attn.wqkv_pack.q_weight",
+ "shape": [
+ 6912,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 7962624,
+ "byteOffset": 22399488
+ },
+ {
+ "name": "model.layers.28.self_attn.wqkv_pack.q_scale",
+ "shape": [
+ 6912,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 995328,
+ "byteOffset": 30362112
+ }
+ ],
+ "md5sum": "72decc56d73760ca281360bdf443e0f9"
+ },
+ {
+ "dataPath": "params_shard_35.bin",
+ "format": "raw-shard",
+ "nbytes": 33352704,
+ "records": [
+ {
+ "name": "model.layers.28.self_attn.o_proj.q_weight",
+ "shape": [
+ 2304,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 2654208,
+ "byteOffset": 0
+ },
+ {
+ "name": "model.layers.28.self_attn.o_proj.q_scale",
+ "shape": [
+ 2304,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 331776,
+ "byteOffset": 2654208
+ },
+ {
+ "name": "model.layers.28.post_attention_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 2985984
+ },
+ {
+ "name": "model.layers.28.mlp.gate_up_proj.q_weight",
+ "shape": [
+ 11520,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 13271040,
+ "byteOffset": 2990592
+ },
+ {
+ "name": "model.layers.28.mlp.gate_up_proj.q_scale",
+ "shape": [
+ 11520,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 1658880,
+ "byteOffset": 16261632
+ },
+ {
+ "name": "model.layers.28.mlp.down_proj.q_weight",
+ "shape": [
+ 2304,
+ 720
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 6635520,
+ "byteOffset": 17920512
+ },
+ {
+ "name": "model.layers.28.mlp.down_proj.q_scale",
+ "shape": [
+ 2304,
+ 180
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 829440,
+ "byteOffset": 24556032
+ },
+ {
+ "name": "model.layers.29.input_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 25385472
+ },
+ {
+ "name": "model.layers.29.self_attn.wqkv_pack.q_weight",
+ "shape": [
+ 6912,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 7962624,
+ "byteOffset": 25390080
+ }
+ ],
+ "md5sum": "881ce27f23d74f6a6f1bab9c0498de53"
+ },
+ {
+ "dataPath": "params_shard_36.bin",
+ "format": "raw-shard",
+ "nbytes": 26385408,
+ "records": [
+ {
+ "name": "model.layers.29.self_attn.wqkv_pack.q_scale",
+ "shape": [
+ 6912,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 995328,
+ "byteOffset": 0
+ },
+ {
+ "name": "model.layers.29.self_attn.o_proj.q_weight",
+ "shape": [
+ 2304,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 2654208,
+ "byteOffset": 995328
+ },
+ {
+ "name": "model.layers.29.self_attn.o_proj.q_scale",
+ "shape": [
+ 2304,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 331776,
+ "byteOffset": 3649536
+ },
+ {
+ "name": "model.layers.29.post_attention_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 3981312
+ },
+ {
+ "name": "model.layers.29.mlp.gate_up_proj.q_weight",
+ "shape": [
+ 11520,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 13271040,
+ "byteOffset": 3985920
+ },
+ {
+ "name": "model.layers.29.mlp.gate_up_proj.q_scale",
+ "shape": [
+ 11520,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 1658880,
+ "byteOffset": 17256960
+ },
+ {
+ "name": "model.layers.29.mlp.down_proj.q_weight",
+ "shape": [
+ 2304,
+ 720
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 6635520,
+ "byteOffset": 18915840
+ },
+ {
+ "name": "model.layers.29.mlp.down_proj.q_scale",
+ "shape": [
+ 2304,
+ 180
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 829440,
+ "byteOffset": 25551360
+ },
+ {
+ "name": "model.layers.30.input_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 26380800
+ }
+ ],
+ "md5sum": "3fed9aeef9535418af46d53f564fe4e0"
+ },
+ {
+ "dataPath": "params_shard_37.bin",
+ "format": "raw-shard",
+ "nbytes": 33513984,
+ "records": [
+ {
+ "name": "model.layers.30.self_attn.wqkv_pack.q_weight",
+ "shape": [
+ 6912,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 7962624,
+ "byteOffset": 0
+ },
+ {
+ "name": "model.layers.30.self_attn.wqkv_pack.q_scale",
+ "shape": [
+ 6912,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 995328,
+ "byteOffset": 7962624
+ },
+ {
+ "name": "model.layers.30.self_attn.o_proj.q_weight",
+ "shape": [
+ 2304,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 2654208,
+ "byteOffset": 8957952
+ },
+ {
+ "name": "model.layers.30.self_attn.o_proj.q_scale",
+ "shape": [
+ 2304,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 331776,
+ "byteOffset": 11612160
+ },
+ {
+ "name": "model.layers.30.post_attention_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 11943936
+ },
+ {
+ "name": "model.layers.30.mlp.gate_up_proj.q_weight",
+ "shape": [
+ 11520,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 13271040,
+ "byteOffset": 11948544
+ },
+ {
+ "name": "model.layers.30.mlp.gate_up_proj.q_scale",
+ "shape": [
+ 11520,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 1658880,
+ "byteOffset": 25219584
+ },
+ {
+ "name": "model.layers.30.mlp.down_proj.q_weight",
+ "shape": [
+ 2304,
+ 720
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 6635520,
+ "byteOffset": 26878464
+ }
+ ],
+ "md5sum": "d99ae46b9de89d42dac9ef0e7fca5e95"
+ },
+ {
+ "dataPath": "params_shard_38.bin",
+ "format": "raw-shard",
+ "nbytes": 27712512,
+ "records": [
+ {
+ "name": "model.layers.30.mlp.down_proj.q_scale",
+ "shape": [
+ 2304,
+ 180
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 829440,
+ "byteOffset": 0
+ },
+ {
+ "name": "model.layers.31.input_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 829440
+ },
+ {
+ "name": "model.layers.31.self_attn.wqkv_pack.q_weight",
+ "shape": [
+ 6912,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 7962624,
+ "byteOffset": 834048
+ },
+ {
+ "name": "model.layers.31.self_attn.wqkv_pack.q_scale",
+ "shape": [
+ 6912,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 995328,
+ "byteOffset": 8796672
+ },
+ {
+ "name": "model.layers.31.self_attn.o_proj.q_weight",
+ "shape": [
+ 2304,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 2654208,
+ "byteOffset": 9792000
+ },
+ {
+ "name": "model.layers.31.self_attn.o_proj.q_scale",
+ "shape": [
+ 2304,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 331776,
+ "byteOffset": 12446208
+ },
+ {
+ "name": "model.layers.31.post_attention_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 12777984
+ },
+ {
+ "name": "model.layers.31.mlp.gate_up_proj.q_weight",
+ "shape": [
+ 11520,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 13271040,
+ "byteOffset": 12782592
+ },
+ {
+ "name": "model.layers.31.mlp.gate_up_proj.q_scale",
+ "shape": [
+ 11520,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 1658880,
+ "byteOffset": 26053632
+ }
+ ],
+ "md5sum": "0e7b880dc531fd473320cc8899cdb236"
+ },
+ {
+ "dataPath": "params_shard_39.bin",
+ "format": "raw-shard",
+ "nbytes": 32689152,
+ "records": [
+ {
+ "name": "model.layers.31.mlp.down_proj.q_weight",
+ "shape": [
+ 2304,
+ 720
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 6635520,
+ "byteOffset": 0
+ },
+ {
+ "name": "model.layers.31.mlp.down_proj.q_scale",
+ "shape": [
+ 2304,
+ 180
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 829440,
+ "byteOffset": 6635520
+ },
+ {
+ "name": "model.layers.32.input_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 7464960
+ },
+ {
+ "name": "model.layers.32.self_attn.wqkv_pack.q_weight",
+ "shape": [
+ 6912,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 7962624,
+ "byteOffset": 7469568
+ },
+ {
+ "name": "model.layers.32.self_attn.wqkv_pack.q_scale",
+ "shape": [
+ 6912,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 995328,
+ "byteOffset": 15432192
+ },
+ {
+ "name": "model.layers.32.self_attn.o_proj.q_weight",
+ "shape": [
+ 2304,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 2654208,
+ "byteOffset": 16427520
+ },
+ {
+ "name": "model.layers.32.self_attn.o_proj.q_scale",
+ "shape": [
+ 2304,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 331776,
+ "byteOffset": 19081728
+ },
+ {
+ "name": "model.layers.32.post_attention_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 19413504
+ },
+ {
+ "name": "model.layers.32.mlp.gate_up_proj.q_weight",
+ "shape": [
+ 11520,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 13271040,
+ "byteOffset": 19418112
+ }
+ ],
+ "md5sum": "22419a5248a7923be57ef1881431ca09"
+ },
+ {
+ "dataPath": "params_shard_40.bin",
+ "format": "raw-shard",
+ "nbytes": 21076992,
+ "records": [
+ {
+ "name": "model.layers.32.mlp.gate_up_proj.q_scale",
+ "shape": [
+ 11520,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 1658880,
+ "byteOffset": 0
+ },
+ {
+ "name": "model.layers.32.mlp.down_proj.q_weight",
+ "shape": [
+ 2304,
+ 720
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 6635520,
+ "byteOffset": 1658880
+ },
+ {
+ "name": "model.layers.32.mlp.down_proj.q_scale",
+ "shape": [
+ 2304,
+ 180
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 829440,
+ "byteOffset": 8294400
+ },
+ {
+ "name": "model.layers.33.input_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 9123840
+ },
+ {
+ "name": "model.layers.33.self_attn.wqkv_pack.q_weight",
+ "shape": [
+ 6912,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 7962624,
+ "byteOffset": 9128448
+ },
+ {
+ "name": "model.layers.33.self_attn.wqkv_pack.q_scale",
+ "shape": [
+ 6912,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 995328,
+ "byteOffset": 17091072
+ },
+ {
+ "name": "model.layers.33.self_attn.o_proj.q_weight",
+ "shape": [
+ 2304,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 2654208,
+ "byteOffset": 18086400
+ },
+ {
+ "name": "model.layers.33.self_attn.o_proj.q_scale",
+ "shape": [
+ 2304,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 331776,
+ "byteOffset": 20740608
+ },
+ {
+ "name": "model.layers.33.post_attention_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 21072384
+ }
+ ],
+ "md5sum": "6ec54fa42c17ef6a45205f6334607485"
+ },
+ {
+ "dataPath": "params_shard_41.bin",
+ "format": "raw-shard",
+ "nbytes": 31357440,
+ "records": [
+ {
+ "name": "model.layers.33.mlp.gate_up_proj.q_weight",
+ "shape": [
+ 11520,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 13271040,
+ "byteOffset": 0
+ },
+ {
+ "name": "model.layers.33.mlp.gate_up_proj.q_scale",
+ "shape": [
+ 11520,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 1658880,
+ "byteOffset": 13271040
+ },
+ {
+ "name": "model.layers.33.mlp.down_proj.q_weight",
+ "shape": [
+ 2304,
+ 720
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 6635520,
+ "byteOffset": 14929920
+ },
+ {
+ "name": "model.layers.33.mlp.down_proj.q_scale",
+ "shape": [
+ 2304,
+ 180
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 829440,
+ "byteOffset": 21565440
+ },
+ {
+ "name": "model.layers.34.input_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 22394880
+ },
+ {
+ "name": "model.layers.34.self_attn.wqkv_pack.q_weight",
+ "shape": [
+ 6912,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 7962624,
+ "byteOffset": 22399488
+ },
+ {
+ "name": "model.layers.34.self_attn.wqkv_pack.q_scale",
+ "shape": [
+ 6912,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 995328,
+ "byteOffset": 30362112
+ }
+ ],
+ "md5sum": "232e89b26e877460804eb1318a895088"
+ },
+ {
+ "dataPath": "params_shard_42.bin",
+ "format": "raw-shard",
+ "nbytes": 33352704,
+ "records": [
+ {
+ "name": "model.layers.34.self_attn.o_proj.q_weight",
+ "shape": [
+ 2304,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 2654208,
+ "byteOffset": 0
+ },
+ {
+ "name": "model.layers.34.self_attn.o_proj.q_scale",
+ "shape": [
+ 2304,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 331776,
+ "byteOffset": 2654208
+ },
+ {
+ "name": "model.layers.34.post_attention_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 2985984
+ },
+ {
+ "name": "model.layers.34.mlp.gate_up_proj.q_weight",
+ "shape": [
+ 11520,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 13271040,
+ "byteOffset": 2990592
+ },
+ {
+ "name": "model.layers.34.mlp.gate_up_proj.q_scale",
+ "shape": [
+ 11520,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 1658880,
+ "byteOffset": 16261632
+ },
+ {
+ "name": "model.layers.34.mlp.down_proj.q_weight",
+ "shape": [
+ 2304,
+ 720
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 6635520,
+ "byteOffset": 17920512
+ },
+ {
+ "name": "model.layers.34.mlp.down_proj.q_scale",
+ "shape": [
+ 2304,
+ 180
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 829440,
+ "byteOffset": 24556032
+ },
+ {
+ "name": "model.layers.35.input_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 25385472
+ },
+ {
+ "name": "model.layers.35.self_attn.wqkv_pack.q_weight",
+ "shape": [
+ 6912,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 7962624,
+ "byteOffset": 25390080
+ }
+ ],
+ "md5sum": "0e6785b2d4a9df38323a555d3f6cebe7"
+ },
+ {
+ "dataPath": "params_shard_43.bin",
+ "format": "raw-shard",
+ "nbytes": 26385408,
+ "records": [
+ {
+ "name": "model.layers.35.self_attn.wqkv_pack.q_scale",
+ "shape": [
+ 6912,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 995328,
+ "byteOffset": 0
+ },
+ {
+ "name": "model.layers.35.self_attn.o_proj.q_weight",
+ "shape": [
+ 2304,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 2654208,
+ "byteOffset": 995328
+ },
+ {
+ "name": "model.layers.35.self_attn.o_proj.q_scale",
+ "shape": [
+ 2304,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 331776,
+ "byteOffset": 3649536
+ },
+ {
+ "name": "model.layers.35.post_attention_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 3981312
+ },
+ {
+ "name": "model.layers.35.mlp.gate_up_proj.q_weight",
+ "shape": [
+ 11520,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 13271040,
+ "byteOffset": 3985920
+ },
+ {
+ "name": "model.layers.35.mlp.gate_up_proj.q_scale",
+ "shape": [
+ 11520,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 1658880,
+ "byteOffset": 17256960
+ },
+ {
+ "name": "model.layers.35.mlp.down_proj.q_weight",
+ "shape": [
+ 2304,
+ 720
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 6635520,
+ "byteOffset": 18915840
+ },
+ {
+ "name": "model.layers.35.mlp.down_proj.q_scale",
+ "shape": [
+ 2304,
+ 180
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 829440,
+ "byteOffset": 25551360
+ },
+ {
+ "name": "model.layers.36.input_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 26380800
+ }
+ ],
+ "md5sum": "210c4b4617f42ada7b1617e09d9dad4d"
+ },
+ {
+ "dataPath": "params_shard_44.bin",
+ "format": "raw-shard",
+ "nbytes": 33513984,
+ "records": [
+ {
+ "name": "model.layers.36.self_attn.wqkv_pack.q_weight",
+ "shape": [
+ 6912,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 7962624,
+ "byteOffset": 0
+ },
+ {
+ "name": "model.layers.36.self_attn.wqkv_pack.q_scale",
+ "shape": [
+ 6912,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 995328,
+ "byteOffset": 7962624
+ },
+ {
+ "name": "model.layers.36.self_attn.o_proj.q_weight",
+ "shape": [
+ 2304,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 2654208,
+ "byteOffset": 8957952
+ },
+ {
+ "name": "model.layers.36.self_attn.o_proj.q_scale",
+ "shape": [
+ 2304,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 331776,
+ "byteOffset": 11612160
+ },
+ {
+ "name": "model.layers.36.post_attention_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 11943936
+ },
+ {
+ "name": "model.layers.36.mlp.gate_up_proj.q_weight",
+ "shape": [
+ 11520,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 13271040,
+ "byteOffset": 11948544
+ },
+ {
+ "name": "model.layers.36.mlp.gate_up_proj.q_scale",
+ "shape": [
+ 11520,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 1658880,
+ "byteOffset": 25219584
+ },
+ {
+ "name": "model.layers.36.mlp.down_proj.q_weight",
+ "shape": [
+ 2304,
+ 720
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 6635520,
+ "byteOffset": 26878464
+ }
+ ],
+ "md5sum": "a9691a5c539761866071fe66ba7c43ab"
+ },
+ {
+ "dataPath": "params_shard_45.bin",
+ "format": "raw-shard",
+ "nbytes": 27712512,
+ "records": [
+ {
+ "name": "model.layers.36.mlp.down_proj.q_scale",
+ "shape": [
+ 2304,
+ 180
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 829440,
+ "byteOffset": 0
+ },
+ {
+ "name": "model.layers.37.input_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 829440
+ },
+ {
+ "name": "model.layers.37.self_attn.wqkv_pack.q_weight",
+ "shape": [
+ 6912,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 7962624,
+ "byteOffset": 834048
+ },
+ {
+ "name": "model.layers.37.self_attn.wqkv_pack.q_scale",
+ "shape": [
+ 6912,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 995328,
+ "byteOffset": 8796672
+ },
+ {
+ "name": "model.layers.37.self_attn.o_proj.q_weight",
+ "shape": [
+ 2304,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 2654208,
+ "byteOffset": 9792000
+ },
+ {
+ "name": "model.layers.37.self_attn.o_proj.q_scale",
+ "shape": [
+ 2304,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 331776,
+ "byteOffset": 12446208
+ },
+ {
+ "name": "model.layers.37.post_attention_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 12777984
+ },
+ {
+ "name": "model.layers.37.mlp.gate_up_proj.q_weight",
+ "shape": [
+ 11520,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 13271040,
+ "byteOffset": 12782592
+ },
+ {
+ "name": "model.layers.37.mlp.gate_up_proj.q_scale",
+ "shape": [
+ 11520,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 1658880,
+ "byteOffset": 26053632
+ }
+ ],
+ "md5sum": "bf0f96571886d57590af8ba7e45cd3cd"
+ },
+ {
+ "dataPath": "params_shard_46.bin",
+ "format": "raw-shard",
+ "nbytes": 32689152,
+ "records": [
+ {
+ "name": "model.layers.37.mlp.down_proj.q_weight",
+ "shape": [
+ 2304,
+ 720
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 6635520,
+ "byteOffset": 0
+ },
+ {
+ "name": "model.layers.37.mlp.down_proj.q_scale",
+ "shape": [
+ 2304,
+ 180
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 829440,
+ "byteOffset": 6635520
+ },
+ {
+ "name": "model.layers.38.input_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 7464960
+ },
+ {
+ "name": "model.layers.38.self_attn.wqkv_pack.q_weight",
+ "shape": [
+ 6912,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 7962624,
+ "byteOffset": 7469568
+ },
+ {
+ "name": "model.layers.38.self_attn.wqkv_pack.q_scale",
+ "shape": [
+ 6912,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 995328,
+ "byteOffset": 15432192
+ },
+ {
+ "name": "model.layers.38.self_attn.o_proj.q_weight",
+ "shape": [
+ 2304,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 2654208,
+ "byteOffset": 16427520
+ },
+ {
+ "name": "model.layers.38.self_attn.o_proj.q_scale",
+ "shape": [
+ 2304,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 331776,
+ "byteOffset": 19081728
+ },
+ {
+ "name": "model.layers.38.post_attention_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 19413504
+ },
+ {
+ "name": "model.layers.38.mlp.gate_up_proj.q_weight",
+ "shape": [
+ 11520,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 13271040,
+ "byteOffset": 19418112
+ }
+ ],
+ "md5sum": "9bd722dd944e5baf8bf29e227152b517"
+ },
+ {
+ "dataPath": "params_shard_47.bin",
+ "format": "raw-shard",
+ "nbytes": 21076992,
+ "records": [
+ {
+ "name": "model.layers.38.mlp.gate_up_proj.q_scale",
+ "shape": [
+ 11520,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 1658880,
+ "byteOffset": 0
+ },
+ {
+ "name": "model.layers.38.mlp.down_proj.q_weight",
+ "shape": [
+ 2304,
+ 720
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 6635520,
+ "byteOffset": 1658880
+ },
+ {
+ "name": "model.layers.38.mlp.down_proj.q_scale",
+ "shape": [
+ 2304,
+ 180
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 829440,
+ "byteOffset": 8294400
+ },
+ {
+ "name": "model.layers.39.input_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 9123840
+ },
+ {
+ "name": "model.layers.39.self_attn.wqkv_pack.q_weight",
+ "shape": [
+ 6912,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 7962624,
+ "byteOffset": 9128448
+ },
+ {
+ "name": "model.layers.39.self_attn.wqkv_pack.q_scale",
+ "shape": [
+ 6912,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 995328,
+ "byteOffset": 17091072
+ },
+ {
+ "name": "model.layers.39.self_attn.o_proj.q_weight",
+ "shape": [
+ 2304,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 2654208,
+ "byteOffset": 18086400
+ },
+ {
+ "name": "model.layers.39.self_attn.o_proj.q_scale",
+ "shape": [
+ 2304,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 331776,
+ "byteOffset": 20740608
+ },
+ {
+ "name": "model.layers.39.post_attention_layernorm.weight",
+ "shape": [
+ 2304
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 4608,
+ "byteOffset": 21072384
+ }
+ ],
+ "md5sum": "47e0601e96a3a32fa93aa1f3c3f80b7e"
+ },
+ {
+ "dataPath": "params_shard_48.bin",
+ "format": "raw-shard",
+ "nbytes": 22394880,
+ "records": [
+ {
+ "name": "model.layers.39.mlp.gate_up_proj.q_weight",
+ "shape": [
+ 11520,
+ 288
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 13271040,
+ "byteOffset": 0
+ },
+ {
+ "name": "model.layers.39.mlp.gate_up_proj.q_scale",
+ "shape": [
+ 11520,
+ 72
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 1658880,
+ "byteOffset": 13271040
+ },
+ {
+ "name": "model.layers.39.mlp.down_proj.q_weight",
+ "shape": [
+ 2304,
+ 720
+ ],
+ "dtype": "uint32",
+ "format": "f32-to-bf16",
+ "nbytes": 6635520,
+ "byteOffset": 14929920
+ },
+ {
+ "name": "model.layers.39.mlp.down_proj.q_scale",
+ "shape": [
+ 2304,
+ 180
+ ],
+ "dtype": "float16",
+ "format": "f32-to-bf16",
+ "nbytes": 829440,
+ "byteOffset": 21565440
+ }
+ ],
+ "md5sum": "003f49fd09fd6767ae29c9d5baf6542f"
+ }
+ ]
+}
\ No newline at end of file
diff --git a/params_shard_0.bin b/params_shard_0.bin
new file mode 100644
index 0000000000000000000000000000000000000000..0c90f51e0bef050df1e3f02d3fb02da6b5a4cdc9
--- /dev/null
+++ b/params_shard_0.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:fa485daa8aba60a6b7bfcb5d48b122d634ff8540ede68988c86b26b969c3cf23
+size 565678080
diff --git a/params_shard_1.bin b/params_shard_1.bin
new file mode 100644
index 0000000000000000000000000000000000000000..0c82cb59fc933791ecded68c1db92abac1fd80c4
--- /dev/null
+++ b/params_shard_1.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:85049f84e40e4f15d149fb54bb8e5f274bc9eeb3e55342bfc26fe00738b34acf
+size 565678080
diff --git a/params_shard_10.bin b/params_shard_10.bin
new file mode 100644
index 0000000000000000000000000000000000000000..bcfe8d136010e59fba2efce133d1c79f2e8ea0da
--- /dev/null
+++ b/params_shard_10.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:be573e6433d3c14a6dae389a7087e03da1bf60032f6c0579d676fa036059a993
+size 27712512
diff --git a/params_shard_11.bin b/params_shard_11.bin
new file mode 100644
index 0000000000000000000000000000000000000000..9e37e10acd4222ef607b05dabe54321258cae898
--- /dev/null
+++ b/params_shard_11.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:3d42993968a7d6a0faf5e975dace9c0d2f5ac76f07905813ecfd043ddffa94c4
+size 32689152
diff --git a/params_shard_12.bin b/params_shard_12.bin
new file mode 100644
index 0000000000000000000000000000000000000000..fc35b1d96f5d3b3150907ea4f1027572344e4368
--- /dev/null
+++ b/params_shard_12.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:54812d712abf89d7c17e77c67299d91bbc064a30e93fd25eee8b2949d1cb8a8c
+size 21076992
diff --git a/params_shard_13.bin b/params_shard_13.bin
new file mode 100644
index 0000000000000000000000000000000000000000..cf63987c20c3d036a9ad2762f945186a8bb4a6e4
--- /dev/null
+++ b/params_shard_13.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:e8fbaa47c1712acc433c344dd90e906436e8a88526a8cc7b6fd797ea32280cb5
+size 31357440
diff --git a/params_shard_14.bin b/params_shard_14.bin
new file mode 100644
index 0000000000000000000000000000000000000000..d12588a6001de12c05c59a9ec87b4ee32ff41268
--- /dev/null
+++ b/params_shard_14.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:b657c8112db4bab7c671a4220b457abd311198301df34a9e036734ce89c15780
+size 33352704
diff --git a/params_shard_15.bin b/params_shard_15.bin
new file mode 100644
index 0000000000000000000000000000000000000000..bf7b347baed7ae18d8e5925655e2f81f18a8460c
--- /dev/null
+++ b/params_shard_15.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:651a79d9b9f7e8a072411146714373a9c6a99c0731e8129ee8d4a5409b043361
+size 26385408
diff --git a/params_shard_16.bin b/params_shard_16.bin
new file mode 100644
index 0000000000000000000000000000000000000000..c46600c02165de732492514d6d47657d7d123f33
--- /dev/null
+++ b/params_shard_16.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:630b42df1c66e9439132dbaba465212f97a2791633f7a6c00ab6f0feae6f14d1
+size 33513984
diff --git a/params_shard_17.bin b/params_shard_17.bin
new file mode 100644
index 0000000000000000000000000000000000000000..e51ec8472f810fd3dba9eaa25ab02e185245e94a
--- /dev/null
+++ b/params_shard_17.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:9fb203637bfd59875c012f4618328f5f7ed9a599e2a243ab75fa7c79355bca04
+size 27712512
diff --git a/params_shard_18.bin b/params_shard_18.bin
new file mode 100644
index 0000000000000000000000000000000000000000..c802bbd278fb69173a9a0fa84383e7a7e829862e
--- /dev/null
+++ b/params_shard_18.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:b3a23fa8ee39505be9075f9a02e003e14c5a6b5f5f3e02a7d8d48ce7b29d095b
+size 32689152
diff --git a/params_shard_19.bin b/params_shard_19.bin
new file mode 100644
index 0000000000000000000000000000000000000000..cd5ebc9e28172af01ade62932230b2cdfdbfd27b
--- /dev/null
+++ b/params_shard_19.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:e1c9f087dd6353106a315a33f99f145de88d595757a7ee22833a6dbc63c3d7aa
+size 21076992
diff --git a/params_shard_2.bin b/params_shard_2.bin
new file mode 100644
index 0000000000000000000000000000000000000000..181b97e7f20c04e215eea16ab51d6f6442af7649
--- /dev/null
+++ b/params_shard_2.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:c23f0dee58fb0a59fa9fac00a9d6971340f1edff501cbcf665c4a27d1e758b4c
+size 33523200
diff --git a/params_shard_20.bin b/params_shard_20.bin
new file mode 100644
index 0000000000000000000000000000000000000000..e138a3db908a47da4e241c310e36ab87f127fff0
--- /dev/null
+++ b/params_shard_20.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:e79968e851f27ab2e431ce7a87ebe2bb8629fdc3839c4f2683f7908ac45ce1bb
+size 31357440
diff --git a/params_shard_21.bin b/params_shard_21.bin
new file mode 100644
index 0000000000000000000000000000000000000000..47a8a7e0fcca0f24c7a3185eaf8a204ff067e86c
--- /dev/null
+++ b/params_shard_21.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:27c3fd144fe212743154d2040771c49c75cba3291344b8c4767881f269a01405
+size 33352704
diff --git a/params_shard_22.bin b/params_shard_22.bin
new file mode 100644
index 0000000000000000000000000000000000000000..713706c77da51e3104450a447f5441f1476e8a32
--- /dev/null
+++ b/params_shard_22.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:71a14492a5ca3b4a6c21052aa9f380e09032071719b555b0e340255632942f7e
+size 26385408
diff --git a/params_shard_23.bin b/params_shard_23.bin
new file mode 100644
index 0000000000000000000000000000000000000000..7099efc915e31317850a9868cd05e640c4d3f0f7
--- /dev/null
+++ b/params_shard_23.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:f4d931b85f3fa0a718dbcb13b08f71c70d04ed6d5d798f4dd6ef4dc8291cb53e
+size 33513984
diff --git a/params_shard_24.bin b/params_shard_24.bin
new file mode 100644
index 0000000000000000000000000000000000000000..1917534de1b7414a682c6eac34bbc53df73181e0
--- /dev/null
+++ b/params_shard_24.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:5c77b2a1287bc6d86c0e75128dbb2f80764395dc7efd5e2f59f3acfb2943f430
+size 27712512
diff --git a/params_shard_25.bin b/params_shard_25.bin
new file mode 100644
index 0000000000000000000000000000000000000000..50538679ba8e30e6fd0306325a3dc98adc24c5c5
--- /dev/null
+++ b/params_shard_25.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:df58040aa555a95c0e05f31f01e4c11e3e2cdc5d9af844174fee788bc80a01df
+size 32689152
diff --git a/params_shard_26.bin b/params_shard_26.bin
new file mode 100644
index 0000000000000000000000000000000000000000..4f34ddea256ed30b4e5b2cb33b243cedee1cbc50
--- /dev/null
+++ b/params_shard_26.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:c80698bc38add6ddb77f95aad13926910d5718275cdc441ff7dd28003409ffd7
+size 21076992
diff --git a/params_shard_27.bin b/params_shard_27.bin
new file mode 100644
index 0000000000000000000000000000000000000000..7ac5df471b4688f131f1456d1bfdb0915e7cd412
--- /dev/null
+++ b/params_shard_27.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:e1b65bf7a78155ca23f51b508c66a1443a383cfab5a5eba2f923966977529c22
+size 31357440
diff --git a/params_shard_28.bin b/params_shard_28.bin
new file mode 100644
index 0000000000000000000000000000000000000000..68e9e0f752d363c771d315e461bc6eae8c81d807
--- /dev/null
+++ b/params_shard_28.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:1279aad7ffeb7333c5c04ad1962a0a4e5a4d6a1785afc36e65d27813d32617cc
+size 33352704
diff --git a/params_shard_29.bin b/params_shard_29.bin
new file mode 100644
index 0000000000000000000000000000000000000000..aab6606e8298c0a80b3f9f61d5cc1c7513b51a8a
--- /dev/null
+++ b/params_shard_29.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:e5d7a38266e905a1107dab8c5f53dd2c04f6125acfa0b18daf56af6671aa6871
+size 26385408
diff --git a/params_shard_3.bin b/params_shard_3.bin
new file mode 100644
index 0000000000000000000000000000000000000000..5b11f4be46d738502f74c023bd848c422efa01a7
--- /dev/null
+++ b/params_shard_3.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:6f26dde1bec9641ed72663d60c95c06dafdbbe47645e0d085684a4bbf234c118
+size 27712512
diff --git a/params_shard_30.bin b/params_shard_30.bin
new file mode 100644
index 0000000000000000000000000000000000000000..ef37aacf53076af8541334a3519ef378c5123e59
--- /dev/null
+++ b/params_shard_30.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:81d7911f54ff567f89e6ded0abe347e06301a6e084fe40e0866db62e8e90d868
+size 33513984
diff --git a/params_shard_31.bin b/params_shard_31.bin
new file mode 100644
index 0000000000000000000000000000000000000000..138a3f93edbffcc6977760f09582f1ea5575a07e
--- /dev/null
+++ b/params_shard_31.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:89624843725fb6bbe31ef97f529546af148a42cb692aa87ab04da7c06d092246
+size 27712512
diff --git a/params_shard_32.bin b/params_shard_32.bin
new file mode 100644
index 0000000000000000000000000000000000000000..28afdd3c56be51fd59b1c8c070d9bc19bb9885d8
--- /dev/null
+++ b/params_shard_32.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:ece512f5b52d0797652515c1851cf597aea926f9a945a2f7dbe875a9da73a5f8
+size 32689152
diff --git a/params_shard_33.bin b/params_shard_33.bin
new file mode 100644
index 0000000000000000000000000000000000000000..625eab514b6234e530b1a070d5e72fe4f7c31880
--- /dev/null
+++ b/params_shard_33.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:62ff8ec0211985b30fc9ba14927e868261ffc0d6d2e95b8a4e0c8b7cb2a1aa69
+size 21076992
diff --git a/params_shard_34.bin b/params_shard_34.bin
new file mode 100644
index 0000000000000000000000000000000000000000..3ad41e6991489c29e1b6182771d7cc021e14a4fb
--- /dev/null
+++ b/params_shard_34.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:c2464e0b3e87b4ca8552ca252dec6884a7ffce2f8cb4be895a8df66c3f5e5d9b
+size 31357440
diff --git a/params_shard_35.bin b/params_shard_35.bin
new file mode 100644
index 0000000000000000000000000000000000000000..14eaf2e06c69fac57ab550af65af8912826a0ec4
--- /dev/null
+++ b/params_shard_35.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:2bd33fc7541ee9c1308291c3c9fa2bbd98801658fcb9b27a085aff7373d554cd
+size 33352704
diff --git a/params_shard_36.bin b/params_shard_36.bin
new file mode 100644
index 0000000000000000000000000000000000000000..a5d8be138d23d83d6bd61ffe2b4a3abc84709883
--- /dev/null
+++ b/params_shard_36.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:081035d55f4817a9eaf402f0b493275c5fa78ea0f098786cf77c6fd7d9865a8b
+size 26385408
diff --git a/params_shard_37.bin b/params_shard_37.bin
new file mode 100644
index 0000000000000000000000000000000000000000..4ef4871fe583615d917d22ee56e9410341c7e774
--- /dev/null
+++ b/params_shard_37.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:73fd48356cfcbb5843d46ee30644fc1e4060648ca20eb5537670f11448b1f1ea
+size 33513984
diff --git a/params_shard_38.bin b/params_shard_38.bin
new file mode 100644
index 0000000000000000000000000000000000000000..5cf857b26fd23766382f9e270159093f25fdd77e
--- /dev/null
+++ b/params_shard_38.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:1a5500b6d20ca190ad0cf20626b67eaa14fa6c3e00933a6bf7b72a50bdfddccf
+size 27712512
diff --git a/params_shard_39.bin b/params_shard_39.bin
new file mode 100644
index 0000000000000000000000000000000000000000..bb8a9a3510231ead31fd6abf1f07047d82d739d3
--- /dev/null
+++ b/params_shard_39.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:721338f950c696e7fe3d360b4a89eb0bfd09c09c89a83a71ee971911d1064f19
+size 32689152
diff --git a/params_shard_4.bin b/params_shard_4.bin
new file mode 100644
index 0000000000000000000000000000000000000000..47817e16d9e80bcc0fe27c7a20f1f0b19a2b6071
--- /dev/null
+++ b/params_shard_4.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:1dfc3ce17c6e353f73f5487edb326e3665b26a8bd6e917327b51e2d3a7ae3df4
+size 32689152
diff --git a/params_shard_40.bin b/params_shard_40.bin
new file mode 100644
index 0000000000000000000000000000000000000000..ba0a8b64c732b58e432abbed2fb67ee728b4ea25
--- /dev/null
+++ b/params_shard_40.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:3260c45b0b37b9aa39af462320f8f18aed815f8a2037b395fd881340c6a967ee
+size 21076992
diff --git a/params_shard_41.bin b/params_shard_41.bin
new file mode 100644
index 0000000000000000000000000000000000000000..2f9be85970c766fd325363d8d09fe1c8de565965
--- /dev/null
+++ b/params_shard_41.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:01bdc3baa26a86cdd2d22654c85668cd05b6152a8a3cbbf93fd26378a4f7bcdc
+size 31357440
diff --git a/params_shard_42.bin b/params_shard_42.bin
new file mode 100644
index 0000000000000000000000000000000000000000..f9eb23845be75608067e5719a0e0cc2fcf413027
--- /dev/null
+++ b/params_shard_42.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:1eceffd11721fde88023bd37bcbf6249ef989b84d9e1ae26779aff5350c36305
+size 33352704
diff --git a/params_shard_43.bin b/params_shard_43.bin
new file mode 100644
index 0000000000000000000000000000000000000000..64338944c4295b08fdc1b6020e036ec3fea13165
--- /dev/null
+++ b/params_shard_43.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:319831a1e2c0062cb2895e22e69587506ca3be7f812df5d668976f63c371a613
+size 26385408
diff --git a/params_shard_44.bin b/params_shard_44.bin
new file mode 100644
index 0000000000000000000000000000000000000000..dc6b06623e4d6b4e6ad2258c3e5e5441b1946d4a
--- /dev/null
+++ b/params_shard_44.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:4db92a4a22630baf2f9fd68b32f1eae4a7a1655387481c606d54128334174d53
+size 33513984
diff --git a/params_shard_45.bin b/params_shard_45.bin
new file mode 100644
index 0000000000000000000000000000000000000000..88b3a860775bbe3ec8e5ae5853d0fa5d9419cf4e
--- /dev/null
+++ b/params_shard_45.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:3a75934db5029e22a715e90713dec3b1f18bf05556405b9241d4439b5e218123
+size 27712512
diff --git a/params_shard_46.bin b/params_shard_46.bin
new file mode 100644
index 0000000000000000000000000000000000000000..f2f89688a2a12abc4f88c88b19f593dc0dafcac2
--- /dev/null
+++ b/params_shard_46.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:cd78ae90605b6acbb00917d66ef46031b65ce361487517e51148a7ffc9c03526
+size 32689152
diff --git a/params_shard_47.bin b/params_shard_47.bin
new file mode 100644
index 0000000000000000000000000000000000000000..7bd7c8a6808467dd717b6a31d9964f25c92415b9
--- /dev/null
+++ b/params_shard_47.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:7a4bbb823c76bc6955c629453012e61f73e17c62dc7b1fd57231673a6230e2bf
+size 21076992
diff --git a/params_shard_48.bin b/params_shard_48.bin
new file mode 100644
index 0000000000000000000000000000000000000000..704b1efc8ad00c2f0713a73b01916157da7cbc16
--- /dev/null
+++ b/params_shard_48.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:81243f223d0c71d8d2725df1139e9951cdd0ae2c112b1db9a00966f7c08fdd6f
+size 22394880
diff --git a/params_shard_5.bin b/params_shard_5.bin
new file mode 100644
index 0000000000000000000000000000000000000000..dc6065f20a2336b0092bb58259675866a8ee2175
--- /dev/null
+++ b/params_shard_5.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:3e1107d79ae4c1bde34b19c60f3e790593c0439ff7111287b1cad381a0a1be1c
+size 21076992
diff --git a/params_shard_6.bin b/params_shard_6.bin
new file mode 100644
index 0000000000000000000000000000000000000000..6bc8bcb47ae335a7a5529b7e8374acbd5c3f92c1
--- /dev/null
+++ b/params_shard_6.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:1e3d2c795741f5285bd680742ce9df3be1d0fb63a3954ecfda3c29849ea8d0f0
+size 31357440
diff --git a/params_shard_7.bin b/params_shard_7.bin
new file mode 100644
index 0000000000000000000000000000000000000000..14d4ad33a3f662d1a796e88e8058d630e1d7fa5d
--- /dev/null
+++ b/params_shard_7.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:727685769de27086361da327db4d882f1bf37adaf402e56443f2e21524a3229d
+size 33352704
diff --git a/params_shard_8.bin b/params_shard_8.bin
new file mode 100644
index 0000000000000000000000000000000000000000..8c551d9122906a99d9874da2bcda065896082588
--- /dev/null
+++ b/params_shard_8.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:c9ac8b52f365855b44ad33de15cd6783c87cb7896c01704641f0de3a087b3832
+size 26385408
diff --git a/params_shard_9.bin b/params_shard_9.bin
new file mode 100644
index 0000000000000000000000000000000000000000..9267e466bbcdbb1bceccab8b9ce38832ed7763a0
--- /dev/null
+++ b/params_shard_9.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:6d15b44345be87064a78cd05ef8f62590232875e79a8a8f8667bd5bb5254761e
+size 33513984
diff --git a/tokenizer.json b/tokenizer.json
new file mode 100644
index 0000000000000000000000000000000000000000..21b7d2808648961dd8cf490cd6cca0567142248d
--- /dev/null
+++ b/tokenizer.json
@@ -0,0 +1,294498 @@
+{
+ "version": "1.0",
+ "truncation": null,
+ "padding": null,
+ "added_tokens": [
+ {
+ "id": 0,
+ "content": "",
+ "single_word": false,
+ "lstrip": false,
+ "rstrip": false,
+ "normalized": false,
+ "special": true
+ },
+ {
+ "id": 1,
+ "content": "",
+ "single_word": false,
+ "lstrip": false,
+ "rstrip": false,
+ "normalized": false,
+ "special": true
+ },
+ {
+ "id": 2,
+ "content": "",
+ "single_word": false,
+ "lstrip": false,
+ "rstrip": false,
+ "normalized": false,
+ "special": true
+ },
+ {
+ "id": 122753,
+ "content": "<|im_end|>",
+ "single_word": false,
+ "lstrip": false,
+ "rstrip": false,
+ "normalized": false,
+ "special": true
+ },
+ {
+ "id": 122754,
+ "content": "▁",
+ "single_word": false,
+ "lstrip": false,
+ "rstrip": false,
+ "normalized": false,
+ "special": true
+ },
+ {
+ "id": 122755,
+ "content": "▁",
+ "single_word": false,
+ "lstrip": false,
+ "rstrip": false,
+ "normalized": false,
+ "special": true
+ },
+ {
+ "id": 122756,
+ "content": "<|tool_call|>",
+ "single_word": false,
+ "lstrip": false,
+ "rstrip": false,
+ "normalized": false,
+ "special": true
+ },
+ {
+ "id": 122757,
+ "content": "<|im_start|>",
+ "single_word": false,
+ "lstrip": false,
+ "rstrip": false,
+ "normalized": false,
+ "special": true
+ },
+ {
+ "id": 122758,
+ "content": "▁",
+ "single_word": false,
+ "lstrip": false,
+ "rstrip": false,
+ "normalized": false,
+ "special": true
+ },
+ {
+ "id": 122759,
+ "content": "▁",
+ "single_word": false,
+ "lstrip": false,
+ "rstrip": false,
+ "normalized": false,
+ "special": true
+ }
+ ],
+ "normalizer": {
+ "type": "Sequence",
+ "normalizers": [
+ {
+ "type": "Prepend",
+ "prepend": "▁"
+ },
+ {
+ "type": "Replace",
+ "pattern": {
+ "String": " "
+ },
+ "content": "▁"
+ }
+ ]
+ },
+ "pre_tokenizer": null,
+ "post_processor": {
+ "type": "TemplateProcessing",
+ "single": [
+ {
+ "SpecialToken": {
+ "id": "",
+ "type_id": 0
+ }
+ },
+ {
+ "Sequence": {
+ "id": "A",
+ "type_id": 0
+ }
+ }
+ ],
+ "pair": [
+ {
+ "SpecialToken": {
+ "id": "",
+ "type_id": 0
+ }
+ },
+ {
+ "Sequence": {
+ "id": "A",
+ "type_id": 0
+ }
+ },
+ {
+ "SpecialToken": {
+ "id": "",
+ "type_id": 1
+ }
+ },
+ {
+ "Sequence": {
+ "id": "B",
+ "type_id": 1
+ }
+ }
+ ],
+ "special_tokens": {
+ "": {
+ "id": "",
+ "ids": [
+ 1
+ ],
+ "tokens": [
+ ""
+ ]
+ }
+ }
+ },
+ "decoder": {
+ "type": "Sequence",
+ "decoders": [
+ {
+ "type": "Replace",
+ "pattern": {
+ "String": "▁"
+ },
+ "content": " "
+ },
+ {
+ "type": "ByteFallback"
+ },
+ {
+ "type": "Fuse"
+ },
+ {
+ "type": "Strip",
+ "content": " ",
+ "start": 1,
+ "stop": 0
+ }
+ ]
+ },
+ "model": {
+ "type": "BPE",
+ "dropout": null,
+ "unk_token": "",
+ "continuing_subword_prefix": null,
+ "end_of_word_suffix": null,
+ "fuse_unk": true,
+ "byte_fallback": true,
+ "vocab": {
+ "": 0,
+ "": 1,
+ "": 2,
+ "": 3,
+ "": 4,
+ "\n": 5,
+ "\t": 6,
+ "": 7,
+ "": 8,
+ "": 9,
+ "": 10,
+ "": 11,
+ "
": 12,
+ "": 13,
+ " | | ": 14,
+ "": 15,
+ "": 16,
+ "": 17,
+ "": 18,
+ "": 21,
+ "": 22,
+ "
": 23,
+ "": 24,
+ "": 25,
+ "": 26,
+ "": 27,
+ "": 28,
+ "": 29,
+ "": 30,
+ "": 31,
+ "": 32,
+ "
": 33,
+ "
": 34,
+ "
": 35,
+ "": 36,
+ "": 37,
+ "": 38,
+ "
": 39,
+ "": 40,
+ "": 41,
+ "
": 42,
+ "": 43,
+ "
": 44,
+ "
": 45,
+ "": 46,
+ "": 47,
+ "
": 48,
+ "": 49,
+ "": 50,
+ "": 51,
+ "0": 52,
+ "1": 53,
+ "2": 54,
+ "3": 55,
+ "4": 56,
+ "5": 57,
+ "6": 58,
+ "7": 59,
+ "8": 60,
+ "9": 61,
+ "+": 62,
+ "-": 63,
+ "=": 64,
+ ",": 65,
+ "。": 66,
+ "!": 67,
+ "?": 68,
+ "、": 69,
+ ":": 70,
+ "¥": 71,
+ ".": 72,
+ "!": 73,
+ "?": 74,
+ "...": 75,
+ "。。。": 76,
+ "。。。。。。": 77,
+ "《": 78,
+ "》": 79,
+ "【": 80,
+ "】": 81,
+ "『": 82,
+ "』": 83,
+ "```": 84,
+ "": 86,
+ "---": 87,
+ "": 88,
+ ";": 89,
+ ".": 90,
+ "=": 91,
+ "<": 92,
+ ">": 93,
+ "-": 94,
+ "+": 95,
+ "%": 96,
+ "‼": 97,
+ "㊣": 98,
+ "/": 99,
+ "|": 100,
+ "": 101,
+ "": 102,
+ "": 103,
+ "": 104,
+ "": 105,
+ "": 106,
+ "": 107,
+ "": 108,
+ "": 109,
+ "": 110,
+ "": 111,
+ "": 112,
+ "": 113,
+ "": 114,
+ "": 115,
+ "": 116,
+ "": 117,
+ "": 118,
+ "": 119,
+ "": 120,
+ "": 121,
+ "": 122,
+ "": 123,
+ "": 124,
+ "": 125,
+ "": 126,
+ "": 127,
+ "": 128,
+ "": 129,
+ "": 130,
+ "": 131,
+ "": 132,
+ "": 133,
+ "": 134,
+ "": 135,
+ "": 136,
+ "": 137,
+ "": 138,
+ "": 139,
+ "": 140,
+ "": 141,
+ "": 142,
+ "": 143,
+ "": 144,
+ "": 145,
+ "": 146,
+ "": 147,
+ "": 148,
+ "": 149,
+ "": 150,
+ "": 151,
+ "": 152,
+ "": 153,
+ "": 154,
+ "": 155,
+ "": 156,
+ "": 157,
+ "": 158,
+ "": 159,
+ "": 160,
+ "": 161,
+ "": 162,
+ "": 163,
+ "": 164,
+ "": 165,
+ "": 166,
+ "": 167,
+ "": 168,
+ "": 169,
+ "": 170,
+ "": 171,
+ "": 172,
+ "": 173,
+ "": 174,
+ "": 175,
+ "": 176,
+ "": 177,
+ "": 178,
+ "": 179,
+ "": 180,
+ "": 181,
+ "": 182,
+ "": 183,
+ "": 184,
+ "": 185,
+ "": 186,
+ "": 187,
+ "": 188,
+ "": 189,
+ "": 190,
+ "": 191,
+ "": 192,
+ "": 193,
+ "": 194,
+ "": 195,
+ "": 196,
+ "": 197,
+ "": 198,
+ "": 199,
+ "": 200,
+ "": 201,
+ "": 202,
+ "": 203,
+ "": 204,
+ "": 205,
+ "": 206,
+ "": 207,
+ "": 208,
+ "": 209,
+ "": 210,
+ "": 211,
+ "": 212,
+ "": 213,
+ "": 214,
+ "": 215,
+ "": 216,
+ "": 217,
+ "": 218,
+ "": 219,
+ "": 220,
+ "": 221,
+ "": 222,
+ "": 223,
+ "": 224,
+ "": 225,
+ "": 226,
+ "": 227,
+ "": 228,
+ "": 229,
+ "": 230,
+ "": 231,
+ "": 232,
+ "": 233,
+ "": 234,
+ "": 235,
+ "": 236,
+ "": 237,
+ "": 238,
+ "": 239,
+ "": 240,
+ "": 241,
+ "": 242,
+ "": 243,
+ "": 244,
+ "": 245,
+ "": 246,
+ "": 247,
+ "": 248,
+ "": 249,
+ "": 250,
+ "": 251,
+ "": 252,
+ "": 253,
+ "": 254,
+ "": 255,
+ "": 256,
+ "": 257,
+ "": 258,
+ "": 259,
+ "": 260,
+ "": 261,
+ "": 262,
+ "": 263,
+ "": 264,
+ "": 265,
+ "": 266,
+ "": 267,
+ "": 268,
+ "": 269,
+ "": 270,
+ "": 271,
+ "": 272,
+ "": 273,
+ "": 274,
+ "": 275,
+ "": 276,
+ "": 277,
+ "": 278,
+ "": 279,
+ "": 280,
+ "": 281,
+ "": 282,
+ "": 283,
+ "": 284,
+ "": 285,
+ "": 286,
+ "": 287,
+ "": 288,
+ "": 289,
+ "": 290,
+ "": 291,
+ "": 292,
+ "": 293,
+ "": 294,
+ "": 295,
+ "": 296,
+ "": 297,
+ "": 298,
+ "": 299,
+ "": 300,
+ "": 301,
+ "": 302,
+ "": 303,
+ "": 304,
+ "": 305,
+ "": 306,
+ "": 307,
+ "": 308,
+ "": 309,
+ "": 310,
+ "": 311,
+ "": 312,
+ "": 313,
+ "": 314,
+ "": 315,
+ "": 316,
+ "": 317,
+ "": 318,
+ "": 319,
+ "": 320,
+ "": 321,
+ "": 322,
+ "": 323,
+ "": 324,
+ "": 325,
+ "": 326,
+ "": 327,
+ "": 328,
+ "": 329,
+ "": 330,
+ "": 331,
+ "": 332,
+ "": 333,
+ "": 334,
+ "": 335,
+ "": 336,
+ "": 337,
+ "": 338,
+ "": 339,
+ "": 340,
+ "": 341,
+ "": 342,
+ "": 343,
+ "": 344,
+ "": 345,
+ "": 346,
+ "": 347,
+ "": 348,
+ "": 349,
+ "": 350,
+ "": 351,
+ "": 352,
+ "": 353,
+ "