riczhou's picture
Initial commit
9caeb03 verified
raw
history blame contribute delete
No virus
103 kB
{
"metadata": {
"ParamSize": 195,
"ParamBytes": 7642159104.0,
"BitsPerParam": 16.0
},
"records": [
{
"dataPath": "params_shard_0.bin",
"format": "raw-shard",
"nbytes": 197001216,
"records": [
{
"name": "lm_head.weight",
"shape": [
32064,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 197001216,
"byteOffset": 0
}
],
"md5sum": "8c2d85352caa7c9a98d74a9ee3cf6a94"
},
{
"dataPath": "params_shard_1.bin",
"format": "raw-shard",
"nbytes": 50331648,
"records": [
{
"name": "transformer.h.21.mlp.down_proj.weight",
"shape": [
3072,
8192
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 50331648,
"byteOffset": 0
}
],
"md5sum": "fbdfcef1bc44437d1e2dd389140c2334"
},
{
"dataPath": "params_shard_2.bin",
"format": "raw-shard",
"nbytes": 100663296,
"records": [
{
"name": "transformer.h.21.mlp.gate_up_proj.weight",
"shape": [
16384,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 100663296,
"byteOffset": 0
}
],
"md5sum": "b14efca889e2938bd0dd70bf01f3e03a"
},
{
"dataPath": "params_shard_3.bin",
"format": "raw-shard",
"nbytes": 56623104,
"records": [
{
"name": "transformer.h.21.mixer.qkv_proj.weight",
"shape": [
9216,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 56623104,
"byteOffset": 0
}
],
"md5sum": "c1a49fa4938c39315cb4cdcc5c2d7d49"
},
{
"dataPath": "params_shard_4.bin",
"format": "raw-shard",
"nbytes": 50331648,
"records": [
{
"name": "transformer.h.22.mlp.down_proj.weight",
"shape": [
3072,
8192
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 50331648,
"byteOffset": 0
}
],
"md5sum": "a2239c27ad68b6f533a7102d82527878"
},
{
"dataPath": "params_shard_5.bin",
"format": "raw-shard",
"nbytes": 100663296,
"records": [
{
"name": "transformer.h.22.mlp.gate_up_proj.weight",
"shape": [
16384,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 100663296,
"byteOffset": 0
}
],
"md5sum": "c56bee40a05fbd4322eade65122c07f3"
},
{
"dataPath": "params_shard_6.bin",
"format": "raw-shard",
"nbytes": 56623104,
"records": [
{
"name": "transformer.h.22.mixer.qkv_proj.weight",
"shape": [
9216,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 56623104,
"byteOffset": 0
}
],
"md5sum": "bfa63c8cb8115ffdfe3d0e9141eee648"
},
{
"dataPath": "params_shard_7.bin",
"format": "raw-shard",
"nbytes": 50331648,
"records": [
{
"name": "transformer.h.23.mlp.down_proj.weight",
"shape": [
3072,
8192
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 50331648,
"byteOffset": 0
}
],
"md5sum": "83521bd8fbce11184521d6ab41a29cea"
},
{
"dataPath": "params_shard_8.bin",
"format": "raw-shard",
"nbytes": 100663296,
"records": [
{
"name": "transformer.h.23.mlp.gate_up_proj.weight",
"shape": [
16384,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 100663296,
"byteOffset": 0
}
],
"md5sum": "2ea3121970667f0f7645c5e728237748"
},
{
"dataPath": "params_shard_9.bin",
"format": "raw-shard",
"nbytes": 18874368,
"records": [
{
"name": "transformer.h.23.mixer.out_proj.weight",
"shape": [
3072,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 18874368,
"byteOffset": 0
}
],
"md5sum": "1934b976730e3bd2abf6d9a8b72529d8"
},
{
"dataPath": "params_shard_10.bin",
"format": "raw-shard",
"nbytes": 56623104,
"records": [
{
"name": "transformer.h.23.mixer.qkv_proj.weight",
"shape": [
9216,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 56623104,
"byteOffset": 0
}
],
"md5sum": "1cf17bdee74a90757820d2d22aabecf7"
},
{
"dataPath": "params_shard_11.bin",
"format": "raw-shard",
"nbytes": 50331648,
"records": [
{
"name": "transformer.h.24.mlp.down_proj.weight",
"shape": [
3072,
8192
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 50331648,
"byteOffset": 0
}
],
"md5sum": "03112ed394c45b3446d495aeb5cbdff0"
},
{
"dataPath": "params_shard_12.bin",
"format": "raw-shard",
"nbytes": 100663296,
"records": [
{
"name": "transformer.h.24.mlp.gate_up_proj.weight",
"shape": [
16384,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 100663296,
"byteOffset": 0
}
],
"md5sum": "49188904aa2c30790064abfd8097672c"
},
{
"dataPath": "params_shard_13.bin",
"format": "raw-shard",
"nbytes": 18874368,
"records": [
{
"name": "transformer.h.24.mixer.out_proj.weight",
"shape": [
3072,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 18874368,
"byteOffset": 0
}
],
"md5sum": "f42a37069ae388cfdaf2d09833d82244"
},
{
"dataPath": "params_shard_14.bin",
"format": "raw-shard",
"nbytes": 56623104,
"records": [
{
"name": "transformer.h.24.mixer.qkv_proj.weight",
"shape": [
9216,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 56623104,
"byteOffset": 0
}
],
"md5sum": "30c14cf99aff89957ec7da295c4ab632"
},
{
"dataPath": "params_shard_15.bin",
"format": "raw-shard",
"nbytes": 50331648,
"records": [
{
"name": "transformer.h.25.mlp.down_proj.weight",
"shape": [
3072,
8192
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 50331648,
"byteOffset": 0
}
],
"md5sum": "b57a23df765fd830804df014adca71e7"
},
{
"dataPath": "params_shard_16.bin",
"format": "raw-shard",
"nbytes": 100663296,
"records": [
{
"name": "transformer.h.25.mlp.gate_up_proj.weight",
"shape": [
16384,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 100663296,
"byteOffset": 0
}
],
"md5sum": "6a8895d6b1ccd90c8e91f75da70bdd3e"
},
{
"dataPath": "params_shard_17.bin",
"format": "raw-shard",
"nbytes": 18874368,
"records": [
{
"name": "transformer.h.25.mixer.out_proj.weight",
"shape": [
3072,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 18874368,
"byteOffset": 0
}
],
"md5sum": "66114f0757c09a36c7ffd636719cc7b3"
},
{
"dataPath": "params_shard_18.bin",
"format": "raw-shard",
"nbytes": 56623104,
"records": [
{
"name": "transformer.h.25.mixer.qkv_proj.weight",
"shape": [
9216,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 56623104,
"byteOffset": 0
}
],
"md5sum": "f9ba7dd1a9ffe04b85583439d91d551a"
},
{
"dataPath": "params_shard_19.bin",
"format": "raw-shard",
"nbytes": 50331648,
"records": [
{
"name": "transformer.h.26.mlp.down_proj.weight",
"shape": [
3072,
8192
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 50331648,
"byteOffset": 0
}
],
"md5sum": "71e79d1578fd6ee4e4524679fb1276ff"
},
{
"dataPath": "params_shard_20.bin",
"format": "raw-shard",
"nbytes": 100663296,
"records": [
{
"name": "transformer.h.26.mlp.gate_up_proj.weight",
"shape": [
16384,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 100663296,
"byteOffset": 0
}
],
"md5sum": "5dc9d0557437fcf6e561ee3951baadae"
},
{
"dataPath": "params_shard_21.bin",
"format": "raw-shard",
"nbytes": 18874368,
"records": [
{
"name": "transformer.h.26.mixer.out_proj.weight",
"shape": [
3072,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 18874368,
"byteOffset": 0
}
],
"md5sum": "297e4f56ec4f407b4e9546c012b09775"
},
{
"dataPath": "params_shard_22.bin",
"format": "raw-shard",
"nbytes": 56623104,
"records": [
{
"name": "transformer.h.26.mixer.qkv_proj.weight",
"shape": [
9216,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 56623104,
"byteOffset": 0
}
],
"md5sum": "2bf6b964d4c2688cc13183649276342a"
},
{
"dataPath": "params_shard_23.bin",
"format": "raw-shard",
"nbytes": 50331648,
"records": [
{
"name": "transformer.h.27.mlp.down_proj.weight",
"shape": [
3072,
8192
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 50331648,
"byteOffset": 0
}
],
"md5sum": "17b63513014e546d7b672228b045718c"
},
{
"dataPath": "params_shard_24.bin",
"format": "raw-shard",
"nbytes": 100663296,
"records": [
{
"name": "transformer.h.27.mlp.gate_up_proj.weight",
"shape": [
16384,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 100663296,
"byteOffset": 0
}
],
"md5sum": "487b046eee7038d7046ba4740a5ca521"
},
{
"dataPath": "params_shard_25.bin",
"format": "raw-shard",
"nbytes": 18874368,
"records": [
{
"name": "transformer.h.27.mixer.out_proj.weight",
"shape": [
3072,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 18874368,
"byteOffset": 0
}
],
"md5sum": "30b04b01a54e57beba23e7fe29957c87"
},
{
"dataPath": "params_shard_26.bin",
"format": "raw-shard",
"nbytes": 56623104,
"records": [
{
"name": "transformer.h.27.mixer.qkv_proj.weight",
"shape": [
9216,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 56623104,
"byteOffset": 0
}
],
"md5sum": "00ed36d8f6e889b0567ff936972055bb"
},
{
"dataPath": "params_shard_27.bin",
"format": "raw-shard",
"nbytes": 50331648,
"records": [
{
"name": "transformer.h.28.mlp.down_proj.weight",
"shape": [
3072,
8192
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 50331648,
"byteOffset": 0
}
],
"md5sum": "d790288dcc53da10b9ee96e9a80571be"
},
{
"dataPath": "params_shard_28.bin",
"format": "raw-shard",
"nbytes": 100663296,
"records": [
{
"name": "transformer.h.28.mlp.gate_up_proj.weight",
"shape": [
16384,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 100663296,
"byteOffset": 0
}
],
"md5sum": "8625bcda8d93dc0529984087f07cccb8"
},
{
"dataPath": "params_shard_29.bin",
"format": "raw-shard",
"nbytes": 18874368,
"records": [
{
"name": "transformer.h.28.mixer.out_proj.weight",
"shape": [
3072,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 18874368,
"byteOffset": 0
}
],
"md5sum": "ab0e01373b835f596283764432bdcea9"
},
{
"dataPath": "params_shard_30.bin",
"format": "raw-shard",
"nbytes": 56623104,
"records": [
{
"name": "transformer.h.28.mixer.qkv_proj.weight",
"shape": [
9216,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 56623104,
"byteOffset": 0
}
],
"md5sum": "3b373274561aeb974d849cc01c2850b0"
},
{
"dataPath": "params_shard_31.bin",
"format": "raw-shard",
"nbytes": 50331648,
"records": [
{
"name": "transformer.h.29.mlp.down_proj.weight",
"shape": [
3072,
8192
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 50331648,
"byteOffset": 0
}
],
"md5sum": "ac9774e76732119d81106d74f7f6ee52"
},
{
"dataPath": "params_shard_32.bin",
"format": "raw-shard",
"nbytes": 100663296,
"records": [
{
"name": "transformer.h.29.mlp.gate_up_proj.weight",
"shape": [
16384,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 100663296,
"byteOffset": 0
}
],
"md5sum": "47013b7213be9e785d22bdaa10590ca4"
},
{
"dataPath": "params_shard_33.bin",
"format": "raw-shard",
"nbytes": 18874368,
"records": [
{
"name": "transformer.h.29.mixer.out_proj.weight",
"shape": [
3072,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 18874368,
"byteOffset": 0
}
],
"md5sum": "aaf205e3e3798acd2b6f1bc92d94d6de"
},
{
"dataPath": "params_shard_34.bin",
"format": "raw-shard",
"nbytes": 56623104,
"records": [
{
"name": "transformer.h.29.mixer.qkv_proj.weight",
"shape": [
9216,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 56623104,
"byteOffset": 0
}
],
"md5sum": "d3b2899e4379328af50315fe46f9e144"
},
{
"dataPath": "params_shard_35.bin",
"format": "raw-shard",
"nbytes": 50331648,
"records": [
{
"name": "transformer.h.30.mlp.down_proj.weight",
"shape": [
3072,
8192
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 50331648,
"byteOffset": 0
}
],
"md5sum": "d5604e270a26008514d6f07651520d7b"
},
{
"dataPath": "params_shard_36.bin",
"format": "raw-shard",
"nbytes": 100663296,
"records": [
{
"name": "transformer.h.30.mlp.gate_up_proj.weight",
"shape": [
16384,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 100663296,
"byteOffset": 0
}
],
"md5sum": "c21012a11160161b3f16436d5e8fbcea"
},
{
"dataPath": "params_shard_37.bin",
"format": "raw-shard",
"nbytes": 18874368,
"records": [
{
"name": "transformer.h.30.mixer.out_proj.weight",
"shape": [
3072,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 18874368,
"byteOffset": 0
}
],
"md5sum": "99ac28b15ff15c0e7dcfada05307ac89"
},
{
"dataPath": "params_shard_38.bin",
"format": "raw-shard",
"nbytes": 56623104,
"records": [
{
"name": "transformer.h.30.mixer.qkv_proj.weight",
"shape": [
9216,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 56623104,
"byteOffset": 0
}
],
"md5sum": "225394d73feea900325e934fd3a1c0b4"
},
{
"dataPath": "params_shard_39.bin",
"format": "raw-shard",
"nbytes": 50331648,
"records": [
{
"name": "transformer.h.31.mlp.down_proj.weight",
"shape": [
3072,
8192
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 50331648,
"byteOffset": 0
}
],
"md5sum": "9d287bb8e3cf1f29df30aefc9c027fd2"
},
{
"dataPath": "params_shard_40.bin",
"format": "raw-shard",
"nbytes": 100663296,
"records": [
{
"name": "transformer.h.31.mlp.gate_up_proj.weight",
"shape": [
16384,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 100663296,
"byteOffset": 0
}
],
"md5sum": "263c716606e3a2afc30802ef6483db82"
},
{
"dataPath": "params_shard_41.bin",
"format": "raw-shard",
"nbytes": 18874368,
"records": [
{
"name": "transformer.h.31.mixer.out_proj.weight",
"shape": [
3072,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 18874368,
"byteOffset": 0
}
],
"md5sum": "f0ecb725e774085ffd140fd5df0089e6"
},
{
"dataPath": "params_shard_42.bin",
"format": "raw-shard",
"nbytes": 56623104,
"records": [
{
"name": "transformer.h.31.mixer.qkv_proj.weight",
"shape": [
9216,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 56623104,
"byteOffset": 0
}
],
"md5sum": "191fa72b252485f0a9db714e8a2fa3e5"
},
{
"dataPath": "params_shard_43.bin",
"format": "raw-shard",
"nbytes": 197001216,
"records": [
{
"name": "transformer.embd.weight",
"shape": [
32064,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 197001216,
"byteOffset": 0
}
],
"md5sum": "6f9e82ee16bf9e48202971f1bc7faddf"
},
{
"dataPath": "params_shard_44.bin",
"format": "raw-shard",
"nbytes": 50331648,
"records": [
{
"name": "transformer.h.0.mlp.down_proj.weight",
"shape": [
3072,
8192
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 50331648,
"byteOffset": 0
}
],
"md5sum": "6d31c842666fb35f49c2826e9469172e"
},
{
"dataPath": "params_shard_45.bin",
"format": "raw-shard",
"nbytes": 100663296,
"records": [
{
"name": "transformer.h.0.mlp.gate_up_proj.weight",
"shape": [
16384,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 100663296,
"byteOffset": 0
}
],
"md5sum": "caf606342f712b439b4540b30b4d4dd8"
},
{
"dataPath": "params_shard_46.bin",
"format": "raw-shard",
"nbytes": 18874368,
"records": [
{
"name": "transformer.h.0.mixer.out_proj.weight",
"shape": [
3072,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 18874368,
"byteOffset": 0
}
],
"md5sum": "3cbc63d65af31aeeb5f0ff6d21b05d91"
},
{
"dataPath": "params_shard_47.bin",
"format": "raw-shard",
"nbytes": 56623104,
"records": [
{
"name": "transformer.h.0.mixer.qkv_proj.weight",
"shape": [
9216,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 56623104,
"byteOffset": 0
}
],
"md5sum": "8011494667ca20936cfbd3fe057f030f"
},
{
"dataPath": "params_shard_48.bin",
"format": "raw-shard",
"nbytes": 50331648,
"records": [
{
"name": "transformer.h.1.mlp.down_proj.weight",
"shape": [
3072,
8192
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 50331648,
"byteOffset": 0
}
],
"md5sum": "d58dba16b580f460e040860a3be560d7"
},
{
"dataPath": "params_shard_49.bin",
"format": "raw-shard",
"nbytes": 100663296,
"records": [
{
"name": "transformer.h.1.mlp.gate_up_proj.weight",
"shape": [
16384,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 100663296,
"byteOffset": 0
}
],
"md5sum": "e3851f228e9eeea429ffdb5bfcf65712"
},
{
"dataPath": "params_shard_50.bin",
"format": "raw-shard",
"nbytes": 18874368,
"records": [
{
"name": "transformer.h.1.mixer.out_proj.weight",
"shape": [
3072,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 18874368,
"byteOffset": 0
}
],
"md5sum": "36a299bba3d83d4b09fe38cea015c920"
},
{
"dataPath": "params_shard_51.bin",
"format": "raw-shard",
"nbytes": 56623104,
"records": [
{
"name": "transformer.h.1.mixer.qkv_proj.weight",
"shape": [
9216,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 56623104,
"byteOffset": 0
}
],
"md5sum": "a2c6ca04c30deb8ca3dd477f9b070990"
},
{
"dataPath": "params_shard_52.bin",
"format": "raw-shard",
"nbytes": 50331648,
"records": [
{
"name": "transformer.h.10.mlp.down_proj.weight",
"shape": [
3072,
8192
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 50331648,
"byteOffset": 0
}
],
"md5sum": "cc7e87530395bcaae2bbaaf14cdccf01"
},
{
"dataPath": "params_shard_53.bin",
"format": "raw-shard",
"nbytes": 100663296,
"records": [
{
"name": "transformer.h.10.mlp.gate_up_proj.weight",
"shape": [
16384,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 100663296,
"byteOffset": 0
}
],
"md5sum": "c39dc653bd8e615ccf216eafefa87cc9"
},
{
"dataPath": "params_shard_54.bin",
"format": "raw-shard",
"nbytes": 18874368,
"records": [
{
"name": "transformer.h.10.mixer.out_proj.weight",
"shape": [
3072,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 18874368,
"byteOffset": 0
}
],
"md5sum": "46da89076ea5cf5e396f3e5b886af15e"
},
{
"dataPath": "params_shard_55.bin",
"format": "raw-shard",
"nbytes": 56623104,
"records": [
{
"name": "transformer.h.10.mixer.qkv_proj.weight",
"shape": [
9216,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 56623104,
"byteOffset": 0
}
],
"md5sum": "dc18d2f273cd86d9bb06d7dcae88c221"
},
{
"dataPath": "params_shard_56.bin",
"format": "raw-shard",
"nbytes": 50331648,
"records": [
{
"name": "transformer.h.11.mlp.down_proj.weight",
"shape": [
3072,
8192
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 50331648,
"byteOffset": 0
}
],
"md5sum": "36bc078291c38abe52684ab9eae3a086"
},
{
"dataPath": "params_shard_57.bin",
"format": "raw-shard",
"nbytes": 100663296,
"records": [
{
"name": "transformer.h.11.mlp.gate_up_proj.weight",
"shape": [
16384,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 100663296,
"byteOffset": 0
}
],
"md5sum": "00b40de1e12839f2fafaf0742585be28"
},
{
"dataPath": "params_shard_58.bin",
"format": "raw-shard",
"nbytes": 18874368,
"records": [
{
"name": "transformer.h.11.mixer.out_proj.weight",
"shape": [
3072,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 18874368,
"byteOffset": 0
}
],
"md5sum": "d8bba7fe7d417dea96d3c791522604d6"
},
{
"dataPath": "params_shard_59.bin",
"format": "raw-shard",
"nbytes": 56623104,
"records": [
{
"name": "transformer.h.11.mixer.qkv_proj.weight",
"shape": [
9216,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 56623104,
"byteOffset": 0
}
],
"md5sum": "652f758c4908ca7d5a76a3d241bb83a0"
},
{
"dataPath": "params_shard_60.bin",
"format": "raw-shard",
"nbytes": 50331648,
"records": [
{
"name": "transformer.h.12.mlp.down_proj.weight",
"shape": [
3072,
8192
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 50331648,
"byteOffset": 0
}
],
"md5sum": "8cf3e143a89cbf61c8620a66b6446075"
},
{
"dataPath": "params_shard_61.bin",
"format": "raw-shard",
"nbytes": 100663296,
"records": [
{
"name": "transformer.h.12.mlp.gate_up_proj.weight",
"shape": [
16384,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 100663296,
"byteOffset": 0
}
],
"md5sum": "9101ba07812357ac26670e4b6a07dadf"
},
{
"dataPath": "params_shard_62.bin",
"format": "raw-shard",
"nbytes": 18874368,
"records": [
{
"name": "transformer.h.12.mixer.out_proj.weight",
"shape": [
3072,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 18874368,
"byteOffset": 0
}
],
"md5sum": "756a1a808db0ed7583500bc0a2b54008"
},
{
"dataPath": "params_shard_63.bin",
"format": "raw-shard",
"nbytes": 56623104,
"records": [
{
"name": "transformer.h.12.mixer.qkv_proj.weight",
"shape": [
9216,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 56623104,
"byteOffset": 0
}
],
"md5sum": "42262b1d88e9d9d49b33fda8283cb753"
},
{
"dataPath": "params_shard_64.bin",
"format": "raw-shard",
"nbytes": 50331648,
"records": [
{
"name": "transformer.h.13.mlp.down_proj.weight",
"shape": [
3072,
8192
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 50331648,
"byteOffset": 0
}
],
"md5sum": "3bd7a6b66767bd6ebabcbeaa81cd0d15"
},
{
"dataPath": "params_shard_65.bin",
"format": "raw-shard",
"nbytes": 100663296,
"records": [
{
"name": "transformer.h.13.mlp.gate_up_proj.weight",
"shape": [
16384,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 100663296,
"byteOffset": 0
}
],
"md5sum": "dae0535b359e750b5168e08edaab2f0a"
},
{
"dataPath": "params_shard_66.bin",
"format": "raw-shard",
"nbytes": 18874368,
"records": [
{
"name": "transformer.h.13.mixer.out_proj.weight",
"shape": [
3072,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 18874368,
"byteOffset": 0
}
],
"md5sum": "71fd9d016830fe9a2e62c82f4727b55f"
},
{
"dataPath": "params_shard_67.bin",
"format": "raw-shard",
"nbytes": 56623104,
"records": [
{
"name": "transformer.h.13.mixer.qkv_proj.weight",
"shape": [
9216,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 56623104,
"byteOffset": 0
}
],
"md5sum": "bdac4468f08f6d325910744e53a73334"
},
{
"dataPath": "params_shard_68.bin",
"format": "raw-shard",
"nbytes": 50331648,
"records": [
{
"name": "transformer.h.14.mlp.down_proj.weight",
"shape": [
3072,
8192
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 50331648,
"byteOffset": 0
}
],
"md5sum": "41331cffbf45ca84270d7f85134ece7c"
},
{
"dataPath": "params_shard_69.bin",
"format": "raw-shard",
"nbytes": 100663296,
"records": [
{
"name": "transformer.h.14.mlp.gate_up_proj.weight",
"shape": [
16384,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 100663296,
"byteOffset": 0
}
],
"md5sum": "22c314f049027022e97d39b3bd9e56fe"
},
{
"dataPath": "params_shard_70.bin",
"format": "raw-shard",
"nbytes": 18874368,
"records": [
{
"name": "transformer.h.14.mixer.out_proj.weight",
"shape": [
3072,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 18874368,
"byteOffset": 0
}
],
"md5sum": "d51c88b65e6d2f35d7bd16e4ec04136b"
},
{
"dataPath": "params_shard_71.bin",
"format": "raw-shard",
"nbytes": 56623104,
"records": [
{
"name": "transformer.h.14.mixer.qkv_proj.weight",
"shape": [
9216,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 56623104,
"byteOffset": 0
}
],
"md5sum": "72e28da4e109231f4a7466c4eec109cb"
},
{
"dataPath": "params_shard_72.bin",
"format": "raw-shard",
"nbytes": 50331648,
"records": [
{
"name": "transformer.h.15.mlp.down_proj.weight",
"shape": [
3072,
8192
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 50331648,
"byteOffset": 0
}
],
"md5sum": "96c63673e9f0a437c157c676b2af5ee9"
},
{
"dataPath": "params_shard_73.bin",
"format": "raw-shard",
"nbytes": 100663296,
"records": [
{
"name": "transformer.h.15.mlp.gate_up_proj.weight",
"shape": [
16384,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 100663296,
"byteOffset": 0
}
],
"md5sum": "5558186b76ecb44c6cf7156980cb8978"
},
{
"dataPath": "params_shard_74.bin",
"format": "raw-shard",
"nbytes": 18874368,
"records": [
{
"name": "transformer.h.15.mixer.out_proj.weight",
"shape": [
3072,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 18874368,
"byteOffset": 0
}
],
"md5sum": "79dc6596ac0f78f6f8ebad933ecb50fd"
},
{
"dataPath": "params_shard_75.bin",
"format": "raw-shard",
"nbytes": 56623104,
"records": [
{
"name": "transformer.h.15.mixer.qkv_proj.weight",
"shape": [
9216,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 56623104,
"byteOffset": 0
}
],
"md5sum": "0b3553b64fa2313e345ee6baa3c258bb"
},
{
"dataPath": "params_shard_76.bin",
"format": "raw-shard",
"nbytes": 50331648,
"records": [
{
"name": "transformer.h.16.mlp.down_proj.weight",
"shape": [
3072,
8192
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 50331648,
"byteOffset": 0
}
],
"md5sum": "e51ca07d5edd7aaae0d992b4e8dcfb21"
},
{
"dataPath": "params_shard_77.bin",
"format": "raw-shard",
"nbytes": 100663296,
"records": [
{
"name": "transformer.h.16.mlp.gate_up_proj.weight",
"shape": [
16384,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 100663296,
"byteOffset": 0
}
],
"md5sum": "7ee0335455612128b3b9171a15700bf2"
},
{
"dataPath": "params_shard_78.bin",
"format": "raw-shard",
"nbytes": 18874368,
"records": [
{
"name": "transformer.h.16.mixer.out_proj.weight",
"shape": [
3072,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 18874368,
"byteOffset": 0
}
],
"md5sum": "18c35ebe336c1a121b23b1b1e4084a25"
},
{
"dataPath": "params_shard_79.bin",
"format": "raw-shard",
"nbytes": 56623104,
"records": [
{
"name": "transformer.h.16.mixer.qkv_proj.weight",
"shape": [
9216,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 56623104,
"byteOffset": 0
}
],
"md5sum": "946796a5a531e3372a9eb3a0c7549430"
},
{
"dataPath": "params_shard_80.bin",
"format": "raw-shard",
"nbytes": 50331648,
"records": [
{
"name": "transformer.h.17.mlp.down_proj.weight",
"shape": [
3072,
8192
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 50331648,
"byteOffset": 0
}
],
"md5sum": "93f0d0f857c235dfeb5ffdc9fab7351a"
},
{
"dataPath": "params_shard_81.bin",
"format": "raw-shard",
"nbytes": 100663296,
"records": [
{
"name": "transformer.h.17.mlp.gate_up_proj.weight",
"shape": [
16384,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 100663296,
"byteOffset": 0
}
],
"md5sum": "c59ecdf2dd9d347fedc9e5708f624d3c"
},
{
"dataPath": "params_shard_82.bin",
"format": "raw-shard",
"nbytes": 18874368,
"records": [
{
"name": "transformer.h.17.mixer.out_proj.weight",
"shape": [
3072,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 18874368,
"byteOffset": 0
}
],
"md5sum": "335d4f2d9379b41144983b72b18a52ea"
},
{
"dataPath": "params_shard_83.bin",
"format": "raw-shard",
"nbytes": 56623104,
"records": [
{
"name": "transformer.h.17.mixer.qkv_proj.weight",
"shape": [
9216,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 56623104,
"byteOffset": 0
}
],
"md5sum": "ad8afa5fe6d4634adadd2c9e8b2f069c"
},
{
"dataPath": "params_shard_84.bin",
"format": "raw-shard",
"nbytes": 50331648,
"records": [
{
"name": "transformer.h.18.mlp.down_proj.weight",
"shape": [
3072,
8192
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 50331648,
"byteOffset": 0
}
],
"md5sum": "d458ff2536af888debbf4a0018dd1767"
},
{
"dataPath": "params_shard_85.bin",
"format": "raw-shard",
"nbytes": 100663296,
"records": [
{
"name": "transformer.h.18.mlp.gate_up_proj.weight",
"shape": [
16384,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 100663296,
"byteOffset": 0
}
],
"md5sum": "e7796e4854cdc538e01d7112ab8a5179"
},
{
"dataPath": "params_shard_86.bin",
"format": "raw-shard",
"nbytes": 18874368,
"records": [
{
"name": "transformer.h.18.mixer.out_proj.weight",
"shape": [
3072,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 18874368,
"byteOffset": 0
}
],
"md5sum": "9df18282bd08212a8ae8fad347721ce2"
},
{
"dataPath": "params_shard_87.bin",
"format": "raw-shard",
"nbytes": 56623104,
"records": [
{
"name": "transformer.h.18.mixer.qkv_proj.weight",
"shape": [
9216,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 56623104,
"byteOffset": 0
}
],
"md5sum": "0763aaa7da6da824f21db13a1bdd3832"
},
{
"dataPath": "params_shard_88.bin",
"format": "raw-shard",
"nbytes": 50331648,
"records": [
{
"name": "transformer.h.19.mlp.down_proj.weight",
"shape": [
3072,
8192
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 50331648,
"byteOffset": 0
}
],
"md5sum": "a49f64fd634619b2ea22b320f9a65720"
},
{
"dataPath": "params_shard_89.bin",
"format": "raw-shard",
"nbytes": 100663296,
"records": [
{
"name": "transformer.h.19.mlp.gate_up_proj.weight",
"shape": [
16384,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 100663296,
"byteOffset": 0
}
],
"md5sum": "9bc101e6065af554c1f77e18793af166"
},
{
"dataPath": "params_shard_90.bin",
"format": "raw-shard",
"nbytes": 18874368,
"records": [
{
"name": "transformer.h.19.mixer.out_proj.weight",
"shape": [
3072,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 18874368,
"byteOffset": 0
}
],
"md5sum": "b2e650b00c4be4a2699e1f4426c1d92f"
},
{
"dataPath": "params_shard_91.bin",
"format": "raw-shard",
"nbytes": 56623104,
"records": [
{
"name": "transformer.h.19.mixer.qkv_proj.weight",
"shape": [
9216,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 56623104,
"byteOffset": 0
}
],
"md5sum": "9454b1638ae97f3d95aa11e6c5368d82"
},
{
"dataPath": "params_shard_92.bin",
"format": "raw-shard",
"nbytes": 50331648,
"records": [
{
"name": "transformer.h.2.mlp.down_proj.weight",
"shape": [
3072,
8192
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 50331648,
"byteOffset": 0
}
],
"md5sum": "36232340846a5ed879134aa5a8336a0e"
},
{
"dataPath": "params_shard_93.bin",
"format": "raw-shard",
"nbytes": 100663296,
"records": [
{
"name": "transformer.h.2.mlp.gate_up_proj.weight",
"shape": [
16384,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 100663296,
"byteOffset": 0
}
],
"md5sum": "2c848444240798a2c8f998ecc0e7444b"
},
{
"dataPath": "params_shard_94.bin",
"format": "raw-shard",
"nbytes": 18874368,
"records": [
{
"name": "transformer.h.2.mixer.out_proj.weight",
"shape": [
3072,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 18874368,
"byteOffset": 0
}
],
"md5sum": "5b0642334e6411b70e7cbfc0a203ead3"
},
{
"dataPath": "params_shard_95.bin",
"format": "raw-shard",
"nbytes": 56623104,
"records": [
{
"name": "transformer.h.2.mixer.qkv_proj.weight",
"shape": [
9216,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 56623104,
"byteOffset": 0
}
],
"md5sum": "6b0e054b7f54523751b04449c6d2f9c7"
},
{
"dataPath": "params_shard_96.bin",
"format": "raw-shard",
"nbytes": 50331648,
"records": [
{
"name": "transformer.h.20.mlp.down_proj.weight",
"shape": [
3072,
8192
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 50331648,
"byteOffset": 0
}
],
"md5sum": "74a5476d9d22d264626d5b9cbcddcb27"
},
{
"dataPath": "params_shard_97.bin",
"format": "raw-shard",
"nbytes": 100663296,
"records": [
{
"name": "transformer.h.20.mlp.gate_up_proj.weight",
"shape": [
16384,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 100663296,
"byteOffset": 0
}
],
"md5sum": "a1eee72b43f1c814da6b1e240aebd8fb"
},
{
"dataPath": "params_shard_98.bin",
"format": "raw-shard",
"nbytes": 18874368,
"records": [
{
"name": "transformer.h.20.mixer.out_proj.weight",
"shape": [
3072,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 18874368,
"byteOffset": 0
}
],
"md5sum": "9274cf6c8ba095bebd1fd97af1a2e642"
},
{
"dataPath": "params_shard_99.bin",
"format": "raw-shard",
"nbytes": 56623104,
"records": [
{
"name": "transformer.h.20.mixer.qkv_proj.weight",
"shape": [
9216,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 56623104,
"byteOffset": 0
}
],
"md5sum": "00b3cc2bc5d7dc8a7063579ecafeca01"
},
{
"dataPath": "params_shard_100.bin",
"format": "raw-shard",
"nbytes": 18874368,
"records": [
{
"name": "transformer.h.21.mixer.out_proj.weight",
"shape": [
3072,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 18874368,
"byteOffset": 0
}
],
"md5sum": "acba950f76ad24093ed04f28c465d198"
},
{
"dataPath": "params_shard_101.bin",
"format": "raw-shard",
"nbytes": 50331648,
"records": [
{
"name": "transformer.h.3.mlp.down_proj.weight",
"shape": [
3072,
8192
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 50331648,
"byteOffset": 0
}
],
"md5sum": "d8e4ac1c861de551f3345bc45c4a9745"
},
{
"dataPath": "params_shard_102.bin",
"format": "raw-shard",
"nbytes": 100663296,
"records": [
{
"name": "transformer.h.3.mlp.gate_up_proj.weight",
"shape": [
16384,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 100663296,
"byteOffset": 0
}
],
"md5sum": "2a020772f91756e032ebc7a920a2ece2"
},
{
"dataPath": "params_shard_103.bin",
"format": "raw-shard",
"nbytes": 18874368,
"records": [
{
"name": "transformer.h.3.mixer.out_proj.weight",
"shape": [
3072,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 18874368,
"byteOffset": 0
}
],
"md5sum": "37a98cccee5ea8f0591dfdd76dbf3adb"
},
{
"dataPath": "params_shard_104.bin",
"format": "raw-shard",
"nbytes": 56623104,
"records": [
{
"name": "transformer.h.3.mixer.qkv_proj.weight",
"shape": [
9216,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 56623104,
"byteOffset": 0
}
],
"md5sum": "67ad1c4d1ae5fa7c2385bedf20c8b73f"
},
{
"dataPath": "params_shard_105.bin",
"format": "raw-shard",
"nbytes": 50331648,
"records": [
{
"name": "transformer.h.4.mlp.down_proj.weight",
"shape": [
3072,
8192
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 50331648,
"byteOffset": 0
}
],
"md5sum": "fe21c57ec308833254a7ab4b0ebbafe3"
},
{
"dataPath": "params_shard_106.bin",
"format": "raw-shard",
"nbytes": 100663296,
"records": [
{
"name": "transformer.h.4.mlp.gate_up_proj.weight",
"shape": [
16384,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 100663296,
"byteOffset": 0
}
],
"md5sum": "b3283f70b20bbe2237821ae571184a9a"
},
{
"dataPath": "params_shard_107.bin",
"format": "raw-shard",
"nbytes": 18874368,
"records": [
{
"name": "transformer.h.4.mixer.out_proj.weight",
"shape": [
3072,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 18874368,
"byteOffset": 0
}
],
"md5sum": "5576fa782022a5c40c568017ba734d99"
},
{
"dataPath": "params_shard_108.bin",
"format": "raw-shard",
"nbytes": 56623104,
"records": [
{
"name": "transformer.h.4.mixer.qkv_proj.weight",
"shape": [
9216,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 56623104,
"byteOffset": 0
}
],
"md5sum": "2920b6adb22a73082d3ac2f8741a974d"
},
{
"dataPath": "params_shard_109.bin",
"format": "raw-shard",
"nbytes": 50331648,
"records": [
{
"name": "transformer.h.5.mlp.down_proj.weight",
"shape": [
3072,
8192
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 50331648,
"byteOffset": 0
}
],
"md5sum": "ffdf79001044f23701e18213bf76f0bd"
},
{
"dataPath": "params_shard_110.bin",
"format": "raw-shard",
"nbytes": 100663296,
"records": [
{
"name": "transformer.h.5.mlp.gate_up_proj.weight",
"shape": [
16384,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 100663296,
"byteOffset": 0
}
],
"md5sum": "715c79ed4715fb3e6d8d2d79c5f56b5e"
},
{
"dataPath": "params_shard_111.bin",
"format": "raw-shard",
"nbytes": 18874368,
"records": [
{
"name": "transformer.h.5.mixer.out_proj.weight",
"shape": [
3072,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 18874368,
"byteOffset": 0
}
],
"md5sum": "654c5931d95b895a9a55ff2afbc699a5"
},
{
"dataPath": "params_shard_112.bin",
"format": "raw-shard",
"nbytes": 56623104,
"records": [
{
"name": "transformer.h.5.mixer.qkv_proj.weight",
"shape": [
9216,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 56623104,
"byteOffset": 0
}
],
"md5sum": "f8eba59f4c24e5d784248fa5d18a42d8"
},
{
"dataPath": "params_shard_113.bin",
"format": "raw-shard",
"nbytes": 50331648,
"records": [
{
"name": "transformer.h.6.mlp.down_proj.weight",
"shape": [
3072,
8192
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 50331648,
"byteOffset": 0
}
],
"md5sum": "2c66892223944f2ec60aaf69d34d2e11"
},
{
"dataPath": "params_shard_114.bin",
"format": "raw-shard",
"nbytes": 100663296,
"records": [
{
"name": "transformer.h.6.mlp.gate_up_proj.weight",
"shape": [
16384,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 100663296,
"byteOffset": 0
}
],
"md5sum": "37cd987183ccc430ce35b48fa7f05db3"
},
{
"dataPath": "params_shard_115.bin",
"format": "raw-shard",
"nbytes": 18874368,
"records": [
{
"name": "transformer.h.6.mixer.out_proj.weight",
"shape": [
3072,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 18874368,
"byteOffset": 0
}
],
"md5sum": "fcf4c68aae219083c5b13dd54bb6a0f7"
},
{
"dataPath": "params_shard_116.bin",
"format": "raw-shard",
"nbytes": 56623104,
"records": [
{
"name": "transformer.h.6.mixer.qkv_proj.weight",
"shape": [
9216,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 56623104,
"byteOffset": 0
}
],
"md5sum": "cfff45bedd5d3f9c1acdec8706ddfd12"
},
{
"dataPath": "params_shard_117.bin",
"format": "raw-shard",
"nbytes": 50331648,
"records": [
{
"name": "transformer.h.7.mlp.down_proj.weight",
"shape": [
3072,
8192
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 50331648,
"byteOffset": 0
}
],
"md5sum": "7a908227d9ab71d3e4f605e44a4c0b84"
},
{
"dataPath": "params_shard_118.bin",
"format": "raw-shard",
"nbytes": 100663296,
"records": [
{
"name": "transformer.h.7.mlp.gate_up_proj.weight",
"shape": [
16384,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 100663296,
"byteOffset": 0
}
],
"md5sum": "a63968773cbb992bdeb65f804200f966"
},
{
"dataPath": "params_shard_119.bin",
"format": "raw-shard",
"nbytes": 18874368,
"records": [
{
"name": "transformer.h.7.mixer.out_proj.weight",
"shape": [
3072,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 18874368,
"byteOffset": 0
}
],
"md5sum": "09a77c5c5de700053b2fd325baee2e3e"
},
{
"dataPath": "params_shard_120.bin",
"format": "raw-shard",
"nbytes": 56623104,
"records": [
{
"name": "transformer.h.7.mixer.qkv_proj.weight",
"shape": [
9216,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 56623104,
"byteOffset": 0
}
],
"md5sum": "1331913d0a139cc0ea7f3c2ef19a1854"
},
{
"dataPath": "params_shard_121.bin",
"format": "raw-shard",
"nbytes": 50331648,
"records": [
{
"name": "transformer.h.8.mlp.down_proj.weight",
"shape": [
3072,
8192
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 50331648,
"byteOffset": 0
}
],
"md5sum": "d0efc02d3f2d395960cc59a6a29941dd"
},
{
"dataPath": "params_shard_122.bin",
"format": "raw-shard",
"nbytes": 100663296,
"records": [
{
"name": "transformer.h.8.mlp.gate_up_proj.weight",
"shape": [
16384,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 100663296,
"byteOffset": 0
}
],
"md5sum": "515287320d750ddef90a2c26d59b45cc"
},
{
"dataPath": "params_shard_123.bin",
"format": "raw-shard",
"nbytes": 18874368,
"records": [
{
"name": "transformer.h.8.mixer.out_proj.weight",
"shape": [
3072,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 18874368,
"byteOffset": 0
}
],
"md5sum": "48386e64e7bc21c2c6d9ec1efc5d348c"
},
{
"dataPath": "params_shard_124.bin",
"format": "raw-shard",
"nbytes": 56623104,
"records": [
{
"name": "transformer.h.8.mixer.qkv_proj.weight",
"shape": [
9216,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 56623104,
"byteOffset": 0
}
],
"md5sum": "fcbd7dd1b3d229bbe556ed019dbb3272"
},
{
"dataPath": "params_shard_125.bin",
"format": "raw-shard",
"nbytes": 50331648,
"records": [
{
"name": "transformer.h.9.mlp.down_proj.weight",
"shape": [
3072,
8192
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 50331648,
"byteOffset": 0
}
],
"md5sum": "d46cfb6c6df8baa7173adcb195a45609"
},
{
"dataPath": "params_shard_126.bin",
"format": "raw-shard",
"nbytes": 100663296,
"records": [
{
"name": "transformer.h.9.mlp.gate_up_proj.weight",
"shape": [
16384,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 100663296,
"byteOffset": 0
}
],
"md5sum": "1e6c430b422fc400e63d33c45f7bd44b"
},
{
"dataPath": "params_shard_127.bin",
"format": "raw-shard",
"nbytes": 18874368,
"records": [
{
"name": "transformer.h.9.mixer.out_proj.weight",
"shape": [
3072,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 18874368,
"byteOffset": 0
}
],
"md5sum": "48d9d3fdff5e72ecf1569b8eec89f3e3"
},
{
"dataPath": "params_shard_128.bin",
"format": "raw-shard",
"nbytes": 56623104,
"records": [
{
"name": "transformer.h.9.mixer.qkv_proj.weight",
"shape": [
9216,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 56623104,
"byteOffset": 0
}
],
"md5sum": "d80713d97db0fe2e884ae7bc0d611235"
},
{
"dataPath": "params_shard_129.bin",
"format": "raw-shard",
"nbytes": 19273728,
"records": [
{
"name": "transformer.h.21.ln.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 0
},
{
"name": "transformer.h.21.post_attention_layernorm.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 6144
},
{
"name": "transformer.h.22.ln.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 12288
},
{
"name": "transformer.h.22.post_attention_layernorm.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 18432
},
{
"name": "transformer.h.22.mixer.out_proj.weight",
"shape": [
3072,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 18874368,
"byteOffset": 24576
},
{
"name": "transformer.h.23.ln.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 18898944
},
{
"name": "transformer.h.23.post_attention_layernorm.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 18905088
},
{
"name": "transformer.h.24.ln.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 18911232
},
{
"name": "transformer.h.24.post_attention_layernorm.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 18917376
},
{
"name": "transformer.h.25.ln.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 18923520
},
{
"name": "transformer.h.25.post_attention_layernorm.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 18929664
},
{
"name": "transformer.h.26.ln.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 18935808
},
{
"name": "transformer.h.26.post_attention_layernorm.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 18941952
},
{
"name": "transformer.h.27.ln.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 18948096
},
{
"name": "transformer.h.27.post_attention_layernorm.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 18954240
},
{
"name": "transformer.h.28.ln.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 18960384
},
{
"name": "transformer.h.28.post_attention_layernorm.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 18966528
},
{
"name": "transformer.h.29.ln.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 18972672
},
{
"name": "transformer.h.29.post_attention_layernorm.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 18978816
},
{
"name": "transformer.h.30.ln.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 18984960
},
{
"name": "transformer.h.30.post_attention_layernorm.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 18991104
},
{
"name": "transformer.h.31.ln.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 18997248
},
{
"name": "transformer.h.31.post_attention_layernorm.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 19003392
},
{
"name": "transformer.norm.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 19009536
},
{
"name": "transformer.h.0.ln.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 19015680
},
{
"name": "transformer.h.0.post_attention_layernorm.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 19021824
},
{
"name": "transformer.h.1.ln.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 19027968
},
{
"name": "transformer.h.1.post_attention_layernorm.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 19034112
},
{
"name": "transformer.h.10.ln.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 19040256
},
{
"name": "transformer.h.10.post_attention_layernorm.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 19046400
},
{
"name": "transformer.h.11.ln.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 19052544
},
{
"name": "transformer.h.11.post_attention_layernorm.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 19058688
},
{
"name": "transformer.h.12.ln.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 19064832
},
{
"name": "transformer.h.12.post_attention_layernorm.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 19070976
},
{
"name": "transformer.h.13.ln.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 19077120
},
{
"name": "transformer.h.13.post_attention_layernorm.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 19083264
},
{
"name": "transformer.h.14.ln.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 19089408
},
{
"name": "transformer.h.14.post_attention_layernorm.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 19095552
},
{
"name": "transformer.h.15.ln.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 19101696
},
{
"name": "transformer.h.15.post_attention_layernorm.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 19107840
},
{
"name": "transformer.h.16.ln.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 19113984
},
{
"name": "transformer.h.16.post_attention_layernorm.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 19120128
},
{
"name": "transformer.h.17.ln.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 19126272
},
{
"name": "transformer.h.17.post_attention_layernorm.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 19132416
},
{
"name": "transformer.h.18.ln.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 19138560
},
{
"name": "transformer.h.18.post_attention_layernorm.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 19144704
},
{
"name": "transformer.h.19.ln.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 19150848
},
{
"name": "transformer.h.19.post_attention_layernorm.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 19156992
},
{
"name": "transformer.h.2.ln.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 19163136
},
{
"name": "transformer.h.2.post_attention_layernorm.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 19169280
},
{
"name": "transformer.h.20.ln.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 19175424
},
{
"name": "transformer.h.20.post_attention_layernorm.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 19181568
},
{
"name": "transformer.h.3.ln.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 19187712
},
{
"name": "transformer.h.3.post_attention_layernorm.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 19193856
},
{
"name": "transformer.h.4.ln.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 19200000
},
{
"name": "transformer.h.4.post_attention_layernorm.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 19206144
},
{
"name": "transformer.h.5.ln.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 19212288
},
{
"name": "transformer.h.5.post_attention_layernorm.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 19218432
},
{
"name": "transformer.h.6.ln.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 19224576
},
{
"name": "transformer.h.6.post_attention_layernorm.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 19230720
},
{
"name": "transformer.h.7.ln.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 19236864
},
{
"name": "transformer.h.7.post_attention_layernorm.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 19243008
},
{
"name": "transformer.h.8.ln.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 19249152
},
{
"name": "transformer.h.8.post_attention_layernorm.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 19255296
},
{
"name": "transformer.h.9.ln.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 19261440
},
{
"name": "transformer.h.9.post_attention_layernorm.weight",
"shape": [
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6144,
"byteOffset": 19267584
}
],
"md5sum": "7d90cc7bee5c8a5aa80ac6fca9f03b61"
}
]
}