{ "metadata": { "ParamSize": 195, "ParamBytes": 7642159104.0, "BitsPerParam": 16.0 }, "records": [ { "dataPath": "params_shard_0.bin", "format": "raw-shard", "nbytes": 197001216, "records": [ { "name": "lm_head.weight", "shape": [ 32064, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 197001216, "byteOffset": 0 } ], "md5sum": "8c2d85352caa7c9a98d74a9ee3cf6a94" }, { "dataPath": "params_shard_1.bin", "format": "raw-shard", "nbytes": 50331648, "records": [ { "name": "transformer.h.21.mlp.down_proj.weight", "shape": [ 3072, 8192 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 50331648, "byteOffset": 0 } ], "md5sum": "fbdfcef1bc44437d1e2dd389140c2334" }, { "dataPath": "params_shard_2.bin", "format": "raw-shard", "nbytes": 100663296, "records": [ { "name": "transformer.h.21.mlp.gate_up_proj.weight", "shape": [ 16384, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 100663296, "byteOffset": 0 } ], "md5sum": "b14efca889e2938bd0dd70bf01f3e03a" }, { "dataPath": "params_shard_3.bin", "format": "raw-shard", "nbytes": 56623104, "records": [ { "name": "transformer.h.21.mixer.qkv_proj.weight", "shape": [ 9216, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 56623104, "byteOffset": 0 } ], "md5sum": "c1a49fa4938c39315cb4cdcc5c2d7d49" }, { "dataPath": "params_shard_4.bin", "format": "raw-shard", "nbytes": 50331648, "records": [ { "name": "transformer.h.22.mlp.down_proj.weight", "shape": [ 3072, 8192 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 50331648, "byteOffset": 0 } ], "md5sum": "a2239c27ad68b6f533a7102d82527878" }, { "dataPath": "params_shard_5.bin", "format": "raw-shard", "nbytes": 100663296, "records": [ { "name": "transformer.h.22.mlp.gate_up_proj.weight", "shape": [ 16384, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 100663296, "byteOffset": 0 } ], "md5sum": "c56bee40a05fbd4322eade65122c07f3" }, { "dataPath": "params_shard_6.bin", "format": "raw-shard", "nbytes": 56623104, "records": [ { "name": "transformer.h.22.mixer.qkv_proj.weight", "shape": [ 9216, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 56623104, "byteOffset": 0 } ], "md5sum": "bfa63c8cb8115ffdfe3d0e9141eee648" }, { "dataPath": "params_shard_7.bin", "format": "raw-shard", "nbytes": 50331648, "records": [ { "name": "transformer.h.23.mlp.down_proj.weight", "shape": [ 3072, 8192 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 50331648, "byteOffset": 0 } ], "md5sum": "83521bd8fbce11184521d6ab41a29cea" }, { "dataPath": "params_shard_8.bin", "format": "raw-shard", "nbytes": 100663296, "records": [ { "name": "transformer.h.23.mlp.gate_up_proj.weight", "shape": [ 16384, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 100663296, "byteOffset": 0 } ], "md5sum": "2ea3121970667f0f7645c5e728237748" }, { "dataPath": "params_shard_9.bin", "format": "raw-shard", "nbytes": 18874368, "records": [ { "name": "transformer.h.23.mixer.out_proj.weight", "shape": [ 3072, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 18874368, "byteOffset": 0 } ], "md5sum": "1934b976730e3bd2abf6d9a8b72529d8" }, { "dataPath": "params_shard_10.bin", "format": "raw-shard", "nbytes": 56623104, "records": [ { "name": "transformer.h.23.mixer.qkv_proj.weight", "shape": [ 9216, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 56623104, "byteOffset": 0 } ], "md5sum": "1cf17bdee74a90757820d2d22aabecf7" }, { "dataPath": "params_shard_11.bin", "format": "raw-shard", "nbytes": 50331648, "records": [ { "name": "transformer.h.24.mlp.down_proj.weight", "shape": [ 3072, 8192 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 50331648, "byteOffset": 0 } ], "md5sum": "03112ed394c45b3446d495aeb5cbdff0" }, { "dataPath": "params_shard_12.bin", "format": "raw-shard", "nbytes": 100663296, "records": [ { "name": "transformer.h.24.mlp.gate_up_proj.weight", "shape": [ 16384, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 100663296, "byteOffset": 0 } ], "md5sum": "49188904aa2c30790064abfd8097672c" }, { "dataPath": "params_shard_13.bin", "format": "raw-shard", "nbytes": 18874368, "records": [ { "name": "transformer.h.24.mixer.out_proj.weight", "shape": [ 3072, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 18874368, "byteOffset": 0 } ], "md5sum": "f42a37069ae388cfdaf2d09833d82244" }, { "dataPath": "params_shard_14.bin", "format": "raw-shard", "nbytes": 56623104, "records": [ { "name": "transformer.h.24.mixer.qkv_proj.weight", "shape": [ 9216, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 56623104, "byteOffset": 0 } ], "md5sum": "30c14cf99aff89957ec7da295c4ab632" }, { "dataPath": "params_shard_15.bin", "format": "raw-shard", "nbytes": 50331648, "records": [ { "name": "transformer.h.25.mlp.down_proj.weight", "shape": [ 3072, 8192 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 50331648, "byteOffset": 0 } ], "md5sum": "b57a23df765fd830804df014adca71e7" }, { "dataPath": "params_shard_16.bin", "format": "raw-shard", "nbytes": 100663296, "records": [ { "name": "transformer.h.25.mlp.gate_up_proj.weight", "shape": [ 16384, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 100663296, "byteOffset": 0 } ], "md5sum": "6a8895d6b1ccd90c8e91f75da70bdd3e" }, { "dataPath": "params_shard_17.bin", "format": "raw-shard", "nbytes": 18874368, "records": [ { "name": "transformer.h.25.mixer.out_proj.weight", "shape": [ 3072, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 18874368, "byteOffset": 0 } ], "md5sum": "66114f0757c09a36c7ffd636719cc7b3" }, { "dataPath": "params_shard_18.bin", "format": "raw-shard", "nbytes": 56623104, "records": [ { "name": "transformer.h.25.mixer.qkv_proj.weight", "shape": [ 9216, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 56623104, "byteOffset": 0 } ], "md5sum": "f9ba7dd1a9ffe04b85583439d91d551a" }, { "dataPath": "params_shard_19.bin", "format": "raw-shard", "nbytes": 50331648, "records": [ { "name": "transformer.h.26.mlp.down_proj.weight", "shape": [ 3072, 8192 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 50331648, "byteOffset": 0 } ], "md5sum": "71e79d1578fd6ee4e4524679fb1276ff" }, { "dataPath": "params_shard_20.bin", "format": "raw-shard", "nbytes": 100663296, "records": [ { "name": "transformer.h.26.mlp.gate_up_proj.weight", "shape": [ 16384, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 100663296, "byteOffset": 0 } ], "md5sum": "5dc9d0557437fcf6e561ee3951baadae" }, { "dataPath": "params_shard_21.bin", "format": "raw-shard", "nbytes": 18874368, "records": [ { "name": "transformer.h.26.mixer.out_proj.weight", "shape": [ 3072, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 18874368, "byteOffset": 0 } ], "md5sum": "297e4f56ec4f407b4e9546c012b09775" }, { "dataPath": "params_shard_22.bin", "format": "raw-shard", "nbytes": 56623104, "records": [ { "name": "transformer.h.26.mixer.qkv_proj.weight", "shape": [ 9216, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 56623104, "byteOffset": 0 } ], "md5sum": "2bf6b964d4c2688cc13183649276342a" }, { "dataPath": "params_shard_23.bin", "format": "raw-shard", "nbytes": 50331648, "records": [ { "name": "transformer.h.27.mlp.down_proj.weight", "shape": [ 3072, 8192 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 50331648, "byteOffset": 0 } ], "md5sum": "17b63513014e546d7b672228b045718c" }, { "dataPath": "params_shard_24.bin", "format": "raw-shard", "nbytes": 100663296, "records": [ { "name": "transformer.h.27.mlp.gate_up_proj.weight", "shape": [ 16384, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 100663296, "byteOffset": 0 } ], "md5sum": "487b046eee7038d7046ba4740a5ca521" }, { "dataPath": "params_shard_25.bin", "format": "raw-shard", "nbytes": 18874368, "records": [ { "name": "transformer.h.27.mixer.out_proj.weight", "shape": [ 3072, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 18874368, "byteOffset": 0 } ], "md5sum": "30b04b01a54e57beba23e7fe29957c87" }, { "dataPath": "params_shard_26.bin", "format": "raw-shard", "nbytes": 56623104, "records": [ { "name": "transformer.h.27.mixer.qkv_proj.weight", "shape": [ 9216, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 56623104, "byteOffset": 0 } ], "md5sum": "00ed36d8f6e889b0567ff936972055bb" }, { "dataPath": "params_shard_27.bin", "format": "raw-shard", "nbytes": 50331648, "records": [ { "name": "transformer.h.28.mlp.down_proj.weight", "shape": [ 3072, 8192 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 50331648, "byteOffset": 0 } ], "md5sum": "d790288dcc53da10b9ee96e9a80571be" }, { "dataPath": "params_shard_28.bin", "format": "raw-shard", "nbytes": 100663296, "records": [ { "name": "transformer.h.28.mlp.gate_up_proj.weight", "shape": [ 16384, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 100663296, "byteOffset": 0 } ], "md5sum": "8625bcda8d93dc0529984087f07cccb8" }, { "dataPath": "params_shard_29.bin", "format": "raw-shard", "nbytes": 18874368, "records": [ { "name": "transformer.h.28.mixer.out_proj.weight", "shape": [ 3072, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 18874368, "byteOffset": 0 } ], "md5sum": "ab0e01373b835f596283764432bdcea9" }, { "dataPath": "params_shard_30.bin", "format": "raw-shard", "nbytes": 56623104, "records": [ { "name": "transformer.h.28.mixer.qkv_proj.weight", "shape": [ 9216, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 56623104, "byteOffset": 0 } ], "md5sum": "3b373274561aeb974d849cc01c2850b0" }, { "dataPath": "params_shard_31.bin", "format": "raw-shard", "nbytes": 50331648, "records": [ { "name": "transformer.h.29.mlp.down_proj.weight", "shape": [ 3072, 8192 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 50331648, "byteOffset": 0 } ], "md5sum": "ac9774e76732119d81106d74f7f6ee52" }, { "dataPath": "params_shard_32.bin", "format": "raw-shard", "nbytes": 100663296, "records": [ { "name": "transformer.h.29.mlp.gate_up_proj.weight", "shape": [ 16384, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 100663296, "byteOffset": 0 } ], "md5sum": "47013b7213be9e785d22bdaa10590ca4" }, { "dataPath": "params_shard_33.bin", "format": "raw-shard", "nbytes": 18874368, "records": [ { "name": "transformer.h.29.mixer.out_proj.weight", "shape": [ 3072, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 18874368, "byteOffset": 0 } ], "md5sum": "aaf205e3e3798acd2b6f1bc92d94d6de" }, { "dataPath": "params_shard_34.bin", "format": "raw-shard", "nbytes": 56623104, "records": [ { "name": "transformer.h.29.mixer.qkv_proj.weight", "shape": [ 9216, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 56623104, "byteOffset": 0 } ], "md5sum": "d3b2899e4379328af50315fe46f9e144" }, { "dataPath": "params_shard_35.bin", "format": "raw-shard", "nbytes": 50331648, "records": [ { "name": "transformer.h.30.mlp.down_proj.weight", "shape": [ 3072, 8192 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 50331648, "byteOffset": 0 } ], "md5sum": "d5604e270a26008514d6f07651520d7b" }, { "dataPath": "params_shard_36.bin", "format": "raw-shard", "nbytes": 100663296, "records": [ { "name": "transformer.h.30.mlp.gate_up_proj.weight", "shape": [ 16384, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 100663296, "byteOffset": 0 } ], "md5sum": "c21012a11160161b3f16436d5e8fbcea" }, { "dataPath": "params_shard_37.bin", "format": "raw-shard", "nbytes": 18874368, "records": [ { "name": "transformer.h.30.mixer.out_proj.weight", "shape": [ 3072, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 18874368, "byteOffset": 0 } ], "md5sum": "99ac28b15ff15c0e7dcfada05307ac89" }, { "dataPath": "params_shard_38.bin", "format": "raw-shard", "nbytes": 56623104, "records": [ { "name": "transformer.h.30.mixer.qkv_proj.weight", "shape": [ 9216, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 56623104, "byteOffset": 0 } ], "md5sum": "225394d73feea900325e934fd3a1c0b4" }, { "dataPath": "params_shard_39.bin", "format": "raw-shard", "nbytes": 50331648, "records": [ { "name": "transformer.h.31.mlp.down_proj.weight", "shape": [ 3072, 8192 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 50331648, "byteOffset": 0 } ], "md5sum": "9d287bb8e3cf1f29df30aefc9c027fd2" }, { "dataPath": "params_shard_40.bin", "format": "raw-shard", "nbytes": 100663296, "records": [ { "name": "transformer.h.31.mlp.gate_up_proj.weight", "shape": [ 16384, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 100663296, "byteOffset": 0 } ], "md5sum": "263c716606e3a2afc30802ef6483db82" }, { "dataPath": "params_shard_41.bin", "format": "raw-shard", "nbytes": 18874368, "records": [ { "name": "transformer.h.31.mixer.out_proj.weight", "shape": [ 3072, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 18874368, "byteOffset": 0 } ], "md5sum": "f0ecb725e774085ffd140fd5df0089e6" }, { "dataPath": "params_shard_42.bin", "format": "raw-shard", "nbytes": 56623104, "records": [ { "name": "transformer.h.31.mixer.qkv_proj.weight", "shape": [ 9216, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 56623104, "byteOffset": 0 } ], "md5sum": "191fa72b252485f0a9db714e8a2fa3e5" }, { "dataPath": "params_shard_43.bin", "format": "raw-shard", "nbytes": 197001216, "records": [ { "name": "transformer.embd.weight", "shape": [ 32064, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 197001216, "byteOffset": 0 } ], "md5sum": "6f9e82ee16bf9e48202971f1bc7faddf" }, { "dataPath": "params_shard_44.bin", "format": "raw-shard", "nbytes": 50331648, "records": [ { "name": "transformer.h.0.mlp.down_proj.weight", "shape": [ 3072, 8192 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 50331648, "byteOffset": 0 } ], "md5sum": "6d31c842666fb35f49c2826e9469172e" }, { "dataPath": "params_shard_45.bin", "format": "raw-shard", "nbytes": 100663296, "records": [ { "name": "transformer.h.0.mlp.gate_up_proj.weight", "shape": [ 16384, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 100663296, "byteOffset": 0 } ], "md5sum": "caf606342f712b439b4540b30b4d4dd8" }, { "dataPath": "params_shard_46.bin", "format": "raw-shard", "nbytes": 18874368, "records": [ { "name": "transformer.h.0.mixer.out_proj.weight", "shape": [ 3072, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 18874368, "byteOffset": 0 } ], "md5sum": "3cbc63d65af31aeeb5f0ff6d21b05d91" }, { "dataPath": "params_shard_47.bin", "format": "raw-shard", "nbytes": 56623104, "records": [ { "name": "transformer.h.0.mixer.qkv_proj.weight", "shape": [ 9216, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 56623104, "byteOffset": 0 } ], "md5sum": "8011494667ca20936cfbd3fe057f030f" }, { "dataPath": "params_shard_48.bin", "format": "raw-shard", "nbytes": 50331648, "records": [ { "name": "transformer.h.1.mlp.down_proj.weight", "shape": [ 3072, 8192 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 50331648, "byteOffset": 0 } ], "md5sum": "d58dba16b580f460e040860a3be560d7" }, { "dataPath": "params_shard_49.bin", "format": "raw-shard", "nbytes": 100663296, "records": [ { "name": "transformer.h.1.mlp.gate_up_proj.weight", "shape": [ 16384, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 100663296, "byteOffset": 0 } ], "md5sum": "e3851f228e9eeea429ffdb5bfcf65712" }, { "dataPath": "params_shard_50.bin", "format": "raw-shard", "nbytes": 18874368, "records": [ { "name": "transformer.h.1.mixer.out_proj.weight", "shape": [ 3072, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 18874368, "byteOffset": 0 } ], "md5sum": "36a299bba3d83d4b09fe38cea015c920" }, { "dataPath": "params_shard_51.bin", "format": "raw-shard", "nbytes": 56623104, "records": [ { "name": "transformer.h.1.mixer.qkv_proj.weight", "shape": [ 9216, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 56623104, "byteOffset": 0 } ], "md5sum": "a2c6ca04c30deb8ca3dd477f9b070990" }, { "dataPath": "params_shard_52.bin", "format": "raw-shard", "nbytes": 50331648, "records": [ { "name": "transformer.h.10.mlp.down_proj.weight", "shape": [ 3072, 8192 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 50331648, "byteOffset": 0 } ], "md5sum": "cc7e87530395bcaae2bbaaf14cdccf01" }, { "dataPath": "params_shard_53.bin", "format": "raw-shard", "nbytes": 100663296, "records": [ { "name": "transformer.h.10.mlp.gate_up_proj.weight", "shape": [ 16384, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 100663296, "byteOffset": 0 } ], "md5sum": "c39dc653bd8e615ccf216eafefa87cc9" }, { "dataPath": "params_shard_54.bin", "format": "raw-shard", "nbytes": 18874368, "records": [ { "name": "transformer.h.10.mixer.out_proj.weight", "shape": [ 3072, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 18874368, "byteOffset": 0 } ], "md5sum": "46da89076ea5cf5e396f3e5b886af15e" }, { "dataPath": "params_shard_55.bin", "format": "raw-shard", "nbytes": 56623104, "records": [ { "name": "transformer.h.10.mixer.qkv_proj.weight", "shape": [ 9216, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 56623104, "byteOffset": 0 } ], "md5sum": "dc18d2f273cd86d9bb06d7dcae88c221" }, { "dataPath": "params_shard_56.bin", "format": "raw-shard", "nbytes": 50331648, "records": [ { "name": "transformer.h.11.mlp.down_proj.weight", "shape": [ 3072, 8192 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 50331648, "byteOffset": 0 } ], "md5sum": "36bc078291c38abe52684ab9eae3a086" }, { "dataPath": "params_shard_57.bin", "format": "raw-shard", "nbytes": 100663296, "records": [ { "name": "transformer.h.11.mlp.gate_up_proj.weight", "shape": [ 16384, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 100663296, "byteOffset": 0 } ], "md5sum": "00b40de1e12839f2fafaf0742585be28" }, { "dataPath": "params_shard_58.bin", "format": "raw-shard", "nbytes": 18874368, "records": [ { "name": "transformer.h.11.mixer.out_proj.weight", "shape": [ 3072, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 18874368, "byteOffset": 0 } ], "md5sum": "d8bba7fe7d417dea96d3c791522604d6" }, { "dataPath": "params_shard_59.bin", "format": "raw-shard", "nbytes": 56623104, "records": [ { "name": "transformer.h.11.mixer.qkv_proj.weight", "shape": [ 9216, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 56623104, "byteOffset": 0 } ], "md5sum": "652f758c4908ca7d5a76a3d241bb83a0" }, { "dataPath": "params_shard_60.bin", "format": "raw-shard", "nbytes": 50331648, "records": [ { "name": "transformer.h.12.mlp.down_proj.weight", "shape": [ 3072, 8192 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 50331648, "byteOffset": 0 } ], "md5sum": "8cf3e143a89cbf61c8620a66b6446075" }, { "dataPath": "params_shard_61.bin", "format": "raw-shard", "nbytes": 100663296, "records": [ { "name": "transformer.h.12.mlp.gate_up_proj.weight", "shape": [ 16384, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 100663296, "byteOffset": 0 } ], "md5sum": "9101ba07812357ac26670e4b6a07dadf" }, { "dataPath": "params_shard_62.bin", "format": "raw-shard", "nbytes": 18874368, "records": [ { "name": "transformer.h.12.mixer.out_proj.weight", "shape": [ 3072, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 18874368, "byteOffset": 0 } ], "md5sum": "756a1a808db0ed7583500bc0a2b54008" }, { "dataPath": "params_shard_63.bin", "format": "raw-shard", "nbytes": 56623104, "records": [ { "name": "transformer.h.12.mixer.qkv_proj.weight", "shape": [ 9216, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 56623104, "byteOffset": 0 } ], "md5sum": "42262b1d88e9d9d49b33fda8283cb753" }, { "dataPath": "params_shard_64.bin", "format": "raw-shard", "nbytes": 50331648, "records": [ { "name": "transformer.h.13.mlp.down_proj.weight", "shape": [ 3072, 8192 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 50331648, "byteOffset": 0 } ], "md5sum": "3bd7a6b66767bd6ebabcbeaa81cd0d15" }, { "dataPath": "params_shard_65.bin", "format": "raw-shard", "nbytes": 100663296, "records": [ { "name": "transformer.h.13.mlp.gate_up_proj.weight", "shape": [ 16384, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 100663296, "byteOffset": 0 } ], "md5sum": "dae0535b359e750b5168e08edaab2f0a" }, { "dataPath": "params_shard_66.bin", "format": "raw-shard", "nbytes": 18874368, "records": [ { "name": "transformer.h.13.mixer.out_proj.weight", "shape": [ 3072, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 18874368, "byteOffset": 0 } ], "md5sum": "71fd9d016830fe9a2e62c82f4727b55f" }, { "dataPath": "params_shard_67.bin", "format": "raw-shard", "nbytes": 56623104, "records": [ { "name": "transformer.h.13.mixer.qkv_proj.weight", "shape": [ 9216, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 56623104, "byteOffset": 0 } ], "md5sum": "bdac4468f08f6d325910744e53a73334" }, { "dataPath": "params_shard_68.bin", "format": "raw-shard", "nbytes": 50331648, "records": [ { "name": "transformer.h.14.mlp.down_proj.weight", "shape": [ 3072, 8192 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 50331648, "byteOffset": 0 } ], "md5sum": "41331cffbf45ca84270d7f85134ece7c" }, { "dataPath": "params_shard_69.bin", "format": "raw-shard", "nbytes": 100663296, "records": [ { "name": "transformer.h.14.mlp.gate_up_proj.weight", "shape": [ 16384, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 100663296, "byteOffset": 0 } ], "md5sum": "22c314f049027022e97d39b3bd9e56fe" }, { "dataPath": "params_shard_70.bin", "format": "raw-shard", "nbytes": 18874368, "records": [ { "name": "transformer.h.14.mixer.out_proj.weight", "shape": [ 3072, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 18874368, "byteOffset": 0 } ], "md5sum": "d51c88b65e6d2f35d7bd16e4ec04136b" }, { "dataPath": "params_shard_71.bin", "format": "raw-shard", "nbytes": 56623104, "records": [ { "name": "transformer.h.14.mixer.qkv_proj.weight", "shape": [ 9216, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 56623104, "byteOffset": 0 } ], "md5sum": "72e28da4e109231f4a7466c4eec109cb" }, { "dataPath": "params_shard_72.bin", "format": "raw-shard", "nbytes": 50331648, "records": [ { "name": "transformer.h.15.mlp.down_proj.weight", "shape": [ 3072, 8192 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 50331648, "byteOffset": 0 } ], "md5sum": "96c63673e9f0a437c157c676b2af5ee9" }, { "dataPath": "params_shard_73.bin", "format": "raw-shard", "nbytes": 100663296, "records": [ { "name": "transformer.h.15.mlp.gate_up_proj.weight", "shape": [ 16384, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 100663296, "byteOffset": 0 } ], "md5sum": "5558186b76ecb44c6cf7156980cb8978" }, { "dataPath": "params_shard_74.bin", "format": "raw-shard", "nbytes": 18874368, "records": [ { "name": "transformer.h.15.mixer.out_proj.weight", "shape": [ 3072, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 18874368, "byteOffset": 0 } ], "md5sum": "79dc6596ac0f78f6f8ebad933ecb50fd" }, { "dataPath": "params_shard_75.bin", "format": "raw-shard", "nbytes": 56623104, "records": [ { "name": "transformer.h.15.mixer.qkv_proj.weight", "shape": [ 9216, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 56623104, "byteOffset": 0 } ], "md5sum": "0b3553b64fa2313e345ee6baa3c258bb" }, { "dataPath": "params_shard_76.bin", "format": "raw-shard", "nbytes": 50331648, "records": [ { "name": "transformer.h.16.mlp.down_proj.weight", "shape": [ 3072, 8192 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 50331648, "byteOffset": 0 } ], "md5sum": "e51ca07d5edd7aaae0d992b4e8dcfb21" }, { "dataPath": "params_shard_77.bin", "format": "raw-shard", "nbytes": 100663296, "records": [ { "name": "transformer.h.16.mlp.gate_up_proj.weight", "shape": [ 16384, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 100663296, "byteOffset": 0 } ], "md5sum": "7ee0335455612128b3b9171a15700bf2" }, { "dataPath": "params_shard_78.bin", "format": "raw-shard", "nbytes": 18874368, "records": [ { "name": "transformer.h.16.mixer.out_proj.weight", "shape": [ 3072, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 18874368, "byteOffset": 0 } ], "md5sum": "18c35ebe336c1a121b23b1b1e4084a25" }, { "dataPath": "params_shard_79.bin", "format": "raw-shard", "nbytes": 56623104, "records": [ { "name": "transformer.h.16.mixer.qkv_proj.weight", "shape": [ 9216, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 56623104, "byteOffset": 0 } ], "md5sum": "946796a5a531e3372a9eb3a0c7549430" }, { "dataPath": "params_shard_80.bin", "format": "raw-shard", "nbytes": 50331648, "records": [ { "name": "transformer.h.17.mlp.down_proj.weight", "shape": [ 3072, 8192 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 50331648, "byteOffset": 0 } ], "md5sum": "93f0d0f857c235dfeb5ffdc9fab7351a" }, { "dataPath": "params_shard_81.bin", "format": "raw-shard", "nbytes": 100663296, "records": [ { "name": "transformer.h.17.mlp.gate_up_proj.weight", "shape": [ 16384, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 100663296, "byteOffset": 0 } ], "md5sum": "c59ecdf2dd9d347fedc9e5708f624d3c" }, { "dataPath": "params_shard_82.bin", "format": "raw-shard", "nbytes": 18874368, "records": [ { "name": "transformer.h.17.mixer.out_proj.weight", "shape": [ 3072, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 18874368, "byteOffset": 0 } ], "md5sum": "335d4f2d9379b41144983b72b18a52ea" }, { "dataPath": "params_shard_83.bin", "format": "raw-shard", "nbytes": 56623104, "records": [ { "name": "transformer.h.17.mixer.qkv_proj.weight", "shape": [ 9216, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 56623104, "byteOffset": 0 } ], "md5sum": "ad8afa5fe6d4634adadd2c9e8b2f069c" }, { "dataPath": "params_shard_84.bin", "format": "raw-shard", "nbytes": 50331648, "records": [ { "name": "transformer.h.18.mlp.down_proj.weight", "shape": [ 3072, 8192 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 50331648, "byteOffset": 0 } ], "md5sum": "d458ff2536af888debbf4a0018dd1767" }, { "dataPath": "params_shard_85.bin", "format": "raw-shard", "nbytes": 100663296, "records": [ { "name": "transformer.h.18.mlp.gate_up_proj.weight", "shape": [ 16384, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 100663296, "byteOffset": 0 } ], "md5sum": "e7796e4854cdc538e01d7112ab8a5179" }, { "dataPath": "params_shard_86.bin", "format": "raw-shard", "nbytes": 18874368, "records": [ { "name": "transformer.h.18.mixer.out_proj.weight", "shape": [ 3072, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 18874368, "byteOffset": 0 } ], "md5sum": "9df18282bd08212a8ae8fad347721ce2" }, { "dataPath": "params_shard_87.bin", "format": "raw-shard", "nbytes": 56623104, "records": [ { "name": "transformer.h.18.mixer.qkv_proj.weight", "shape": [ 9216, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 56623104, "byteOffset": 0 } ], "md5sum": "0763aaa7da6da824f21db13a1bdd3832" }, { "dataPath": "params_shard_88.bin", "format": "raw-shard", "nbytes": 50331648, "records": [ { "name": "transformer.h.19.mlp.down_proj.weight", "shape": [ 3072, 8192 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 50331648, "byteOffset": 0 } ], "md5sum": "a49f64fd634619b2ea22b320f9a65720" }, { "dataPath": "params_shard_89.bin", "format": "raw-shard", "nbytes": 100663296, "records": [ { "name": "transformer.h.19.mlp.gate_up_proj.weight", "shape": [ 16384, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 100663296, "byteOffset": 0 } ], "md5sum": "9bc101e6065af554c1f77e18793af166" }, { "dataPath": "params_shard_90.bin", "format": "raw-shard", "nbytes": 18874368, "records": [ { "name": "transformer.h.19.mixer.out_proj.weight", "shape": [ 3072, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 18874368, "byteOffset": 0 } ], "md5sum": "b2e650b00c4be4a2699e1f4426c1d92f" }, { "dataPath": "params_shard_91.bin", "format": "raw-shard", "nbytes": 56623104, "records": [ { "name": "transformer.h.19.mixer.qkv_proj.weight", "shape": [ 9216, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 56623104, "byteOffset": 0 } ], "md5sum": "9454b1638ae97f3d95aa11e6c5368d82" }, { "dataPath": "params_shard_92.bin", "format": "raw-shard", "nbytes": 50331648, "records": [ { "name": "transformer.h.2.mlp.down_proj.weight", "shape": [ 3072, 8192 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 50331648, "byteOffset": 0 } ], "md5sum": "36232340846a5ed879134aa5a8336a0e" }, { "dataPath": "params_shard_93.bin", "format": "raw-shard", "nbytes": 100663296, "records": [ { "name": "transformer.h.2.mlp.gate_up_proj.weight", "shape": [ 16384, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 100663296, "byteOffset": 0 } ], "md5sum": "2c848444240798a2c8f998ecc0e7444b" }, { "dataPath": "params_shard_94.bin", "format": "raw-shard", "nbytes": 18874368, "records": [ { "name": "transformer.h.2.mixer.out_proj.weight", "shape": [ 3072, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 18874368, "byteOffset": 0 } ], "md5sum": "5b0642334e6411b70e7cbfc0a203ead3" }, { "dataPath": "params_shard_95.bin", "format": "raw-shard", "nbytes": 56623104, "records": [ { "name": "transformer.h.2.mixer.qkv_proj.weight", "shape": [ 9216, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 56623104, "byteOffset": 0 } ], "md5sum": "6b0e054b7f54523751b04449c6d2f9c7" }, { "dataPath": "params_shard_96.bin", "format": "raw-shard", "nbytes": 50331648, "records": [ { "name": "transformer.h.20.mlp.down_proj.weight", "shape": [ 3072, 8192 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 50331648, "byteOffset": 0 } ], "md5sum": "74a5476d9d22d264626d5b9cbcddcb27" }, { "dataPath": "params_shard_97.bin", "format": "raw-shard", "nbytes": 100663296, "records": [ { "name": "transformer.h.20.mlp.gate_up_proj.weight", "shape": [ 16384, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 100663296, "byteOffset": 0 } ], "md5sum": "a1eee72b43f1c814da6b1e240aebd8fb" }, { "dataPath": "params_shard_98.bin", "format": "raw-shard", "nbytes": 18874368, "records": [ { "name": "transformer.h.20.mixer.out_proj.weight", "shape": [ 3072, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 18874368, "byteOffset": 0 } ], "md5sum": "9274cf6c8ba095bebd1fd97af1a2e642" }, { "dataPath": "params_shard_99.bin", "format": "raw-shard", "nbytes": 56623104, "records": [ { "name": "transformer.h.20.mixer.qkv_proj.weight", "shape": [ 9216, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 56623104, "byteOffset": 0 } ], "md5sum": "00b3cc2bc5d7dc8a7063579ecafeca01" }, { "dataPath": "params_shard_100.bin", "format": "raw-shard", "nbytes": 18874368, "records": [ { "name": "transformer.h.21.mixer.out_proj.weight", "shape": [ 3072, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 18874368, "byteOffset": 0 } ], "md5sum": "acba950f76ad24093ed04f28c465d198" }, { "dataPath": "params_shard_101.bin", "format": "raw-shard", "nbytes": 50331648, "records": [ { "name": "transformer.h.3.mlp.down_proj.weight", "shape": [ 3072, 8192 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 50331648, "byteOffset": 0 } ], "md5sum": "d8e4ac1c861de551f3345bc45c4a9745" }, { "dataPath": "params_shard_102.bin", "format": "raw-shard", "nbytes": 100663296, "records": [ { "name": "transformer.h.3.mlp.gate_up_proj.weight", "shape": [ 16384, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 100663296, "byteOffset": 0 } ], "md5sum": "2a020772f91756e032ebc7a920a2ece2" }, { "dataPath": "params_shard_103.bin", "format": "raw-shard", "nbytes": 18874368, "records": [ { "name": "transformer.h.3.mixer.out_proj.weight", "shape": [ 3072, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 18874368, "byteOffset": 0 } ], "md5sum": "37a98cccee5ea8f0591dfdd76dbf3adb" }, { "dataPath": "params_shard_104.bin", "format": "raw-shard", "nbytes": 56623104, "records": [ { "name": "transformer.h.3.mixer.qkv_proj.weight", "shape": [ 9216, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 56623104, "byteOffset": 0 } ], "md5sum": "67ad1c4d1ae5fa7c2385bedf20c8b73f" }, { "dataPath": "params_shard_105.bin", "format": "raw-shard", "nbytes": 50331648, "records": [ { "name": "transformer.h.4.mlp.down_proj.weight", "shape": [ 3072, 8192 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 50331648, "byteOffset": 0 } ], "md5sum": "fe21c57ec308833254a7ab4b0ebbafe3" }, { "dataPath": "params_shard_106.bin", "format": "raw-shard", "nbytes": 100663296, "records": [ { "name": "transformer.h.4.mlp.gate_up_proj.weight", "shape": [ 16384, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 100663296, "byteOffset": 0 } ], "md5sum": "b3283f70b20bbe2237821ae571184a9a" }, { "dataPath": "params_shard_107.bin", "format": "raw-shard", "nbytes": 18874368, "records": [ { "name": "transformer.h.4.mixer.out_proj.weight", "shape": [ 3072, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 18874368, "byteOffset": 0 } ], "md5sum": "5576fa782022a5c40c568017ba734d99" }, { "dataPath": "params_shard_108.bin", "format": "raw-shard", "nbytes": 56623104, "records": [ { "name": "transformer.h.4.mixer.qkv_proj.weight", "shape": [ 9216, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 56623104, "byteOffset": 0 } ], "md5sum": "2920b6adb22a73082d3ac2f8741a974d" }, { "dataPath": "params_shard_109.bin", "format": "raw-shard", "nbytes": 50331648, "records": [ { "name": "transformer.h.5.mlp.down_proj.weight", "shape": [ 3072, 8192 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 50331648, "byteOffset": 0 } ], "md5sum": "ffdf79001044f23701e18213bf76f0bd" }, { "dataPath": "params_shard_110.bin", "format": "raw-shard", "nbytes": 100663296, "records": [ { "name": "transformer.h.5.mlp.gate_up_proj.weight", "shape": [ 16384, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 100663296, "byteOffset": 0 } ], "md5sum": "715c79ed4715fb3e6d8d2d79c5f56b5e" }, { "dataPath": "params_shard_111.bin", "format": "raw-shard", "nbytes": 18874368, "records": [ { "name": "transformer.h.5.mixer.out_proj.weight", "shape": [ 3072, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 18874368, "byteOffset": 0 } ], "md5sum": "654c5931d95b895a9a55ff2afbc699a5" }, { "dataPath": "params_shard_112.bin", "format": "raw-shard", "nbytes": 56623104, "records": [ { "name": "transformer.h.5.mixer.qkv_proj.weight", "shape": [ 9216, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 56623104, "byteOffset": 0 } ], "md5sum": "f8eba59f4c24e5d784248fa5d18a42d8" }, { "dataPath": "params_shard_113.bin", "format": "raw-shard", "nbytes": 50331648, "records": [ { "name": "transformer.h.6.mlp.down_proj.weight", "shape": [ 3072, 8192 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 50331648, "byteOffset": 0 } ], "md5sum": "2c66892223944f2ec60aaf69d34d2e11" }, { "dataPath": "params_shard_114.bin", "format": "raw-shard", "nbytes": 100663296, "records": [ { "name": "transformer.h.6.mlp.gate_up_proj.weight", "shape": [ 16384, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 100663296, "byteOffset": 0 } ], "md5sum": "37cd987183ccc430ce35b48fa7f05db3" }, { "dataPath": "params_shard_115.bin", "format": "raw-shard", "nbytes": 18874368, "records": [ { "name": "transformer.h.6.mixer.out_proj.weight", "shape": [ 3072, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 18874368, "byteOffset": 0 } ], "md5sum": "fcf4c68aae219083c5b13dd54bb6a0f7" }, { "dataPath": "params_shard_116.bin", "format": "raw-shard", "nbytes": 56623104, "records": [ { "name": "transformer.h.6.mixer.qkv_proj.weight", "shape": [ 9216, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 56623104, "byteOffset": 0 } ], "md5sum": "cfff45bedd5d3f9c1acdec8706ddfd12" }, { "dataPath": "params_shard_117.bin", "format": "raw-shard", "nbytes": 50331648, "records": [ { "name": "transformer.h.7.mlp.down_proj.weight", "shape": [ 3072, 8192 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 50331648, "byteOffset": 0 } ], "md5sum": "7a908227d9ab71d3e4f605e44a4c0b84" }, { "dataPath": "params_shard_118.bin", "format": "raw-shard", "nbytes": 100663296, "records": [ { "name": "transformer.h.7.mlp.gate_up_proj.weight", "shape": [ 16384, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 100663296, "byteOffset": 0 } ], "md5sum": "a63968773cbb992bdeb65f804200f966" }, { "dataPath": "params_shard_119.bin", "format": "raw-shard", "nbytes": 18874368, "records": [ { "name": "transformer.h.7.mixer.out_proj.weight", "shape": [ 3072, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 18874368, "byteOffset": 0 } ], "md5sum": "09a77c5c5de700053b2fd325baee2e3e" }, { "dataPath": "params_shard_120.bin", "format": "raw-shard", "nbytes": 56623104, "records": [ { "name": "transformer.h.7.mixer.qkv_proj.weight", "shape": [ 9216, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 56623104, "byteOffset": 0 } ], "md5sum": "1331913d0a139cc0ea7f3c2ef19a1854" }, { "dataPath": "params_shard_121.bin", "format": "raw-shard", "nbytes": 50331648, "records": [ { "name": "transformer.h.8.mlp.down_proj.weight", "shape": [ 3072, 8192 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 50331648, "byteOffset": 0 } ], "md5sum": "d0efc02d3f2d395960cc59a6a29941dd" }, { "dataPath": "params_shard_122.bin", "format": "raw-shard", "nbytes": 100663296, "records": [ { "name": "transformer.h.8.mlp.gate_up_proj.weight", "shape": [ 16384, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 100663296, "byteOffset": 0 } ], "md5sum": "515287320d750ddef90a2c26d59b45cc" }, { "dataPath": "params_shard_123.bin", "format": "raw-shard", "nbytes": 18874368, "records": [ { "name": "transformer.h.8.mixer.out_proj.weight", "shape": [ 3072, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 18874368, "byteOffset": 0 } ], "md5sum": "48386e64e7bc21c2c6d9ec1efc5d348c" }, { "dataPath": "params_shard_124.bin", "format": "raw-shard", "nbytes": 56623104, "records": [ { "name": "transformer.h.8.mixer.qkv_proj.weight", "shape": [ 9216, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 56623104, "byteOffset": 0 } ], "md5sum": "fcbd7dd1b3d229bbe556ed019dbb3272" }, { "dataPath": "params_shard_125.bin", "format": "raw-shard", "nbytes": 50331648, "records": [ { "name": "transformer.h.9.mlp.down_proj.weight", "shape": [ 3072, 8192 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 50331648, "byteOffset": 0 } ], "md5sum": "d46cfb6c6df8baa7173adcb195a45609" }, { "dataPath": "params_shard_126.bin", "format": "raw-shard", "nbytes": 100663296, "records": [ { "name": "transformer.h.9.mlp.gate_up_proj.weight", "shape": [ 16384, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 100663296, "byteOffset": 0 } ], "md5sum": "1e6c430b422fc400e63d33c45f7bd44b" }, { "dataPath": "params_shard_127.bin", "format": "raw-shard", "nbytes": 18874368, "records": [ { "name": "transformer.h.9.mixer.out_proj.weight", "shape": [ 3072, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 18874368, "byteOffset": 0 } ], "md5sum": "48d9d3fdff5e72ecf1569b8eec89f3e3" }, { "dataPath": "params_shard_128.bin", "format": "raw-shard", "nbytes": 56623104, "records": [ { "name": "transformer.h.9.mixer.qkv_proj.weight", "shape": [ 9216, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 56623104, "byteOffset": 0 } ], "md5sum": "d80713d97db0fe2e884ae7bc0d611235" }, { "dataPath": "params_shard_129.bin", "format": "raw-shard", "nbytes": 19273728, "records": [ { "name": "transformer.h.21.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 0 }, { "name": "transformer.h.21.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 6144 }, { "name": "transformer.h.22.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 12288 }, { "name": "transformer.h.22.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 18432 }, { "name": "transformer.h.22.mixer.out_proj.weight", "shape": [ 3072, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 18874368, "byteOffset": 24576 }, { "name": "transformer.h.23.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 18898944 }, { "name": "transformer.h.23.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 18905088 }, { "name": "transformer.h.24.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 18911232 }, { "name": "transformer.h.24.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 18917376 }, { "name": "transformer.h.25.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 18923520 }, { "name": "transformer.h.25.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 18929664 }, { "name": "transformer.h.26.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 18935808 }, { "name": "transformer.h.26.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 18941952 }, { "name": "transformer.h.27.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 18948096 }, { "name": "transformer.h.27.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 18954240 }, { "name": "transformer.h.28.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 18960384 }, { "name": "transformer.h.28.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 18966528 }, { "name": "transformer.h.29.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 18972672 }, { "name": "transformer.h.29.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 18978816 }, { "name": "transformer.h.30.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 18984960 }, { "name": "transformer.h.30.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 18991104 }, { "name": "transformer.h.31.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 18997248 }, { "name": "transformer.h.31.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 19003392 }, { "name": "transformer.norm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 19009536 }, { "name": "transformer.h.0.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 19015680 }, { "name": "transformer.h.0.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 19021824 }, { "name": "transformer.h.1.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 19027968 }, { "name": "transformer.h.1.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 19034112 }, { "name": "transformer.h.10.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 19040256 }, { "name": "transformer.h.10.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 19046400 }, { "name": "transformer.h.11.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 19052544 }, { "name": "transformer.h.11.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 19058688 }, { "name": "transformer.h.12.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 19064832 }, { "name": "transformer.h.12.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 19070976 }, { "name": "transformer.h.13.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 19077120 }, { "name": "transformer.h.13.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 19083264 }, { "name": "transformer.h.14.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 19089408 }, { "name": "transformer.h.14.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 19095552 }, { "name": "transformer.h.15.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 19101696 }, { "name": "transformer.h.15.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 19107840 }, { "name": "transformer.h.16.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 19113984 }, { "name": "transformer.h.16.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 19120128 }, { "name": "transformer.h.17.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 19126272 }, { "name": "transformer.h.17.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 19132416 }, { "name": "transformer.h.18.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 19138560 }, { "name": "transformer.h.18.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 19144704 }, { "name": "transformer.h.19.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 19150848 }, { "name": "transformer.h.19.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 19156992 }, { "name": "transformer.h.2.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 19163136 }, { "name": "transformer.h.2.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 19169280 }, { "name": "transformer.h.20.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 19175424 }, { "name": "transformer.h.20.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 19181568 }, { "name": "transformer.h.3.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 19187712 }, { "name": "transformer.h.3.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 19193856 }, { "name": "transformer.h.4.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 19200000 }, { "name": "transformer.h.4.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 19206144 }, { "name": "transformer.h.5.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 19212288 }, { "name": "transformer.h.5.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 19218432 }, { "name": "transformer.h.6.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 19224576 }, { "name": "transformer.h.6.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 19230720 }, { "name": "transformer.h.7.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 19236864 }, { "name": "transformer.h.7.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 19243008 }, { "name": "transformer.h.8.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 19249152 }, { "name": "transformer.h.8.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 19255296 }, { "name": "transformer.h.9.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 19261440 }, { "name": "transformer.h.9.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 19267584 } ], "md5sum": "7d90cc7bee5c8a5aa80ac6fca9f03b61" } ] }