internlm2_5-7b-chat-q4f16_1-MLC / ndarray-cache.json
riczhou's picture
Upload folder using huggingface_hub
a776767 verified
{
"metadata": {
"ParamSize": 325,
"ParamBytes": 4352843776.0,
"BitsPerParam": 4.500395693373896
},
"records": [
{
"dataPath": "params_shard_0.bin",
"format": "raw-shard",
"nbytes": 58720256,
"records": [
{
"name": "model.layers.0.feed_forward.gate_up_proj.q_weight",
"shape": [
28672,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 58720256,
"byteOffset": 0
}
],
"md5sum": "a8fb734318d546cb1d3adc84efa8fea8"
},
{
"dataPath": "params_shard_1.bin",
"format": "raw-shard",
"nbytes": 29360128,
"records": [
{
"name": "model.layers.0.feed_forward.w2.q_weight",
"shape": [
4096,
1792
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 29360128,
"byteOffset": 0
}
],
"md5sum": "11aafd3d6e604c73a3e1c4cee09bc00a"
},
{
"dataPath": "params_shard_2.bin",
"format": "raw-shard",
"nbytes": 30941184,
"records": [
{
"name": "model.layers.0.attention.wo.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 0
},
{
"name": "model.layers.0.attention.wo.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 8388608
},
{
"name": "model.layers.0.attention.wqkv.q_weight",
"shape": [
6144,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 12582912,
"byteOffset": 9437184
},
{
"name": "model.layers.0.attention.wqkv.q_scale",
"shape": [
6144,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1572864,
"byteOffset": 22020096
},
{
"name": "model.layers.0.attention_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 23592960
},
{
"name": "model.layers.0.feed_forward.gate_up_proj.q_scale",
"shape": [
28672,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 7340032,
"byteOffset": 23601152
}
],
"md5sum": "7d1208035674e722b485c21fdd5bf978"
},
{
"dataPath": "params_shard_3.bin",
"format": "raw-shard",
"nbytes": 58720256,
"records": [
{
"name": "model.layers.1.feed_forward.gate_up_proj.q_weight",
"shape": [
28672,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 58720256,
"byteOffset": 0
}
],
"md5sum": "8ac161c26e3edf50db7875ab4a03c1cf"
},
{
"dataPath": "params_shard_4.bin",
"format": "raw-shard",
"nbytes": 27279360,
"records": [
{
"name": "model.layers.0.feed_forward.w2.q_scale",
"shape": [
4096,
448
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3670016,
"byteOffset": 0
},
{
"name": "model.layers.0.ffn_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 3670016
},
{
"name": "model.layers.1.attention.wo.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 3678208
},
{
"name": "model.layers.1.attention.wo.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 12066816
},
{
"name": "model.layers.1.attention.wqkv.q_weight",
"shape": [
6144,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 12582912,
"byteOffset": 13115392
},
{
"name": "model.layers.1.attention.wqkv.q_scale",
"shape": [
6144,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1572864,
"byteOffset": 25698304
},
{
"name": "model.layers.1.attention_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 27271168
}
],
"md5sum": "01e5741b604a75b8ad03bc505222b39a"
},
{
"dataPath": "params_shard_5.bin",
"format": "raw-shard",
"nbytes": 29360128,
"records": [
{
"name": "model.layers.1.feed_forward.w2.q_weight",
"shape": [
4096,
1792
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 29360128,
"byteOffset": 0
}
],
"md5sum": "54b5b9c66896b86e0258a9b0a119bbe2"
},
{
"dataPath": "params_shard_6.bin",
"format": "raw-shard",
"nbytes": 33038336,
"records": [
{
"name": "model.layers.1.feed_forward.gate_up_proj.q_scale",
"shape": [
28672,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 7340032,
"byteOffset": 0
},
{
"name": "model.layers.1.feed_forward.w2.q_scale",
"shape": [
4096,
448
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3670016,
"byteOffset": 7340032
},
{
"name": "model.layers.1.ffn_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 11010048
},
{
"name": "model.layers.2.attention.wo.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 11018240
},
{
"name": "model.layers.2.attention.wo.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 19406848
},
{
"name": "model.layers.2.attention.wqkv.q_weight",
"shape": [
6144,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 12582912,
"byteOffset": 20455424
}
],
"md5sum": "fbe69cb7f64520b292a6976551fc787b"
},
{
"dataPath": "params_shard_7.bin",
"format": "raw-shard",
"nbytes": 58720256,
"records": [
{
"name": "model.layers.2.feed_forward.gate_up_proj.q_weight",
"shape": [
28672,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 58720256,
"byteOffset": 0
}
],
"md5sum": "91585ac15188fb577aa8b16459eca13a"
},
{
"dataPath": "params_shard_8.bin",
"format": "raw-shard",
"nbytes": 189530112,
"records": [
{
"name": "model.tok_embeddings.q_weight",
"shape": [
92544,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 189530112,
"byteOffset": 0
}
],
"md5sum": "e7d7a62dad1fa85cb9458952b1893a63"
},
{
"dataPath": "params_shard_9.bin",
"format": "raw-shard",
"nbytes": 32604160,
"records": [
{
"name": "model.layers.2.attention.wqkv.q_scale",
"shape": [
6144,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1572864,
"byteOffset": 0
},
{
"name": "model.layers.2.feed_forward.gate_up_proj.q_scale",
"shape": [
28672,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 7340032,
"byteOffset": 1572864
},
{
"name": "model.tok_embeddings.q_scale",
"shape": [
92544,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 23691264,
"byteOffset": 8912896
}
],
"md5sum": "409a08c82128942d08255311592a5c2d"
},
{
"dataPath": "params_shard_10.bin",
"format": "raw-shard",
"nbytes": 58720256,
"records": [
{
"name": "model.layers.10.feed_forward.gate_up_proj.q_weight",
"shape": [
28672,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 58720256,
"byteOffset": 0
}
],
"md5sum": "192cbd81ff510ac76fa8e43f4674bc8d"
},
{
"dataPath": "params_shard_11.bin",
"format": "raw-shard",
"nbytes": 29360128,
"records": [
{
"name": "model.layers.10.feed_forward.w2.q_weight",
"shape": [
4096,
1792
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 29360128,
"byteOffset": 0
}
],
"md5sum": "0bf26b9b3421eaa1851a9649615a384b"
},
{
"dataPath": "params_shard_12.bin",
"format": "raw-shard",
"nbytes": 30941184,
"records": [
{
"name": "model.layers.10.attention.wo.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 0
},
{
"name": "model.layers.10.attention.wo.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 8388608
},
{
"name": "model.layers.10.attention.wqkv.q_weight",
"shape": [
6144,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 12582912,
"byteOffset": 9437184
},
{
"name": "model.layers.10.attention.wqkv.q_scale",
"shape": [
6144,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1572864,
"byteOffset": 22020096
},
{
"name": "model.layers.10.attention_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 23592960
},
{
"name": "model.layers.10.feed_forward.gate_up_proj.q_scale",
"shape": [
28672,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 7340032,
"byteOffset": 23601152
}
],
"md5sum": "c1924047cd3441baaaa01207f098e97b"
},
{
"dataPath": "params_shard_13.bin",
"format": "raw-shard",
"nbytes": 58720256,
"records": [
{
"name": "model.layers.11.feed_forward.gate_up_proj.q_weight",
"shape": [
28672,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 58720256,
"byteOffset": 0
}
],
"md5sum": "e3cb549ad53b4bdec29112cb49c0514c"
},
{
"dataPath": "params_shard_14.bin",
"format": "raw-shard",
"nbytes": 27271168,
"records": [
{
"name": "model.layers.10.feed_forward.w2.q_scale",
"shape": [
4096,
448
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3670016,
"byteOffset": 0
},
{
"name": "model.layers.10.ffn_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 3670016
},
{
"name": "model.layers.11.attention.wo.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 3678208
},
{
"name": "model.layers.11.attention.wo.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 12066816
},
{
"name": "model.layers.11.attention.wqkv.q_weight",
"shape": [
6144,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 12582912,
"byteOffset": 13115392
},
{
"name": "model.layers.11.attention.wqkv.q_scale",
"shape": [
6144,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1572864,
"byteOffset": 25698304
}
],
"md5sum": "a4c31bfe1c0f28ac3fa5d74ae67fb37d"
},
{
"dataPath": "params_shard_15.bin",
"format": "raw-shard",
"nbytes": 58720256,
"records": [
{
"name": "model.layers.7.feed_forward.gate_up_proj.q_weight",
"shape": [
28672,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 58720256,
"byteOffset": 0
}
],
"md5sum": "bc8744e922ff9a489e34d46ecf002dbf"
},
{
"dataPath": "params_shard_16.bin",
"format": "raw-shard",
"nbytes": 29360128,
"records": [
{
"name": "model.layers.7.feed_forward.w2.q_weight",
"shape": [
4096,
1792
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 29360128,
"byteOffset": 0
}
],
"md5sum": "92e98d6b7029bcbda46a6bbd4b8471f1"
},
{
"dataPath": "params_shard_17.bin",
"format": "raw-shard",
"nbytes": 27803648,
"records": [
{
"name": "model.layers.11.feed_forward.gate_up_proj.q_scale",
"shape": [
28672,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 7340032,
"byteOffset": 0
},
{
"name": "model.layers.7.attention_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 7340032
},
{
"name": "model.layers.7.feed_forward.gate_up_proj.q_scale",
"shape": [
28672,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 7340032,
"byteOffset": 7348224
},
{
"name": "model.layers.7.feed_forward.w2.q_scale",
"shape": [
4096,
448
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3670016,
"byteOffset": 14688256
},
{
"name": "model.layers.7.ffn_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 18358272
},
{
"name": "model.layers.8.attention.wo.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 18366464
},
{
"name": "model.layers.8.attention.wo.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 26755072
}
],
"md5sum": "3a0f09bd2b7fc14bc1d203c0cd8331bc"
},
{
"dataPath": "params_shard_18.bin",
"format": "raw-shard",
"nbytes": 58720256,
"records": [
{
"name": "model.layers.8.feed_forward.gate_up_proj.q_weight",
"shape": [
28672,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 58720256,
"byteOffset": 0
}
],
"md5sum": "837ac5dfec2030cd105544724b817333"
},
{
"dataPath": "params_shard_19.bin",
"format": "raw-shard",
"nbytes": 29360128,
"records": [
{
"name": "model.layers.8.feed_forward.w2.q_weight",
"shape": [
4096,
1792
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 29360128,
"byteOffset": 0
}
],
"md5sum": "0911c8a622086d4ac104e0f03f779646"
},
{
"dataPath": "params_shard_20.bin",
"format": "raw-shard",
"nbytes": 25182208,
"records": [
{
"name": "model.layers.8.attention.wqkv.q_weight",
"shape": [
6144,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 12582912,
"byteOffset": 0
},
{
"name": "model.layers.8.attention.wqkv.q_scale",
"shape": [
6144,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1572864,
"byteOffset": 12582912
},
{
"name": "model.layers.8.attention_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 14155776
},
{
"name": "model.layers.8.feed_forward.gate_up_proj.q_scale",
"shape": [
28672,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 7340032,
"byteOffset": 14163968
},
{
"name": "model.layers.8.feed_forward.w2.q_scale",
"shape": [
4096,
448
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3670016,
"byteOffset": 21504000
},
{
"name": "model.layers.8.ffn_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 25174016
}
],
"md5sum": "4411c751de9da387aa2cf997ed769c49"
},
{
"dataPath": "params_shard_21.bin",
"format": "raw-shard",
"nbytes": 58720256,
"records": [
{
"name": "model.layers.9.feed_forward.gate_up_proj.q_weight",
"shape": [
28672,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 58720256,
"byteOffset": 0
}
],
"md5sum": "b943df398efd5086944ab8eb3d200fb6"
},
{
"dataPath": "params_shard_22.bin",
"format": "raw-shard",
"nbytes": 29360128,
"records": [
{
"name": "model.layers.9.feed_forward.w2.q_weight",
"shape": [
4096,
1792
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 29360128,
"byteOffset": 0
}
],
"md5sum": "1da237c2d1d5dbc3cdf24b0a4abd9a46"
},
{
"dataPath": "params_shard_23.bin",
"format": "raw-shard",
"nbytes": 30941184,
"records": [
{
"name": "model.layers.9.attention.wo.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 0
},
{
"name": "model.layers.9.attention.wo.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 8388608
},
{
"name": "model.layers.9.attention.wqkv.q_weight",
"shape": [
6144,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 12582912,
"byteOffset": 9437184
},
{
"name": "model.layers.9.attention.wqkv.q_scale",
"shape": [
6144,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1572864,
"byteOffset": 22020096
},
{
"name": "model.layers.9.attention_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 23592960
},
{
"name": "model.layers.9.feed_forward.gate_up_proj.q_scale",
"shape": [
28672,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 7340032,
"byteOffset": 23601152
}
],
"md5sum": "e8d363c062da59ffa22c7aca32437794"
},
{
"dataPath": "params_shard_24.bin",
"format": "raw-shard",
"nbytes": 33046528,
"records": [
{
"name": "model.layers.9.feed_forward.w2.q_scale",
"shape": [
4096,
448
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3670016,
"byteOffset": 0
},
{
"name": "model.layers.9.ffn_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 3670016
},
{
"name": "model.layers.11.attention_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 3678208
},
{
"name": "model.layers.11.feed_forward.w2.q_weight",
"shape": [
4096,
1792
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 29360128,
"byteOffset": 3686400
}
],
"md5sum": "fe3c4514539fb80dd0108944faa9a9b3"
},
{
"dataPath": "params_shard_25.bin",
"format": "raw-shard",
"nbytes": 58720256,
"records": [
{
"name": "model.layers.12.feed_forward.gate_up_proj.q_weight",
"shape": [
28672,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 58720256,
"byteOffset": 0
}
],
"md5sum": "bd1e3a374e79c62ba8196660829f3664"
},
{
"dataPath": "params_shard_26.bin",
"format": "raw-shard",
"nbytes": 27279360,
"records": [
{
"name": "model.layers.11.feed_forward.w2.q_scale",
"shape": [
4096,
448
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3670016,
"byteOffset": 0
},
{
"name": "model.layers.11.ffn_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 3670016
},
{
"name": "model.layers.12.attention.wo.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 3678208
},
{
"name": "model.layers.12.attention.wo.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 12066816
},
{
"name": "model.layers.12.attention.wqkv.q_weight",
"shape": [
6144,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 12582912,
"byteOffset": 13115392
},
{
"name": "model.layers.12.attention.wqkv.q_scale",
"shape": [
6144,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1572864,
"byteOffset": 25698304
},
{
"name": "model.layers.12.attention_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 27271168
}
],
"md5sum": "3dc074fabf3e7162bef8010448047ea8"
},
{
"dataPath": "params_shard_27.bin",
"format": "raw-shard",
"nbytes": 29360128,
"records": [
{
"name": "model.layers.12.feed_forward.w2.q_weight",
"shape": [
4096,
1792
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 29360128,
"byteOffset": 0
}
],
"md5sum": "fb43369281fc809964a041b466ac613e"
},
{
"dataPath": "params_shard_28.bin",
"format": "raw-shard",
"nbytes": 33038336,
"records": [
{
"name": "model.layers.12.feed_forward.gate_up_proj.q_scale",
"shape": [
28672,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 7340032,
"byteOffset": 0
},
{
"name": "model.layers.12.feed_forward.w2.q_scale",
"shape": [
4096,
448
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3670016,
"byteOffset": 7340032
},
{
"name": "model.layers.12.ffn_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 11010048
},
{
"name": "model.layers.13.attention.wo.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 11018240
},
{
"name": "model.layers.13.attention.wo.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 19406848
},
{
"name": "model.layers.13.attention.wqkv.q_weight",
"shape": [
6144,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 12582912,
"byteOffset": 20455424
}
],
"md5sum": "6abc281de58322d13e4e1ad6cd8fd97a"
},
{
"dataPath": "params_shard_29.bin",
"format": "raw-shard",
"nbytes": 58720256,
"records": [
{
"name": "model.layers.13.feed_forward.gate_up_proj.q_weight",
"shape": [
28672,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 58720256,
"byteOffset": 0
}
],
"md5sum": "1e988abdaa623d6cd505559555da5b2b"
},
{
"dataPath": "params_shard_30.bin",
"format": "raw-shard",
"nbytes": 29360128,
"records": [
{
"name": "model.layers.13.feed_forward.w2.q_weight",
"shape": [
4096,
1792
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 29360128,
"byteOffset": 0
}
],
"md5sum": "8f2c30b448bf566a6e2fb4b57244bf69"
},
{
"dataPath": "params_shard_31.bin",
"format": "raw-shard",
"nbytes": 22036480,
"records": [
{
"name": "model.layers.13.attention.wqkv.q_scale",
"shape": [
6144,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1572864,
"byteOffset": 0
},
{
"name": "model.layers.13.attention_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 1572864
},
{
"name": "model.layers.13.feed_forward.gate_up_proj.q_scale",
"shape": [
28672,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 7340032,
"byteOffset": 1581056
},
{
"name": "model.layers.13.feed_forward.w2.q_scale",
"shape": [
4096,
448
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3670016,
"byteOffset": 8921088
},
{
"name": "model.layers.13.ffn_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 12591104
},
{
"name": "model.layers.14.attention.wo.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 12599296
},
{
"name": "model.layers.14.attention.wo.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 20987904
}
],
"md5sum": "30a1049fe58440140adad1d658c27854"
},
{
"dataPath": "params_shard_32.bin",
"format": "raw-shard",
"nbytes": 58720256,
"records": [
{
"name": "model.layers.14.feed_forward.gate_up_proj.q_weight",
"shape": [
28672,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 58720256,
"byteOffset": 0
}
],
"md5sum": "f2a07100a63bc206de2095536742397a"
},
{
"dataPath": "params_shard_33.bin",
"format": "raw-shard",
"nbytes": 29360128,
"records": [
{
"name": "model.layers.14.feed_forward.w2.q_weight",
"shape": [
4096,
1792
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 29360128,
"byteOffset": 0
}
],
"md5sum": "16ca036658657e6c1bbca703d5eebafe"
},
{
"dataPath": "params_shard_34.bin",
"format": "raw-shard",
"nbytes": 25182208,
"records": [
{
"name": "model.layers.14.attention.wqkv.q_weight",
"shape": [
6144,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 12582912,
"byteOffset": 0
},
{
"name": "model.layers.14.attention.wqkv.q_scale",
"shape": [
6144,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1572864,
"byteOffset": 12582912
},
{
"name": "model.layers.14.attention_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 14155776
},
{
"name": "model.layers.14.feed_forward.gate_up_proj.q_scale",
"shape": [
28672,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 7340032,
"byteOffset": 14163968
},
{
"name": "model.layers.14.feed_forward.w2.q_scale",
"shape": [
4096,
448
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3670016,
"byteOffset": 21504000
},
{
"name": "model.layers.14.ffn_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 25174016
}
],
"md5sum": "d4bbe7eedd99925a9a278b5703061142"
},
{
"dataPath": "params_shard_35.bin",
"format": "raw-shard",
"nbytes": 58720256,
"records": [
{
"name": "model.layers.15.feed_forward.gate_up_proj.q_weight",
"shape": [
28672,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 58720256,
"byteOffset": 0
}
],
"md5sum": "6fbae9e7077d1bd4404fa341ddf4db07"
},
{
"dataPath": "params_shard_36.bin",
"format": "raw-shard",
"nbytes": 29360128,
"records": [
{
"name": "model.layers.15.feed_forward.w2.q_weight",
"shape": [
4096,
1792
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 29360128,
"byteOffset": 0
}
],
"md5sum": "d2d4d9b3392d3d9547cf57991a29d4e8"
},
{
"dataPath": "params_shard_37.bin",
"format": "raw-shard",
"nbytes": 30941184,
"records": [
{
"name": "model.layers.15.attention.wo.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 0
},
{
"name": "model.layers.15.attention.wo.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 8388608
},
{
"name": "model.layers.15.attention.wqkv.q_weight",
"shape": [
6144,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 12582912,
"byteOffset": 9437184
},
{
"name": "model.layers.15.attention.wqkv.q_scale",
"shape": [
6144,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1572864,
"byteOffset": 22020096
},
{
"name": "model.layers.15.attention_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 23592960
},
{
"name": "model.layers.15.feed_forward.gate_up_proj.q_scale",
"shape": [
28672,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 7340032,
"byteOffset": 23601152
}
],
"md5sum": "88f0b0c312b921135223e3109a0b1701"
},
{
"dataPath": "params_shard_38.bin",
"format": "raw-shard",
"nbytes": 58720256,
"records": [
{
"name": "model.layers.16.feed_forward.gate_up_proj.q_weight",
"shape": [
28672,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 58720256,
"byteOffset": 0
}
],
"md5sum": "e654b3d06fe3a7eb221d4d9affc5bf7f"
},
{
"dataPath": "params_shard_39.bin",
"format": "raw-shard",
"nbytes": 27279360,
"records": [
{
"name": "model.layers.15.feed_forward.w2.q_scale",
"shape": [
4096,
448
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3670016,
"byteOffset": 0
},
{
"name": "model.layers.15.ffn_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 3670016
},
{
"name": "model.layers.16.attention.wo.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 3678208
},
{
"name": "model.layers.16.attention.wo.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 12066816
},
{
"name": "model.layers.16.attention.wqkv.q_weight",
"shape": [
6144,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 12582912,
"byteOffset": 13115392
},
{
"name": "model.layers.16.attention.wqkv.q_scale",
"shape": [
6144,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1572864,
"byteOffset": 25698304
},
{
"name": "model.layers.16.attention_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 27271168
}
],
"md5sum": "8f45e9759239d727714c0d6abe2f5cce"
},
{
"dataPath": "params_shard_40.bin",
"format": "raw-shard",
"nbytes": 29360128,
"records": [
{
"name": "model.layers.16.feed_forward.w2.q_weight",
"shape": [
4096,
1792
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 29360128,
"byteOffset": 0
}
],
"md5sum": "a64a9245e009a9fff6821187f723634d"
},
{
"dataPath": "params_shard_41.bin",
"format": "raw-shard",
"nbytes": 33038336,
"records": [
{
"name": "model.layers.16.feed_forward.gate_up_proj.q_scale",
"shape": [
28672,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 7340032,
"byteOffset": 0
},
{
"name": "model.layers.16.feed_forward.w2.q_scale",
"shape": [
4096,
448
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3670016,
"byteOffset": 7340032
},
{
"name": "model.layers.16.ffn_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 11010048
},
{
"name": "model.layers.17.attention.wo.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 11018240
},
{
"name": "model.layers.17.attention.wo.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 19406848
},
{
"name": "model.layers.17.attention.wqkv.q_weight",
"shape": [
6144,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 12582912,
"byteOffset": 20455424
}
],
"md5sum": "c59678336125e6ec68a7fe05ceefc12e"
},
{
"dataPath": "params_shard_42.bin",
"format": "raw-shard",
"nbytes": 58720256,
"records": [
{
"name": "model.layers.17.feed_forward.gate_up_proj.q_weight",
"shape": [
28672,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 58720256,
"byteOffset": 0
}
],
"md5sum": "9ffde77b2c14e3dcaa58998cf3422057"
},
{
"dataPath": "params_shard_43.bin",
"format": "raw-shard",
"nbytes": 29360128,
"records": [
{
"name": "model.layers.17.feed_forward.w2.q_weight",
"shape": [
4096,
1792
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 29360128,
"byteOffset": 0
}
],
"md5sum": "12220e5f1096095605445254b5260398"
},
{
"dataPath": "params_shard_44.bin",
"format": "raw-shard",
"nbytes": 22036480,
"records": [
{
"name": "model.layers.17.attention.wqkv.q_scale",
"shape": [
6144,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1572864,
"byteOffset": 0
},
{
"name": "model.layers.17.attention_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 1572864
},
{
"name": "model.layers.17.feed_forward.gate_up_proj.q_scale",
"shape": [
28672,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 7340032,
"byteOffset": 1581056
},
{
"name": "model.layers.17.feed_forward.w2.q_scale",
"shape": [
4096,
448
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3670016,
"byteOffset": 8921088
},
{
"name": "model.layers.17.ffn_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 12591104
},
{
"name": "model.layers.18.attention.wo.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 12599296
},
{
"name": "model.layers.18.attention.wo.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 20987904
}
],
"md5sum": "9721f91852a1eb0e1b5b70cd74d42ee1"
},
{
"dataPath": "params_shard_45.bin",
"format": "raw-shard",
"nbytes": 58720256,
"records": [
{
"name": "model.layers.18.feed_forward.gate_up_proj.q_weight",
"shape": [
28672,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 58720256,
"byteOffset": 0
}
],
"md5sum": "570cff552709d5682597828c94e0c63d"
},
{
"dataPath": "params_shard_46.bin",
"format": "raw-shard",
"nbytes": 29360128,
"records": [
{
"name": "model.layers.18.feed_forward.w2.q_weight",
"shape": [
4096,
1792
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 29360128,
"byteOffset": 0
}
],
"md5sum": "99bbf413150630a3d66ad3e88320e597"
},
{
"dataPath": "params_shard_47.bin",
"format": "raw-shard",
"nbytes": 25182208,
"records": [
{
"name": "model.layers.18.attention.wqkv.q_weight",
"shape": [
6144,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 12582912,
"byteOffset": 0
},
{
"name": "model.layers.18.attention.wqkv.q_scale",
"shape": [
6144,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1572864,
"byteOffset": 12582912
},
{
"name": "model.layers.18.attention_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 14155776
},
{
"name": "model.layers.18.feed_forward.gate_up_proj.q_scale",
"shape": [
28672,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 7340032,
"byteOffset": 14163968
},
{
"name": "model.layers.18.feed_forward.w2.q_scale",
"shape": [
4096,
448
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3670016,
"byteOffset": 21504000
},
{
"name": "model.layers.18.ffn_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 25174016
}
],
"md5sum": "56c371d3eb7cc3f8c27bff9e77e6b265"
},
{
"dataPath": "params_shard_48.bin",
"format": "raw-shard",
"nbytes": 58720256,
"records": [
{
"name": "model.layers.19.feed_forward.gate_up_proj.q_weight",
"shape": [
28672,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 58720256,
"byteOffset": 0
}
],
"md5sum": "b32aa22e0c8821187e69d0ad7aa100e6"
},
{
"dataPath": "params_shard_49.bin",
"format": "raw-shard",
"nbytes": 29360128,
"records": [
{
"name": "model.layers.19.feed_forward.w2.q_weight",
"shape": [
4096,
1792
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 29360128,
"byteOffset": 0
}
],
"md5sum": "258aa6da64b19b93cf89dc77cacc9962"
},
{
"dataPath": "params_shard_50.bin",
"format": "raw-shard",
"nbytes": 30941184,
"records": [
{
"name": "model.layers.19.attention.wo.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 0
},
{
"name": "model.layers.19.attention.wo.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 8388608
},
{
"name": "model.layers.19.attention.wqkv.q_weight",
"shape": [
6144,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 12582912,
"byteOffset": 9437184
},
{
"name": "model.layers.19.attention.wqkv.q_scale",
"shape": [
6144,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1572864,
"byteOffset": 22020096
},
{
"name": "model.layers.19.attention_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 23592960
},
{
"name": "model.layers.19.feed_forward.gate_up_proj.q_scale",
"shape": [
28672,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 7340032,
"byteOffset": 23601152
}
],
"md5sum": "8f60926616686027d6370ad90f1c3731"
},
{
"dataPath": "params_shard_51.bin",
"format": "raw-shard",
"nbytes": 58720256,
"records": [
{
"name": "model.layers.20.feed_forward.gate_up_proj.q_weight",
"shape": [
28672,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 58720256,
"byteOffset": 0
}
],
"md5sum": "9c26f34c662cd511c4ee08f31da886c3"
},
{
"dataPath": "params_shard_52.bin",
"format": "raw-shard",
"nbytes": 27271168,
"records": [
{
"name": "model.layers.19.feed_forward.w2.q_scale",
"shape": [
4096,
448
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3670016,
"byteOffset": 0
},
{
"name": "model.layers.19.ffn_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 3670016
},
{
"name": "model.layers.20.attention.wo.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 3678208
},
{
"name": "model.layers.20.attention.wo.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 12066816
},
{
"name": "model.layers.20.attention.wqkv.q_weight",
"shape": [
6144,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 12582912,
"byteOffset": 13115392
},
{
"name": "model.layers.20.attention.wqkv.q_scale",
"shape": [
6144,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1572864,
"byteOffset": 25698304
}
],
"md5sum": "af986a801ca1b805e12155d65e440831"
},
{
"dataPath": "params_shard_53.bin",
"format": "raw-shard",
"nbytes": 29360128,
"records": [
{
"name": "model.layers.2.feed_forward.w2.q_weight",
"shape": [
4096,
1792
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 29360128,
"byteOffset": 0
}
],
"md5sum": "70d0eebfad7f407be34e6f73f7ab257e"
},
{
"dataPath": "params_shard_54.bin",
"format": "raw-shard",
"nbytes": 33046528,
"records": [
{
"name": "model.layers.20.feed_forward.gate_up_proj.q_scale",
"shape": [
28672,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 7340032,
"byteOffset": 0
},
{
"name": "model.layers.2.attention_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 7340032
},
{
"name": "model.layers.2.feed_forward.w2.q_scale",
"shape": [
4096,
448
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3670016,
"byteOffset": 7348224
},
{
"name": "model.layers.2.ffn_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 11018240
},
{
"name": "model.layers.3.attention.wo.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 11026432
},
{
"name": "model.layers.3.attention.wo.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 19415040
},
{
"name": "model.layers.3.attention.wqkv.q_weight",
"shape": [
6144,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 12582912,
"byteOffset": 20463616
}
],
"md5sum": "8d210b95e5c4aeb3d7d612dee3737cd6"
},
{
"dataPath": "params_shard_55.bin",
"format": "raw-shard",
"nbytes": 58720256,
"records": [
{
"name": "model.layers.3.feed_forward.gate_up_proj.q_weight",
"shape": [
28672,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 58720256,
"byteOffset": 0
}
],
"md5sum": "df6a5411cfe831123e9aabb1b8dabab2"
},
{
"dataPath": "params_shard_56.bin",
"format": "raw-shard",
"nbytes": 29360128,
"records": [
{
"name": "model.layers.3.feed_forward.w2.q_weight",
"shape": [
4096,
1792
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 29360128,
"byteOffset": 0
}
],
"md5sum": "386d0d1fdfc0684c9e94a32734b2a8b5"
},
{
"dataPath": "params_shard_57.bin",
"format": "raw-shard",
"nbytes": 22036480,
"records": [
{
"name": "model.layers.3.attention.wqkv.q_scale",
"shape": [
6144,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1572864,
"byteOffset": 0
},
{
"name": "model.layers.3.attention_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 1572864
},
{
"name": "model.layers.3.feed_forward.gate_up_proj.q_scale",
"shape": [
28672,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 7340032,
"byteOffset": 1581056
},
{
"name": "model.layers.3.feed_forward.w2.q_scale",
"shape": [
4096,
448
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3670016,
"byteOffset": 8921088
},
{
"name": "model.layers.3.ffn_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 12591104
},
{
"name": "model.layers.4.attention.wo.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 12599296
},
{
"name": "model.layers.4.attention.wo.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 20987904
}
],
"md5sum": "e5e2f150468f0d721120e2afbf789392"
},
{
"dataPath": "params_shard_58.bin",
"format": "raw-shard",
"nbytes": 58720256,
"records": [
{
"name": "model.layers.4.feed_forward.gate_up_proj.q_weight",
"shape": [
28672,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 58720256,
"byteOffset": 0
}
],
"md5sum": "d1f056e5c9efa51f9a4a7af9826a8920"
},
{
"dataPath": "params_shard_59.bin",
"format": "raw-shard",
"nbytes": 29360128,
"records": [
{
"name": "model.layers.4.feed_forward.w2.q_weight",
"shape": [
4096,
1792
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 29360128,
"byteOffset": 0
}
],
"md5sum": "29ec6526ccde32fb8ebf7d41dc41d904"
},
{
"dataPath": "params_shard_60.bin",
"format": "raw-shard",
"nbytes": 25182208,
"records": [
{
"name": "model.layers.4.attention.wqkv.q_weight",
"shape": [
6144,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 12582912,
"byteOffset": 0
},
{
"name": "model.layers.4.attention.wqkv.q_scale",
"shape": [
6144,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1572864,
"byteOffset": 12582912
},
{
"name": "model.layers.4.attention_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 14155776
},
{
"name": "model.layers.4.feed_forward.gate_up_proj.q_scale",
"shape": [
28672,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 7340032,
"byteOffset": 14163968
},
{
"name": "model.layers.4.feed_forward.w2.q_scale",
"shape": [
4096,
448
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3670016,
"byteOffset": 21504000
},
{
"name": "model.layers.4.ffn_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 25174016
}
],
"md5sum": "c5ab182f25a315f95e8221e10d315b6f"
},
{
"dataPath": "params_shard_61.bin",
"format": "raw-shard",
"nbytes": 58720256,
"records": [
{
"name": "model.layers.5.feed_forward.gate_up_proj.q_weight",
"shape": [
28672,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 58720256,
"byteOffset": 0
}
],
"md5sum": "74916aacd1c0fae2a28659c5ca8907ae"
},
{
"dataPath": "params_shard_62.bin",
"format": "raw-shard",
"nbytes": 29360128,
"records": [
{
"name": "model.layers.5.feed_forward.w2.q_weight",
"shape": [
4096,
1792
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 29360128,
"byteOffset": 0
}
],
"md5sum": "72874946269793678f0856213236a032"
},
{
"dataPath": "params_shard_63.bin",
"format": "raw-shard",
"nbytes": 30941184,
"records": [
{
"name": "model.layers.5.attention.wo.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 0
},
{
"name": "model.layers.5.attention.wo.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 8388608
},
{
"name": "model.layers.5.attention.wqkv.q_weight",
"shape": [
6144,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 12582912,
"byteOffset": 9437184
},
{
"name": "model.layers.5.attention.wqkv.q_scale",
"shape": [
6144,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1572864,
"byteOffset": 22020096
},
{
"name": "model.layers.5.attention_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 23592960
},
{
"name": "model.layers.5.feed_forward.gate_up_proj.q_scale",
"shape": [
28672,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 7340032,
"byteOffset": 23601152
}
],
"md5sum": "4296876e2fcd7a10504bffe90e20c79a"
},
{
"dataPath": "params_shard_64.bin",
"format": "raw-shard",
"nbytes": 58720256,
"records": [
{
"name": "model.layers.6.feed_forward.gate_up_proj.q_weight",
"shape": [
28672,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 58720256,
"byteOffset": 0
}
],
"md5sum": "a2169e57f0b56ce3d623ce5f0c05f231"
},
{
"dataPath": "params_shard_65.bin",
"format": "raw-shard",
"nbytes": 27279360,
"records": [
{
"name": "model.layers.5.feed_forward.w2.q_scale",
"shape": [
4096,
448
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3670016,
"byteOffset": 0
},
{
"name": "model.layers.5.ffn_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 3670016
},
{
"name": "model.layers.6.attention.wo.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 3678208
},
{
"name": "model.layers.6.attention.wo.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 12066816
},
{
"name": "model.layers.6.attention.wqkv.q_weight",
"shape": [
6144,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 12582912,
"byteOffset": 13115392
},
{
"name": "model.layers.6.attention.wqkv.q_scale",
"shape": [
6144,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1572864,
"byteOffset": 25698304
},
{
"name": "model.layers.6.attention_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 27271168
}
],
"md5sum": "c72f8b4cba925bbae7deeac026efd5c7"
},
{
"dataPath": "params_shard_66.bin",
"format": "raw-shard",
"nbytes": 29360128,
"records": [
{
"name": "model.layers.6.feed_forward.w2.q_weight",
"shape": [
4096,
1792
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 29360128,
"byteOffset": 0
}
],
"md5sum": "b015ea288ba240e16a0d206ebb61f769"
},
{
"dataPath": "params_shard_67.bin",
"format": "raw-shard",
"nbytes": 33038336,
"records": [
{
"name": "model.layers.6.feed_forward.gate_up_proj.q_scale",
"shape": [
28672,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 7340032,
"byteOffset": 0
},
{
"name": "model.layers.6.feed_forward.w2.q_scale",
"shape": [
4096,
448
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3670016,
"byteOffset": 7340032
},
{
"name": "model.layers.6.ffn_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 11010048
},
{
"name": "model.layers.7.attention.wo.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 11018240
},
{
"name": "model.layers.7.attention.wo.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 19406848
},
{
"name": "model.layers.7.attention.wqkv.q_weight",
"shape": [
6144,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 12582912,
"byteOffset": 20455424
}
],
"md5sum": "768a816dd80a725790e1828505473aa8"
},
{
"dataPath": "params_shard_68.bin",
"format": "raw-shard",
"nbytes": 30941184,
"records": [
{
"name": "model.layers.7.attention.wqkv.q_scale",
"shape": [
6144,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1572864,
"byteOffset": 0
},
{
"name": "model.layers.20.attention_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 1572864
},
{
"name": "model.layers.20.feed_forward.w2.q_weight",
"shape": [
4096,
1792
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 29360128,
"byteOffset": 1581056
}
],
"md5sum": "12beb8400baeaeb7863dcbb2ec64ff9e"
},
{
"dataPath": "params_shard_69.bin",
"format": "raw-shard",
"nbytes": 58720256,
"records": [
{
"name": "model.layers.21.feed_forward.gate_up_proj.q_weight",
"shape": [
28672,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 58720256,
"byteOffset": 0
}
],
"md5sum": "72d0d6928a395811d4a93be4a9d22ca1"
},
{
"dataPath": "params_shard_70.bin",
"format": "raw-shard",
"nbytes": 27279360,
"records": [
{
"name": "model.layers.20.feed_forward.w2.q_scale",
"shape": [
4096,
448
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3670016,
"byteOffset": 0
},
{
"name": "model.layers.20.ffn_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 3670016
},
{
"name": "model.layers.21.attention.wo.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 3678208
},
{
"name": "model.layers.21.attention.wo.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 12066816
},
{
"name": "model.layers.21.attention.wqkv.q_weight",
"shape": [
6144,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 12582912,
"byteOffset": 13115392
},
{
"name": "model.layers.21.attention.wqkv.q_scale",
"shape": [
6144,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1572864,
"byteOffset": 25698304
},
{
"name": "model.layers.21.attention_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 27271168
}
],
"md5sum": "02e215b925d5d5d95ff5949b74580937"
},
{
"dataPath": "params_shard_71.bin",
"format": "raw-shard",
"nbytes": 29360128,
"records": [
{
"name": "model.layers.21.feed_forward.w2.q_weight",
"shape": [
4096,
1792
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 29360128,
"byteOffset": 0
}
],
"md5sum": "9d73026df7d7eb66da42cf45228d260d"
},
{
"dataPath": "params_shard_72.bin",
"format": "raw-shard",
"nbytes": 33038336,
"records": [
{
"name": "model.layers.21.feed_forward.gate_up_proj.q_scale",
"shape": [
28672,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 7340032,
"byteOffset": 0
},
{
"name": "model.layers.21.feed_forward.w2.q_scale",
"shape": [
4096,
448
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3670016,
"byteOffset": 7340032
},
{
"name": "model.layers.21.ffn_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 11010048
},
{
"name": "model.layers.22.attention.wo.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 11018240
},
{
"name": "model.layers.22.attention.wo.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 19406848
},
{
"name": "model.layers.22.attention.wqkv.q_weight",
"shape": [
6144,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 12582912,
"byteOffset": 20455424
}
],
"md5sum": "2c51526dd092af29b34a194876ee9214"
},
{
"dataPath": "params_shard_73.bin",
"format": "raw-shard",
"nbytes": 58720256,
"records": [
{
"name": "model.layers.22.feed_forward.gate_up_proj.q_weight",
"shape": [
28672,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 58720256,
"byteOffset": 0
}
],
"md5sum": "6f88a1b8734e951515c5568938488527"
},
{
"dataPath": "params_shard_74.bin",
"format": "raw-shard",
"nbytes": 29360128,
"records": [
{
"name": "model.layers.22.feed_forward.w2.q_weight",
"shape": [
4096,
1792
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 29360128,
"byteOffset": 0
}
],
"md5sum": "26e169961d6bf7c1dd40f17e40bcba36"
},
{
"dataPath": "params_shard_75.bin",
"format": "raw-shard",
"nbytes": 22036480,
"records": [
{
"name": "model.layers.22.attention.wqkv.q_scale",
"shape": [
6144,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1572864,
"byteOffset": 0
},
{
"name": "model.layers.22.attention_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 1572864
},
{
"name": "model.layers.22.feed_forward.gate_up_proj.q_scale",
"shape": [
28672,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 7340032,
"byteOffset": 1581056
},
{
"name": "model.layers.22.feed_forward.w2.q_scale",
"shape": [
4096,
448
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3670016,
"byteOffset": 8921088
},
{
"name": "model.layers.22.ffn_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 12591104
},
{
"name": "model.layers.23.attention.wo.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 12599296
},
{
"name": "model.layers.23.attention.wo.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 20987904
}
],
"md5sum": "330d1da3733ec37ba659f1c726c2f2ae"
},
{
"dataPath": "params_shard_76.bin",
"format": "raw-shard",
"nbytes": 58720256,
"records": [
{
"name": "model.layers.23.feed_forward.gate_up_proj.q_weight",
"shape": [
28672,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 58720256,
"byteOffset": 0
}
],
"md5sum": "5a34028252202efe79037e3a5d62a99d"
},
{
"dataPath": "params_shard_77.bin",
"format": "raw-shard",
"nbytes": 29360128,
"records": [
{
"name": "model.layers.23.feed_forward.w2.q_weight",
"shape": [
4096,
1792
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 29360128,
"byteOffset": 0
}
],
"md5sum": "6570a7c7e4b2498975b87b80bb72949b"
},
{
"dataPath": "params_shard_78.bin",
"format": "raw-shard",
"nbytes": 25182208,
"records": [
{
"name": "model.layers.23.attention.wqkv.q_weight",
"shape": [
6144,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 12582912,
"byteOffset": 0
},
{
"name": "model.layers.23.attention.wqkv.q_scale",
"shape": [
6144,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1572864,
"byteOffset": 12582912
},
{
"name": "model.layers.23.attention_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 14155776
},
{
"name": "model.layers.23.feed_forward.gate_up_proj.q_scale",
"shape": [
28672,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 7340032,
"byteOffset": 14163968
},
{
"name": "model.layers.23.feed_forward.w2.q_scale",
"shape": [
4096,
448
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3670016,
"byteOffset": 21504000
},
{
"name": "model.layers.23.ffn_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 25174016
}
],
"md5sum": "0fb043db9ce3fef9c096942fa143d9b6"
},
{
"dataPath": "params_shard_79.bin",
"format": "raw-shard",
"nbytes": 58720256,
"records": [
{
"name": "model.layers.24.feed_forward.gate_up_proj.q_weight",
"shape": [
28672,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 58720256,
"byteOffset": 0
}
],
"md5sum": "43383acd4ec75cdb2c86f629482b645c"
},
{
"dataPath": "params_shard_80.bin",
"format": "raw-shard",
"nbytes": 29360128,
"records": [
{
"name": "model.layers.24.feed_forward.w2.q_weight",
"shape": [
4096,
1792
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 29360128,
"byteOffset": 0
}
],
"md5sum": "0730d83081f6124e0b8eb12f38bde68d"
},
{
"dataPath": "params_shard_81.bin",
"format": "raw-shard",
"nbytes": 30941184,
"records": [
{
"name": "model.layers.24.attention.wo.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 0
},
{
"name": "model.layers.24.attention.wo.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 8388608
},
{
"name": "model.layers.24.attention.wqkv.q_weight",
"shape": [
6144,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 12582912,
"byteOffset": 9437184
},
{
"name": "model.layers.24.attention.wqkv.q_scale",
"shape": [
6144,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1572864,
"byteOffset": 22020096
},
{
"name": "model.layers.24.attention_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 23592960
},
{
"name": "model.layers.24.feed_forward.gate_up_proj.q_scale",
"shape": [
28672,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 7340032,
"byteOffset": 23601152
}
],
"md5sum": "083ca0fb718ce1bce6d901fac94ba836"
},
{
"dataPath": "params_shard_82.bin",
"format": "raw-shard",
"nbytes": 58720256,
"records": [
{
"name": "model.layers.25.feed_forward.gate_up_proj.q_weight",
"shape": [
28672,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 58720256,
"byteOffset": 0
}
],
"md5sum": "761c98f5c84cc7011b49788009a49f46"
},
{
"dataPath": "params_shard_83.bin",
"format": "raw-shard",
"nbytes": 27279360,
"records": [
{
"name": "model.layers.24.feed_forward.w2.q_scale",
"shape": [
4096,
448
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3670016,
"byteOffset": 0
},
{
"name": "model.layers.24.ffn_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 3670016
},
{
"name": "model.layers.25.attention.wo.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 3678208
},
{
"name": "model.layers.25.attention.wo.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 12066816
},
{
"name": "model.layers.25.attention.wqkv.q_weight",
"shape": [
6144,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 12582912,
"byteOffset": 13115392
},
{
"name": "model.layers.25.attention.wqkv.q_scale",
"shape": [
6144,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1572864,
"byteOffset": 25698304
},
{
"name": "model.layers.25.attention_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 27271168
}
],
"md5sum": "531ef434a8ad1c181b57a946ee8d5b3e"
},
{
"dataPath": "params_shard_84.bin",
"format": "raw-shard",
"nbytes": 29360128,
"records": [
{
"name": "model.layers.25.feed_forward.w2.q_weight",
"shape": [
4096,
1792
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 29360128,
"byteOffset": 0
}
],
"md5sum": "ea8ed182ea5effa0858587b65b6bc4de"
},
{
"dataPath": "params_shard_85.bin",
"format": "raw-shard",
"nbytes": 33038336,
"records": [
{
"name": "model.layers.25.feed_forward.gate_up_proj.q_scale",
"shape": [
28672,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 7340032,
"byteOffset": 0
},
{
"name": "model.layers.25.feed_forward.w2.q_scale",
"shape": [
4096,
448
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3670016,
"byteOffset": 7340032
},
{
"name": "model.layers.25.ffn_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 11010048
},
{
"name": "model.layers.26.attention.wo.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 11018240
},
{
"name": "model.layers.26.attention.wo.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 19406848
},
{
"name": "model.layers.26.attention.wqkv.q_weight",
"shape": [
6144,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 12582912,
"byteOffset": 20455424
}
],
"md5sum": "563704527f8899622c744522f91328d6"
},
{
"dataPath": "params_shard_86.bin",
"format": "raw-shard",
"nbytes": 58720256,
"records": [
{
"name": "model.layers.26.feed_forward.gate_up_proj.q_weight",
"shape": [
28672,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 58720256,
"byteOffset": 0
}
],
"md5sum": "5adeae8b6f05c9c359c07eb4bb4521ad"
},
{
"dataPath": "params_shard_87.bin",
"format": "raw-shard",
"nbytes": 29360128,
"records": [
{
"name": "model.layers.26.feed_forward.w2.q_weight",
"shape": [
4096,
1792
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 29360128,
"byteOffset": 0
}
],
"md5sum": "6c2cfb1b9d3b3f15b1b15d88ff381e06"
},
{
"dataPath": "params_shard_88.bin",
"format": "raw-shard",
"nbytes": 22036480,
"records": [
{
"name": "model.layers.26.attention.wqkv.q_scale",
"shape": [
6144,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1572864,
"byteOffset": 0
},
{
"name": "model.layers.26.attention_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 1572864
},
{
"name": "model.layers.26.feed_forward.gate_up_proj.q_scale",
"shape": [
28672,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 7340032,
"byteOffset": 1581056
},
{
"name": "model.layers.26.feed_forward.w2.q_scale",
"shape": [
4096,
448
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3670016,
"byteOffset": 8921088
},
{
"name": "model.layers.26.ffn_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 12591104
},
{
"name": "model.layers.27.attention.wo.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 12599296
},
{
"name": "model.layers.27.attention.wo.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 20987904
}
],
"md5sum": "d45fa8e54ac2c1e8e8e9f77a011f593d"
},
{
"dataPath": "params_shard_89.bin",
"format": "raw-shard",
"nbytes": 58720256,
"records": [
{
"name": "model.layers.27.feed_forward.gate_up_proj.q_weight",
"shape": [
28672,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 58720256,
"byteOffset": 0
}
],
"md5sum": "cd958ff22a449f3af6fd27ca09e073a1"
},
{
"dataPath": "params_shard_90.bin",
"format": "raw-shard",
"nbytes": 29360128,
"records": [
{
"name": "model.layers.27.feed_forward.w2.q_weight",
"shape": [
4096,
1792
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 29360128,
"byteOffset": 0
}
],
"md5sum": "58ca4a2190283f3ad47951dbd8dc9fcd"
},
{
"dataPath": "params_shard_91.bin",
"format": "raw-shard",
"nbytes": 25182208,
"records": [
{
"name": "model.layers.27.attention.wqkv.q_weight",
"shape": [
6144,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 12582912,
"byteOffset": 0
},
{
"name": "model.layers.27.attention.wqkv.q_scale",
"shape": [
6144,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1572864,
"byteOffset": 12582912
},
{
"name": "model.layers.27.attention_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 14155776
},
{
"name": "model.layers.27.feed_forward.gate_up_proj.q_scale",
"shape": [
28672,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 7340032,
"byteOffset": 14163968
},
{
"name": "model.layers.27.feed_forward.w2.q_scale",
"shape": [
4096,
448
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3670016,
"byteOffset": 21504000
},
{
"name": "model.layers.27.ffn_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 25174016
}
],
"md5sum": "e4f7f265147bc0c325a2ac1c64bd7f93"
},
{
"dataPath": "params_shard_92.bin",
"format": "raw-shard",
"nbytes": 58720256,
"records": [
{
"name": "model.layers.28.feed_forward.gate_up_proj.q_weight",
"shape": [
28672,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 58720256,
"byteOffset": 0
}
],
"md5sum": "0ac702362ddc58cd301700f98058150f"
},
{
"dataPath": "params_shard_93.bin",
"format": "raw-shard",
"nbytes": 29360128,
"records": [
{
"name": "model.layers.28.feed_forward.w2.q_weight",
"shape": [
4096,
1792
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 29360128,
"byteOffset": 0
}
],
"md5sum": "1bfe154bfa639f29ed2d995bcde9de6b"
},
{
"dataPath": "params_shard_94.bin",
"format": "raw-shard",
"nbytes": 30941184,
"records": [
{
"name": "model.layers.28.attention.wo.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 0
},
{
"name": "model.layers.28.attention.wo.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 8388608
},
{
"name": "model.layers.28.attention.wqkv.q_weight",
"shape": [
6144,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 12582912,
"byteOffset": 9437184
},
{
"name": "model.layers.28.attention.wqkv.q_scale",
"shape": [
6144,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1572864,
"byteOffset": 22020096
},
{
"name": "model.layers.28.attention_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 23592960
},
{
"name": "model.layers.28.feed_forward.gate_up_proj.q_scale",
"shape": [
28672,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 7340032,
"byteOffset": 23601152
}
],
"md5sum": "371cac4d2802b011c8788f98911f4a62"
},
{
"dataPath": "params_shard_95.bin",
"format": "raw-shard",
"nbytes": 58720256,
"records": [
{
"name": "model.layers.29.feed_forward.gate_up_proj.q_weight",
"shape": [
28672,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 58720256,
"byteOffset": 0
}
],
"md5sum": "30874186743719f09886f660a8f32bf8"
},
{
"dataPath": "params_shard_96.bin",
"format": "raw-shard",
"nbytes": 27271168,
"records": [
{
"name": "model.layers.28.feed_forward.w2.q_scale",
"shape": [
4096,
448
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3670016,
"byteOffset": 0
},
{
"name": "model.layers.28.ffn_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 3670016
},
{
"name": "model.layers.29.attention.wo.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 3678208
},
{
"name": "model.layers.29.attention.wo.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 12066816
},
{
"name": "model.layers.29.attention.wqkv.q_weight",
"shape": [
6144,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 12582912,
"byteOffset": 13115392
},
{
"name": "model.layers.29.attention.wqkv.q_scale",
"shape": [
6144,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1572864,
"byteOffset": 25698304
}
],
"md5sum": "823b75876a4e954575a9d63fadd218a9"
},
{
"dataPath": "params_shard_97.bin",
"format": "raw-shard",
"nbytes": 29360128,
"records": [
{
"name": "model.layers.29.feed_forward.w2.q_weight",
"shape": [
4096,
1792
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 29360128,
"byteOffset": 0
}
],
"md5sum": "7d96d447321ebbf67d7c5838e4e4f73f"
},
{
"dataPath": "params_shard_98.bin",
"format": "raw-shard",
"nbytes": 33046528,
"records": [
{
"name": "model.layers.29.feed_forward.gate_up_proj.q_scale",
"shape": [
28672,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 7340032,
"byteOffset": 0
},
{
"name": "model.layers.29.attention_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 7340032
},
{
"name": "model.layers.29.feed_forward.w2.q_scale",
"shape": [
4096,
448
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3670016,
"byteOffset": 7348224
},
{
"name": "model.layers.29.ffn_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 11018240
},
{
"name": "model.layers.30.attention.wo.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 11026432
},
{
"name": "model.layers.30.attention.wo.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 19415040
},
{
"name": "model.layers.30.attention.wqkv.q_weight",
"shape": [
6144,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 12582912,
"byteOffset": 20463616
}
],
"md5sum": "522fd6276fa044524287a7667d6dbfc3"
},
{
"dataPath": "params_shard_99.bin",
"format": "raw-shard",
"nbytes": 58720256,
"records": [
{
"name": "model.layers.30.feed_forward.gate_up_proj.q_weight",
"shape": [
28672,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 58720256,
"byteOffset": 0
}
],
"md5sum": "1201112897c57b26fbc0cf3c6a4930a3"
},
{
"dataPath": "params_shard_100.bin",
"format": "raw-shard",
"nbytes": 29360128,
"records": [
{
"name": "model.layers.30.feed_forward.w2.q_weight",
"shape": [
4096,
1792
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 29360128,
"byteOffset": 0
}
],
"md5sum": "26f67ccbda1d50dc733c5f86a906e8fd"
},
{
"dataPath": "params_shard_101.bin",
"format": "raw-shard",
"nbytes": 22036480,
"records": [
{
"name": "model.layers.30.attention.wqkv.q_scale",
"shape": [
6144,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1572864,
"byteOffset": 0
},
{
"name": "model.layers.30.attention_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 1572864
},
{
"name": "model.layers.30.feed_forward.gate_up_proj.q_scale",
"shape": [
28672,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 7340032,
"byteOffset": 1581056
},
{
"name": "model.layers.30.feed_forward.w2.q_scale",
"shape": [
4096,
448
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3670016,
"byteOffset": 8921088
},
{
"name": "model.layers.30.ffn_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 12591104
},
{
"name": "model.layers.31.attention.wo.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 12599296
},
{
"name": "model.layers.31.attention.wo.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 20987904
}
],
"md5sum": "419c3a1cc6e5d9cd0eb3393b0286a207"
},
{
"dataPath": "params_shard_102.bin",
"format": "raw-shard",
"nbytes": 58720256,
"records": [
{
"name": "model.layers.31.feed_forward.gate_up_proj.q_weight",
"shape": [
28672,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 58720256,
"byteOffset": 0
}
],
"md5sum": "bb479bed3af9787b7bdbb203972b76a2"
},
{
"dataPath": "params_shard_103.bin",
"format": "raw-shard",
"nbytes": 29360128,
"records": [
{
"name": "model.layers.31.feed_forward.w2.q_weight",
"shape": [
4096,
1792
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 29360128,
"byteOffset": 0
}
],
"md5sum": "d0db478d96363151015506e0f2d6151a"
},
{
"dataPath": "params_shard_104.bin",
"format": "raw-shard",
"nbytes": 189530112,
"records": [
{
"name": "output.q_weight",
"shape": [
92544,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 189530112,
"byteOffset": 0
}
],
"md5sum": "014abeacd2422b9301942654c7cdbd3e"
},
{
"dataPath": "params_shard_105.bin",
"format": "raw-shard",
"nbytes": 23691264,
"records": [
{
"name": "output.q_scale",
"shape": [
92544,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 23691264,
"byteOffset": 0
}
],
"md5sum": "f9e498870595df54d1722f3d9ef3006e"
},
{
"dataPath": "params_shard_106.bin",
"format": "raw-shard",
"nbytes": 25190400,
"records": [
{
"name": "model.layers.31.attention.wqkv.q_weight",
"shape": [
6144,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 12582912,
"byteOffset": 0
},
{
"name": "model.layers.31.attention.wqkv.q_scale",
"shape": [
6144,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1572864,
"byteOffset": 12582912
},
{
"name": "model.layers.31.attention_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 14155776
},
{
"name": "model.layers.31.feed_forward.gate_up_proj.q_scale",
"shape": [
28672,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 7340032,
"byteOffset": 14163968
},
{
"name": "model.layers.31.feed_forward.w2.q_scale",
"shape": [
4096,
448
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3670016,
"byteOffset": 21504000
},
{
"name": "model.layers.31.ffn_norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 25174016
},
{
"name": "model.norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 25182208
}
],
"md5sum": "614dbcf6b8b4d8c0c0ee852bf2aad289"
}
]
}