mlc-q4f16-phi-2-orange-v2 / ndarray-cache.json
Felladrin's picture
Upload folder using huggingface_hub
f4a2ac7 verified
{
"metadata": {
"ParamSize": 455,
"ParamBytes": 1564948480.0,
"BitsPerParam": 4.503961083574167
},
"records": [
{
"dataPath": "params_shard_0.bin",
"format": "raw-shard",
"nbytes": 65536000,
"records": [
{
"name": "lm_head.linear.q_weight",
"shape": [
51200,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 65536000,
"byteOffset": 0
}
],
"md5sum": "0563df3b218670c0c1dfdb3fe1672a14"
},
{
"dataPath": "params_shard_1.bin",
"format": "raw-shard",
"nbytes": 23086080,
"records": [
{
"name": "lm_head.linear.bias",
"shape": [
51200
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 102400,
"byteOffset": 0
},
{
"name": "lm_head.linear.q_scale",
"shape": [
51200,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192000,
"byteOffset": 102400
},
{
"name": "lm_head.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 8294400
},
{
"name": "lm_head.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 8299520
},
{
"name": "transformer.h.30.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 8304640
},
{
"name": "transformer.h.30.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 8309760
},
{
"name": "transformer.h.30.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 8314880
},
{
"name": "transformer.h.30.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 8335360
},
{
"name": "transformer.h.30.mlp.fc1.q_scale",
"shape": [
10240,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 21442560
},
{
"name": "transformer.h.30.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 23080960
}
],
"md5sum": "a714c91cd079b7fed2a5011849e20e68"
},
{
"dataPath": "params_shard_2.bin",
"format": "raw-shard",
"nbytes": 29542400,
"records": [
{
"name": "transformer.h.30.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.30.mlp.fc2.q_scale",
"shape": [
2560,
320
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "transformer.h.30.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 14745600
},
{
"name": "transformer.h.30.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 14750720
},
{
"name": "transformer.h.30.mixer.out_proj.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 18027520
},
{
"name": "transformer.h.30.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 18437120
},
{
"name": "transformer.h.30.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 18452480
},
{
"name": "transformer.h.30.mixer.Wqkv.q_scale",
"shape": [
7680,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1228800,
"byteOffset": 28282880
},
{
"name": "transformer.h.31.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 29511680
},
{
"name": "transformer.h.31.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 29516800
},
{
"name": "transformer.h.31.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 29521920
}
],
"md5sum": "0f889f66f58b43624314826a2bcd36da"
},
{
"dataPath": "params_shard_3.bin",
"format": "raw-shard",
"nbytes": 33203200,
"records": [
{
"name": "transformer.h.31.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.31.mlp.fc1.q_scale",
"shape": [
10240,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "transformer.h.31.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 14745600
},
{
"name": "transformer.h.31.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 14750720
},
{
"name": "transformer.h.31.mlp.fc2.q_scale",
"shape": [
2560,
320
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 27857920
},
{
"name": "transformer.h.31.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 29496320
},
{
"name": "transformer.h.31.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 29501440
},
{
"name": "transformer.h.31.mixer.out_proj.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 32778240
},
{
"name": "transformer.h.31.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 33187840
}
],
"md5sum": "95ded8e295f3f48b300527bd4d6a56db"
},
{
"dataPath": "params_shard_4.bin",
"format": "raw-shard",
"nbytes": 65536000,
"records": [
{
"name": "transformer.embd.q_weight",
"shape": [
51200,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 65536000,
"byteOffset": 0
}
],
"md5sum": "80e3dc5dad0a019ee3ae2074f29b4681"
},
{
"dataPath": "params_shard_5.bin",
"format": "raw-shard",
"nbytes": 32389120,
"records": [
{
"name": "transformer.h.31.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 0
},
{
"name": "transformer.h.31.mixer.Wqkv.q_scale",
"shape": [
7680,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1228800,
"byteOffset": 9830400
},
{
"name": "transformer.embd.q_scale",
"shape": [
51200,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192000,
"byteOffset": 11059200
},
{
"name": "transformer.h.0.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 19251200
},
{
"name": "transformer.h.0.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 19256320
},
{
"name": "transformer.h.0.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 19261440
},
{
"name": "transformer.h.0.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 19281920
}
],
"md5sum": "40b02551b299ba59b11e57c5edf2d46a"
},
{
"dataPath": "params_shard_6.bin",
"format": "raw-shard",
"nbytes": 31185920,
"records": [
{
"name": "transformer.h.0.mlp.fc1.q_scale",
"shape": [
10240,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 0
},
{
"name": "transformer.h.0.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 1638400
},
{
"name": "transformer.h.0.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 1643520
},
{
"name": "transformer.h.0.mlp.fc2.q_scale",
"shape": [
2560,
320
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 14750720
},
{
"name": "transformer.h.0.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 16389120
},
{
"name": "transformer.h.0.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 16394240
},
{
"name": "transformer.h.0.mixer.out_proj.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 19671040
},
{
"name": "transformer.h.0.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 20080640
},
{
"name": "transformer.h.0.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 20096000
},
{
"name": "transformer.h.0.mixer.Wqkv.q_scale",
"shape": [
7680,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1228800,
"byteOffset": 29926400
},
{
"name": "transformer.h.1.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 31155200
},
{
"name": "transformer.h.1.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 31160320
},
{
"name": "transformer.h.1.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 31165440
}
],
"md5sum": "99984a390b07b95c9c2dbbe76184a314"
},
{
"dataPath": "params_shard_7.bin",
"format": "raw-shard",
"nbytes": 33203200,
"records": [
{
"name": "transformer.h.1.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.1.mlp.fc1.q_scale",
"shape": [
10240,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "transformer.h.1.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 14745600
},
{
"name": "transformer.h.1.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 14750720
},
{
"name": "transformer.h.1.mlp.fc2.q_scale",
"shape": [
2560,
320
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 27857920
},
{
"name": "transformer.h.1.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 29496320
},
{
"name": "transformer.h.1.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 29501440
},
{
"name": "transformer.h.1.mixer.out_proj.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 32778240
},
{
"name": "transformer.h.1.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 33187840
}
],
"md5sum": "a66a83a85f8b402c70b70cca4c60e770"
},
{
"dataPath": "params_shard_8.bin",
"format": "raw-shard",
"nbytes": 25840640,
"records": [
{
"name": "transformer.h.1.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 0
},
{
"name": "transformer.h.1.mixer.Wqkv.q_scale",
"shape": [
7680,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1228800,
"byteOffset": 9830400
},
{
"name": "transformer.h.10.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 11059200
},
{
"name": "transformer.h.10.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 11064320
},
{
"name": "transformer.h.10.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 11069440
},
{
"name": "transformer.h.10.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 11089920
},
{
"name": "transformer.h.10.mlp.fc1.q_scale",
"shape": [
10240,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 24197120
},
{
"name": "transformer.h.10.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 25835520
}
],
"md5sum": "49079c4e9ff8b3b46276e49f37d91648"
},
{
"dataPath": "params_shard_9.bin",
"format": "raw-shard",
"nbytes": 29542400,
"records": [
{
"name": "transformer.h.10.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.10.mlp.fc2.q_scale",
"shape": [
2560,
320
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "transformer.h.10.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 14745600
},
{
"name": "transformer.h.10.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 14750720
},
{
"name": "transformer.h.10.mixer.out_proj.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 18027520
},
{
"name": "transformer.h.10.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 18437120
},
{
"name": "transformer.h.10.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 18452480
},
{
"name": "transformer.h.10.mixer.Wqkv.q_scale",
"shape": [
7680,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1228800,
"byteOffset": 28282880
},
{
"name": "transformer.h.11.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 29511680
},
{
"name": "transformer.h.11.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 29516800
},
{
"name": "transformer.h.11.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 29521920
}
],
"md5sum": "f3ce8bfc8128dfc4898770e6f65ae250"
},
{
"dataPath": "params_shard_10.bin",
"format": "raw-shard",
"nbytes": 33203200,
"records": [
{
"name": "transformer.h.11.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.11.mlp.fc1.q_scale",
"shape": [
10240,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "transformer.h.11.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 14745600
},
{
"name": "transformer.h.11.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 14750720
},
{
"name": "transformer.h.11.mlp.fc2.q_scale",
"shape": [
2560,
320
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 27857920
},
{
"name": "transformer.h.11.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 29496320
},
{
"name": "transformer.h.11.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 29501440
},
{
"name": "transformer.h.11.mixer.out_proj.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 32778240
},
{
"name": "transformer.h.11.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 33187840
}
],
"md5sum": "3be603bb7d783d7032f569b1b7ff2595"
},
{
"dataPath": "params_shard_11.bin",
"format": "raw-shard",
"nbytes": 25840640,
"records": [
{
"name": "transformer.h.11.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 0
},
{
"name": "transformer.h.11.mixer.Wqkv.q_scale",
"shape": [
7680,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1228800,
"byteOffset": 9830400
},
{
"name": "transformer.h.12.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 11059200
},
{
"name": "transformer.h.12.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 11064320
},
{
"name": "transformer.h.12.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 11069440
},
{
"name": "transformer.h.12.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 11089920
},
{
"name": "transformer.h.12.mlp.fc1.q_scale",
"shape": [
10240,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 24197120
},
{
"name": "transformer.h.12.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 25835520
}
],
"md5sum": "796af2114eb841352b72f06f81f6abe5"
},
{
"dataPath": "params_shard_12.bin",
"format": "raw-shard",
"nbytes": 29542400,
"records": [
{
"name": "transformer.h.12.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.12.mlp.fc2.q_scale",
"shape": [
2560,
320
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "transformer.h.12.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 14745600
},
{
"name": "transformer.h.12.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 14750720
},
{
"name": "transformer.h.12.mixer.out_proj.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 18027520
},
{
"name": "transformer.h.12.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 18437120
},
{
"name": "transformer.h.12.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 18452480
},
{
"name": "transformer.h.12.mixer.Wqkv.q_scale",
"shape": [
7680,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1228800,
"byteOffset": 28282880
},
{
"name": "transformer.h.13.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 29511680
},
{
"name": "transformer.h.13.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 29516800
},
{
"name": "transformer.h.13.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 29521920
}
],
"md5sum": "129949c46ad7766ea17a44c2b27c62ba"
},
{
"dataPath": "params_shard_13.bin",
"format": "raw-shard",
"nbytes": 33203200,
"records": [
{
"name": "transformer.h.13.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.13.mlp.fc1.q_scale",
"shape": [
10240,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "transformer.h.13.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 14745600
},
{
"name": "transformer.h.13.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 14750720
},
{
"name": "transformer.h.13.mlp.fc2.q_scale",
"shape": [
2560,
320
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 27857920
},
{
"name": "transformer.h.13.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 29496320
},
{
"name": "transformer.h.13.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 29501440
},
{
"name": "transformer.h.13.mixer.out_proj.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 32778240
},
{
"name": "transformer.h.13.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 33187840
}
],
"md5sum": "4ab23dff3b4135960fb5eb45b73dbf9c"
},
{
"dataPath": "params_shard_14.bin",
"format": "raw-shard",
"nbytes": 25840640,
"records": [
{
"name": "transformer.h.13.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 0
},
{
"name": "transformer.h.13.mixer.Wqkv.q_scale",
"shape": [
7680,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1228800,
"byteOffset": 9830400
},
{
"name": "transformer.h.14.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 11059200
},
{
"name": "transformer.h.14.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 11064320
},
{
"name": "transformer.h.14.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 11069440
},
{
"name": "transformer.h.14.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 11089920
},
{
"name": "transformer.h.14.mlp.fc1.q_scale",
"shape": [
10240,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 24197120
},
{
"name": "transformer.h.14.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 25835520
}
],
"md5sum": "cf718371bfdfce35844cb984ee15a469"
},
{
"dataPath": "params_shard_15.bin",
"format": "raw-shard",
"nbytes": 29542400,
"records": [
{
"name": "transformer.h.14.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.14.mlp.fc2.q_scale",
"shape": [
2560,
320
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "transformer.h.14.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 14745600
},
{
"name": "transformer.h.14.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 14750720
},
{
"name": "transformer.h.14.mixer.out_proj.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 18027520
},
{
"name": "transformer.h.14.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 18437120
},
{
"name": "transformer.h.14.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 18452480
},
{
"name": "transformer.h.14.mixer.Wqkv.q_scale",
"shape": [
7680,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1228800,
"byteOffset": 28282880
},
{
"name": "transformer.h.15.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 29511680
},
{
"name": "transformer.h.15.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 29516800
},
{
"name": "transformer.h.15.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 29521920
}
],
"md5sum": "8a32e2ded62644151597011db1884f43"
},
{
"dataPath": "params_shard_16.bin",
"format": "raw-shard",
"nbytes": 33203200,
"records": [
{
"name": "transformer.h.15.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.15.mlp.fc1.q_scale",
"shape": [
10240,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "transformer.h.15.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 14745600
},
{
"name": "transformer.h.15.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 14750720
},
{
"name": "transformer.h.15.mlp.fc2.q_scale",
"shape": [
2560,
320
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 27857920
},
{
"name": "transformer.h.15.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 29496320
},
{
"name": "transformer.h.15.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 29501440
},
{
"name": "transformer.h.15.mixer.out_proj.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 32778240
},
{
"name": "transformer.h.15.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 33187840
}
],
"md5sum": "52d1999e12d42135ccafd6e5fcf19fe7"
},
{
"dataPath": "params_shard_17.bin",
"format": "raw-shard",
"nbytes": 25840640,
"records": [
{
"name": "transformer.h.15.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 0
},
{
"name": "transformer.h.15.mixer.Wqkv.q_scale",
"shape": [
7680,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1228800,
"byteOffset": 9830400
},
{
"name": "transformer.h.16.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 11059200
},
{
"name": "transformer.h.16.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 11064320
},
{
"name": "transformer.h.16.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 11069440
},
{
"name": "transformer.h.16.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 11089920
},
{
"name": "transformer.h.16.mlp.fc1.q_scale",
"shape": [
10240,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 24197120
},
{
"name": "transformer.h.16.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 25835520
}
],
"md5sum": "28d7575edfd6c63c2b40008ae364f83a"
},
{
"dataPath": "params_shard_18.bin",
"format": "raw-shard",
"nbytes": 29542400,
"records": [
{
"name": "transformer.h.16.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.16.mlp.fc2.q_scale",
"shape": [
2560,
320
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "transformer.h.16.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 14745600
},
{
"name": "transformer.h.16.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 14750720
},
{
"name": "transformer.h.16.mixer.out_proj.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 18027520
},
{
"name": "transformer.h.16.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 18437120
},
{
"name": "transformer.h.16.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 18452480
},
{
"name": "transformer.h.16.mixer.Wqkv.q_scale",
"shape": [
7680,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1228800,
"byteOffset": 28282880
},
{
"name": "transformer.h.17.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 29511680
},
{
"name": "transformer.h.17.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 29516800
},
{
"name": "transformer.h.17.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 29521920
}
],
"md5sum": "c72988bbef97357d8954974d3476387e"
},
{
"dataPath": "params_shard_19.bin",
"format": "raw-shard",
"nbytes": 33203200,
"records": [
{
"name": "transformer.h.17.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.17.mlp.fc1.q_scale",
"shape": [
10240,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "transformer.h.17.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 14745600
},
{
"name": "transformer.h.17.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 14750720
},
{
"name": "transformer.h.17.mlp.fc2.q_scale",
"shape": [
2560,
320
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 27857920
},
{
"name": "transformer.h.17.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 29496320
},
{
"name": "transformer.h.17.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 29501440
},
{
"name": "transformer.h.17.mixer.out_proj.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 32778240
},
{
"name": "transformer.h.17.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 33187840
}
],
"md5sum": "47ec02a16e40f962cdf8e807b386e824"
},
{
"dataPath": "params_shard_20.bin",
"format": "raw-shard",
"nbytes": 25840640,
"records": [
{
"name": "transformer.h.17.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 0
},
{
"name": "transformer.h.17.mixer.Wqkv.q_scale",
"shape": [
7680,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1228800,
"byteOffset": 9830400
},
{
"name": "transformer.h.18.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 11059200
},
{
"name": "transformer.h.18.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 11064320
},
{
"name": "transformer.h.18.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 11069440
},
{
"name": "transformer.h.18.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 11089920
},
{
"name": "transformer.h.18.mlp.fc1.q_scale",
"shape": [
10240,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 24197120
},
{
"name": "transformer.h.18.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 25835520
}
],
"md5sum": "2e8424d7d4f62e4f5b715138b34afb3f"
},
{
"dataPath": "params_shard_21.bin",
"format": "raw-shard",
"nbytes": 29542400,
"records": [
{
"name": "transformer.h.18.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.18.mlp.fc2.q_scale",
"shape": [
2560,
320
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "transformer.h.18.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 14745600
},
{
"name": "transformer.h.18.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 14750720
},
{
"name": "transformer.h.18.mixer.out_proj.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 18027520
},
{
"name": "transformer.h.18.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 18437120
},
{
"name": "transformer.h.18.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 18452480
},
{
"name": "transformer.h.18.mixer.Wqkv.q_scale",
"shape": [
7680,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1228800,
"byteOffset": 28282880
},
{
"name": "transformer.h.19.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 29511680
},
{
"name": "transformer.h.19.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 29516800
},
{
"name": "transformer.h.19.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 29521920
}
],
"md5sum": "8481fee88fa3e97c07cb9d4efa012d8d"
},
{
"dataPath": "params_shard_22.bin",
"format": "raw-shard",
"nbytes": 33203200,
"records": [
{
"name": "transformer.h.19.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.19.mlp.fc1.q_scale",
"shape": [
10240,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "transformer.h.19.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 14745600
},
{
"name": "transformer.h.19.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 14750720
},
{
"name": "transformer.h.19.mlp.fc2.q_scale",
"shape": [
2560,
320
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 27857920
},
{
"name": "transformer.h.19.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 29496320
},
{
"name": "transformer.h.19.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 29501440
},
{
"name": "transformer.h.19.mixer.out_proj.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 32778240
},
{
"name": "transformer.h.19.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 33187840
}
],
"md5sum": "6322db68d1c3e7222fa9e5eec77f3f22"
},
{
"dataPath": "params_shard_23.bin",
"format": "raw-shard",
"nbytes": 25840640,
"records": [
{
"name": "transformer.h.19.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 0
},
{
"name": "transformer.h.19.mixer.Wqkv.q_scale",
"shape": [
7680,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1228800,
"byteOffset": 9830400
},
{
"name": "transformer.h.2.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 11059200
},
{
"name": "transformer.h.2.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 11064320
},
{
"name": "transformer.h.2.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 11069440
},
{
"name": "transformer.h.2.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 11089920
},
{
"name": "transformer.h.2.mlp.fc1.q_scale",
"shape": [
10240,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 24197120
},
{
"name": "transformer.h.2.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 25835520
}
],
"md5sum": "e9cafdbc37285044e64da523307fd296"
},
{
"dataPath": "params_shard_24.bin",
"format": "raw-shard",
"nbytes": 29542400,
"records": [
{
"name": "transformer.h.2.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.2.mlp.fc2.q_scale",
"shape": [
2560,
320
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "transformer.h.2.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 14745600
},
{
"name": "transformer.h.2.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 14750720
},
{
"name": "transformer.h.2.mixer.out_proj.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 18027520
},
{
"name": "transformer.h.2.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 18437120
},
{
"name": "transformer.h.2.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 18452480
},
{
"name": "transformer.h.2.mixer.Wqkv.q_scale",
"shape": [
7680,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1228800,
"byteOffset": 28282880
},
{
"name": "transformer.h.20.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 29511680
},
{
"name": "transformer.h.20.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 29516800
},
{
"name": "transformer.h.20.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 29521920
}
],
"md5sum": "dcdc364d267d02d159e7a4ac6ba22dbb"
},
{
"dataPath": "params_shard_25.bin",
"format": "raw-shard",
"nbytes": 33203200,
"records": [
{
"name": "transformer.h.20.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.20.mlp.fc1.q_scale",
"shape": [
10240,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "transformer.h.20.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 14745600
},
{
"name": "transformer.h.20.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 14750720
},
{
"name": "transformer.h.20.mlp.fc2.q_scale",
"shape": [
2560,
320
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 27857920
},
{
"name": "transformer.h.20.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 29496320
},
{
"name": "transformer.h.20.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 29501440
},
{
"name": "transformer.h.20.mixer.out_proj.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 32778240
},
{
"name": "transformer.h.20.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 33187840
}
],
"md5sum": "0a04ff23a7aa6e926772c8e40d57378e"
},
{
"dataPath": "params_shard_26.bin",
"format": "raw-shard",
"nbytes": 25840640,
"records": [
{
"name": "transformer.h.20.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 0
},
{
"name": "transformer.h.20.mixer.Wqkv.q_scale",
"shape": [
7680,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1228800,
"byteOffset": 9830400
},
{
"name": "transformer.h.21.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 11059200
},
{
"name": "transformer.h.21.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 11064320
},
{
"name": "transformer.h.21.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 11069440
},
{
"name": "transformer.h.21.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 11089920
},
{
"name": "transformer.h.21.mlp.fc1.q_scale",
"shape": [
10240,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 24197120
},
{
"name": "transformer.h.21.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 25835520
}
],
"md5sum": "566a4710714ac274397b4bc2df35cb4f"
},
{
"dataPath": "params_shard_27.bin",
"format": "raw-shard",
"nbytes": 29542400,
"records": [
{
"name": "transformer.h.21.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.21.mlp.fc2.q_scale",
"shape": [
2560,
320
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "transformer.h.21.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 14745600
},
{
"name": "transformer.h.21.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 14750720
},
{
"name": "transformer.h.21.mixer.out_proj.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 18027520
},
{
"name": "transformer.h.21.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 18437120
},
{
"name": "transformer.h.21.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 18452480
},
{
"name": "transformer.h.21.mixer.Wqkv.q_scale",
"shape": [
7680,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1228800,
"byteOffset": 28282880
},
{
"name": "transformer.h.22.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 29511680
},
{
"name": "transformer.h.22.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 29516800
},
{
"name": "transformer.h.22.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 29521920
}
],
"md5sum": "08f141af33952f5789f7fe97e4741f38"
},
{
"dataPath": "params_shard_28.bin",
"format": "raw-shard",
"nbytes": 33203200,
"records": [
{
"name": "transformer.h.22.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.22.mlp.fc1.q_scale",
"shape": [
10240,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "transformer.h.22.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 14745600
},
{
"name": "transformer.h.22.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 14750720
},
{
"name": "transformer.h.22.mlp.fc2.q_scale",
"shape": [
2560,
320
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 27857920
},
{
"name": "transformer.h.22.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 29496320
},
{
"name": "transformer.h.22.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 29501440
},
{
"name": "transformer.h.22.mixer.out_proj.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 32778240
},
{
"name": "transformer.h.22.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 33187840
}
],
"md5sum": "280fcdaa44a75664f3232653cfdd2c61"
},
{
"dataPath": "params_shard_29.bin",
"format": "raw-shard",
"nbytes": 25840640,
"records": [
{
"name": "transformer.h.22.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 0
},
{
"name": "transformer.h.22.mixer.Wqkv.q_scale",
"shape": [
7680,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1228800,
"byteOffset": 9830400
},
{
"name": "transformer.h.23.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 11059200
},
{
"name": "transformer.h.23.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 11064320
},
{
"name": "transformer.h.23.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 11069440
},
{
"name": "transformer.h.23.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 11089920
},
{
"name": "transformer.h.23.mlp.fc1.q_scale",
"shape": [
10240,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 24197120
},
{
"name": "transformer.h.23.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 25835520
}
],
"md5sum": "8efe07fbca7cf83bb926f3fa9753c247"
},
{
"dataPath": "params_shard_30.bin",
"format": "raw-shard",
"nbytes": 29542400,
"records": [
{
"name": "transformer.h.23.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.23.mlp.fc2.q_scale",
"shape": [
2560,
320
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "transformer.h.23.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 14745600
},
{
"name": "transformer.h.23.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 14750720
},
{
"name": "transformer.h.23.mixer.out_proj.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 18027520
},
{
"name": "transformer.h.23.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 18437120
},
{
"name": "transformer.h.23.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 18452480
},
{
"name": "transformer.h.23.mixer.Wqkv.q_scale",
"shape": [
7680,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1228800,
"byteOffset": 28282880
},
{
"name": "transformer.h.24.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 29511680
},
{
"name": "transformer.h.24.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 29516800
},
{
"name": "transformer.h.24.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 29521920
}
],
"md5sum": "31f146b919a0d0019f305d8b7c35d9e3"
},
{
"dataPath": "params_shard_31.bin",
"format": "raw-shard",
"nbytes": 33203200,
"records": [
{
"name": "transformer.h.24.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.24.mlp.fc1.q_scale",
"shape": [
10240,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "transformer.h.24.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 14745600
},
{
"name": "transformer.h.24.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 14750720
},
{
"name": "transformer.h.24.mlp.fc2.q_scale",
"shape": [
2560,
320
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 27857920
},
{
"name": "transformer.h.24.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 29496320
},
{
"name": "transformer.h.24.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 29501440
},
{
"name": "transformer.h.24.mixer.out_proj.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 32778240
},
{
"name": "transformer.h.24.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 33187840
}
],
"md5sum": "5c6ad46a2a4d2f171eedae23dcdc2d73"
},
{
"dataPath": "params_shard_32.bin",
"format": "raw-shard",
"nbytes": 25840640,
"records": [
{
"name": "transformer.h.24.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 0
},
{
"name": "transformer.h.24.mixer.Wqkv.q_scale",
"shape": [
7680,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1228800,
"byteOffset": 9830400
},
{
"name": "transformer.h.25.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 11059200
},
{
"name": "transformer.h.25.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 11064320
},
{
"name": "transformer.h.25.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 11069440
},
{
"name": "transformer.h.25.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 11089920
},
{
"name": "transformer.h.25.mlp.fc1.q_scale",
"shape": [
10240,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 24197120
},
{
"name": "transformer.h.25.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 25835520
}
],
"md5sum": "5c1bc33181cb493922a7c79c7de75c58"
},
{
"dataPath": "params_shard_33.bin",
"format": "raw-shard",
"nbytes": 29542400,
"records": [
{
"name": "transformer.h.25.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.25.mlp.fc2.q_scale",
"shape": [
2560,
320
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "transformer.h.25.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 14745600
},
{
"name": "transformer.h.25.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 14750720
},
{
"name": "transformer.h.25.mixer.out_proj.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 18027520
},
{
"name": "transformer.h.25.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 18437120
},
{
"name": "transformer.h.25.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 18452480
},
{
"name": "transformer.h.25.mixer.Wqkv.q_scale",
"shape": [
7680,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1228800,
"byteOffset": 28282880
},
{
"name": "transformer.h.26.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 29511680
},
{
"name": "transformer.h.26.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 29516800
},
{
"name": "transformer.h.26.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 29521920
}
],
"md5sum": "10a4e27221c13bfa92d6c8a8ca5ac190"
},
{
"dataPath": "params_shard_34.bin",
"format": "raw-shard",
"nbytes": 33203200,
"records": [
{
"name": "transformer.h.26.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.26.mlp.fc1.q_scale",
"shape": [
10240,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "transformer.h.26.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 14745600
},
{
"name": "transformer.h.26.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 14750720
},
{
"name": "transformer.h.26.mlp.fc2.q_scale",
"shape": [
2560,
320
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 27857920
},
{
"name": "transformer.h.26.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 29496320
},
{
"name": "transformer.h.26.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 29501440
},
{
"name": "transformer.h.26.mixer.out_proj.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 32778240
},
{
"name": "transformer.h.26.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 33187840
}
],
"md5sum": "4894d28044f62855957d11be1208284a"
},
{
"dataPath": "params_shard_35.bin",
"format": "raw-shard",
"nbytes": 25840640,
"records": [
{
"name": "transformer.h.26.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 0
},
{
"name": "transformer.h.26.mixer.Wqkv.q_scale",
"shape": [
7680,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1228800,
"byteOffset": 9830400
},
{
"name": "transformer.h.27.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 11059200
},
{
"name": "transformer.h.27.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 11064320
},
{
"name": "transformer.h.27.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 11069440
},
{
"name": "transformer.h.27.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 11089920
},
{
"name": "transformer.h.27.mlp.fc1.q_scale",
"shape": [
10240,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 24197120
},
{
"name": "transformer.h.27.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 25835520
}
],
"md5sum": "dfa7635717a460b55f1936cc5077a36b"
},
{
"dataPath": "params_shard_36.bin",
"format": "raw-shard",
"nbytes": 29542400,
"records": [
{
"name": "transformer.h.27.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.27.mlp.fc2.q_scale",
"shape": [
2560,
320
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "transformer.h.27.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 14745600
},
{
"name": "transformer.h.27.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 14750720
},
{
"name": "transformer.h.27.mixer.out_proj.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 18027520
},
{
"name": "transformer.h.27.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 18437120
},
{
"name": "transformer.h.27.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 18452480
},
{
"name": "transformer.h.27.mixer.Wqkv.q_scale",
"shape": [
7680,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1228800,
"byteOffset": 28282880
},
{
"name": "transformer.h.28.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 29511680
},
{
"name": "transformer.h.28.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 29516800
},
{
"name": "transformer.h.28.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 29521920
}
],
"md5sum": "49931f843af4558fbd60cf24d46a5dab"
},
{
"dataPath": "params_shard_37.bin",
"format": "raw-shard",
"nbytes": 33203200,
"records": [
{
"name": "transformer.h.28.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.28.mlp.fc1.q_scale",
"shape": [
10240,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "transformer.h.28.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 14745600
},
{
"name": "transformer.h.28.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 14750720
},
{
"name": "transformer.h.28.mlp.fc2.q_scale",
"shape": [
2560,
320
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 27857920
},
{
"name": "transformer.h.28.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 29496320
},
{
"name": "transformer.h.28.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 29501440
},
{
"name": "transformer.h.28.mixer.out_proj.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 32778240
},
{
"name": "transformer.h.28.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 33187840
}
],
"md5sum": "e523639011c93a1e2bf3a09b3521f905"
},
{
"dataPath": "params_shard_38.bin",
"format": "raw-shard",
"nbytes": 25840640,
"records": [
{
"name": "transformer.h.28.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 0
},
{
"name": "transformer.h.28.mixer.Wqkv.q_scale",
"shape": [
7680,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1228800,
"byteOffset": 9830400
},
{
"name": "transformer.h.29.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 11059200
},
{
"name": "transformer.h.29.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 11064320
},
{
"name": "transformer.h.29.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 11069440
},
{
"name": "transformer.h.29.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 11089920
},
{
"name": "transformer.h.29.mlp.fc1.q_scale",
"shape": [
10240,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 24197120
},
{
"name": "transformer.h.29.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 25835520
}
],
"md5sum": "e9af0b1efe2bc2c5150d8ad3a8729a40"
},
{
"dataPath": "params_shard_39.bin",
"format": "raw-shard",
"nbytes": 29542400,
"records": [
{
"name": "transformer.h.29.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.29.mlp.fc2.q_scale",
"shape": [
2560,
320
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "transformer.h.29.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 14745600
},
{
"name": "transformer.h.29.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 14750720
},
{
"name": "transformer.h.29.mixer.out_proj.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 18027520
},
{
"name": "transformer.h.29.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 18437120
},
{
"name": "transformer.h.29.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 18452480
},
{
"name": "transformer.h.29.mixer.Wqkv.q_scale",
"shape": [
7680,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1228800,
"byteOffset": 28282880
},
{
"name": "transformer.h.3.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 29511680
},
{
"name": "transformer.h.3.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 29516800
},
{
"name": "transformer.h.3.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 29521920
}
],
"md5sum": "a1aa7d9d817e4f26d54bee22197bd127"
},
{
"dataPath": "params_shard_40.bin",
"format": "raw-shard",
"nbytes": 33203200,
"records": [
{
"name": "transformer.h.3.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.3.mlp.fc1.q_scale",
"shape": [
10240,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "transformer.h.3.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 14745600
},
{
"name": "transformer.h.3.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 14750720
},
{
"name": "transformer.h.3.mlp.fc2.q_scale",
"shape": [
2560,
320
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 27857920
},
{
"name": "transformer.h.3.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 29496320
},
{
"name": "transformer.h.3.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 29501440
},
{
"name": "transformer.h.3.mixer.out_proj.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 32778240
},
{
"name": "transformer.h.3.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 33187840
}
],
"md5sum": "de95c27564d286ead197cb9b795a9acb"
},
{
"dataPath": "params_shard_41.bin",
"format": "raw-shard",
"nbytes": 25840640,
"records": [
{
"name": "transformer.h.3.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 0
},
{
"name": "transformer.h.3.mixer.Wqkv.q_scale",
"shape": [
7680,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1228800,
"byteOffset": 9830400
},
{
"name": "transformer.h.4.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 11059200
},
{
"name": "transformer.h.4.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 11064320
},
{
"name": "transformer.h.4.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 11069440
},
{
"name": "transformer.h.4.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 11089920
},
{
"name": "transformer.h.4.mlp.fc1.q_scale",
"shape": [
10240,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 24197120
},
{
"name": "transformer.h.4.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 25835520
}
],
"md5sum": "599ba1675095c9ff39abbb1eef68f242"
},
{
"dataPath": "params_shard_42.bin",
"format": "raw-shard",
"nbytes": 29542400,
"records": [
{
"name": "transformer.h.4.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.4.mlp.fc2.q_scale",
"shape": [
2560,
320
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "transformer.h.4.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 14745600
},
{
"name": "transformer.h.4.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 14750720
},
{
"name": "transformer.h.4.mixer.out_proj.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 18027520
},
{
"name": "transformer.h.4.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 18437120
},
{
"name": "transformer.h.4.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 18452480
},
{
"name": "transformer.h.4.mixer.Wqkv.q_scale",
"shape": [
7680,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1228800,
"byteOffset": 28282880
},
{
"name": "transformer.h.5.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 29511680
},
{
"name": "transformer.h.5.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 29516800
},
{
"name": "transformer.h.5.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 29521920
}
],
"md5sum": "570bfb81a2183e7ffb4a75d92d508293"
},
{
"dataPath": "params_shard_43.bin",
"format": "raw-shard",
"nbytes": 33203200,
"records": [
{
"name": "transformer.h.5.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.5.mlp.fc1.q_scale",
"shape": [
10240,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "transformer.h.5.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 14745600
},
{
"name": "transformer.h.5.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 14750720
},
{
"name": "transformer.h.5.mlp.fc2.q_scale",
"shape": [
2560,
320
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 27857920
},
{
"name": "transformer.h.5.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 29496320
},
{
"name": "transformer.h.5.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 29501440
},
{
"name": "transformer.h.5.mixer.out_proj.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 32778240
},
{
"name": "transformer.h.5.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 33187840
}
],
"md5sum": "922f1d794294b53935c4adccc099ada8"
},
{
"dataPath": "params_shard_44.bin",
"format": "raw-shard",
"nbytes": 25840640,
"records": [
{
"name": "transformer.h.5.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 0
},
{
"name": "transformer.h.5.mixer.Wqkv.q_scale",
"shape": [
7680,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1228800,
"byteOffset": 9830400
},
{
"name": "transformer.h.6.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 11059200
},
{
"name": "transformer.h.6.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 11064320
},
{
"name": "transformer.h.6.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 11069440
},
{
"name": "transformer.h.6.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 11089920
},
{
"name": "transformer.h.6.mlp.fc1.q_scale",
"shape": [
10240,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 24197120
},
{
"name": "transformer.h.6.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 25835520
}
],
"md5sum": "aaeab85841fef67c6205e1959990f82b"
},
{
"dataPath": "params_shard_45.bin",
"format": "raw-shard",
"nbytes": 29542400,
"records": [
{
"name": "transformer.h.6.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.6.mlp.fc2.q_scale",
"shape": [
2560,
320
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "transformer.h.6.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 14745600
},
{
"name": "transformer.h.6.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 14750720
},
{
"name": "transformer.h.6.mixer.out_proj.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 18027520
},
{
"name": "transformer.h.6.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 18437120
},
{
"name": "transformer.h.6.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 18452480
},
{
"name": "transformer.h.6.mixer.Wqkv.q_scale",
"shape": [
7680,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1228800,
"byteOffset": 28282880
},
{
"name": "transformer.h.7.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 29511680
},
{
"name": "transformer.h.7.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 29516800
},
{
"name": "transformer.h.7.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 29521920
}
],
"md5sum": "6d325248a17262a8f32c5d38bb1e52b1"
},
{
"dataPath": "params_shard_46.bin",
"format": "raw-shard",
"nbytes": 33203200,
"records": [
{
"name": "transformer.h.7.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.7.mlp.fc1.q_scale",
"shape": [
10240,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "transformer.h.7.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 14745600
},
{
"name": "transformer.h.7.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 14750720
},
{
"name": "transformer.h.7.mlp.fc2.q_scale",
"shape": [
2560,
320
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 27857920
},
{
"name": "transformer.h.7.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 29496320
},
{
"name": "transformer.h.7.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 29501440
},
{
"name": "transformer.h.7.mixer.out_proj.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 32778240
},
{
"name": "transformer.h.7.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 33187840
}
],
"md5sum": "f9ff28ec9b7125356dc6b3fce2d86d76"
},
{
"dataPath": "params_shard_47.bin",
"format": "raw-shard",
"nbytes": 25840640,
"records": [
{
"name": "transformer.h.7.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 0
},
{
"name": "transformer.h.7.mixer.Wqkv.q_scale",
"shape": [
7680,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1228800,
"byteOffset": 9830400
},
{
"name": "transformer.h.8.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 11059200
},
{
"name": "transformer.h.8.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 11064320
},
{
"name": "transformer.h.8.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 11069440
},
{
"name": "transformer.h.8.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 11089920
},
{
"name": "transformer.h.8.mlp.fc1.q_scale",
"shape": [
10240,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 24197120
},
{
"name": "transformer.h.8.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 25835520
}
],
"md5sum": "a3a13949ea5831f0f6649b0f9ad2483b"
},
{
"dataPath": "params_shard_48.bin",
"format": "raw-shard",
"nbytes": 29542400,
"records": [
{
"name": "transformer.h.8.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.8.mlp.fc2.q_scale",
"shape": [
2560,
320
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "transformer.h.8.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 14745600
},
{
"name": "transformer.h.8.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 14750720
},
{
"name": "transformer.h.8.mixer.out_proj.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 18027520
},
{
"name": "transformer.h.8.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 18437120
},
{
"name": "transformer.h.8.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 18452480
},
{
"name": "transformer.h.8.mixer.Wqkv.q_scale",
"shape": [
7680,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1228800,
"byteOffset": 28282880
},
{
"name": "transformer.h.9.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 29511680
},
{
"name": "transformer.h.9.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 29516800
},
{
"name": "transformer.h.9.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 29521920
}
],
"md5sum": "28a81b5359fec6bb6f21dbf7224dbabd"
},
{
"dataPath": "params_shard_49.bin",
"format": "raw-shard",
"nbytes": 33203200,
"records": [
{
"name": "transformer.h.9.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.9.mlp.fc1.q_scale",
"shape": [
10240,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "transformer.h.9.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 14745600
},
{
"name": "transformer.h.9.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 14750720
},
{
"name": "transformer.h.9.mlp.fc2.q_scale",
"shape": [
2560,
320
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 27857920
},
{
"name": "transformer.h.9.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 29496320
},
{
"name": "transformer.h.9.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 29501440
},
{
"name": "transformer.h.9.mixer.out_proj.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 32778240
},
{
"name": "transformer.h.9.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 33187840
}
],
"md5sum": "07ad236eab63c07cb01c8186da8f2908"
},
{
"dataPath": "params_shard_50.bin",
"format": "raw-shard",
"nbytes": 11059200,
"records": [
{
"name": "transformer.h.9.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 0
},
{
"name": "transformer.h.9.mixer.Wqkv.q_scale",
"shape": [
7680,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1228800,
"byteOffset": 9830400
}
],
"md5sum": "17f86263b975491c6bfd64946915060f"
}
]
}