phi-2-orange-w4a16g128asym / ndarray-cache.json
numen-tech's picture
Update config
6d9833c
{
"metadata": {
"ParamSize": 455,
"ParamBytes": 1434695680.0,
"BitsPerParam": 4.071494604849007
},
"records": [
{
"dataPath": "params_shard_0.bin",
"format": "compressed-shard",
"nbytes": 65536000,
"records": [
{
"name": "lm_head.linear.q_weight",
"shape": [
51200,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 65536000,
"byteOffset": 0
}
],
"md5sum": "1c6ea922aadba0369650cd3c977ee1d5"
},
{
"dataPath": "params_shard_1.bin",
"format": "compressed-shard",
"nbytes": 29245440,
"records": [
{
"name": "lm_head.linear.bias",
"shape": [
51200
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 102400,
"byteOffset": 0
},
{
"name": "lm_head.linear.q_scale",
"shape": [
51200,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2048000,
"byteOffset": 102400
},
{
"name": "lm_head.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 2150400
},
{
"name": "lm_head.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 2155520
},
{
"name": "transformer.h.29.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 2160640
},
{
"name": "transformer.h.29.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 2181120
},
{
"name": "transformer.h.29.mlp.fc1.q_scale",
"shape": [
10240,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 15288320
},
{
"name": "transformer.h.29.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 15697920
},
{
"name": "transformer.h.29.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 15703040
},
{
"name": "transformer.h.29.mlp.fc2.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 28810240
},
{
"name": "transformer.h.30.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 29219840
},
{
"name": "transformer.h.30.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 29224960
},
{
"name": "transformer.h.30.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 29230080
}
],
"md5sum": "2ccda816080ff281661d749049272378"
},
{
"dataPath": "params_shard_2.bin",
"format": "compressed-shard",
"nbytes": 27064320,
"records": [
{
"name": "transformer.h.30.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 0
},
{
"name": "transformer.h.30.mixer.Wqkv.q_scale",
"shape": [
7680,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 307200,
"byteOffset": 9830400
},
{
"name": "transformer.h.30.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 10137600
},
{
"name": "transformer.h.30.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 10142720
},
{
"name": "transformer.h.30.mixer.out_proj.q_scale",
"shape": [
2560,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 102400,
"byteOffset": 13419520
},
{
"name": "transformer.h.30.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 13521920
},
{
"name": "transformer.h.30.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 13542400
},
{
"name": "transformer.h.30.mlp.fc1.q_scale",
"shape": [
10240,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 26649600
},
{
"name": "transformer.h.30.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 27059200
}
],
"md5sum": "1ae597c9ed3255f188159b94695af3ba"
},
{
"dataPath": "params_shard_3.bin",
"format": "compressed-shard",
"nbytes": 27084800,
"records": [
{
"name": "transformer.h.30.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.30.mlp.fc2.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 13107200
},
{
"name": "transformer.h.31.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 13516800
},
{
"name": "transformer.h.31.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 13521920
},
{
"name": "transformer.h.31.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 13527040
},
{
"name": "transformer.h.31.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 13542400
},
{
"name": "transformer.h.31.mixer.Wqkv.q_scale",
"shape": [
7680,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 307200,
"byteOffset": 23372800
},
{
"name": "transformer.h.31.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 23680000
},
{
"name": "transformer.h.31.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 23685120
},
{
"name": "transformer.h.31.mixer.out_proj.q_scale",
"shape": [
2560,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 102400,
"byteOffset": 26961920
},
{
"name": "transformer.h.31.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 27064320
}
],
"md5sum": "2ad34d2d557601c1723658c33c0ed951"
},
{
"dataPath": "params_shard_4.bin",
"format": "compressed-shard",
"nbytes": 65536000,
"records": [
{
"name": "transformer.embd.q_weight",
"shape": [
51200,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 65536000,
"byteOffset": 0
}
],
"md5sum": "b3bb224393e472b7b51cc7636d44c6d7"
},
{
"dataPath": "params_shard_5.bin",
"format": "compressed-shard",
"nbytes": 29112320,
"records": [
{
"name": "transformer.h.31.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.31.mlp.fc1.q_scale",
"shape": [
10240,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 13107200
},
{
"name": "transformer.h.31.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 13516800
},
{
"name": "transformer.h.31.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 13521920
},
{
"name": "transformer.h.31.mlp.fc2.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 26629120
},
{
"name": "transformer.embd.q_scale",
"shape": [
51200,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2048000,
"byteOffset": 27038720
},
{
"name": "transformer.h.0.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 29086720
},
{
"name": "transformer.h.0.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 29091840
},
{
"name": "transformer.h.0.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 29096960
}
],
"md5sum": "bc8dca255292bd99e98c97ca61eec935"
},
{
"dataPath": "params_shard_6.bin",
"format": "compressed-shard",
"nbytes": 27064320,
"records": [
{
"name": "transformer.h.0.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 0
},
{
"name": "transformer.h.0.mixer.Wqkv.q_scale",
"shape": [
7680,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 307200,
"byteOffset": 9830400
},
{
"name": "transformer.h.0.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 10137600
},
{
"name": "transformer.h.0.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 10142720
},
{
"name": "transformer.h.0.mixer.out_proj.q_scale",
"shape": [
2560,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 102400,
"byteOffset": 13419520
},
{
"name": "transformer.h.0.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 13521920
},
{
"name": "transformer.h.0.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 13542400
},
{
"name": "transformer.h.0.mlp.fc1.q_scale",
"shape": [
10240,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 26649600
},
{
"name": "transformer.h.0.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 27059200
}
],
"md5sum": "1bf50690267f8c3b5b6b51a97dfaadde"
},
{
"dataPath": "params_shard_7.bin",
"format": "compressed-shard",
"nbytes": 27084800,
"records": [
{
"name": "transformer.h.0.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.0.mlp.fc2.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 13107200
},
{
"name": "transformer.h.1.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 13516800
},
{
"name": "transformer.h.1.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 13521920
},
{
"name": "transformer.h.1.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 13527040
},
{
"name": "transformer.h.1.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 13542400
},
{
"name": "transformer.h.1.mixer.Wqkv.q_scale",
"shape": [
7680,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 307200,
"byteOffset": 23372800
},
{
"name": "transformer.h.1.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 23680000
},
{
"name": "transformer.h.1.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 23685120
},
{
"name": "transformer.h.1.mixer.out_proj.q_scale",
"shape": [
2560,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 102400,
"byteOffset": 26961920
},
{
"name": "transformer.h.1.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 27064320
}
],
"md5sum": "bcfa3c4ed22d9c6c9e7da25d619afd71"
},
{
"dataPath": "params_shard_8.bin",
"format": "compressed-shard",
"nbytes": 27064320,
"records": [
{
"name": "transformer.h.1.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.1.mlp.fc1.q_scale",
"shape": [
10240,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 13107200
},
{
"name": "transformer.h.1.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 13516800
},
{
"name": "transformer.h.1.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 13521920
},
{
"name": "transformer.h.1.mlp.fc2.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 26629120
},
{
"name": "transformer.h.10.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 27038720
},
{
"name": "transformer.h.10.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 27043840
},
{
"name": "transformer.h.10.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 27048960
}
],
"md5sum": "fe3e3a3a6e06b04228cc4ae23171c4bf"
},
{
"dataPath": "params_shard_9.bin",
"format": "compressed-shard",
"nbytes": 27064320,
"records": [
{
"name": "transformer.h.10.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 0
},
{
"name": "transformer.h.10.mixer.Wqkv.q_scale",
"shape": [
7680,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 307200,
"byteOffset": 9830400
},
{
"name": "transformer.h.10.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 10137600
},
{
"name": "transformer.h.10.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 10142720
},
{
"name": "transformer.h.10.mixer.out_proj.q_scale",
"shape": [
2560,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 102400,
"byteOffset": 13419520
},
{
"name": "transformer.h.10.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 13521920
},
{
"name": "transformer.h.10.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 13542400
},
{
"name": "transformer.h.10.mlp.fc1.q_scale",
"shape": [
10240,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 26649600
},
{
"name": "transformer.h.10.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 27059200
}
],
"md5sum": "7407feac12aa8deea8006f860325bae2"
},
{
"dataPath": "params_shard_10.bin",
"format": "compressed-shard",
"nbytes": 27084800,
"records": [
{
"name": "transformer.h.10.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.10.mlp.fc2.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 13107200
},
{
"name": "transformer.h.11.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 13516800
},
{
"name": "transformer.h.11.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 13521920
},
{
"name": "transformer.h.11.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 13527040
},
{
"name": "transformer.h.11.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 13542400
},
{
"name": "transformer.h.11.mixer.Wqkv.q_scale",
"shape": [
7680,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 307200,
"byteOffset": 23372800
},
{
"name": "transformer.h.11.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 23680000
},
{
"name": "transformer.h.11.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 23685120
},
{
"name": "transformer.h.11.mixer.out_proj.q_scale",
"shape": [
2560,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 102400,
"byteOffset": 26961920
},
{
"name": "transformer.h.11.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 27064320
}
],
"md5sum": "3c3c648d196405e485b141107fd82940"
},
{
"dataPath": "params_shard_11.bin",
"format": "compressed-shard",
"nbytes": 27064320,
"records": [
{
"name": "transformer.h.11.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.11.mlp.fc1.q_scale",
"shape": [
10240,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 13107200
},
{
"name": "transformer.h.11.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 13516800
},
{
"name": "transformer.h.11.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 13521920
},
{
"name": "transformer.h.11.mlp.fc2.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 26629120
},
{
"name": "transformer.h.12.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 27038720
},
{
"name": "transformer.h.12.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 27043840
},
{
"name": "transformer.h.12.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 27048960
}
],
"md5sum": "39684d8a986e8d1b5241ce31200102b6"
},
{
"dataPath": "params_shard_12.bin",
"format": "compressed-shard",
"nbytes": 27064320,
"records": [
{
"name": "transformer.h.12.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 0
},
{
"name": "transformer.h.12.mixer.Wqkv.q_scale",
"shape": [
7680,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 307200,
"byteOffset": 9830400
},
{
"name": "transformer.h.12.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 10137600
},
{
"name": "transformer.h.12.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 10142720
},
{
"name": "transformer.h.12.mixer.out_proj.q_scale",
"shape": [
2560,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 102400,
"byteOffset": 13419520
},
{
"name": "transformer.h.12.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 13521920
},
{
"name": "transformer.h.12.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 13542400
},
{
"name": "transformer.h.12.mlp.fc1.q_scale",
"shape": [
10240,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 26649600
},
{
"name": "transformer.h.12.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 27059200
}
],
"md5sum": "44f2ccfdd1e22639c2a6f327f4b30e02"
},
{
"dataPath": "params_shard_13.bin",
"format": "compressed-shard",
"nbytes": 27084800,
"records": [
{
"name": "transformer.h.12.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.12.mlp.fc2.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 13107200
},
{
"name": "transformer.h.13.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 13516800
},
{
"name": "transformer.h.13.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 13521920
},
{
"name": "transformer.h.13.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 13527040
},
{
"name": "transformer.h.13.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 13542400
},
{
"name": "transformer.h.13.mixer.Wqkv.q_scale",
"shape": [
7680,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 307200,
"byteOffset": 23372800
},
{
"name": "transformer.h.13.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 23680000
},
{
"name": "transformer.h.13.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 23685120
},
{
"name": "transformer.h.13.mixer.out_proj.q_scale",
"shape": [
2560,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 102400,
"byteOffset": 26961920
},
{
"name": "transformer.h.13.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 27064320
}
],
"md5sum": "b46367f465d50d983e4882e367e9c2c6"
},
{
"dataPath": "params_shard_14.bin",
"format": "compressed-shard",
"nbytes": 27064320,
"records": [
{
"name": "transformer.h.13.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.13.mlp.fc1.q_scale",
"shape": [
10240,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 13107200
},
{
"name": "transformer.h.13.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 13516800
},
{
"name": "transformer.h.13.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 13521920
},
{
"name": "transformer.h.13.mlp.fc2.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 26629120
},
{
"name": "transformer.h.14.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 27038720
},
{
"name": "transformer.h.14.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 27043840
},
{
"name": "transformer.h.14.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 27048960
}
],
"md5sum": "e68319a53ecbdab6ac06e5297191773b"
},
{
"dataPath": "params_shard_15.bin",
"format": "compressed-shard",
"nbytes": 27064320,
"records": [
{
"name": "transformer.h.14.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 0
},
{
"name": "transformer.h.14.mixer.Wqkv.q_scale",
"shape": [
7680,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 307200,
"byteOffset": 9830400
},
{
"name": "transformer.h.14.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 10137600
},
{
"name": "transformer.h.14.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 10142720
},
{
"name": "transformer.h.14.mixer.out_proj.q_scale",
"shape": [
2560,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 102400,
"byteOffset": 13419520
},
{
"name": "transformer.h.14.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 13521920
},
{
"name": "transformer.h.14.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 13542400
},
{
"name": "transformer.h.14.mlp.fc1.q_scale",
"shape": [
10240,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 26649600
},
{
"name": "transformer.h.14.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 27059200
}
],
"md5sum": "2540ee26e1efa5e4e0bbd000911d835b"
},
{
"dataPath": "params_shard_16.bin",
"format": "compressed-shard",
"nbytes": 27084800,
"records": [
{
"name": "transformer.h.14.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.14.mlp.fc2.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 13107200
},
{
"name": "transformer.h.15.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 13516800
},
{
"name": "transformer.h.15.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 13521920
},
{
"name": "transformer.h.15.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 13527040
},
{
"name": "transformer.h.15.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 13542400
},
{
"name": "transformer.h.15.mixer.Wqkv.q_scale",
"shape": [
7680,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 307200,
"byteOffset": 23372800
},
{
"name": "transformer.h.15.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 23680000
},
{
"name": "transformer.h.15.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 23685120
},
{
"name": "transformer.h.15.mixer.out_proj.q_scale",
"shape": [
2560,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 102400,
"byteOffset": 26961920
},
{
"name": "transformer.h.15.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 27064320
}
],
"md5sum": "bdcaeeb0ea4fd762e8de5d020e79c709"
},
{
"dataPath": "params_shard_17.bin",
"format": "compressed-shard",
"nbytes": 27064320,
"records": [
{
"name": "transformer.h.15.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.15.mlp.fc1.q_scale",
"shape": [
10240,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 13107200
},
{
"name": "transformer.h.15.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 13516800
},
{
"name": "transformer.h.15.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 13521920
},
{
"name": "transformer.h.15.mlp.fc2.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 26629120
},
{
"name": "transformer.h.16.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 27038720
},
{
"name": "transformer.h.16.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 27043840
},
{
"name": "transformer.h.16.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 27048960
}
],
"md5sum": "ce30fd44576b15ceee4bfb782350bea8"
},
{
"dataPath": "params_shard_18.bin",
"format": "compressed-shard",
"nbytes": 27064320,
"records": [
{
"name": "transformer.h.16.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 0
},
{
"name": "transformer.h.16.mixer.Wqkv.q_scale",
"shape": [
7680,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 307200,
"byteOffset": 9830400
},
{
"name": "transformer.h.16.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 10137600
},
{
"name": "transformer.h.16.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 10142720
},
{
"name": "transformer.h.16.mixer.out_proj.q_scale",
"shape": [
2560,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 102400,
"byteOffset": 13419520
},
{
"name": "transformer.h.16.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 13521920
},
{
"name": "transformer.h.16.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 13542400
},
{
"name": "transformer.h.16.mlp.fc1.q_scale",
"shape": [
10240,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 26649600
},
{
"name": "transformer.h.16.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 27059200
}
],
"md5sum": "aad86806f4fed5ee3eb2ca63423bc999"
},
{
"dataPath": "params_shard_19.bin",
"format": "compressed-shard",
"nbytes": 27084800,
"records": [
{
"name": "transformer.h.16.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.16.mlp.fc2.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 13107200
},
{
"name": "transformer.h.17.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 13516800
},
{
"name": "transformer.h.17.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 13521920
},
{
"name": "transformer.h.17.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 13527040
},
{
"name": "transformer.h.17.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 13542400
},
{
"name": "transformer.h.17.mixer.Wqkv.q_scale",
"shape": [
7680,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 307200,
"byteOffset": 23372800
},
{
"name": "transformer.h.17.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 23680000
},
{
"name": "transformer.h.17.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 23685120
},
{
"name": "transformer.h.17.mixer.out_proj.q_scale",
"shape": [
2560,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 102400,
"byteOffset": 26961920
},
{
"name": "transformer.h.17.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 27064320
}
],
"md5sum": "9a6051009007cfbdbd368a6d7d19a552"
},
{
"dataPath": "params_shard_20.bin",
"format": "compressed-shard",
"nbytes": 27064320,
"records": [
{
"name": "transformer.h.17.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.17.mlp.fc1.q_scale",
"shape": [
10240,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 13107200
},
{
"name": "transformer.h.17.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 13516800
},
{
"name": "transformer.h.17.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 13521920
},
{
"name": "transformer.h.17.mlp.fc2.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 26629120
},
{
"name": "transformer.h.18.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 27038720
},
{
"name": "transformer.h.18.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 27043840
},
{
"name": "transformer.h.18.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 27048960
}
],
"md5sum": "467ae645ee412afc22f2de28a0a3cc5b"
},
{
"dataPath": "params_shard_21.bin",
"format": "compressed-shard",
"nbytes": 27064320,
"records": [
{
"name": "transformer.h.18.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 0
},
{
"name": "transformer.h.18.mixer.Wqkv.q_scale",
"shape": [
7680,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 307200,
"byteOffset": 9830400
},
{
"name": "transformer.h.18.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 10137600
},
{
"name": "transformer.h.18.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 10142720
},
{
"name": "transformer.h.18.mixer.out_proj.q_scale",
"shape": [
2560,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 102400,
"byteOffset": 13419520
},
{
"name": "transformer.h.18.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 13521920
},
{
"name": "transformer.h.18.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 13542400
},
{
"name": "transformer.h.18.mlp.fc1.q_scale",
"shape": [
10240,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 26649600
},
{
"name": "transformer.h.18.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 27059200
}
],
"md5sum": "39fc9ef8851bc236f5ac58b9fc64ef5a"
},
{
"dataPath": "params_shard_22.bin",
"format": "compressed-shard",
"nbytes": 27084800,
"records": [
{
"name": "transformer.h.18.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.18.mlp.fc2.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 13107200
},
{
"name": "transformer.h.19.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 13516800
},
{
"name": "transformer.h.19.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 13521920
},
{
"name": "transformer.h.19.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 13527040
},
{
"name": "transformer.h.19.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 13542400
},
{
"name": "transformer.h.19.mixer.Wqkv.q_scale",
"shape": [
7680,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 307200,
"byteOffset": 23372800
},
{
"name": "transformer.h.19.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 23680000
},
{
"name": "transformer.h.19.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 23685120
},
{
"name": "transformer.h.19.mixer.out_proj.q_scale",
"shape": [
2560,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 102400,
"byteOffset": 26961920
},
{
"name": "transformer.h.19.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 27064320
}
],
"md5sum": "c87023db62f6a30621b5ea43b43cb0b6"
},
{
"dataPath": "params_shard_23.bin",
"format": "compressed-shard",
"nbytes": 27064320,
"records": [
{
"name": "transformer.h.19.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.19.mlp.fc1.q_scale",
"shape": [
10240,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 13107200
},
{
"name": "transformer.h.19.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 13516800
},
{
"name": "transformer.h.19.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 13521920
},
{
"name": "transformer.h.19.mlp.fc2.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 26629120
},
{
"name": "transformer.h.2.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 27038720
},
{
"name": "transformer.h.2.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 27043840
},
{
"name": "transformer.h.2.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 27048960
}
],
"md5sum": "102537956bff3c37e4c0673c42c27999"
},
{
"dataPath": "params_shard_24.bin",
"format": "compressed-shard",
"nbytes": 27064320,
"records": [
{
"name": "transformer.h.2.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 0
},
{
"name": "transformer.h.2.mixer.Wqkv.q_scale",
"shape": [
7680,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 307200,
"byteOffset": 9830400
},
{
"name": "transformer.h.2.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 10137600
},
{
"name": "transformer.h.2.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 10142720
},
{
"name": "transformer.h.2.mixer.out_proj.q_scale",
"shape": [
2560,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 102400,
"byteOffset": 13419520
},
{
"name": "transformer.h.2.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 13521920
},
{
"name": "transformer.h.2.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 13542400
},
{
"name": "transformer.h.2.mlp.fc1.q_scale",
"shape": [
10240,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 26649600
},
{
"name": "transformer.h.2.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 27059200
}
],
"md5sum": "eb8edbec70caaf36e6e97de0d128c275"
},
{
"dataPath": "params_shard_25.bin",
"format": "compressed-shard",
"nbytes": 27084800,
"records": [
{
"name": "transformer.h.2.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.2.mlp.fc2.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 13107200
},
{
"name": "transformer.h.20.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 13516800
},
{
"name": "transformer.h.20.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 13521920
},
{
"name": "transformer.h.20.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 13527040
},
{
"name": "transformer.h.20.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 13542400
},
{
"name": "transformer.h.20.mixer.Wqkv.q_scale",
"shape": [
7680,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 307200,
"byteOffset": 23372800
},
{
"name": "transformer.h.20.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 23680000
},
{
"name": "transformer.h.20.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 23685120
},
{
"name": "transformer.h.20.mixer.out_proj.q_scale",
"shape": [
2560,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 102400,
"byteOffset": 26961920
},
{
"name": "transformer.h.20.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 27064320
}
],
"md5sum": "d8364003461a1c2903b8e3082d40c9d4"
},
{
"dataPath": "params_shard_26.bin",
"format": "compressed-shard",
"nbytes": 27064320,
"records": [
{
"name": "transformer.h.20.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.20.mlp.fc1.q_scale",
"shape": [
10240,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 13107200
},
{
"name": "transformer.h.20.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 13516800
},
{
"name": "transformer.h.20.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 13521920
},
{
"name": "transformer.h.20.mlp.fc2.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 26629120
},
{
"name": "transformer.h.21.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 27038720
},
{
"name": "transformer.h.21.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 27043840
},
{
"name": "transformer.h.21.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 27048960
}
],
"md5sum": "c4ccc6bb97f7abe44b4dd87f7c43ee4f"
},
{
"dataPath": "params_shard_27.bin",
"format": "compressed-shard",
"nbytes": 27064320,
"records": [
{
"name": "transformer.h.21.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 0
},
{
"name": "transformer.h.21.mixer.Wqkv.q_scale",
"shape": [
7680,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 307200,
"byteOffset": 9830400
},
{
"name": "transformer.h.21.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 10137600
},
{
"name": "transformer.h.21.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 10142720
},
{
"name": "transformer.h.21.mixer.out_proj.q_scale",
"shape": [
2560,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 102400,
"byteOffset": 13419520
},
{
"name": "transformer.h.21.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 13521920
},
{
"name": "transformer.h.21.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 13542400
},
{
"name": "transformer.h.21.mlp.fc1.q_scale",
"shape": [
10240,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 26649600
},
{
"name": "transformer.h.21.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 27059200
}
],
"md5sum": "81ca930db9522020400c2ce9187d8e77"
},
{
"dataPath": "params_shard_28.bin",
"format": "compressed-shard",
"nbytes": 27084800,
"records": [
{
"name": "transformer.h.21.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.21.mlp.fc2.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 13107200
},
{
"name": "transformer.h.22.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 13516800
},
{
"name": "transformer.h.22.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 13521920
},
{
"name": "transformer.h.22.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 13527040
},
{
"name": "transformer.h.22.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 13542400
},
{
"name": "transformer.h.22.mixer.Wqkv.q_scale",
"shape": [
7680,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 307200,
"byteOffset": 23372800
},
{
"name": "transformer.h.22.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 23680000
},
{
"name": "transformer.h.22.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 23685120
},
{
"name": "transformer.h.22.mixer.out_proj.q_scale",
"shape": [
2560,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 102400,
"byteOffset": 26961920
},
{
"name": "transformer.h.22.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 27064320
}
],
"md5sum": "5fe3d4b67dc674ec561963945255131f"
},
{
"dataPath": "params_shard_29.bin",
"format": "compressed-shard",
"nbytes": 27064320,
"records": [
{
"name": "transformer.h.22.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.22.mlp.fc1.q_scale",
"shape": [
10240,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 13107200
},
{
"name": "transformer.h.22.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 13516800
},
{
"name": "transformer.h.22.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 13521920
},
{
"name": "transformer.h.22.mlp.fc2.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 26629120
},
{
"name": "transformer.h.23.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 27038720
},
{
"name": "transformer.h.23.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 27043840
},
{
"name": "transformer.h.23.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 27048960
}
],
"md5sum": "8099737cd1fc47616dc819a999f9f7b8"
},
{
"dataPath": "params_shard_30.bin",
"format": "compressed-shard",
"nbytes": 27064320,
"records": [
{
"name": "transformer.h.23.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 0
},
{
"name": "transformer.h.23.mixer.Wqkv.q_scale",
"shape": [
7680,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 307200,
"byteOffset": 9830400
},
{
"name": "transformer.h.23.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 10137600
},
{
"name": "transformer.h.23.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 10142720
},
{
"name": "transformer.h.23.mixer.out_proj.q_scale",
"shape": [
2560,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 102400,
"byteOffset": 13419520
},
{
"name": "transformer.h.23.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 13521920
},
{
"name": "transformer.h.23.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 13542400
},
{
"name": "transformer.h.23.mlp.fc1.q_scale",
"shape": [
10240,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 26649600
},
{
"name": "transformer.h.23.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 27059200
}
],
"md5sum": "6ea1996741061d0b3c48ab0a3ffe0709"
},
{
"dataPath": "params_shard_31.bin",
"format": "compressed-shard",
"nbytes": 27084800,
"records": [
{
"name": "transformer.h.23.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.23.mlp.fc2.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 13107200
},
{
"name": "transformer.h.24.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 13516800
},
{
"name": "transformer.h.24.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 13521920
},
{
"name": "transformer.h.24.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 13527040
},
{
"name": "transformer.h.24.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 13542400
},
{
"name": "transformer.h.24.mixer.Wqkv.q_scale",
"shape": [
7680,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 307200,
"byteOffset": 23372800
},
{
"name": "transformer.h.24.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 23680000
},
{
"name": "transformer.h.24.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 23685120
},
{
"name": "transformer.h.24.mixer.out_proj.q_scale",
"shape": [
2560,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 102400,
"byteOffset": 26961920
},
{
"name": "transformer.h.24.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 27064320
}
],
"md5sum": "fc0d6646718d853204079fbc12e9d968"
},
{
"dataPath": "params_shard_32.bin",
"format": "compressed-shard",
"nbytes": 27064320,
"records": [
{
"name": "transformer.h.24.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.24.mlp.fc1.q_scale",
"shape": [
10240,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 13107200
},
{
"name": "transformer.h.24.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 13516800
},
{
"name": "transformer.h.24.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 13521920
},
{
"name": "transformer.h.24.mlp.fc2.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 26629120
},
{
"name": "transformer.h.25.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 27038720
},
{
"name": "transformer.h.25.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 27043840
},
{
"name": "transformer.h.25.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 27048960
}
],
"md5sum": "2a95f3cd50afd6a231c7b74d5d166178"
},
{
"dataPath": "params_shard_33.bin",
"format": "compressed-shard",
"nbytes": 27064320,
"records": [
{
"name": "transformer.h.25.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 0
},
{
"name": "transformer.h.25.mixer.Wqkv.q_scale",
"shape": [
7680,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 307200,
"byteOffset": 9830400
},
{
"name": "transformer.h.25.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 10137600
},
{
"name": "transformer.h.25.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 10142720
},
{
"name": "transformer.h.25.mixer.out_proj.q_scale",
"shape": [
2560,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 102400,
"byteOffset": 13419520
},
{
"name": "transformer.h.25.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 13521920
},
{
"name": "transformer.h.25.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 13542400
},
{
"name": "transformer.h.25.mlp.fc1.q_scale",
"shape": [
10240,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 26649600
},
{
"name": "transformer.h.25.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 27059200
}
],
"md5sum": "4dac38b06ac30e9079935f56ea707669"
},
{
"dataPath": "params_shard_34.bin",
"format": "compressed-shard",
"nbytes": 27084800,
"records": [
{
"name": "transformer.h.25.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.25.mlp.fc2.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 13107200
},
{
"name": "transformer.h.26.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 13516800
},
{
"name": "transformer.h.26.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 13521920
},
{
"name": "transformer.h.26.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 13527040
},
{
"name": "transformer.h.26.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 13542400
},
{
"name": "transformer.h.26.mixer.Wqkv.q_scale",
"shape": [
7680,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 307200,
"byteOffset": 23372800
},
{
"name": "transformer.h.26.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 23680000
},
{
"name": "transformer.h.26.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 23685120
},
{
"name": "transformer.h.26.mixer.out_proj.q_scale",
"shape": [
2560,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 102400,
"byteOffset": 26961920
},
{
"name": "transformer.h.26.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 27064320
}
],
"md5sum": "b48bbb6157549131c16ad6e841add9ef"
},
{
"dataPath": "params_shard_35.bin",
"format": "compressed-shard",
"nbytes": 27064320,
"records": [
{
"name": "transformer.h.26.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.26.mlp.fc1.q_scale",
"shape": [
10240,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 13107200
},
{
"name": "transformer.h.26.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 13516800
},
{
"name": "transformer.h.26.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 13521920
},
{
"name": "transformer.h.26.mlp.fc2.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 26629120
},
{
"name": "transformer.h.27.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 27038720
},
{
"name": "transformer.h.27.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 27043840
},
{
"name": "transformer.h.27.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 27048960
}
],
"md5sum": "b543038abd2cc89abc8a3fba960aa9c8"
},
{
"dataPath": "params_shard_36.bin",
"format": "compressed-shard",
"nbytes": 27064320,
"records": [
{
"name": "transformer.h.27.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 0
},
{
"name": "transformer.h.27.mixer.Wqkv.q_scale",
"shape": [
7680,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 307200,
"byteOffset": 9830400
},
{
"name": "transformer.h.27.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 10137600
},
{
"name": "transformer.h.27.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 10142720
},
{
"name": "transformer.h.27.mixer.out_proj.q_scale",
"shape": [
2560,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 102400,
"byteOffset": 13419520
},
{
"name": "transformer.h.27.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 13521920
},
{
"name": "transformer.h.27.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 13542400
},
{
"name": "transformer.h.27.mlp.fc1.q_scale",
"shape": [
10240,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 26649600
},
{
"name": "transformer.h.27.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 27059200
}
],
"md5sum": "b48d9b687eac7178c5421afe794f7e96"
},
{
"dataPath": "params_shard_37.bin",
"format": "compressed-shard",
"nbytes": 27084800,
"records": [
{
"name": "transformer.h.27.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.27.mlp.fc2.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 13107200
},
{
"name": "transformer.h.28.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 13516800
},
{
"name": "transformer.h.28.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 13521920
},
{
"name": "transformer.h.28.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 13527040
},
{
"name": "transformer.h.28.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 13542400
},
{
"name": "transformer.h.28.mixer.Wqkv.q_scale",
"shape": [
7680,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 307200,
"byteOffset": 23372800
},
{
"name": "transformer.h.28.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 23680000
},
{
"name": "transformer.h.28.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 23685120
},
{
"name": "transformer.h.28.mixer.out_proj.q_scale",
"shape": [
2560,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 102400,
"byteOffset": 26961920
},
{
"name": "transformer.h.28.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 27064320
}
],
"md5sum": "6dc3c51a48580a401d7edb3d58ecc887"
},
{
"dataPath": "params_shard_38.bin",
"format": "compressed-shard",
"nbytes": 27064320,
"records": [
{
"name": "transformer.h.28.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.28.mlp.fc1.q_scale",
"shape": [
10240,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 13107200
},
{
"name": "transformer.h.28.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 13516800
},
{
"name": "transformer.h.28.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 13521920
},
{
"name": "transformer.h.28.mlp.fc2.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 26629120
},
{
"name": "transformer.h.29.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 27038720
},
{
"name": "transformer.h.29.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 27043840
},
{
"name": "transformer.h.29.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 27048960
}
],
"md5sum": "531163003c677a1b4a22ae629a8abf3f"
},
{
"dataPath": "params_shard_39.bin",
"format": "compressed-shard",
"nbytes": 27089920,
"records": [
{
"name": "transformer.h.29.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 0
},
{
"name": "transformer.h.29.mixer.Wqkv.q_scale",
"shape": [
7680,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 307200,
"byteOffset": 9830400
},
{
"name": "transformer.h.29.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 10137600
},
{
"name": "transformer.h.29.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 10142720
},
{
"name": "transformer.h.29.mixer.out_proj.q_scale",
"shape": [
2560,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 102400,
"byteOffset": 13419520
},
{
"name": "transformer.h.3.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 13521920
},
{
"name": "transformer.h.3.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 13527040
},
{
"name": "transformer.h.3.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 13532160
},
{
"name": "transformer.h.3.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 13547520
},
{
"name": "transformer.h.3.mixer.Wqkv.q_scale",
"shape": [
7680,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 307200,
"byteOffset": 23377920
},
{
"name": "transformer.h.3.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 23685120
},
{
"name": "transformer.h.3.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 23690240
},
{
"name": "transformer.h.3.mixer.out_proj.q_scale",
"shape": [
2560,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 102400,
"byteOffset": 26967040
},
{
"name": "transformer.h.3.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 27069440
}
],
"md5sum": "1bb35e2054170d9b8e963836daeff23a"
},
{
"dataPath": "params_shard_40.bin",
"format": "compressed-shard",
"nbytes": 27064320,
"records": [
{
"name": "transformer.h.3.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.3.mlp.fc1.q_scale",
"shape": [
10240,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 13107200
},
{
"name": "transformer.h.3.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 13516800
},
{
"name": "transformer.h.3.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 13521920
},
{
"name": "transformer.h.3.mlp.fc2.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 26629120
},
{
"name": "transformer.h.4.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 27038720
},
{
"name": "transformer.h.4.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 27043840
},
{
"name": "transformer.h.4.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 27048960
}
],
"md5sum": "f033155933f51e22094b87877641704f"
},
{
"dataPath": "params_shard_41.bin",
"format": "compressed-shard",
"nbytes": 27064320,
"records": [
{
"name": "transformer.h.4.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 0
},
{
"name": "transformer.h.4.mixer.Wqkv.q_scale",
"shape": [
7680,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 307200,
"byteOffset": 9830400
},
{
"name": "transformer.h.4.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 10137600
},
{
"name": "transformer.h.4.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 10142720
},
{
"name": "transformer.h.4.mixer.out_proj.q_scale",
"shape": [
2560,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 102400,
"byteOffset": 13419520
},
{
"name": "transformer.h.4.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 13521920
},
{
"name": "transformer.h.4.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 13542400
},
{
"name": "transformer.h.4.mlp.fc1.q_scale",
"shape": [
10240,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 26649600
},
{
"name": "transformer.h.4.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 27059200
}
],
"md5sum": "749a2cd58c87c6dfc2b686052ace3548"
},
{
"dataPath": "params_shard_42.bin",
"format": "compressed-shard",
"nbytes": 27084800,
"records": [
{
"name": "transformer.h.4.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.4.mlp.fc2.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 13107200
},
{
"name": "transformer.h.5.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 13516800
},
{
"name": "transformer.h.5.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 13521920
},
{
"name": "transformer.h.5.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 13527040
},
{
"name": "transformer.h.5.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 13542400
},
{
"name": "transformer.h.5.mixer.Wqkv.q_scale",
"shape": [
7680,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 307200,
"byteOffset": 23372800
},
{
"name": "transformer.h.5.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 23680000
},
{
"name": "transformer.h.5.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 23685120
},
{
"name": "transformer.h.5.mixer.out_proj.q_scale",
"shape": [
2560,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 102400,
"byteOffset": 26961920
},
{
"name": "transformer.h.5.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 27064320
}
],
"md5sum": "f722028950dcd5b537065740489128c0"
},
{
"dataPath": "params_shard_43.bin",
"format": "compressed-shard",
"nbytes": 27064320,
"records": [
{
"name": "transformer.h.5.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.5.mlp.fc1.q_scale",
"shape": [
10240,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 13107200
},
{
"name": "transformer.h.5.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 13516800
},
{
"name": "transformer.h.5.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 13521920
},
{
"name": "transformer.h.5.mlp.fc2.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 26629120
},
{
"name": "transformer.h.6.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 27038720
},
{
"name": "transformer.h.6.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 27043840
},
{
"name": "transformer.h.6.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 27048960
}
],
"md5sum": "7a344f9074b1b8c88c6ba5d54f3fbde2"
},
{
"dataPath": "params_shard_44.bin",
"format": "compressed-shard",
"nbytes": 27064320,
"records": [
{
"name": "transformer.h.6.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 0
},
{
"name": "transformer.h.6.mixer.Wqkv.q_scale",
"shape": [
7680,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 307200,
"byteOffset": 9830400
},
{
"name": "transformer.h.6.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 10137600
},
{
"name": "transformer.h.6.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 10142720
},
{
"name": "transformer.h.6.mixer.out_proj.q_scale",
"shape": [
2560,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 102400,
"byteOffset": 13419520
},
{
"name": "transformer.h.6.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 13521920
},
{
"name": "transformer.h.6.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 13542400
},
{
"name": "transformer.h.6.mlp.fc1.q_scale",
"shape": [
10240,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 26649600
},
{
"name": "transformer.h.6.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 27059200
}
],
"md5sum": "05911dbb2ba0e67894fa639e47eb6f37"
},
{
"dataPath": "params_shard_45.bin",
"format": "compressed-shard",
"nbytes": 27084800,
"records": [
{
"name": "transformer.h.6.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.6.mlp.fc2.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 13107200
},
{
"name": "transformer.h.7.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 13516800
},
{
"name": "transformer.h.7.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 13521920
},
{
"name": "transformer.h.7.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 13527040
},
{
"name": "transformer.h.7.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 13542400
},
{
"name": "transformer.h.7.mixer.Wqkv.q_scale",
"shape": [
7680,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 307200,
"byteOffset": 23372800
},
{
"name": "transformer.h.7.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 23680000
},
{
"name": "transformer.h.7.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 23685120
},
{
"name": "transformer.h.7.mixer.out_proj.q_scale",
"shape": [
2560,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 102400,
"byteOffset": 26961920
},
{
"name": "transformer.h.7.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 27064320
}
],
"md5sum": "9ea0935206a429f6d8ceeb3b094beb18"
},
{
"dataPath": "params_shard_46.bin",
"format": "compressed-shard",
"nbytes": 27064320,
"records": [
{
"name": "transformer.h.7.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.7.mlp.fc1.q_scale",
"shape": [
10240,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 13107200
},
{
"name": "transformer.h.7.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 13516800
},
{
"name": "transformer.h.7.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 13521920
},
{
"name": "transformer.h.7.mlp.fc2.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 26629120
},
{
"name": "transformer.h.8.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 27038720
},
{
"name": "transformer.h.8.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 27043840
},
{
"name": "transformer.h.8.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 27048960
}
],
"md5sum": "b9466585d50c945190e38014cc6c0154"
},
{
"dataPath": "params_shard_47.bin",
"format": "compressed-shard",
"nbytes": 27064320,
"records": [
{
"name": "transformer.h.8.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 0
},
{
"name": "transformer.h.8.mixer.Wqkv.q_scale",
"shape": [
7680,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 307200,
"byteOffset": 9830400
},
{
"name": "transformer.h.8.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 10137600
},
{
"name": "transformer.h.8.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 10142720
},
{
"name": "transformer.h.8.mixer.out_proj.q_scale",
"shape": [
2560,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 102400,
"byteOffset": 13419520
},
{
"name": "transformer.h.8.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 13521920
},
{
"name": "transformer.h.8.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 13542400
},
{
"name": "transformer.h.8.mlp.fc1.q_scale",
"shape": [
10240,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 26649600
},
{
"name": "transformer.h.8.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 27059200
}
],
"md5sum": "745802e3405b0a9cbde195c90e375999"
},
{
"dataPath": "params_shard_48.bin",
"format": "compressed-shard",
"nbytes": 27084800,
"records": [
{
"name": "transformer.h.8.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.8.mlp.fc2.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 13107200
},
{
"name": "transformer.h.9.ln.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 13516800
},
{
"name": "transformer.h.9.ln.weight",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 13521920
},
{
"name": "transformer.h.9.mixer.Wqkv.bias",
"shape": [
7680
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 15360,
"byteOffset": 13527040
},
{
"name": "transformer.h.9.mixer.Wqkv.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 13542400
},
{
"name": "transformer.h.9.mixer.Wqkv.q_scale",
"shape": [
7680,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 307200,
"byteOffset": 23372800
},
{
"name": "transformer.h.9.mixer.out_proj.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 23680000
},
{
"name": "transformer.h.9.mixer.out_proj.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 23685120
},
{
"name": "transformer.h.9.mixer.out_proj.q_scale",
"shape": [
2560,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 102400,
"byteOffset": 26961920
},
{
"name": "transformer.h.9.mlp.fc1.bias",
"shape": [
10240
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 20480,
"byteOffset": 27064320
}
],
"md5sum": "8eca0b6181877867e814ffb9ce691964"
},
{
"dataPath": "params_shard_49.bin",
"format": "compressed-shard",
"nbytes": 27038720,
"records": [
{
"name": "transformer.h.9.mlp.fc1.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "transformer.h.9.mlp.fc1.q_scale",
"shape": [
10240,
20
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 13107200
},
{
"name": "transformer.h.9.mlp.fc2.bias",
"shape": [
2560
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5120,
"byteOffset": 13516800
},
{
"name": "transformer.h.9.mlp.fc2.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 13521920
},
{
"name": "transformer.h.9.mlp.fc2.q_scale",
"shape": [
2560,
80
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 26629120
}
],
"md5sum": "05c91953a697d13b3dfdb3793ce2a4c8"
}
]
}