{ "metadata": { "ParamSize": 455, "ParamBytes": 1434695680.0, "BitsPerParam": 4.071494604849007 }, "records": [ { "dataPath": "params_shard_0.bin", "format": "compressed-shard", "nbytes": 65536000, "records": [ { "name": "lm_head.linear.q_weight", "shape": [ 51200, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 65536000, "byteOffset": 0 } ], "md5sum": "1c6ea922aadba0369650cd3c977ee1d5" }, { "dataPath": "params_shard_1.bin", "format": "compressed-shard", "nbytes": 29245440, "records": [ { "name": "lm_head.linear.bias", "shape": [ 51200 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 102400, "byteOffset": 0 }, { "name": "lm_head.linear.q_scale", "shape": [ 51200, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2048000, "byteOffset": 102400 }, { "name": "lm_head.ln.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 2150400 }, { "name": "lm_head.ln.weight", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 2155520 }, { "name": "transformer.h.29.mlp.fc1.bias", "shape": [ 10240 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 2160640 }, { "name": "transformer.h.29.mlp.fc1.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 2181120 }, { "name": "transformer.h.29.mlp.fc1.q_scale", "shape": [ 10240, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 15288320 }, { "name": "transformer.h.29.mlp.fc2.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 15697920 }, { "name": "transformer.h.29.mlp.fc2.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 15703040 }, { "name": "transformer.h.29.mlp.fc2.q_scale", "shape": [ 2560, 80 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 28810240 }, { "name": "transformer.h.30.ln.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29219840 }, { "name": "transformer.h.30.ln.weight", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29224960 }, { "name": "transformer.h.30.mixer.Wqkv.bias", "shape": [ 7680 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 29230080 } ], "md5sum": "2ccda816080ff281661d749049272378" }, { "dataPath": "params_shard_2.bin", "format": "compressed-shard", "nbytes": 27064320, "records": [ { "name": "transformer.h.30.mixer.Wqkv.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 0 }, { "name": "transformer.h.30.mixer.Wqkv.q_scale", "shape": [ 7680, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 307200, "byteOffset": 9830400 }, { "name": "transformer.h.30.mixer.out_proj.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 10137600 }, { "name": "transformer.h.30.mixer.out_proj.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 10142720 }, { "name": "transformer.h.30.mixer.out_proj.q_scale", "shape": [ 2560, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 102400, "byteOffset": 13419520 }, { "name": "transformer.h.30.mlp.fc1.bias", "shape": [ 10240 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 13521920 }, { "name": "transformer.h.30.mlp.fc1.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 13542400 }, { "name": "transformer.h.30.mlp.fc1.q_scale", "shape": [ 10240, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 26649600 }, { "name": "transformer.h.30.mlp.fc2.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 27059200 } ], "md5sum": "1ae597c9ed3255f188159b94695af3ba" }, { "dataPath": "params_shard_3.bin", "format": "compressed-shard", "nbytes": 27084800, "records": [ { "name": "transformer.h.30.mlp.fc2.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "transformer.h.30.mlp.fc2.q_scale", "shape": [ 2560, 80 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "transformer.h.31.ln.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 13516800 }, { "name": "transformer.h.31.ln.weight", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 13521920 }, { "name": "transformer.h.31.mixer.Wqkv.bias", "shape": [ 7680 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 13527040 }, { "name": "transformer.h.31.mixer.Wqkv.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 13542400 }, { "name": "transformer.h.31.mixer.Wqkv.q_scale", "shape": [ 7680, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 307200, "byteOffset": 23372800 }, { "name": "transformer.h.31.mixer.out_proj.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 23680000 }, { "name": "transformer.h.31.mixer.out_proj.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 23685120 }, { "name": "transformer.h.31.mixer.out_proj.q_scale", "shape": [ 2560, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 102400, "byteOffset": 26961920 }, { "name": "transformer.h.31.mlp.fc1.bias", "shape": [ 10240 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 27064320 } ], "md5sum": "2ad34d2d557601c1723658c33c0ed951" }, { "dataPath": "params_shard_4.bin", "format": "compressed-shard", "nbytes": 65536000, "records": [ { "name": "transformer.embd.q_weight", "shape": [ 51200, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 65536000, "byteOffset": 0 } ], "md5sum": "b3bb224393e472b7b51cc7636d44c6d7" }, { "dataPath": "params_shard_5.bin", "format": "compressed-shard", "nbytes": 29112320, "records": [ { "name": "transformer.h.31.mlp.fc1.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "transformer.h.31.mlp.fc1.q_scale", "shape": [ 10240, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "transformer.h.31.mlp.fc2.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 13516800 }, { "name": "transformer.h.31.mlp.fc2.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 13521920 }, { "name": "transformer.h.31.mlp.fc2.q_scale", "shape": [ 2560, 80 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 26629120 }, { "name": "transformer.embd.q_scale", "shape": [ 51200, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2048000, "byteOffset": 27038720 }, { "name": "transformer.h.0.ln.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29086720 }, { "name": "transformer.h.0.ln.weight", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29091840 }, { "name": "transformer.h.0.mixer.Wqkv.bias", "shape": [ 7680 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 29096960 } ], "md5sum": "bc8dca255292bd99e98c97ca61eec935" }, { "dataPath": "params_shard_6.bin", "format": "compressed-shard", "nbytes": 27064320, "records": [ { "name": "transformer.h.0.mixer.Wqkv.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 0 }, { "name": "transformer.h.0.mixer.Wqkv.q_scale", "shape": [ 7680, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 307200, "byteOffset": 9830400 }, { "name": "transformer.h.0.mixer.out_proj.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 10137600 }, { "name": "transformer.h.0.mixer.out_proj.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 10142720 }, { "name": "transformer.h.0.mixer.out_proj.q_scale", "shape": [ 2560, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 102400, "byteOffset": 13419520 }, { "name": "transformer.h.0.mlp.fc1.bias", "shape": [ 10240 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 13521920 }, { "name": "transformer.h.0.mlp.fc1.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 13542400 }, { "name": "transformer.h.0.mlp.fc1.q_scale", "shape": [ 10240, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 26649600 }, { "name": "transformer.h.0.mlp.fc2.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 27059200 } ], "md5sum": "1bf50690267f8c3b5b6b51a97dfaadde" }, { "dataPath": "params_shard_7.bin", "format": "compressed-shard", "nbytes": 27084800, "records": [ { "name": "transformer.h.0.mlp.fc2.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "transformer.h.0.mlp.fc2.q_scale", "shape": [ 2560, 80 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "transformer.h.1.ln.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 13516800 }, { "name": "transformer.h.1.ln.weight", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 13521920 }, { "name": "transformer.h.1.mixer.Wqkv.bias", "shape": [ 7680 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 13527040 }, { "name": "transformer.h.1.mixer.Wqkv.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 13542400 }, { "name": "transformer.h.1.mixer.Wqkv.q_scale", "shape": [ 7680, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 307200, "byteOffset": 23372800 }, { "name": "transformer.h.1.mixer.out_proj.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 23680000 }, { "name": "transformer.h.1.mixer.out_proj.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 23685120 }, { "name": "transformer.h.1.mixer.out_proj.q_scale", "shape": [ 2560, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 102400, "byteOffset": 26961920 }, { "name": "transformer.h.1.mlp.fc1.bias", "shape": [ 10240 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 27064320 } ], "md5sum": "bcfa3c4ed22d9c6c9e7da25d619afd71" }, { "dataPath": "params_shard_8.bin", "format": "compressed-shard", "nbytes": 27064320, "records": [ { "name": "transformer.h.1.mlp.fc1.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "transformer.h.1.mlp.fc1.q_scale", "shape": [ 10240, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "transformer.h.1.mlp.fc2.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 13516800 }, { "name": "transformer.h.1.mlp.fc2.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 13521920 }, { "name": "transformer.h.1.mlp.fc2.q_scale", "shape": [ 2560, 80 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 26629120 }, { "name": "transformer.h.10.ln.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 27038720 }, { "name": "transformer.h.10.ln.weight", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 27043840 }, { "name": "transformer.h.10.mixer.Wqkv.bias", "shape": [ 7680 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 27048960 } ], "md5sum": "fe3e3a3a6e06b04228cc4ae23171c4bf" }, { "dataPath": "params_shard_9.bin", "format": "compressed-shard", "nbytes": 27064320, "records": [ { "name": "transformer.h.10.mixer.Wqkv.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 0 }, { "name": "transformer.h.10.mixer.Wqkv.q_scale", "shape": [ 7680, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 307200, "byteOffset": 9830400 }, { "name": "transformer.h.10.mixer.out_proj.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 10137600 }, { "name": "transformer.h.10.mixer.out_proj.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 10142720 }, { "name": "transformer.h.10.mixer.out_proj.q_scale", "shape": [ 2560, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 102400, "byteOffset": 13419520 }, { "name": "transformer.h.10.mlp.fc1.bias", "shape": [ 10240 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 13521920 }, { "name": "transformer.h.10.mlp.fc1.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 13542400 }, { "name": "transformer.h.10.mlp.fc1.q_scale", "shape": [ 10240, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 26649600 }, { "name": "transformer.h.10.mlp.fc2.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 27059200 } ], "md5sum": "7407feac12aa8deea8006f860325bae2" }, { "dataPath": "params_shard_10.bin", "format": "compressed-shard", "nbytes": 27084800, "records": [ { "name": "transformer.h.10.mlp.fc2.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "transformer.h.10.mlp.fc2.q_scale", "shape": [ 2560, 80 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "transformer.h.11.ln.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 13516800 }, { "name": "transformer.h.11.ln.weight", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 13521920 }, { "name": "transformer.h.11.mixer.Wqkv.bias", "shape": [ 7680 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 13527040 }, { "name": "transformer.h.11.mixer.Wqkv.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 13542400 }, { "name": "transformer.h.11.mixer.Wqkv.q_scale", "shape": [ 7680, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 307200, "byteOffset": 23372800 }, { "name": "transformer.h.11.mixer.out_proj.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 23680000 }, { "name": "transformer.h.11.mixer.out_proj.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 23685120 }, { "name": "transformer.h.11.mixer.out_proj.q_scale", "shape": [ 2560, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 102400, "byteOffset": 26961920 }, { "name": "transformer.h.11.mlp.fc1.bias", "shape": [ 10240 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 27064320 } ], "md5sum": "3c3c648d196405e485b141107fd82940" }, { "dataPath": "params_shard_11.bin", "format": "compressed-shard", "nbytes": 27064320, "records": [ { "name": "transformer.h.11.mlp.fc1.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "transformer.h.11.mlp.fc1.q_scale", "shape": [ 10240, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "transformer.h.11.mlp.fc2.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 13516800 }, { "name": "transformer.h.11.mlp.fc2.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 13521920 }, { "name": "transformer.h.11.mlp.fc2.q_scale", "shape": [ 2560, 80 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 26629120 }, { "name": "transformer.h.12.ln.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 27038720 }, { "name": "transformer.h.12.ln.weight", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 27043840 }, { "name": "transformer.h.12.mixer.Wqkv.bias", "shape": [ 7680 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 27048960 } ], "md5sum": "39684d8a986e8d1b5241ce31200102b6" }, { "dataPath": "params_shard_12.bin", "format": "compressed-shard", "nbytes": 27064320, "records": [ { "name": "transformer.h.12.mixer.Wqkv.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 0 }, { "name": "transformer.h.12.mixer.Wqkv.q_scale", "shape": [ 7680, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 307200, "byteOffset": 9830400 }, { "name": "transformer.h.12.mixer.out_proj.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 10137600 }, { "name": "transformer.h.12.mixer.out_proj.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 10142720 }, { "name": "transformer.h.12.mixer.out_proj.q_scale", "shape": [ 2560, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 102400, "byteOffset": 13419520 }, { "name": "transformer.h.12.mlp.fc1.bias", "shape": [ 10240 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 13521920 }, { "name": "transformer.h.12.mlp.fc1.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 13542400 }, { "name": "transformer.h.12.mlp.fc1.q_scale", "shape": [ 10240, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 26649600 }, { "name": "transformer.h.12.mlp.fc2.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 27059200 } ], "md5sum": "44f2ccfdd1e22639c2a6f327f4b30e02" }, { "dataPath": "params_shard_13.bin", "format": "compressed-shard", "nbytes": 27084800, "records": [ { "name": "transformer.h.12.mlp.fc2.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "transformer.h.12.mlp.fc2.q_scale", "shape": [ 2560, 80 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "transformer.h.13.ln.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 13516800 }, { "name": "transformer.h.13.ln.weight", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 13521920 }, { "name": "transformer.h.13.mixer.Wqkv.bias", "shape": [ 7680 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 13527040 }, { "name": "transformer.h.13.mixer.Wqkv.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 13542400 }, { "name": "transformer.h.13.mixer.Wqkv.q_scale", "shape": [ 7680, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 307200, "byteOffset": 23372800 }, { "name": "transformer.h.13.mixer.out_proj.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 23680000 }, { "name": "transformer.h.13.mixer.out_proj.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 23685120 }, { "name": "transformer.h.13.mixer.out_proj.q_scale", "shape": [ 2560, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 102400, "byteOffset": 26961920 }, { "name": "transformer.h.13.mlp.fc1.bias", "shape": [ 10240 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 27064320 } ], "md5sum": "b46367f465d50d983e4882e367e9c2c6" }, { "dataPath": "params_shard_14.bin", "format": "compressed-shard", "nbytes": 27064320, "records": [ { "name": "transformer.h.13.mlp.fc1.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "transformer.h.13.mlp.fc1.q_scale", "shape": [ 10240, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "transformer.h.13.mlp.fc2.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 13516800 }, { "name": "transformer.h.13.mlp.fc2.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 13521920 }, { "name": "transformer.h.13.mlp.fc2.q_scale", "shape": [ 2560, 80 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 26629120 }, { "name": "transformer.h.14.ln.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 27038720 }, { "name": "transformer.h.14.ln.weight", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 27043840 }, { "name": "transformer.h.14.mixer.Wqkv.bias", "shape": [ 7680 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 27048960 } ], "md5sum": "e68319a53ecbdab6ac06e5297191773b" }, { "dataPath": "params_shard_15.bin", "format": "compressed-shard", "nbytes": 27064320, "records": [ { "name": "transformer.h.14.mixer.Wqkv.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 0 }, { "name": "transformer.h.14.mixer.Wqkv.q_scale", "shape": [ 7680, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 307200, "byteOffset": 9830400 }, { "name": "transformer.h.14.mixer.out_proj.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 10137600 }, { "name": "transformer.h.14.mixer.out_proj.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 10142720 }, { "name": "transformer.h.14.mixer.out_proj.q_scale", "shape": [ 2560, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 102400, "byteOffset": 13419520 }, { "name": "transformer.h.14.mlp.fc1.bias", "shape": [ 10240 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 13521920 }, { "name": "transformer.h.14.mlp.fc1.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 13542400 }, { "name": "transformer.h.14.mlp.fc1.q_scale", "shape": [ 10240, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 26649600 }, { "name": "transformer.h.14.mlp.fc2.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 27059200 } ], "md5sum": "2540ee26e1efa5e4e0bbd000911d835b" }, { "dataPath": "params_shard_16.bin", "format": "compressed-shard", "nbytes": 27084800, "records": [ { "name": "transformer.h.14.mlp.fc2.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "transformer.h.14.mlp.fc2.q_scale", "shape": [ 2560, 80 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "transformer.h.15.ln.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 13516800 }, { "name": "transformer.h.15.ln.weight", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 13521920 }, { "name": "transformer.h.15.mixer.Wqkv.bias", "shape": [ 7680 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 13527040 }, { "name": "transformer.h.15.mixer.Wqkv.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 13542400 }, { "name": "transformer.h.15.mixer.Wqkv.q_scale", "shape": [ 7680, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 307200, "byteOffset": 23372800 }, { "name": "transformer.h.15.mixer.out_proj.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 23680000 }, { "name": "transformer.h.15.mixer.out_proj.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 23685120 }, { "name": "transformer.h.15.mixer.out_proj.q_scale", "shape": [ 2560, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 102400, "byteOffset": 26961920 }, { "name": "transformer.h.15.mlp.fc1.bias", "shape": [ 10240 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 27064320 } ], "md5sum": "bdcaeeb0ea4fd762e8de5d020e79c709" }, { "dataPath": "params_shard_17.bin", "format": "compressed-shard", "nbytes": 27064320, "records": [ { "name": "transformer.h.15.mlp.fc1.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "transformer.h.15.mlp.fc1.q_scale", "shape": [ 10240, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "transformer.h.15.mlp.fc2.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 13516800 }, { "name": "transformer.h.15.mlp.fc2.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 13521920 }, { "name": "transformer.h.15.mlp.fc2.q_scale", "shape": [ 2560, 80 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 26629120 }, { "name": "transformer.h.16.ln.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 27038720 }, { "name": "transformer.h.16.ln.weight", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 27043840 }, { "name": "transformer.h.16.mixer.Wqkv.bias", "shape": [ 7680 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 27048960 } ], "md5sum": "ce30fd44576b15ceee4bfb782350bea8" }, { "dataPath": "params_shard_18.bin", "format": "compressed-shard", "nbytes": 27064320, "records": [ { "name": "transformer.h.16.mixer.Wqkv.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 0 }, { "name": "transformer.h.16.mixer.Wqkv.q_scale", "shape": [ 7680, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 307200, "byteOffset": 9830400 }, { "name": "transformer.h.16.mixer.out_proj.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 10137600 }, { "name": "transformer.h.16.mixer.out_proj.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 10142720 }, { "name": "transformer.h.16.mixer.out_proj.q_scale", "shape": [ 2560, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 102400, "byteOffset": 13419520 }, { "name": "transformer.h.16.mlp.fc1.bias", "shape": [ 10240 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 13521920 }, { "name": "transformer.h.16.mlp.fc1.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 13542400 }, { "name": "transformer.h.16.mlp.fc1.q_scale", "shape": [ 10240, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 26649600 }, { "name": "transformer.h.16.mlp.fc2.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 27059200 } ], "md5sum": "aad86806f4fed5ee3eb2ca63423bc999" }, { "dataPath": "params_shard_19.bin", "format": "compressed-shard", "nbytes": 27084800, "records": [ { "name": "transformer.h.16.mlp.fc2.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "transformer.h.16.mlp.fc2.q_scale", "shape": [ 2560, 80 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "transformer.h.17.ln.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 13516800 }, { "name": "transformer.h.17.ln.weight", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 13521920 }, { "name": "transformer.h.17.mixer.Wqkv.bias", "shape": [ 7680 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 13527040 }, { "name": "transformer.h.17.mixer.Wqkv.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 13542400 }, { "name": "transformer.h.17.mixer.Wqkv.q_scale", "shape": [ 7680, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 307200, "byteOffset": 23372800 }, { "name": "transformer.h.17.mixer.out_proj.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 23680000 }, { "name": "transformer.h.17.mixer.out_proj.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 23685120 }, { "name": "transformer.h.17.mixer.out_proj.q_scale", "shape": [ 2560, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 102400, "byteOffset": 26961920 }, { "name": "transformer.h.17.mlp.fc1.bias", "shape": [ 10240 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 27064320 } ], "md5sum": "9a6051009007cfbdbd368a6d7d19a552" }, { "dataPath": "params_shard_20.bin", "format": "compressed-shard", "nbytes": 27064320, "records": [ { "name": "transformer.h.17.mlp.fc1.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "transformer.h.17.mlp.fc1.q_scale", "shape": [ 10240, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "transformer.h.17.mlp.fc2.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 13516800 }, { "name": "transformer.h.17.mlp.fc2.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 13521920 }, { "name": "transformer.h.17.mlp.fc2.q_scale", "shape": [ 2560, 80 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 26629120 }, { "name": "transformer.h.18.ln.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 27038720 }, { "name": "transformer.h.18.ln.weight", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 27043840 }, { "name": "transformer.h.18.mixer.Wqkv.bias", "shape": [ 7680 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 27048960 } ], "md5sum": "467ae645ee412afc22f2de28a0a3cc5b" }, { "dataPath": "params_shard_21.bin", "format": "compressed-shard", "nbytes": 27064320, "records": [ { "name": "transformer.h.18.mixer.Wqkv.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 0 }, { "name": "transformer.h.18.mixer.Wqkv.q_scale", "shape": [ 7680, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 307200, "byteOffset": 9830400 }, { "name": "transformer.h.18.mixer.out_proj.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 10137600 }, { "name": "transformer.h.18.mixer.out_proj.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 10142720 }, { "name": "transformer.h.18.mixer.out_proj.q_scale", "shape": [ 2560, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 102400, "byteOffset": 13419520 }, { "name": "transformer.h.18.mlp.fc1.bias", "shape": [ 10240 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 13521920 }, { "name": "transformer.h.18.mlp.fc1.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 13542400 }, { "name": "transformer.h.18.mlp.fc1.q_scale", "shape": [ 10240, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 26649600 }, { "name": "transformer.h.18.mlp.fc2.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 27059200 } ], "md5sum": "39fc9ef8851bc236f5ac58b9fc64ef5a" }, { "dataPath": "params_shard_22.bin", "format": "compressed-shard", "nbytes": 27084800, "records": [ { "name": "transformer.h.18.mlp.fc2.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "transformer.h.18.mlp.fc2.q_scale", "shape": [ 2560, 80 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "transformer.h.19.ln.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 13516800 }, { "name": "transformer.h.19.ln.weight", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 13521920 }, { "name": "transformer.h.19.mixer.Wqkv.bias", "shape": [ 7680 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 13527040 }, { "name": "transformer.h.19.mixer.Wqkv.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 13542400 }, { "name": "transformer.h.19.mixer.Wqkv.q_scale", "shape": [ 7680, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 307200, "byteOffset": 23372800 }, { "name": "transformer.h.19.mixer.out_proj.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 23680000 }, { "name": "transformer.h.19.mixer.out_proj.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 23685120 }, { "name": "transformer.h.19.mixer.out_proj.q_scale", "shape": [ 2560, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 102400, "byteOffset": 26961920 }, { "name": "transformer.h.19.mlp.fc1.bias", "shape": [ 10240 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 27064320 } ], "md5sum": "c87023db62f6a30621b5ea43b43cb0b6" }, { "dataPath": "params_shard_23.bin", "format": "compressed-shard", "nbytes": 27064320, "records": [ { "name": "transformer.h.19.mlp.fc1.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "transformer.h.19.mlp.fc1.q_scale", "shape": [ 10240, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "transformer.h.19.mlp.fc2.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 13516800 }, { "name": "transformer.h.19.mlp.fc2.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 13521920 }, { "name": "transformer.h.19.mlp.fc2.q_scale", "shape": [ 2560, 80 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 26629120 }, { "name": "transformer.h.2.ln.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 27038720 }, { "name": "transformer.h.2.ln.weight", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 27043840 }, { "name": "transformer.h.2.mixer.Wqkv.bias", "shape": [ 7680 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 27048960 } ], "md5sum": "102537956bff3c37e4c0673c42c27999" }, { "dataPath": "params_shard_24.bin", "format": "compressed-shard", "nbytes": 27064320, "records": [ { "name": "transformer.h.2.mixer.Wqkv.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 0 }, { "name": "transformer.h.2.mixer.Wqkv.q_scale", "shape": [ 7680, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 307200, "byteOffset": 9830400 }, { "name": "transformer.h.2.mixer.out_proj.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 10137600 }, { "name": "transformer.h.2.mixer.out_proj.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 10142720 }, { "name": "transformer.h.2.mixer.out_proj.q_scale", "shape": [ 2560, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 102400, "byteOffset": 13419520 }, { "name": "transformer.h.2.mlp.fc1.bias", "shape": [ 10240 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 13521920 }, { "name": "transformer.h.2.mlp.fc1.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 13542400 }, { "name": "transformer.h.2.mlp.fc1.q_scale", "shape": [ 10240, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 26649600 }, { "name": "transformer.h.2.mlp.fc2.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 27059200 } ], "md5sum": "eb8edbec70caaf36e6e97de0d128c275" }, { "dataPath": "params_shard_25.bin", "format": "compressed-shard", "nbytes": 27084800, "records": [ { "name": "transformer.h.2.mlp.fc2.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "transformer.h.2.mlp.fc2.q_scale", "shape": [ 2560, 80 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "transformer.h.20.ln.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 13516800 }, { "name": "transformer.h.20.ln.weight", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 13521920 }, { "name": "transformer.h.20.mixer.Wqkv.bias", "shape": [ 7680 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 13527040 }, { "name": "transformer.h.20.mixer.Wqkv.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 13542400 }, { "name": "transformer.h.20.mixer.Wqkv.q_scale", "shape": [ 7680, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 307200, "byteOffset": 23372800 }, { "name": "transformer.h.20.mixer.out_proj.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 23680000 }, { "name": "transformer.h.20.mixer.out_proj.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 23685120 }, { "name": "transformer.h.20.mixer.out_proj.q_scale", "shape": [ 2560, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 102400, "byteOffset": 26961920 }, { "name": "transformer.h.20.mlp.fc1.bias", "shape": [ 10240 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 27064320 } ], "md5sum": "d8364003461a1c2903b8e3082d40c9d4" }, { "dataPath": "params_shard_26.bin", "format": "compressed-shard", "nbytes": 27064320, "records": [ { "name": "transformer.h.20.mlp.fc1.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "transformer.h.20.mlp.fc1.q_scale", "shape": [ 10240, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "transformer.h.20.mlp.fc2.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 13516800 }, { "name": "transformer.h.20.mlp.fc2.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 13521920 }, { "name": "transformer.h.20.mlp.fc2.q_scale", "shape": [ 2560, 80 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 26629120 }, { "name": "transformer.h.21.ln.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 27038720 }, { "name": "transformer.h.21.ln.weight", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 27043840 }, { "name": "transformer.h.21.mixer.Wqkv.bias", "shape": [ 7680 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 27048960 } ], "md5sum": "c4ccc6bb97f7abe44b4dd87f7c43ee4f" }, { "dataPath": "params_shard_27.bin", "format": "compressed-shard", "nbytes": 27064320, "records": [ { "name": "transformer.h.21.mixer.Wqkv.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 0 }, { "name": "transformer.h.21.mixer.Wqkv.q_scale", "shape": [ 7680, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 307200, "byteOffset": 9830400 }, { "name": "transformer.h.21.mixer.out_proj.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 10137600 }, { "name": "transformer.h.21.mixer.out_proj.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 10142720 }, { "name": "transformer.h.21.mixer.out_proj.q_scale", "shape": [ 2560, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 102400, "byteOffset": 13419520 }, { "name": "transformer.h.21.mlp.fc1.bias", "shape": [ 10240 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 13521920 }, { "name": "transformer.h.21.mlp.fc1.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 13542400 }, { "name": "transformer.h.21.mlp.fc1.q_scale", "shape": [ 10240, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 26649600 }, { "name": "transformer.h.21.mlp.fc2.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 27059200 } ], "md5sum": "81ca930db9522020400c2ce9187d8e77" }, { "dataPath": "params_shard_28.bin", "format": "compressed-shard", "nbytes": 27084800, "records": [ { "name": "transformer.h.21.mlp.fc2.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "transformer.h.21.mlp.fc2.q_scale", "shape": [ 2560, 80 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "transformer.h.22.ln.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 13516800 }, { "name": "transformer.h.22.ln.weight", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 13521920 }, { "name": "transformer.h.22.mixer.Wqkv.bias", "shape": [ 7680 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 13527040 }, { "name": "transformer.h.22.mixer.Wqkv.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 13542400 }, { "name": "transformer.h.22.mixer.Wqkv.q_scale", "shape": [ 7680, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 307200, "byteOffset": 23372800 }, { "name": "transformer.h.22.mixer.out_proj.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 23680000 }, { "name": "transformer.h.22.mixer.out_proj.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 23685120 }, { "name": "transformer.h.22.mixer.out_proj.q_scale", "shape": [ 2560, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 102400, "byteOffset": 26961920 }, { "name": "transformer.h.22.mlp.fc1.bias", "shape": [ 10240 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 27064320 } ], "md5sum": "5fe3d4b67dc674ec561963945255131f" }, { "dataPath": "params_shard_29.bin", "format": "compressed-shard", "nbytes": 27064320, "records": [ { "name": "transformer.h.22.mlp.fc1.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "transformer.h.22.mlp.fc1.q_scale", "shape": [ 10240, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "transformer.h.22.mlp.fc2.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 13516800 }, { "name": "transformer.h.22.mlp.fc2.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 13521920 }, { "name": "transformer.h.22.mlp.fc2.q_scale", "shape": [ 2560, 80 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 26629120 }, { "name": "transformer.h.23.ln.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 27038720 }, { "name": "transformer.h.23.ln.weight", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 27043840 }, { "name": "transformer.h.23.mixer.Wqkv.bias", "shape": [ 7680 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 27048960 } ], "md5sum": "8099737cd1fc47616dc819a999f9f7b8" }, { "dataPath": "params_shard_30.bin", "format": "compressed-shard", "nbytes": 27064320, "records": [ { "name": "transformer.h.23.mixer.Wqkv.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 0 }, { "name": "transformer.h.23.mixer.Wqkv.q_scale", "shape": [ 7680, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 307200, "byteOffset": 9830400 }, { "name": "transformer.h.23.mixer.out_proj.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 10137600 }, { "name": "transformer.h.23.mixer.out_proj.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 10142720 }, { "name": "transformer.h.23.mixer.out_proj.q_scale", "shape": [ 2560, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 102400, "byteOffset": 13419520 }, { "name": "transformer.h.23.mlp.fc1.bias", "shape": [ 10240 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 13521920 }, { "name": "transformer.h.23.mlp.fc1.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 13542400 }, { "name": "transformer.h.23.mlp.fc1.q_scale", "shape": [ 10240, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 26649600 }, { "name": "transformer.h.23.mlp.fc2.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 27059200 } ], "md5sum": "6ea1996741061d0b3c48ab0a3ffe0709" }, { "dataPath": "params_shard_31.bin", "format": "compressed-shard", "nbytes": 27084800, "records": [ { "name": "transformer.h.23.mlp.fc2.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "transformer.h.23.mlp.fc2.q_scale", "shape": [ 2560, 80 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "transformer.h.24.ln.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 13516800 }, { "name": "transformer.h.24.ln.weight", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 13521920 }, { "name": "transformer.h.24.mixer.Wqkv.bias", "shape": [ 7680 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 13527040 }, { "name": "transformer.h.24.mixer.Wqkv.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 13542400 }, { "name": "transformer.h.24.mixer.Wqkv.q_scale", "shape": [ 7680, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 307200, "byteOffset": 23372800 }, { "name": "transformer.h.24.mixer.out_proj.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 23680000 }, { "name": "transformer.h.24.mixer.out_proj.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 23685120 }, { "name": "transformer.h.24.mixer.out_proj.q_scale", "shape": [ 2560, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 102400, "byteOffset": 26961920 }, { "name": "transformer.h.24.mlp.fc1.bias", "shape": [ 10240 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 27064320 } ], "md5sum": "fc0d6646718d853204079fbc12e9d968" }, { "dataPath": "params_shard_32.bin", "format": "compressed-shard", "nbytes": 27064320, "records": [ { "name": "transformer.h.24.mlp.fc1.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "transformer.h.24.mlp.fc1.q_scale", "shape": [ 10240, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "transformer.h.24.mlp.fc2.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 13516800 }, { "name": "transformer.h.24.mlp.fc2.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 13521920 }, { "name": "transformer.h.24.mlp.fc2.q_scale", "shape": [ 2560, 80 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 26629120 }, { "name": "transformer.h.25.ln.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 27038720 }, { "name": "transformer.h.25.ln.weight", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 27043840 }, { "name": "transformer.h.25.mixer.Wqkv.bias", "shape": [ 7680 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 27048960 } ], "md5sum": "2a95f3cd50afd6a231c7b74d5d166178" }, { "dataPath": "params_shard_33.bin", "format": "compressed-shard", "nbytes": 27064320, "records": [ { "name": "transformer.h.25.mixer.Wqkv.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 0 }, { "name": "transformer.h.25.mixer.Wqkv.q_scale", "shape": [ 7680, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 307200, "byteOffset": 9830400 }, { "name": "transformer.h.25.mixer.out_proj.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 10137600 }, { "name": "transformer.h.25.mixer.out_proj.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 10142720 }, { "name": "transformer.h.25.mixer.out_proj.q_scale", "shape": [ 2560, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 102400, "byteOffset": 13419520 }, { "name": "transformer.h.25.mlp.fc1.bias", "shape": [ 10240 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 13521920 }, { "name": "transformer.h.25.mlp.fc1.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 13542400 }, { "name": "transformer.h.25.mlp.fc1.q_scale", "shape": [ 10240, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 26649600 }, { "name": "transformer.h.25.mlp.fc2.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 27059200 } ], "md5sum": "4dac38b06ac30e9079935f56ea707669" }, { "dataPath": "params_shard_34.bin", "format": "compressed-shard", "nbytes": 27084800, "records": [ { "name": "transformer.h.25.mlp.fc2.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "transformer.h.25.mlp.fc2.q_scale", "shape": [ 2560, 80 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "transformer.h.26.ln.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 13516800 }, { "name": "transformer.h.26.ln.weight", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 13521920 }, { "name": "transformer.h.26.mixer.Wqkv.bias", "shape": [ 7680 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 13527040 }, { "name": "transformer.h.26.mixer.Wqkv.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 13542400 }, { "name": "transformer.h.26.mixer.Wqkv.q_scale", "shape": [ 7680, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 307200, "byteOffset": 23372800 }, { "name": "transformer.h.26.mixer.out_proj.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 23680000 }, { "name": "transformer.h.26.mixer.out_proj.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 23685120 }, { "name": "transformer.h.26.mixer.out_proj.q_scale", "shape": [ 2560, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 102400, "byteOffset": 26961920 }, { "name": "transformer.h.26.mlp.fc1.bias", "shape": [ 10240 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 27064320 } ], "md5sum": "b48bbb6157549131c16ad6e841add9ef" }, { "dataPath": "params_shard_35.bin", "format": "compressed-shard", "nbytes": 27064320, "records": [ { "name": "transformer.h.26.mlp.fc1.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "transformer.h.26.mlp.fc1.q_scale", "shape": [ 10240, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "transformer.h.26.mlp.fc2.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 13516800 }, { "name": "transformer.h.26.mlp.fc2.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 13521920 }, { "name": "transformer.h.26.mlp.fc2.q_scale", "shape": [ 2560, 80 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 26629120 }, { "name": "transformer.h.27.ln.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 27038720 }, { "name": "transformer.h.27.ln.weight", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 27043840 }, { "name": "transformer.h.27.mixer.Wqkv.bias", "shape": [ 7680 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 27048960 } ], "md5sum": "b543038abd2cc89abc8a3fba960aa9c8" }, { "dataPath": "params_shard_36.bin", "format": "compressed-shard", "nbytes": 27064320, "records": [ { "name": "transformer.h.27.mixer.Wqkv.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 0 }, { "name": "transformer.h.27.mixer.Wqkv.q_scale", "shape": [ 7680, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 307200, "byteOffset": 9830400 }, { "name": "transformer.h.27.mixer.out_proj.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 10137600 }, { "name": "transformer.h.27.mixer.out_proj.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 10142720 }, { "name": "transformer.h.27.mixer.out_proj.q_scale", "shape": [ 2560, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 102400, "byteOffset": 13419520 }, { "name": "transformer.h.27.mlp.fc1.bias", "shape": [ 10240 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 13521920 }, { "name": "transformer.h.27.mlp.fc1.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 13542400 }, { "name": "transformer.h.27.mlp.fc1.q_scale", "shape": [ 10240, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 26649600 }, { "name": "transformer.h.27.mlp.fc2.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 27059200 } ], "md5sum": "b48d9b687eac7178c5421afe794f7e96" }, { "dataPath": "params_shard_37.bin", "format": "compressed-shard", "nbytes": 27084800, "records": [ { "name": "transformer.h.27.mlp.fc2.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "transformer.h.27.mlp.fc2.q_scale", "shape": [ 2560, 80 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "transformer.h.28.ln.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 13516800 }, { "name": "transformer.h.28.ln.weight", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 13521920 }, { "name": "transformer.h.28.mixer.Wqkv.bias", "shape": [ 7680 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 13527040 }, { "name": "transformer.h.28.mixer.Wqkv.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 13542400 }, { "name": "transformer.h.28.mixer.Wqkv.q_scale", "shape": [ 7680, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 307200, "byteOffset": 23372800 }, { "name": "transformer.h.28.mixer.out_proj.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 23680000 }, { "name": "transformer.h.28.mixer.out_proj.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 23685120 }, { "name": "transformer.h.28.mixer.out_proj.q_scale", "shape": [ 2560, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 102400, "byteOffset": 26961920 }, { "name": "transformer.h.28.mlp.fc1.bias", "shape": [ 10240 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 27064320 } ], "md5sum": "6dc3c51a48580a401d7edb3d58ecc887" }, { "dataPath": "params_shard_38.bin", "format": "compressed-shard", "nbytes": 27064320, "records": [ { "name": "transformer.h.28.mlp.fc1.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "transformer.h.28.mlp.fc1.q_scale", "shape": [ 10240, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "transformer.h.28.mlp.fc2.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 13516800 }, { "name": "transformer.h.28.mlp.fc2.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 13521920 }, { "name": "transformer.h.28.mlp.fc2.q_scale", "shape": [ 2560, 80 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 26629120 }, { "name": "transformer.h.29.ln.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 27038720 }, { "name": "transformer.h.29.ln.weight", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 27043840 }, { "name": "transformer.h.29.mixer.Wqkv.bias", "shape": [ 7680 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 27048960 } ], "md5sum": "531163003c677a1b4a22ae629a8abf3f" }, { "dataPath": "params_shard_39.bin", "format": "compressed-shard", "nbytes": 27089920, "records": [ { "name": "transformer.h.29.mixer.Wqkv.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 0 }, { "name": "transformer.h.29.mixer.Wqkv.q_scale", "shape": [ 7680, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 307200, "byteOffset": 9830400 }, { "name": "transformer.h.29.mixer.out_proj.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 10137600 }, { "name": "transformer.h.29.mixer.out_proj.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 10142720 }, { "name": "transformer.h.29.mixer.out_proj.q_scale", "shape": [ 2560, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 102400, "byteOffset": 13419520 }, { "name": "transformer.h.3.ln.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 13521920 }, { "name": "transformer.h.3.ln.weight", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 13527040 }, { "name": "transformer.h.3.mixer.Wqkv.bias", "shape": [ 7680 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 13532160 }, { "name": "transformer.h.3.mixer.Wqkv.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 13547520 }, { "name": "transformer.h.3.mixer.Wqkv.q_scale", "shape": [ 7680, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 307200, "byteOffset": 23377920 }, { "name": "transformer.h.3.mixer.out_proj.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 23685120 }, { "name": "transformer.h.3.mixer.out_proj.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 23690240 }, { "name": "transformer.h.3.mixer.out_proj.q_scale", "shape": [ 2560, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 102400, "byteOffset": 26967040 }, { "name": "transformer.h.3.mlp.fc1.bias", "shape": [ 10240 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 27069440 } ], "md5sum": "1bb35e2054170d9b8e963836daeff23a" }, { "dataPath": "params_shard_40.bin", "format": "compressed-shard", "nbytes": 27064320, "records": [ { "name": "transformer.h.3.mlp.fc1.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "transformer.h.3.mlp.fc1.q_scale", "shape": [ 10240, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "transformer.h.3.mlp.fc2.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 13516800 }, { "name": "transformer.h.3.mlp.fc2.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 13521920 }, { "name": "transformer.h.3.mlp.fc2.q_scale", "shape": [ 2560, 80 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 26629120 }, { "name": "transformer.h.4.ln.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 27038720 }, { "name": "transformer.h.4.ln.weight", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 27043840 }, { "name": "transformer.h.4.mixer.Wqkv.bias", "shape": [ 7680 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 27048960 } ], "md5sum": "f033155933f51e22094b87877641704f" }, { "dataPath": "params_shard_41.bin", "format": "compressed-shard", "nbytes": 27064320, "records": [ { "name": "transformer.h.4.mixer.Wqkv.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 0 }, { "name": "transformer.h.4.mixer.Wqkv.q_scale", "shape": [ 7680, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 307200, "byteOffset": 9830400 }, { "name": "transformer.h.4.mixer.out_proj.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 10137600 }, { "name": "transformer.h.4.mixer.out_proj.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 10142720 }, { "name": "transformer.h.4.mixer.out_proj.q_scale", "shape": [ 2560, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 102400, "byteOffset": 13419520 }, { "name": "transformer.h.4.mlp.fc1.bias", "shape": [ 10240 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 13521920 }, { "name": "transformer.h.4.mlp.fc1.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 13542400 }, { "name": "transformer.h.4.mlp.fc1.q_scale", "shape": [ 10240, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 26649600 }, { "name": "transformer.h.4.mlp.fc2.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 27059200 } ], "md5sum": "749a2cd58c87c6dfc2b686052ace3548" }, { "dataPath": "params_shard_42.bin", "format": "compressed-shard", "nbytes": 27084800, "records": [ { "name": "transformer.h.4.mlp.fc2.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "transformer.h.4.mlp.fc2.q_scale", "shape": [ 2560, 80 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "transformer.h.5.ln.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 13516800 }, { "name": "transformer.h.5.ln.weight", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 13521920 }, { "name": "transformer.h.5.mixer.Wqkv.bias", "shape": [ 7680 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 13527040 }, { "name": "transformer.h.5.mixer.Wqkv.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 13542400 }, { "name": "transformer.h.5.mixer.Wqkv.q_scale", "shape": [ 7680, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 307200, "byteOffset": 23372800 }, { "name": "transformer.h.5.mixer.out_proj.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 23680000 }, { "name": "transformer.h.5.mixer.out_proj.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 23685120 }, { "name": "transformer.h.5.mixer.out_proj.q_scale", "shape": [ 2560, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 102400, "byteOffset": 26961920 }, { "name": "transformer.h.5.mlp.fc1.bias", "shape": [ 10240 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 27064320 } ], "md5sum": "f722028950dcd5b537065740489128c0" }, { "dataPath": "params_shard_43.bin", "format": "compressed-shard", "nbytes": 27064320, "records": [ { "name": "transformer.h.5.mlp.fc1.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "transformer.h.5.mlp.fc1.q_scale", "shape": [ 10240, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "transformer.h.5.mlp.fc2.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 13516800 }, { "name": "transformer.h.5.mlp.fc2.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 13521920 }, { "name": "transformer.h.5.mlp.fc2.q_scale", "shape": [ 2560, 80 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 26629120 }, { "name": "transformer.h.6.ln.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 27038720 }, { "name": "transformer.h.6.ln.weight", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 27043840 }, { "name": "transformer.h.6.mixer.Wqkv.bias", "shape": [ 7680 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 27048960 } ], "md5sum": "7a344f9074b1b8c88c6ba5d54f3fbde2" }, { "dataPath": "params_shard_44.bin", "format": "compressed-shard", "nbytes": 27064320, "records": [ { "name": "transformer.h.6.mixer.Wqkv.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 0 }, { "name": "transformer.h.6.mixer.Wqkv.q_scale", "shape": [ 7680, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 307200, "byteOffset": 9830400 }, { "name": "transformer.h.6.mixer.out_proj.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 10137600 }, { "name": "transformer.h.6.mixer.out_proj.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 10142720 }, { "name": "transformer.h.6.mixer.out_proj.q_scale", "shape": [ 2560, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 102400, "byteOffset": 13419520 }, { "name": "transformer.h.6.mlp.fc1.bias", "shape": [ 10240 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 13521920 }, { "name": "transformer.h.6.mlp.fc1.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 13542400 }, { "name": "transformer.h.6.mlp.fc1.q_scale", "shape": [ 10240, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 26649600 }, { "name": "transformer.h.6.mlp.fc2.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 27059200 } ], "md5sum": "05911dbb2ba0e67894fa639e47eb6f37" }, { "dataPath": "params_shard_45.bin", "format": "compressed-shard", "nbytes": 27084800, "records": [ { "name": "transformer.h.6.mlp.fc2.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "transformer.h.6.mlp.fc2.q_scale", "shape": [ 2560, 80 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "transformer.h.7.ln.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 13516800 }, { "name": "transformer.h.7.ln.weight", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 13521920 }, { "name": "transformer.h.7.mixer.Wqkv.bias", "shape": [ 7680 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 13527040 }, { "name": "transformer.h.7.mixer.Wqkv.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 13542400 }, { "name": "transformer.h.7.mixer.Wqkv.q_scale", "shape": [ 7680, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 307200, "byteOffset": 23372800 }, { "name": "transformer.h.7.mixer.out_proj.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 23680000 }, { "name": "transformer.h.7.mixer.out_proj.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 23685120 }, { "name": "transformer.h.7.mixer.out_proj.q_scale", "shape": [ 2560, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 102400, "byteOffset": 26961920 }, { "name": "transformer.h.7.mlp.fc1.bias", "shape": [ 10240 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 27064320 } ], "md5sum": "9ea0935206a429f6d8ceeb3b094beb18" }, { "dataPath": "params_shard_46.bin", "format": "compressed-shard", "nbytes": 27064320, "records": [ { "name": "transformer.h.7.mlp.fc1.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "transformer.h.7.mlp.fc1.q_scale", "shape": [ 10240, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "transformer.h.7.mlp.fc2.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 13516800 }, { "name": "transformer.h.7.mlp.fc2.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 13521920 }, { "name": "transformer.h.7.mlp.fc2.q_scale", "shape": [ 2560, 80 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 26629120 }, { "name": "transformer.h.8.ln.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 27038720 }, { "name": "transformer.h.8.ln.weight", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 27043840 }, { "name": "transformer.h.8.mixer.Wqkv.bias", "shape": [ 7680 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 27048960 } ], "md5sum": "b9466585d50c945190e38014cc6c0154" }, { "dataPath": "params_shard_47.bin", "format": "compressed-shard", "nbytes": 27064320, "records": [ { "name": "transformer.h.8.mixer.Wqkv.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 0 }, { "name": "transformer.h.8.mixer.Wqkv.q_scale", "shape": [ 7680, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 307200, "byteOffset": 9830400 }, { "name": "transformer.h.8.mixer.out_proj.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 10137600 }, { "name": "transformer.h.8.mixer.out_proj.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 10142720 }, { "name": "transformer.h.8.mixer.out_proj.q_scale", "shape": [ 2560, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 102400, "byteOffset": 13419520 }, { "name": "transformer.h.8.mlp.fc1.bias", "shape": [ 10240 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 13521920 }, { "name": "transformer.h.8.mlp.fc1.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 13542400 }, { "name": "transformer.h.8.mlp.fc1.q_scale", "shape": [ 10240, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 26649600 }, { "name": "transformer.h.8.mlp.fc2.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 27059200 } ], "md5sum": "745802e3405b0a9cbde195c90e375999" }, { "dataPath": "params_shard_48.bin", "format": "compressed-shard", "nbytes": 27084800, "records": [ { "name": "transformer.h.8.mlp.fc2.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "transformer.h.8.mlp.fc2.q_scale", "shape": [ 2560, 80 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "transformer.h.9.ln.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 13516800 }, { "name": "transformer.h.9.ln.weight", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 13521920 }, { "name": "transformer.h.9.mixer.Wqkv.bias", "shape": [ 7680 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 13527040 }, { "name": "transformer.h.9.mixer.Wqkv.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 13542400 }, { "name": "transformer.h.9.mixer.Wqkv.q_scale", "shape": [ 7680, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 307200, "byteOffset": 23372800 }, { "name": "transformer.h.9.mixer.out_proj.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 23680000 }, { "name": "transformer.h.9.mixer.out_proj.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 23685120 }, { "name": "transformer.h.9.mixer.out_proj.q_scale", "shape": [ 2560, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 102400, "byteOffset": 26961920 }, { "name": "transformer.h.9.mlp.fc1.bias", "shape": [ 10240 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 27064320 } ], "md5sum": "8eca0b6181877867e814ffb9ce691964" }, { "dataPath": "params_shard_49.bin", "format": "compressed-shard", "nbytes": 27038720, "records": [ { "name": "transformer.h.9.mlp.fc1.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "transformer.h.9.mlp.fc1.q_scale", "shape": [ 10240, 20 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "transformer.h.9.mlp.fc2.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 13516800 }, { "name": "transformer.h.9.mlp.fc2.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 13521920 }, { "name": "transformer.h.9.mlp.fc2.q_scale", "shape": [ 2560, 80 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 26629120 } ], "md5sum": "05c91953a697d13b3dfdb3793ce2a4c8" } ] }