{ "metadata": { "ParamSize": 323, "ParamBytes": 2432833536.0, "BitsPerParam": 5.093499892671169 }, "records": [ { "dataPath": "params_shard_0.bin", "format": "raw-shard", "nbytes": 197001216, "records": [ { "name": "lm_head.weight", "shape": [ 32064, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 197001216, "byteOffset": 0 } ], "md5sum": "8c2d85352caa7c9a98d74a9ee3cf6a94" }, { "dataPath": "params_shard_1.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "transformer.h.21.mlp.gate_up_proj.q_weight", "shape": [ 16384, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "90064ab197e2b701de79e935bc7f73e3" }, { "dataPath": "params_shard_2.bin", "format": "raw-shard", "nbytes": 33245184, "records": [ { "name": "transformer.h.21.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 0 }, { "name": "transformer.h.21.mlp.down_proj.q_weight", "shape": [ 3072, 1024 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 12582912, "byteOffset": 6144 }, { "name": "transformer.h.21.mlp.down_proj.q_scale", "shape": [ 3072, 256 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1572864, "byteOffset": 12589056 }, { "name": "transformer.h.21.mlp.gate_up_proj.q_scale", "shape": [ 16384, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3145728, "byteOffset": 14161920 }, { "name": "transformer.h.21.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 17307648 }, { "name": "transformer.h.21.mixer.qkv_proj.q_weight", "shape": [ 9216, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 14155776, "byteOffset": 17313792 }, { "name": "transformer.h.21.mixer.qkv_proj.q_scale", "shape": [ 9216, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1769472, "byteOffset": 31469568 }, { "name": "transformer.h.22.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 33239040 } ], "md5sum": "cf181094af30537db5593b17038f55fa" }, { "dataPath": "params_shard_3.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "transformer.h.22.mlp.gate_up_proj.q_weight", "shape": [ 16384, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "6c426fe4d21283c43d216f8adee2f132" }, { "dataPath": "params_shard_4.bin", "format": "raw-shard", "nbytes": 22616064, "records": [ { "name": "transformer.h.22.mlp.down_proj.q_weight", "shape": [ 3072, 1024 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 12582912, "byteOffset": 0 }, { "name": "transformer.h.22.mlp.down_proj.q_scale", "shape": [ 3072, 256 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1572864, "byteOffset": 12582912 }, { "name": "transformer.h.22.mlp.gate_up_proj.q_scale", "shape": [ 16384, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3145728, "byteOffset": 14155776 }, { "name": "transformer.h.22.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 17301504 }, { "name": "transformer.h.22.mixer.out_proj.q_weight", "shape": [ 3072, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 17307648 }, { "name": "transformer.h.22.mixer.out_proj.q_scale", "shape": [ 3072, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 589824, "byteOffset": 22026240 } ], "md5sum": "69b10d42fb7884e7092244ed74ab0d3e" }, { "dataPath": "params_shard_5.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "transformer.h.23.mlp.gate_up_proj.q_weight", "shape": [ 16384, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "8933fcd99220a1a6a07946589dd300e2" }, { "dataPath": "params_shard_6.bin", "format": "raw-shard", "nbytes": 33239040, "records": [ { "name": "transformer.h.22.mixer.qkv_proj.q_weight", "shape": [ 9216, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 14155776, "byteOffset": 0 }, { "name": "transformer.h.22.mixer.qkv_proj.q_scale", "shape": [ 9216, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1769472, "byteOffset": 14155776 }, { "name": "transformer.h.23.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 15925248 }, { "name": "transformer.h.23.mlp.down_proj.q_weight", "shape": [ 3072, 1024 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 12582912, "byteOffset": 15931392 }, { "name": "transformer.h.23.mlp.down_proj.q_scale", "shape": [ 3072, 256 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1572864, "byteOffset": 28514304 }, { "name": "transformer.h.23.mlp.gate_up_proj.q_scale", "shape": [ 16384, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3145728, "byteOffset": 30087168 }, { "name": "transformer.h.23.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 33232896 } ], "md5sum": "dcd9069b78842a46882a4a82455239eb" }, { "dataPath": "params_shard_7.bin", "format": "raw-shard", "nbytes": 21239808, "records": [ { "name": "transformer.h.23.mixer.out_proj.q_weight", "shape": [ 3072, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 0 }, { "name": "transformer.h.23.mixer.out_proj.q_scale", "shape": [ 3072, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 589824, "byteOffset": 4718592 }, { "name": "transformer.h.23.mixer.qkv_proj.q_weight", "shape": [ 9216, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 14155776, "byteOffset": 5308416 }, { "name": "transformer.h.23.mixer.qkv_proj.q_scale", "shape": [ 9216, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1769472, "byteOffset": 19464192 }, { "name": "transformer.h.24.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 21233664 } ], "md5sum": "7f64871f45c7e6cdfa02f420d017d71f" }, { "dataPath": "params_shard_8.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "transformer.h.24.mlp.gate_up_proj.q_weight", "shape": [ 16384, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "e07db77431a4b6fb768f0c014ff3e110" }, { "dataPath": "params_shard_9.bin", "format": "raw-shard", "nbytes": 22616064, "records": [ { "name": "transformer.h.24.mlp.down_proj.q_weight", "shape": [ 3072, 1024 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 12582912, "byteOffset": 0 }, { "name": "transformer.h.24.mlp.down_proj.q_scale", "shape": [ 3072, 256 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1572864, "byteOffset": 12582912 }, { "name": "transformer.h.24.mlp.gate_up_proj.q_scale", "shape": [ 16384, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3145728, "byteOffset": 14155776 }, { "name": "transformer.h.24.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 17301504 }, { "name": "transformer.h.24.mixer.out_proj.q_weight", "shape": [ 3072, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 17307648 }, { "name": "transformer.h.24.mixer.out_proj.q_scale", "shape": [ 3072, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 589824, "byteOffset": 22026240 } ], "md5sum": "0e3b6fe177c0a3d428ac2cc5de10a45a" }, { "dataPath": "params_shard_10.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "transformer.h.25.mlp.gate_up_proj.q_weight", "shape": [ 16384, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "b1dffdc55fe050f07dea1cbfea2e4cdd" }, { "dataPath": "params_shard_11.bin", "format": "raw-shard", "nbytes": 33239040, "records": [ { "name": "transformer.h.24.mixer.qkv_proj.q_weight", "shape": [ 9216, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 14155776, "byteOffset": 0 }, { "name": "transformer.h.24.mixer.qkv_proj.q_scale", "shape": [ 9216, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1769472, "byteOffset": 14155776 }, { "name": "transformer.h.25.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 15925248 }, { "name": "transformer.h.25.mlp.down_proj.q_weight", "shape": [ 3072, 1024 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 12582912, "byteOffset": 15931392 }, { "name": "transformer.h.25.mlp.down_proj.q_scale", "shape": [ 3072, 256 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1572864, "byteOffset": 28514304 }, { "name": "transformer.h.25.mlp.gate_up_proj.q_scale", "shape": [ 16384, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3145728, "byteOffset": 30087168 }, { "name": "transformer.h.25.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 33232896 } ], "md5sum": "738f3c219adef5193ff9a4d3787a1529" }, { "dataPath": "params_shard_12.bin", "format": "raw-shard", "nbytes": 21239808, "records": [ { "name": "transformer.h.25.mixer.out_proj.q_weight", "shape": [ 3072, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 0 }, { "name": "transformer.h.25.mixer.out_proj.q_scale", "shape": [ 3072, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 589824, "byteOffset": 4718592 }, { "name": "transformer.h.25.mixer.qkv_proj.q_weight", "shape": [ 9216, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 14155776, "byteOffset": 5308416 }, { "name": "transformer.h.25.mixer.qkv_proj.q_scale", "shape": [ 9216, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1769472, "byteOffset": 19464192 }, { "name": "transformer.h.26.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 21233664 } ], "md5sum": "426dff6da0a4b86efb349129881a7780" }, { "dataPath": "params_shard_13.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "transformer.h.26.mlp.gate_up_proj.q_weight", "shape": [ 16384, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "c2feefdb20aba2e2903e61572ae56632" }, { "dataPath": "params_shard_14.bin", "format": "raw-shard", "nbytes": 22616064, "records": [ { "name": "transformer.h.26.mlp.down_proj.q_weight", "shape": [ 3072, 1024 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 12582912, "byteOffset": 0 }, { "name": "transformer.h.26.mlp.down_proj.q_scale", "shape": [ 3072, 256 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1572864, "byteOffset": 12582912 }, { "name": "transformer.h.26.mlp.gate_up_proj.q_scale", "shape": [ 16384, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3145728, "byteOffset": 14155776 }, { "name": "transformer.h.26.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 17301504 }, { "name": "transformer.h.26.mixer.out_proj.q_weight", "shape": [ 3072, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 17307648 }, { "name": "transformer.h.26.mixer.out_proj.q_scale", "shape": [ 3072, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 589824, "byteOffset": 22026240 } ], "md5sum": "49300e430938362d55c28a9b3c9c6094" }, { "dataPath": "params_shard_15.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "transformer.h.27.mlp.gate_up_proj.q_weight", "shape": [ 16384, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "ab950ef1df4c15b05aeb8fd509c9109e" }, { "dataPath": "params_shard_16.bin", "format": "raw-shard", "nbytes": 33239040, "records": [ { "name": "transformer.h.26.mixer.qkv_proj.q_weight", "shape": [ 9216, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 14155776, "byteOffset": 0 }, { "name": "transformer.h.26.mixer.qkv_proj.q_scale", "shape": [ 9216, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1769472, "byteOffset": 14155776 }, { "name": "transformer.h.27.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 15925248 }, { "name": "transformer.h.27.mlp.down_proj.q_weight", "shape": [ 3072, 1024 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 12582912, "byteOffset": 15931392 }, { "name": "transformer.h.27.mlp.down_proj.q_scale", "shape": [ 3072, 256 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1572864, "byteOffset": 28514304 }, { "name": "transformer.h.27.mlp.gate_up_proj.q_scale", "shape": [ 16384, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3145728, "byteOffset": 30087168 }, { "name": "transformer.h.27.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 33232896 } ], "md5sum": "241b48480ec9cf23c4e3cd7b72eb8946" }, { "dataPath": "params_shard_17.bin", "format": "raw-shard", "nbytes": 21239808, "records": [ { "name": "transformer.h.27.mixer.out_proj.q_weight", "shape": [ 3072, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 0 }, { "name": "transformer.h.27.mixer.out_proj.q_scale", "shape": [ 3072, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 589824, "byteOffset": 4718592 }, { "name": "transformer.h.27.mixer.qkv_proj.q_weight", "shape": [ 9216, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 14155776, "byteOffset": 5308416 }, { "name": "transformer.h.27.mixer.qkv_proj.q_scale", "shape": [ 9216, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1769472, "byteOffset": 19464192 }, { "name": "transformer.h.28.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 21233664 } ], "md5sum": "4f45f2e000cfb51eb22e229e0fc6d5e3" }, { "dataPath": "params_shard_18.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "transformer.h.28.mlp.gate_up_proj.q_weight", "shape": [ 16384, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "e8ec32d0936dff7570adeefd6f033148" }, { "dataPath": "params_shard_19.bin", "format": "raw-shard", "nbytes": 22616064, "records": [ { "name": "transformer.h.28.mlp.down_proj.q_weight", "shape": [ 3072, 1024 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 12582912, "byteOffset": 0 }, { "name": "transformer.h.28.mlp.down_proj.q_scale", "shape": [ 3072, 256 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1572864, "byteOffset": 12582912 }, { "name": "transformer.h.28.mlp.gate_up_proj.q_scale", "shape": [ 16384, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3145728, "byteOffset": 14155776 }, { "name": "transformer.h.28.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 17301504 }, { "name": "transformer.h.28.mixer.out_proj.q_weight", "shape": [ 3072, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 17307648 }, { "name": "transformer.h.28.mixer.out_proj.q_scale", "shape": [ 3072, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 589824, "byteOffset": 22026240 } ], "md5sum": "2bba1b9299c2605193c884cb4637f01d" }, { "dataPath": "params_shard_20.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "transformer.h.29.mlp.gate_up_proj.q_weight", "shape": [ 16384, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "2de74bd8d206b921f02c038a3ceb57da" }, { "dataPath": "params_shard_21.bin", "format": "raw-shard", "nbytes": 33239040, "records": [ { "name": "transformer.h.28.mixer.qkv_proj.q_weight", "shape": [ 9216, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 14155776, "byteOffset": 0 }, { "name": "transformer.h.28.mixer.qkv_proj.q_scale", "shape": [ 9216, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1769472, "byteOffset": 14155776 }, { "name": "transformer.h.29.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 15925248 }, { "name": "transformer.h.29.mlp.down_proj.q_weight", "shape": [ 3072, 1024 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 12582912, "byteOffset": 15931392 }, { "name": "transformer.h.29.mlp.down_proj.q_scale", "shape": [ 3072, 256 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1572864, "byteOffset": 28514304 }, { "name": "transformer.h.29.mlp.gate_up_proj.q_scale", "shape": [ 16384, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3145728, "byteOffset": 30087168 }, { "name": "transformer.h.29.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 33232896 } ], "md5sum": "be17e94b3415f5e5ce9d76fb4c000af4" }, { "dataPath": "params_shard_22.bin", "format": "raw-shard", "nbytes": 21239808, "records": [ { "name": "transformer.h.29.mixer.out_proj.q_weight", "shape": [ 3072, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 0 }, { "name": "transformer.h.29.mixer.out_proj.q_scale", "shape": [ 3072, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 589824, "byteOffset": 4718592 }, { "name": "transformer.h.29.mixer.qkv_proj.q_weight", "shape": [ 9216, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 14155776, "byteOffset": 5308416 }, { "name": "transformer.h.29.mixer.qkv_proj.q_scale", "shape": [ 9216, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1769472, "byteOffset": 19464192 }, { "name": "transformer.h.30.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 21233664 } ], "md5sum": "3a8c0c2d5005b16692d226315811cd1e" }, { "dataPath": "params_shard_23.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "transformer.h.30.mlp.gate_up_proj.q_weight", "shape": [ 16384, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "f64d3b1ab3b45719ec6aaf05bbd8003b" }, { "dataPath": "params_shard_24.bin", "format": "raw-shard", "nbytes": 22616064, "records": [ { "name": "transformer.h.30.mlp.down_proj.q_weight", "shape": [ 3072, 1024 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 12582912, "byteOffset": 0 }, { "name": "transformer.h.30.mlp.down_proj.q_scale", "shape": [ 3072, 256 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1572864, "byteOffset": 12582912 }, { "name": "transformer.h.30.mlp.gate_up_proj.q_scale", "shape": [ 16384, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3145728, "byteOffset": 14155776 }, { "name": "transformer.h.30.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 17301504 }, { "name": "transformer.h.30.mixer.out_proj.q_weight", "shape": [ 3072, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 17307648 }, { "name": "transformer.h.30.mixer.out_proj.q_scale", "shape": [ 3072, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 589824, "byteOffset": 22026240 } ], "md5sum": "a77b053ef34c45e78ecb29da965ddf7c" }, { "dataPath": "params_shard_25.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "transformer.h.31.mlp.gate_up_proj.q_weight", "shape": [ 16384, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "1941a06d26f3677c43e9187cc432b25f" }, { "dataPath": "params_shard_26.bin", "format": "raw-shard", "nbytes": 33239040, "records": [ { "name": "transformer.h.30.mixer.qkv_proj.q_weight", "shape": [ 9216, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 14155776, "byteOffset": 0 }, { "name": "transformer.h.30.mixer.qkv_proj.q_scale", "shape": [ 9216, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1769472, "byteOffset": 14155776 }, { "name": "transformer.h.31.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 15925248 }, { "name": "transformer.h.31.mlp.down_proj.q_weight", "shape": [ 3072, 1024 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 12582912, "byteOffset": 15931392 }, { "name": "transformer.h.31.mlp.down_proj.q_scale", "shape": [ 3072, 256 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1572864, "byteOffset": 28514304 }, { "name": "transformer.h.31.mlp.gate_up_proj.q_scale", "shape": [ 16384, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3145728, "byteOffset": 30087168 }, { "name": "transformer.h.31.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 33232896 } ], "md5sum": "b8a526800503a9ad8264188387e0f617" }, { "dataPath": "params_shard_27.bin", "format": "raw-shard", "nbytes": 197001216, "records": [ { "name": "transformer.embd.weight", "shape": [ 32064, 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 197001216, "byteOffset": 0 } ], "md5sum": "6f9e82ee16bf9e48202971f1bc7faddf" }, { "dataPath": "params_shard_28.bin", "format": "raw-shard", "nbytes": 21245952, "records": [ { "name": "transformer.h.31.mixer.out_proj.q_weight", "shape": [ 3072, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 0 }, { "name": "transformer.h.31.mixer.out_proj.q_scale", "shape": [ 3072, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 589824, "byteOffset": 4718592 }, { "name": "transformer.h.31.mixer.qkv_proj.q_weight", "shape": [ 9216, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 14155776, "byteOffset": 5308416 }, { "name": "transformer.h.31.mixer.qkv_proj.q_scale", "shape": [ 9216, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1769472, "byteOffset": 19464192 }, { "name": "transformer.norm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 21233664 }, { "name": "transformer.h.0.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 21239808 } ], "md5sum": "c9780ddc4b6e6f9c6d43de652824fc8b" }, { "dataPath": "params_shard_29.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "transformer.h.0.mlp.gate_up_proj.q_weight", "shape": [ 16384, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "1b8d952fdb67b24c9feebd566e76b7e5" }, { "dataPath": "params_shard_30.bin", "format": "raw-shard", "nbytes": 22616064, "records": [ { "name": "transformer.h.0.mlp.down_proj.q_weight", "shape": [ 3072, 1024 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 12582912, "byteOffset": 0 }, { "name": "transformer.h.0.mlp.down_proj.q_scale", "shape": [ 3072, 256 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1572864, "byteOffset": 12582912 }, { "name": "transformer.h.0.mlp.gate_up_proj.q_scale", "shape": [ 16384, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3145728, "byteOffset": 14155776 }, { "name": "transformer.h.0.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 17301504 }, { "name": "transformer.h.0.mixer.out_proj.q_weight", "shape": [ 3072, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 17307648 }, { "name": "transformer.h.0.mixer.out_proj.q_scale", "shape": [ 3072, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 589824, "byteOffset": 22026240 } ], "md5sum": "47e0b6ed387f09a49a5acdb79dfeb39c" }, { "dataPath": "params_shard_31.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "transformer.h.1.mlp.gate_up_proj.q_weight", "shape": [ 16384, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "f923db91e2479135f949a7b943dabdab" }, { "dataPath": "params_shard_32.bin", "format": "raw-shard", "nbytes": 33239040, "records": [ { "name": "transformer.h.0.mixer.qkv_proj.q_weight", "shape": [ 9216, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 14155776, "byteOffset": 0 }, { "name": "transformer.h.0.mixer.qkv_proj.q_scale", "shape": [ 9216, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1769472, "byteOffset": 14155776 }, { "name": "transformer.h.1.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 15925248 }, { "name": "transformer.h.1.mlp.down_proj.q_weight", "shape": [ 3072, 1024 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 12582912, "byteOffset": 15931392 }, { "name": "transformer.h.1.mlp.down_proj.q_scale", "shape": [ 3072, 256 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1572864, "byteOffset": 28514304 }, { "name": "transformer.h.1.mlp.gate_up_proj.q_scale", "shape": [ 16384, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3145728, "byteOffset": 30087168 }, { "name": "transformer.h.1.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 33232896 } ], "md5sum": "cbaf336f7de99a44242e92f8d8cbca24" }, { "dataPath": "params_shard_33.bin", "format": "raw-shard", "nbytes": 21239808, "records": [ { "name": "transformer.h.1.mixer.out_proj.q_weight", "shape": [ 3072, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 0 }, { "name": "transformer.h.1.mixer.out_proj.q_scale", "shape": [ 3072, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 589824, "byteOffset": 4718592 }, { "name": "transformer.h.1.mixer.qkv_proj.q_weight", "shape": [ 9216, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 14155776, "byteOffset": 5308416 }, { "name": "transformer.h.1.mixer.qkv_proj.q_scale", "shape": [ 9216, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1769472, "byteOffset": 19464192 }, { "name": "transformer.h.10.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 21233664 } ], "md5sum": "e2067d3a2080614fe56f8767fb20b534" }, { "dataPath": "params_shard_34.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "transformer.h.10.mlp.gate_up_proj.q_weight", "shape": [ 16384, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "c05d6a6a23aa9233b5b3d72e8584002b" }, { "dataPath": "params_shard_35.bin", "format": "raw-shard", "nbytes": 22616064, "records": [ { "name": "transformer.h.10.mlp.down_proj.q_weight", "shape": [ 3072, 1024 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 12582912, "byteOffset": 0 }, { "name": "transformer.h.10.mlp.down_proj.q_scale", "shape": [ 3072, 256 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1572864, "byteOffset": 12582912 }, { "name": "transformer.h.10.mlp.gate_up_proj.q_scale", "shape": [ 16384, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3145728, "byteOffset": 14155776 }, { "name": "transformer.h.10.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 17301504 }, { "name": "transformer.h.10.mixer.out_proj.q_weight", "shape": [ 3072, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 17307648 }, { "name": "transformer.h.10.mixer.out_proj.q_scale", "shape": [ 3072, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 589824, "byteOffset": 22026240 } ], "md5sum": "f58097cc82e1a6732a054515856cec5c" }, { "dataPath": "params_shard_36.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "transformer.h.11.mlp.gate_up_proj.q_weight", "shape": [ 16384, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "cf9309dd6a367b9bc233f1ed66c7b404" }, { "dataPath": "params_shard_37.bin", "format": "raw-shard", "nbytes": 33239040, "records": [ { "name": "transformer.h.10.mixer.qkv_proj.q_weight", "shape": [ 9216, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 14155776, "byteOffset": 0 }, { "name": "transformer.h.10.mixer.qkv_proj.q_scale", "shape": [ 9216, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1769472, "byteOffset": 14155776 }, { "name": "transformer.h.11.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 15925248 }, { "name": "transformer.h.11.mlp.down_proj.q_weight", "shape": [ 3072, 1024 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 12582912, "byteOffset": 15931392 }, { "name": "transformer.h.11.mlp.down_proj.q_scale", "shape": [ 3072, 256 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1572864, "byteOffset": 28514304 }, { "name": "transformer.h.11.mlp.gate_up_proj.q_scale", "shape": [ 16384, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3145728, "byteOffset": 30087168 }, { "name": "transformer.h.11.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 33232896 } ], "md5sum": "240b70aa9f8dc1ac5a6538dec4d6e21f" }, { "dataPath": "params_shard_38.bin", "format": "raw-shard", "nbytes": 21239808, "records": [ { "name": "transformer.h.11.mixer.out_proj.q_weight", "shape": [ 3072, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 0 }, { "name": "transformer.h.11.mixer.out_proj.q_scale", "shape": [ 3072, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 589824, "byteOffset": 4718592 }, { "name": "transformer.h.11.mixer.qkv_proj.q_weight", "shape": [ 9216, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 14155776, "byteOffset": 5308416 }, { "name": "transformer.h.11.mixer.qkv_proj.q_scale", "shape": [ 9216, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1769472, "byteOffset": 19464192 }, { "name": "transformer.h.12.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 21233664 } ], "md5sum": "a8cde7cf79c229e2bd9efc0b7f54b0c2" }, { "dataPath": "params_shard_39.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "transformer.h.12.mlp.gate_up_proj.q_weight", "shape": [ 16384, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "db2f113811e2f795bb3c86a40655e6ea" }, { "dataPath": "params_shard_40.bin", "format": "raw-shard", "nbytes": 22616064, "records": [ { "name": "transformer.h.12.mlp.down_proj.q_weight", "shape": [ 3072, 1024 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 12582912, "byteOffset": 0 }, { "name": "transformer.h.12.mlp.down_proj.q_scale", "shape": [ 3072, 256 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1572864, "byteOffset": 12582912 }, { "name": "transformer.h.12.mlp.gate_up_proj.q_scale", "shape": [ 16384, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3145728, "byteOffset": 14155776 }, { "name": "transformer.h.12.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 17301504 }, { "name": "transformer.h.12.mixer.out_proj.q_weight", "shape": [ 3072, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 17307648 }, { "name": "transformer.h.12.mixer.out_proj.q_scale", "shape": [ 3072, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 589824, "byteOffset": 22026240 } ], "md5sum": "514b2b1b0a3a152589b06bdfc1d87c7c" }, { "dataPath": "params_shard_41.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "transformer.h.13.mlp.gate_up_proj.q_weight", "shape": [ 16384, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "19b782ceafa01a90ee393014e9a8aaad" }, { "dataPath": "params_shard_42.bin", "format": "raw-shard", "nbytes": 33239040, "records": [ { "name": "transformer.h.12.mixer.qkv_proj.q_weight", "shape": [ 9216, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 14155776, "byteOffset": 0 }, { "name": "transformer.h.12.mixer.qkv_proj.q_scale", "shape": [ 9216, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1769472, "byteOffset": 14155776 }, { "name": "transformer.h.13.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 15925248 }, { "name": "transformer.h.13.mlp.down_proj.q_weight", "shape": [ 3072, 1024 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 12582912, "byteOffset": 15931392 }, { "name": "transformer.h.13.mlp.down_proj.q_scale", "shape": [ 3072, 256 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1572864, "byteOffset": 28514304 }, { "name": "transformer.h.13.mlp.gate_up_proj.q_scale", "shape": [ 16384, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3145728, "byteOffset": 30087168 }, { "name": "transformer.h.13.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 33232896 } ], "md5sum": "8b2c1a3d8aec01c4e93904cc555afdda" }, { "dataPath": "params_shard_43.bin", "format": "raw-shard", "nbytes": 21239808, "records": [ { "name": "transformer.h.13.mixer.out_proj.q_weight", "shape": [ 3072, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 0 }, { "name": "transformer.h.13.mixer.out_proj.q_scale", "shape": [ 3072, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 589824, "byteOffset": 4718592 }, { "name": "transformer.h.13.mixer.qkv_proj.q_weight", "shape": [ 9216, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 14155776, "byteOffset": 5308416 }, { "name": "transformer.h.13.mixer.qkv_proj.q_scale", "shape": [ 9216, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1769472, "byteOffset": 19464192 }, { "name": "transformer.h.14.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 21233664 } ], "md5sum": "4ab7430fbea44358b72aab2ee4539b7a" }, { "dataPath": "params_shard_44.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "transformer.h.14.mlp.gate_up_proj.q_weight", "shape": [ 16384, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "5152fc7b02c4ba5b509d46fcecff6d81" }, { "dataPath": "params_shard_45.bin", "format": "raw-shard", "nbytes": 22616064, "records": [ { "name": "transformer.h.14.mlp.down_proj.q_weight", "shape": [ 3072, 1024 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 12582912, "byteOffset": 0 }, { "name": "transformer.h.14.mlp.down_proj.q_scale", "shape": [ 3072, 256 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1572864, "byteOffset": 12582912 }, { "name": "transformer.h.14.mlp.gate_up_proj.q_scale", "shape": [ 16384, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3145728, "byteOffset": 14155776 }, { "name": "transformer.h.14.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 17301504 }, { "name": "transformer.h.14.mixer.out_proj.q_weight", "shape": [ 3072, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 17307648 }, { "name": "transformer.h.14.mixer.out_proj.q_scale", "shape": [ 3072, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 589824, "byteOffset": 22026240 } ], "md5sum": "779c46e492f7b81439f6976f59bc7209" }, { "dataPath": "params_shard_46.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "transformer.h.15.mlp.gate_up_proj.q_weight", "shape": [ 16384, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "b2f2c5838bc641fa05674d020ec30019" }, { "dataPath": "params_shard_47.bin", "format": "raw-shard", "nbytes": 33239040, "records": [ { "name": "transformer.h.14.mixer.qkv_proj.q_weight", "shape": [ 9216, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 14155776, "byteOffset": 0 }, { "name": "transformer.h.14.mixer.qkv_proj.q_scale", "shape": [ 9216, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1769472, "byteOffset": 14155776 }, { "name": "transformer.h.15.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 15925248 }, { "name": "transformer.h.15.mlp.down_proj.q_weight", "shape": [ 3072, 1024 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 12582912, "byteOffset": 15931392 }, { "name": "transformer.h.15.mlp.down_proj.q_scale", "shape": [ 3072, 256 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1572864, "byteOffset": 28514304 }, { "name": "transformer.h.15.mlp.gate_up_proj.q_scale", "shape": [ 16384, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3145728, "byteOffset": 30087168 }, { "name": "transformer.h.15.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 33232896 } ], "md5sum": "f9e5d71c651438c7c79b5a241ba79a9b" }, { "dataPath": "params_shard_48.bin", "format": "raw-shard", "nbytes": 21239808, "records": [ { "name": "transformer.h.15.mixer.out_proj.q_weight", "shape": [ 3072, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 0 }, { "name": "transformer.h.15.mixer.out_proj.q_scale", "shape": [ 3072, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 589824, "byteOffset": 4718592 }, { "name": "transformer.h.15.mixer.qkv_proj.q_weight", "shape": [ 9216, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 14155776, "byteOffset": 5308416 }, { "name": "transformer.h.15.mixer.qkv_proj.q_scale", "shape": [ 9216, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1769472, "byteOffset": 19464192 }, { "name": "transformer.h.16.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 21233664 } ], "md5sum": "7eb2caef4cc1d945d329aaa5bc513480" }, { "dataPath": "params_shard_49.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "transformer.h.16.mlp.gate_up_proj.q_weight", "shape": [ 16384, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "aafe2e4720584bb8cbaec7371d21f6f1" }, { "dataPath": "params_shard_50.bin", "format": "raw-shard", "nbytes": 22616064, "records": [ { "name": "transformer.h.16.mlp.down_proj.q_weight", "shape": [ 3072, 1024 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 12582912, "byteOffset": 0 }, { "name": "transformer.h.16.mlp.down_proj.q_scale", "shape": [ 3072, 256 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1572864, "byteOffset": 12582912 }, { "name": "transformer.h.16.mlp.gate_up_proj.q_scale", "shape": [ 16384, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3145728, "byteOffset": 14155776 }, { "name": "transformer.h.16.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 17301504 }, { "name": "transformer.h.16.mixer.out_proj.q_weight", "shape": [ 3072, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 17307648 }, { "name": "transformer.h.16.mixer.out_proj.q_scale", "shape": [ 3072, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 589824, "byteOffset": 22026240 } ], "md5sum": "6ebf829958e55cefb33b6ca886405c01" }, { "dataPath": "params_shard_51.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "transformer.h.17.mlp.gate_up_proj.q_weight", "shape": [ 16384, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "200d5a4dd189510c0f97fb57455036cc" }, { "dataPath": "params_shard_52.bin", "format": "raw-shard", "nbytes": 33239040, "records": [ { "name": "transformer.h.16.mixer.qkv_proj.q_weight", "shape": [ 9216, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 14155776, "byteOffset": 0 }, { "name": "transformer.h.16.mixer.qkv_proj.q_scale", "shape": [ 9216, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1769472, "byteOffset": 14155776 }, { "name": "transformer.h.17.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 15925248 }, { "name": "transformer.h.17.mlp.down_proj.q_weight", "shape": [ 3072, 1024 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 12582912, "byteOffset": 15931392 }, { "name": "transformer.h.17.mlp.down_proj.q_scale", "shape": [ 3072, 256 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1572864, "byteOffset": 28514304 }, { "name": "transformer.h.17.mlp.gate_up_proj.q_scale", "shape": [ 16384, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3145728, "byteOffset": 30087168 }, { "name": "transformer.h.17.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 33232896 } ], "md5sum": "1bdd9972c4a6e9cf1f29e157fb9722d4" }, { "dataPath": "params_shard_53.bin", "format": "raw-shard", "nbytes": 21239808, "records": [ { "name": "transformer.h.17.mixer.out_proj.q_weight", "shape": [ 3072, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 0 }, { "name": "transformer.h.17.mixer.out_proj.q_scale", "shape": [ 3072, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 589824, "byteOffset": 4718592 }, { "name": "transformer.h.17.mixer.qkv_proj.q_weight", "shape": [ 9216, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 14155776, "byteOffset": 5308416 }, { "name": "transformer.h.17.mixer.qkv_proj.q_scale", "shape": [ 9216, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1769472, "byteOffset": 19464192 }, { "name": "transformer.h.18.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 21233664 } ], "md5sum": "90fb84478077ae1bf6ba2cacc250f28f" }, { "dataPath": "params_shard_54.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "transformer.h.18.mlp.gate_up_proj.q_weight", "shape": [ 16384, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "404113231bee9be0d25b6c1468951120" }, { "dataPath": "params_shard_55.bin", "format": "raw-shard", "nbytes": 22616064, "records": [ { "name": "transformer.h.18.mlp.down_proj.q_weight", "shape": [ 3072, 1024 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 12582912, "byteOffset": 0 }, { "name": "transformer.h.18.mlp.down_proj.q_scale", "shape": [ 3072, 256 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1572864, "byteOffset": 12582912 }, { "name": "transformer.h.18.mlp.gate_up_proj.q_scale", "shape": [ 16384, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3145728, "byteOffset": 14155776 }, { "name": "transformer.h.18.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 17301504 }, { "name": "transformer.h.18.mixer.out_proj.q_weight", "shape": [ 3072, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 17307648 }, { "name": "transformer.h.18.mixer.out_proj.q_scale", "shape": [ 3072, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 589824, "byteOffset": 22026240 } ], "md5sum": "b09deabc64a6f0cad5df8dfc6025fc84" }, { "dataPath": "params_shard_56.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "transformer.h.19.mlp.gate_up_proj.q_weight", "shape": [ 16384, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "834b8b028a2d26f586c0e630f4bec147" }, { "dataPath": "params_shard_57.bin", "format": "raw-shard", "nbytes": 33239040, "records": [ { "name": "transformer.h.18.mixer.qkv_proj.q_weight", "shape": [ 9216, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 14155776, "byteOffset": 0 }, { "name": "transformer.h.18.mixer.qkv_proj.q_scale", "shape": [ 9216, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1769472, "byteOffset": 14155776 }, { "name": "transformer.h.19.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 15925248 }, { "name": "transformer.h.19.mlp.down_proj.q_weight", "shape": [ 3072, 1024 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 12582912, "byteOffset": 15931392 }, { "name": "transformer.h.19.mlp.down_proj.q_scale", "shape": [ 3072, 256 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1572864, "byteOffset": 28514304 }, { "name": "transformer.h.19.mlp.gate_up_proj.q_scale", "shape": [ 16384, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3145728, "byteOffset": 30087168 }, { "name": "transformer.h.19.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 33232896 } ], "md5sum": "c4537a3792050b7959c42202a1c3ef4d" }, { "dataPath": "params_shard_58.bin", "format": "raw-shard", "nbytes": 21239808, "records": [ { "name": "transformer.h.19.mixer.out_proj.q_weight", "shape": [ 3072, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 0 }, { "name": "transformer.h.19.mixer.out_proj.q_scale", "shape": [ 3072, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 589824, "byteOffset": 4718592 }, { "name": "transformer.h.19.mixer.qkv_proj.q_weight", "shape": [ 9216, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 14155776, "byteOffset": 5308416 }, { "name": "transformer.h.19.mixer.qkv_proj.q_scale", "shape": [ 9216, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1769472, "byteOffset": 19464192 }, { "name": "transformer.h.2.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 21233664 } ], "md5sum": "3b732ec558d6be7ad5aa96efcb7def63" }, { "dataPath": "params_shard_59.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "transformer.h.2.mlp.gate_up_proj.q_weight", "shape": [ 16384, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "70bd2422e7abada045cf3f4cec0ad360" }, { "dataPath": "params_shard_60.bin", "format": "raw-shard", "nbytes": 22616064, "records": [ { "name": "transformer.h.2.mlp.down_proj.q_weight", "shape": [ 3072, 1024 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 12582912, "byteOffset": 0 }, { "name": "transformer.h.2.mlp.down_proj.q_scale", "shape": [ 3072, 256 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1572864, "byteOffset": 12582912 }, { "name": "transformer.h.2.mlp.gate_up_proj.q_scale", "shape": [ 16384, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3145728, "byteOffset": 14155776 }, { "name": "transformer.h.2.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 17301504 }, { "name": "transformer.h.2.mixer.out_proj.q_weight", "shape": [ 3072, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 17307648 }, { "name": "transformer.h.2.mixer.out_proj.q_scale", "shape": [ 3072, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 589824, "byteOffset": 22026240 } ], "md5sum": "f71edac4d6374e21aa5c0485ca5d5e72" }, { "dataPath": "params_shard_61.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "transformer.h.20.mlp.gate_up_proj.q_weight", "shape": [ 16384, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "5b6bb6c2958b26a757fdf999d9e10aa1" }, { "dataPath": "params_shard_62.bin", "format": "raw-shard", "nbytes": 33239040, "records": [ { "name": "transformer.h.2.mixer.qkv_proj.q_weight", "shape": [ 9216, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 14155776, "byteOffset": 0 }, { "name": "transformer.h.2.mixer.qkv_proj.q_scale", "shape": [ 9216, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1769472, "byteOffset": 14155776 }, { "name": "transformer.h.20.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 15925248 }, { "name": "transformer.h.20.mlp.down_proj.q_weight", "shape": [ 3072, 1024 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 12582912, "byteOffset": 15931392 }, { "name": "transformer.h.20.mlp.down_proj.q_scale", "shape": [ 3072, 256 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1572864, "byteOffset": 28514304 }, { "name": "transformer.h.20.mlp.gate_up_proj.q_scale", "shape": [ 16384, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3145728, "byteOffset": 30087168 }, { "name": "transformer.h.20.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 33232896 } ], "md5sum": "43e3a8359c747ea8ee19e8221652fa62" }, { "dataPath": "params_shard_63.bin", "format": "raw-shard", "nbytes": 26548224, "records": [ { "name": "transformer.h.20.mixer.out_proj.q_weight", "shape": [ 3072, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 0 }, { "name": "transformer.h.20.mixer.out_proj.q_scale", "shape": [ 3072, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 589824, "byteOffset": 4718592 }, { "name": "transformer.h.20.mixer.qkv_proj.q_weight", "shape": [ 9216, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 14155776, "byteOffset": 5308416 }, { "name": "transformer.h.20.mixer.qkv_proj.q_scale", "shape": [ 9216, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1769472, "byteOffset": 19464192 }, { "name": "transformer.h.21.mixer.out_proj.q_weight", "shape": [ 3072, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 21233664 }, { "name": "transformer.h.21.mixer.out_proj.q_scale", "shape": [ 3072, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 589824, "byteOffset": 25952256 }, { "name": "transformer.h.3.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 26542080 } ], "md5sum": "39db77d9665e5d21490eabafadc55b28" }, { "dataPath": "params_shard_64.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "transformer.h.3.mlp.gate_up_proj.q_weight", "shape": [ 16384, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "7823c2e099eeab4b05f4e3c12240ce32" }, { "dataPath": "params_shard_65.bin", "format": "raw-shard", "nbytes": 22616064, "records": [ { "name": "transformer.h.3.mlp.down_proj.q_weight", "shape": [ 3072, 1024 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 12582912, "byteOffset": 0 }, { "name": "transformer.h.3.mlp.down_proj.q_scale", "shape": [ 3072, 256 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1572864, "byteOffset": 12582912 }, { "name": "transformer.h.3.mlp.gate_up_proj.q_scale", "shape": [ 16384, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3145728, "byteOffset": 14155776 }, { "name": "transformer.h.3.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 17301504 }, { "name": "transformer.h.3.mixer.out_proj.q_weight", "shape": [ 3072, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 17307648 }, { "name": "transformer.h.3.mixer.out_proj.q_scale", "shape": [ 3072, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 589824, "byteOffset": 22026240 } ], "md5sum": "7d690e254c29d546738574dbc1bbcd7c" }, { "dataPath": "params_shard_66.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "transformer.h.4.mlp.gate_up_proj.q_weight", "shape": [ 16384, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "511ad1276c55198257596b7073c6fa44" }, { "dataPath": "params_shard_67.bin", "format": "raw-shard", "nbytes": 33239040, "records": [ { "name": "transformer.h.3.mixer.qkv_proj.q_weight", "shape": [ 9216, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 14155776, "byteOffset": 0 }, { "name": "transformer.h.3.mixer.qkv_proj.q_scale", "shape": [ 9216, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1769472, "byteOffset": 14155776 }, { "name": "transformer.h.4.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 15925248 }, { "name": "transformer.h.4.mlp.down_proj.q_weight", "shape": [ 3072, 1024 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 12582912, "byteOffset": 15931392 }, { "name": "transformer.h.4.mlp.down_proj.q_scale", "shape": [ 3072, 256 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1572864, "byteOffset": 28514304 }, { "name": "transformer.h.4.mlp.gate_up_proj.q_scale", "shape": [ 16384, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3145728, "byteOffset": 30087168 }, { "name": "transformer.h.4.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 33232896 } ], "md5sum": "abd5e9132151fce941848eb7a8faf8d5" }, { "dataPath": "params_shard_68.bin", "format": "raw-shard", "nbytes": 21239808, "records": [ { "name": "transformer.h.4.mixer.out_proj.q_weight", "shape": [ 3072, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 0 }, { "name": "transformer.h.4.mixer.out_proj.q_scale", "shape": [ 3072, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 589824, "byteOffset": 4718592 }, { "name": "transformer.h.4.mixer.qkv_proj.q_weight", "shape": [ 9216, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 14155776, "byteOffset": 5308416 }, { "name": "transformer.h.4.mixer.qkv_proj.q_scale", "shape": [ 9216, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1769472, "byteOffset": 19464192 }, { "name": "transformer.h.5.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 21233664 } ], "md5sum": "3f8f66b9da8806c22ec96fba184daa1c" }, { "dataPath": "params_shard_69.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "transformer.h.5.mlp.gate_up_proj.q_weight", "shape": [ 16384, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "cc9ff2fc1b2497f83581c6e8669200ec" }, { "dataPath": "params_shard_70.bin", "format": "raw-shard", "nbytes": 22616064, "records": [ { "name": "transformer.h.5.mlp.down_proj.q_weight", "shape": [ 3072, 1024 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 12582912, "byteOffset": 0 }, { "name": "transformer.h.5.mlp.down_proj.q_scale", "shape": [ 3072, 256 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1572864, "byteOffset": 12582912 }, { "name": "transformer.h.5.mlp.gate_up_proj.q_scale", "shape": [ 16384, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3145728, "byteOffset": 14155776 }, { "name": "transformer.h.5.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 17301504 }, { "name": "transformer.h.5.mixer.out_proj.q_weight", "shape": [ 3072, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 17307648 }, { "name": "transformer.h.5.mixer.out_proj.q_scale", "shape": [ 3072, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 589824, "byteOffset": 22026240 } ], "md5sum": "7e4d7c2a189600ae410151b59cf05ec2" }, { "dataPath": "params_shard_71.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "transformer.h.6.mlp.gate_up_proj.q_weight", "shape": [ 16384, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "f927377a01a9e92d473d1579654926f4" }, { "dataPath": "params_shard_72.bin", "format": "raw-shard", "nbytes": 33239040, "records": [ { "name": "transformer.h.5.mixer.qkv_proj.q_weight", "shape": [ 9216, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 14155776, "byteOffset": 0 }, { "name": "transformer.h.5.mixer.qkv_proj.q_scale", "shape": [ 9216, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1769472, "byteOffset": 14155776 }, { "name": "transformer.h.6.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 15925248 }, { "name": "transformer.h.6.mlp.down_proj.q_weight", "shape": [ 3072, 1024 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 12582912, "byteOffset": 15931392 }, { "name": "transformer.h.6.mlp.down_proj.q_scale", "shape": [ 3072, 256 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1572864, "byteOffset": 28514304 }, { "name": "transformer.h.6.mlp.gate_up_proj.q_scale", "shape": [ 16384, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3145728, "byteOffset": 30087168 }, { "name": "transformer.h.6.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 33232896 } ], "md5sum": "76152a18f5db0807bab111e742a71528" }, { "dataPath": "params_shard_73.bin", "format": "raw-shard", "nbytes": 21239808, "records": [ { "name": "transformer.h.6.mixer.out_proj.q_weight", "shape": [ 3072, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 0 }, { "name": "transformer.h.6.mixer.out_proj.q_scale", "shape": [ 3072, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 589824, "byteOffset": 4718592 }, { "name": "transformer.h.6.mixer.qkv_proj.q_weight", "shape": [ 9216, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 14155776, "byteOffset": 5308416 }, { "name": "transformer.h.6.mixer.qkv_proj.q_scale", "shape": [ 9216, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1769472, "byteOffset": 19464192 }, { "name": "transformer.h.7.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 21233664 } ], "md5sum": "6e745a570a226f19b85743daa84503e4" }, { "dataPath": "params_shard_74.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "transformer.h.7.mlp.gate_up_proj.q_weight", "shape": [ 16384, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "d5254e5ed587132cdfff4e53fd430f26" }, { "dataPath": "params_shard_75.bin", "format": "raw-shard", "nbytes": 22616064, "records": [ { "name": "transformer.h.7.mlp.down_proj.q_weight", "shape": [ 3072, 1024 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 12582912, "byteOffset": 0 }, { "name": "transformer.h.7.mlp.down_proj.q_scale", "shape": [ 3072, 256 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1572864, "byteOffset": 12582912 }, { "name": "transformer.h.7.mlp.gate_up_proj.q_scale", "shape": [ 16384, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3145728, "byteOffset": 14155776 }, { "name": "transformer.h.7.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 17301504 }, { "name": "transformer.h.7.mixer.out_proj.q_weight", "shape": [ 3072, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 17307648 }, { "name": "transformer.h.7.mixer.out_proj.q_scale", "shape": [ 3072, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 589824, "byteOffset": 22026240 } ], "md5sum": "22d2f4e872a142c2db38a98ae8253d2b" }, { "dataPath": "params_shard_76.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "transformer.h.8.mlp.gate_up_proj.q_weight", "shape": [ 16384, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "61cdf38238dfd94b45739921252823e5" }, { "dataPath": "params_shard_77.bin", "format": "raw-shard", "nbytes": 33239040, "records": [ { "name": "transformer.h.7.mixer.qkv_proj.q_weight", "shape": [ 9216, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 14155776, "byteOffset": 0 }, { "name": "transformer.h.7.mixer.qkv_proj.q_scale", "shape": [ 9216, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1769472, "byteOffset": 14155776 }, { "name": "transformer.h.8.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 15925248 }, { "name": "transformer.h.8.mlp.down_proj.q_weight", "shape": [ 3072, 1024 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 12582912, "byteOffset": 15931392 }, { "name": "transformer.h.8.mlp.down_proj.q_scale", "shape": [ 3072, 256 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1572864, "byteOffset": 28514304 }, { "name": "transformer.h.8.mlp.gate_up_proj.q_scale", "shape": [ 16384, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3145728, "byteOffset": 30087168 }, { "name": "transformer.h.8.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 33232896 } ], "md5sum": "755b9f633f3f5c2a1a3907e78332d785" }, { "dataPath": "params_shard_78.bin", "format": "raw-shard", "nbytes": 21239808, "records": [ { "name": "transformer.h.8.mixer.out_proj.q_weight", "shape": [ 3072, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 0 }, { "name": "transformer.h.8.mixer.out_proj.q_scale", "shape": [ 3072, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 589824, "byteOffset": 4718592 }, { "name": "transformer.h.8.mixer.qkv_proj.q_weight", "shape": [ 9216, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 14155776, "byteOffset": 5308416 }, { "name": "transformer.h.8.mixer.qkv_proj.q_scale", "shape": [ 9216, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1769472, "byteOffset": 19464192 }, { "name": "transformer.h.9.ln.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 21233664 } ], "md5sum": "0b17c6765c2a4536d0ce4942995e711a" }, { "dataPath": "params_shard_79.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "transformer.h.9.mlp.gate_up_proj.q_weight", "shape": [ 16384, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "29dbb15e0ad0a182c1a32f5d5ff87eee" }, { "dataPath": "params_shard_80.bin", "format": "raw-shard", "nbytes": 22616064, "records": [ { "name": "transformer.h.9.mlp.down_proj.q_weight", "shape": [ 3072, 1024 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 12582912, "byteOffset": 0 }, { "name": "transformer.h.9.mlp.down_proj.q_scale", "shape": [ 3072, 256 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1572864, "byteOffset": 12582912 }, { "name": "transformer.h.9.mlp.gate_up_proj.q_scale", "shape": [ 16384, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3145728, "byteOffset": 14155776 }, { "name": "transformer.h.9.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 17301504 }, { "name": "transformer.h.9.mixer.out_proj.q_weight", "shape": [ 3072, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 17307648 }, { "name": "transformer.h.9.mixer.out_proj.q_scale", "shape": [ 3072, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 589824, "byteOffset": 22026240 } ], "md5sum": "1ecaf69f1abbd8196467e5b009db3fd8" }, { "dataPath": "params_shard_81.bin", "format": "raw-shard", "nbytes": 15925248, "records": [ { "name": "transformer.h.9.mixer.qkv_proj.q_weight", "shape": [ 9216, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 14155776, "byteOffset": 0 }, { "name": "transformer.h.9.mixer.qkv_proj.q_scale", "shape": [ 9216, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1769472, "byteOffset": 14155776 } ], "md5sum": "30091b5479da58ca028a44ce5ee7d5fd" } ] }