numen-tech's picture
Add weights
854023f
raw
history blame contribute delete
No virus
146 kB
{
"metadata": {
"ParamSize": 325,
"ParamBytes": 3631664128.0,
"BitsPerParam": 2.6739310072364444
},
"records": [
{
"dataPath": "params_shard_0.bin",
"format": "raw-shard",
"nbytes": 211365888,
"records": [
{
"name": "lm_head.q_weight",
"shape": [
128256,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 211365888,
"byteOffset": 0
}
],
"md5sum": "bc1caa034e9be7b960c17d20a964ffbe"
},
{
"dataPath": "params_shard_1.bin",
"format": "raw-shard",
"nbytes": 23527424,
"records": [
{
"name": "model.layers.30.mlp.down_proj.q_weight",
"shape": [
4096,
1436
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 23527424,
"byteOffset": 0
}
],
"md5sum": "b72097bdc7b6f25ae8124ea7dbdeb606"
},
{
"dataPath": "params_shard_2.bin",
"format": "raw-shard",
"nbytes": 47251456,
"records": [
{
"name": "model.layers.30.mlp.gate_up_proj.q_weight",
"shape": [
28672,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 47251456,
"byteOffset": 0
}
],
"md5sum": "5289a6a92358d5decf108088fffb2425"
},
{
"dataPath": "params_shard_3.bin",
"format": "raw-shard",
"nbytes": 29369856,
"records": [
{
"name": "lm_head.q_scale",
"shape": [
128256,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 26420736,
"byteOffset": 0
},
{
"name": "model.layers.30.input_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 26420736
},
{
"name": "model.layers.30.mlp.down_proj.q_scale",
"shape": [
4096,
359
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2940928,
"byteOffset": 26428928
}
],
"md5sum": "cdfff4537642f4e73deb7d784074a43d"
},
{
"dataPath": "params_shard_4.bin",
"format": "raw-shard",
"nbytes": 23527424,
"records": [
{
"name": "model.layers.31.mlp.down_proj.q_weight",
"shape": [
4096,
1436
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 23527424,
"byteOffset": 0
}
],
"md5sum": "994097f1c312567981b7446e0be3ce1f"
},
{
"dataPath": "params_shard_5.bin",
"format": "raw-shard",
"nbytes": 47251456,
"records": [
{
"name": "model.layers.31.mlp.gate_up_proj.q_weight",
"shape": [
28672,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 47251456,
"byteOffset": 0
}
],
"md5sum": "7bda08525877aa317958bd76fb446691"
},
{
"dataPath": "params_shard_6.bin",
"format": "raw-shard",
"nbytes": 27848704,
"records": [
{
"name": "model.layers.30.mlp.gate_up_proj.q_scale",
"shape": [
28672,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5906432,
"byteOffset": 0
},
{
"name": "model.layers.30.post_attention_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 5906432
},
{
"name": "model.layers.30.self_attn.o_proj.q_weight",
"shape": [
4096,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 6750208,
"byteOffset": 5914624
},
{
"name": "model.layers.30.self_attn.o_proj.q_scale",
"shape": [
4096,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 843776,
"byteOffset": 12664832
},
{
"name": "model.layers.30.self_attn.qkv_proj.q_weight",
"shape": [
6144,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 10125312,
"byteOffset": 13508608
},
{
"name": "model.layers.30.self_attn.qkv_proj.q_scale",
"shape": [
6144,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1265664,
"byteOffset": 23633920
},
{
"name": "model.layers.31.input_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 24899584
},
{
"name": "model.layers.31.mlp.down_proj.q_scale",
"shape": [
4096,
359
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2940928,
"byteOffset": 24907776
}
],
"md5sum": "11b0bcc36b45c000229f8934461850b0"
},
{
"dataPath": "params_shard_7.bin",
"format": "raw-shard",
"nbytes": 211365888,
"records": [
{
"name": "model.embed_tokens.q_weight",
"shape": [
128256,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 211365888,
"byteOffset": 0
}
],
"md5sum": "a3034216fdbc9d7d9482ac5e32a3aad3"
},
{
"dataPath": "params_shard_8.bin",
"format": "raw-shard",
"nbytes": 26420736,
"records": [
{
"name": "model.embed_tokens.q_scale",
"shape": [
128256,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 26420736,
"byteOffset": 0
}
],
"md5sum": "b4fd887902fd0ac50f3cd3ee3612c13a"
},
{
"dataPath": "params_shard_9.bin",
"format": "raw-shard",
"nbytes": 23527424,
"records": [
{
"name": "model.layers.0.mlp.down_proj.q_weight",
"shape": [
4096,
1436
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 23527424,
"byteOffset": 0
}
],
"md5sum": "13438183cb1836d58ea9dce9940d56bb"
},
{
"dataPath": "params_shard_10.bin",
"format": "raw-shard",
"nbytes": 47251456,
"records": [
{
"name": "model.layers.0.mlp.gate_up_proj.q_weight",
"shape": [
28672,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 47251456,
"byteOffset": 0
}
],
"md5sum": "a8d37c1383226a0e7cd799284d4f6f7b"
},
{
"dataPath": "params_shard_11.bin",
"format": "raw-shard",
"nbytes": 27856896,
"records": [
{
"name": "model.layers.31.mlp.gate_up_proj.q_scale",
"shape": [
28672,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5906432,
"byteOffset": 0
},
{
"name": "model.layers.31.post_attention_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 5906432
},
{
"name": "model.layers.31.self_attn.qkv_proj.q_weight",
"shape": [
6144,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 10125312,
"byteOffset": 5914624
},
{
"name": "model.layers.31.self_attn.qkv_proj.q_scale",
"shape": [
6144,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1265664,
"byteOffset": 16039936
},
{
"name": "model.layers.31.self_attn.o_proj.q_weight",
"shape": [
4096,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 6750208,
"byteOffset": 17305600
},
{
"name": "model.layers.31.self_attn.o_proj.q_scale",
"shape": [
4096,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 843776,
"byteOffset": 24055808
},
{
"name": "model.norm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 24899584
},
{
"name": "model.layers.0.input_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 24907776
},
{
"name": "model.layers.0.mlp.down_proj.q_scale",
"shape": [
4096,
359
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2940928,
"byteOffset": 24915968
}
],
"md5sum": "e55683ef0dd18578028fc831df47dae4"
},
{
"dataPath": "params_shard_12.bin",
"format": "raw-shard",
"nbytes": 23527424,
"records": [
{
"name": "model.layers.1.mlp.down_proj.q_weight",
"shape": [
4096,
1436
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 23527424,
"byteOffset": 0
}
],
"md5sum": "7336bf57df1a360dbda936ad90d0fc2b"
},
{
"dataPath": "params_shard_13.bin",
"format": "raw-shard",
"nbytes": 47251456,
"records": [
{
"name": "model.layers.1.mlp.gate_up_proj.q_weight",
"shape": [
28672,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 47251456,
"byteOffset": 0
}
],
"md5sum": "b5a5238f4cc91a0a349b3ee9e8e28c7b"
},
{
"dataPath": "params_shard_14.bin",
"format": "raw-shard",
"nbytes": 27848704,
"records": [
{
"name": "model.layers.0.mlp.gate_up_proj.q_scale",
"shape": [
28672,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5906432,
"byteOffset": 0
},
{
"name": "model.layers.0.post_attention_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 5906432
},
{
"name": "model.layers.0.self_attn.qkv_proj.q_weight",
"shape": [
6144,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 10125312,
"byteOffset": 5914624
},
{
"name": "model.layers.0.self_attn.qkv_proj.q_scale",
"shape": [
6144,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1265664,
"byteOffset": 16039936
},
{
"name": "model.layers.0.self_attn.o_proj.q_weight",
"shape": [
4096,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 6750208,
"byteOffset": 17305600
},
{
"name": "model.layers.0.self_attn.o_proj.q_scale",
"shape": [
4096,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 843776,
"byteOffset": 24055808
},
{
"name": "model.layers.1.input_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 24899584
},
{
"name": "model.layers.1.mlp.down_proj.q_scale",
"shape": [
4096,
359
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2940928,
"byteOffset": 24907776
}
],
"md5sum": "0bafb2166a462959db6da52e1d003cda"
},
{
"dataPath": "params_shard_15.bin",
"format": "raw-shard",
"nbytes": 23527424,
"records": [
{
"name": "model.layers.2.mlp.down_proj.q_weight",
"shape": [
4096,
1436
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 23527424,
"byteOffset": 0
}
],
"md5sum": "9ba77af20624ea514165f6ccbe3f202d"
},
{
"dataPath": "params_shard_16.bin",
"format": "raw-shard",
"nbytes": 47251456,
"records": [
{
"name": "model.layers.2.mlp.gate_up_proj.q_weight",
"shape": [
28672,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 47251456,
"byteOffset": 0
}
],
"md5sum": "2cbf5914e128f04443908a50df3067ab"
},
{
"dataPath": "params_shard_17.bin",
"format": "raw-shard",
"nbytes": 27848704,
"records": [
{
"name": "model.layers.1.mlp.gate_up_proj.q_scale",
"shape": [
28672,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5906432,
"byteOffset": 0
},
{
"name": "model.layers.1.post_attention_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 5906432
},
{
"name": "model.layers.1.self_attn.qkv_proj.q_weight",
"shape": [
6144,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 10125312,
"byteOffset": 5914624
},
{
"name": "model.layers.1.self_attn.qkv_proj.q_scale",
"shape": [
6144,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1265664,
"byteOffset": 16039936
},
{
"name": "model.layers.1.self_attn.o_proj.q_weight",
"shape": [
4096,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 6750208,
"byteOffset": 17305600
},
{
"name": "model.layers.1.self_attn.o_proj.q_scale",
"shape": [
4096,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 843776,
"byteOffset": 24055808
},
{
"name": "model.layers.2.input_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 24899584
},
{
"name": "model.layers.2.mlp.down_proj.q_scale",
"shape": [
4096,
359
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2940928,
"byteOffset": 24907776
}
],
"md5sum": "ac8908208926d208f383ddde32ec6ff4"
},
{
"dataPath": "params_shard_18.bin",
"format": "raw-shard",
"nbytes": 23527424,
"records": [
{
"name": "model.layers.3.mlp.down_proj.q_weight",
"shape": [
4096,
1436
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 23527424,
"byteOffset": 0
}
],
"md5sum": "d3df3cc397e61f77a820603ae753d7bb"
},
{
"dataPath": "params_shard_19.bin",
"format": "raw-shard",
"nbytes": 47251456,
"records": [
{
"name": "model.layers.3.mlp.gate_up_proj.q_weight",
"shape": [
28672,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 47251456,
"byteOffset": 0
}
],
"md5sum": "92de54341f0bafb5aec49ae277213746"
},
{
"dataPath": "params_shard_20.bin",
"format": "raw-shard",
"nbytes": 27848704,
"records": [
{
"name": "model.layers.2.mlp.gate_up_proj.q_scale",
"shape": [
28672,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5906432,
"byteOffset": 0
},
{
"name": "model.layers.2.post_attention_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 5906432
},
{
"name": "model.layers.2.self_attn.qkv_proj.q_weight",
"shape": [
6144,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 10125312,
"byteOffset": 5914624
},
{
"name": "model.layers.2.self_attn.qkv_proj.q_scale",
"shape": [
6144,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1265664,
"byteOffset": 16039936
},
{
"name": "model.layers.2.self_attn.o_proj.q_weight",
"shape": [
4096,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 6750208,
"byteOffset": 17305600
},
{
"name": "model.layers.2.self_attn.o_proj.q_scale",
"shape": [
4096,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 843776,
"byteOffset": 24055808
},
{
"name": "model.layers.3.input_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 24899584
},
{
"name": "model.layers.3.mlp.down_proj.q_scale",
"shape": [
4096,
359
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2940928,
"byteOffset": 24907776
}
],
"md5sum": "641c61b86e53c2ee679e9805d95b9f9f"
},
{
"dataPath": "params_shard_21.bin",
"format": "raw-shard",
"nbytes": 23527424,
"records": [
{
"name": "model.layers.4.mlp.down_proj.q_weight",
"shape": [
4096,
1436
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 23527424,
"byteOffset": 0
}
],
"md5sum": "59c3e40e84a7bb3cd90d7022590cc345"
},
{
"dataPath": "params_shard_22.bin",
"format": "raw-shard",
"nbytes": 47251456,
"records": [
{
"name": "model.layers.4.mlp.gate_up_proj.q_weight",
"shape": [
28672,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 47251456,
"byteOffset": 0
}
],
"md5sum": "9b149db0a95c3e9e7cf2053a9ac5848c"
},
{
"dataPath": "params_shard_23.bin",
"format": "raw-shard",
"nbytes": 27848704,
"records": [
{
"name": "model.layers.3.mlp.gate_up_proj.q_scale",
"shape": [
28672,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5906432,
"byteOffset": 0
},
{
"name": "model.layers.3.post_attention_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 5906432
},
{
"name": "model.layers.3.self_attn.qkv_proj.q_weight",
"shape": [
6144,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 10125312,
"byteOffset": 5914624
},
{
"name": "model.layers.3.self_attn.qkv_proj.q_scale",
"shape": [
6144,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1265664,
"byteOffset": 16039936
},
{
"name": "model.layers.3.self_attn.o_proj.q_weight",
"shape": [
4096,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 6750208,
"byteOffset": 17305600
},
{
"name": "model.layers.3.self_attn.o_proj.q_scale",
"shape": [
4096,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 843776,
"byteOffset": 24055808
},
{
"name": "model.layers.4.input_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 24899584
},
{
"name": "model.layers.4.mlp.down_proj.q_scale",
"shape": [
4096,
359
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2940928,
"byteOffset": 24907776
}
],
"md5sum": "788e4191bf64ea1e1d9e89e7a14e28ae"
},
{
"dataPath": "params_shard_24.bin",
"format": "raw-shard",
"nbytes": 23527424,
"records": [
{
"name": "model.layers.5.mlp.down_proj.q_weight",
"shape": [
4096,
1436
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 23527424,
"byteOffset": 0
}
],
"md5sum": "15c8502255c8222bdcde3a822f761f14"
},
{
"dataPath": "params_shard_25.bin",
"format": "raw-shard",
"nbytes": 47251456,
"records": [
{
"name": "model.layers.5.mlp.gate_up_proj.q_weight",
"shape": [
28672,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 47251456,
"byteOffset": 0
}
],
"md5sum": "fe32bffa1c793deb2a79f352ea465f68"
},
{
"dataPath": "params_shard_26.bin",
"format": "raw-shard",
"nbytes": 27848704,
"records": [
{
"name": "model.layers.4.mlp.gate_up_proj.q_scale",
"shape": [
28672,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5906432,
"byteOffset": 0
},
{
"name": "model.layers.4.post_attention_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 5906432
},
{
"name": "model.layers.4.self_attn.qkv_proj.q_weight",
"shape": [
6144,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 10125312,
"byteOffset": 5914624
},
{
"name": "model.layers.4.self_attn.qkv_proj.q_scale",
"shape": [
6144,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1265664,
"byteOffset": 16039936
},
{
"name": "model.layers.4.self_attn.o_proj.q_weight",
"shape": [
4096,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 6750208,
"byteOffset": 17305600
},
{
"name": "model.layers.4.self_attn.o_proj.q_scale",
"shape": [
4096,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 843776,
"byteOffset": 24055808
},
{
"name": "model.layers.5.input_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 24899584
},
{
"name": "model.layers.5.mlp.down_proj.q_scale",
"shape": [
4096,
359
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2940928,
"byteOffset": 24907776
}
],
"md5sum": "820eddcbcdd7b5897e7b6765237be11d"
},
{
"dataPath": "params_shard_27.bin",
"format": "raw-shard",
"nbytes": 23527424,
"records": [
{
"name": "model.layers.6.mlp.down_proj.q_weight",
"shape": [
4096,
1436
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 23527424,
"byteOffset": 0
}
],
"md5sum": "b9b4f7aff9c63f853afb87b6456fb033"
},
{
"dataPath": "params_shard_28.bin",
"format": "raw-shard",
"nbytes": 47251456,
"records": [
{
"name": "model.layers.6.mlp.gate_up_proj.q_weight",
"shape": [
28672,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 47251456,
"byteOffset": 0
}
],
"md5sum": "0210c26c345e41a187bc54de921a4202"
},
{
"dataPath": "params_shard_29.bin",
"format": "raw-shard",
"nbytes": 27848704,
"records": [
{
"name": "model.layers.5.mlp.gate_up_proj.q_scale",
"shape": [
28672,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5906432,
"byteOffset": 0
},
{
"name": "model.layers.5.post_attention_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 5906432
},
{
"name": "model.layers.5.self_attn.qkv_proj.q_weight",
"shape": [
6144,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 10125312,
"byteOffset": 5914624
},
{
"name": "model.layers.5.self_attn.qkv_proj.q_scale",
"shape": [
6144,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1265664,
"byteOffset": 16039936
},
{
"name": "model.layers.5.self_attn.o_proj.q_weight",
"shape": [
4096,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 6750208,
"byteOffset": 17305600
},
{
"name": "model.layers.5.self_attn.o_proj.q_scale",
"shape": [
4096,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 843776,
"byteOffset": 24055808
},
{
"name": "model.layers.6.input_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 24899584
},
{
"name": "model.layers.6.mlp.down_proj.q_scale",
"shape": [
4096,
359
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2940928,
"byteOffset": 24907776
}
],
"md5sum": "80ad5d6ea5a1defd1c0742e0bfecd309"
},
{
"dataPath": "params_shard_30.bin",
"format": "raw-shard",
"nbytes": 23527424,
"records": [
{
"name": "model.layers.7.mlp.down_proj.q_weight",
"shape": [
4096,
1436
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 23527424,
"byteOffset": 0
}
],
"md5sum": "066d7d6ba8d063b936abd04872011f58"
},
{
"dataPath": "params_shard_31.bin",
"format": "raw-shard",
"nbytes": 47251456,
"records": [
{
"name": "model.layers.7.mlp.gate_up_proj.q_weight",
"shape": [
28672,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 47251456,
"byteOffset": 0
}
],
"md5sum": "54844886cbb6f5aefc681e4482981f48"
},
{
"dataPath": "params_shard_32.bin",
"format": "raw-shard",
"nbytes": 27848704,
"records": [
{
"name": "model.layers.6.mlp.gate_up_proj.q_scale",
"shape": [
28672,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5906432,
"byteOffset": 0
},
{
"name": "model.layers.6.post_attention_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 5906432
},
{
"name": "model.layers.6.self_attn.qkv_proj.q_weight",
"shape": [
6144,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 10125312,
"byteOffset": 5914624
},
{
"name": "model.layers.6.self_attn.qkv_proj.q_scale",
"shape": [
6144,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1265664,
"byteOffset": 16039936
},
{
"name": "model.layers.6.self_attn.o_proj.q_weight",
"shape": [
4096,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 6750208,
"byteOffset": 17305600
},
{
"name": "model.layers.6.self_attn.o_proj.q_scale",
"shape": [
4096,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 843776,
"byteOffset": 24055808
},
{
"name": "model.layers.7.input_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 24899584
},
{
"name": "model.layers.7.mlp.down_proj.q_scale",
"shape": [
4096,
359
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2940928,
"byteOffset": 24907776
}
],
"md5sum": "39e82cadf945f5a03c03d40f7b2585d2"
},
{
"dataPath": "params_shard_33.bin",
"format": "raw-shard",
"nbytes": 47251456,
"records": [
{
"name": "model.layers.8.mlp.gate_up_proj.q_weight",
"shape": [
28672,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 47251456,
"byteOffset": 0
}
],
"md5sum": "ea9cffc80cad7ec29cb3ef4e26eeceea"
},
{
"dataPath": "params_shard_34.bin",
"format": "raw-shard",
"nbytes": 30806016,
"records": [
{
"name": "model.layers.7.mlp.gate_up_proj.q_scale",
"shape": [
28672,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5906432,
"byteOffset": 0
},
{
"name": "model.layers.7.post_attention_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 5906432
},
{
"name": "model.layers.7.self_attn.qkv_proj.q_weight",
"shape": [
6144,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 10125312,
"byteOffset": 5914624
},
{
"name": "model.layers.7.self_attn.qkv_proj.q_scale",
"shape": [
6144,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1265664,
"byteOffset": 16039936
},
{
"name": "model.layers.7.self_attn.o_proj.q_weight",
"shape": [
4096,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 6750208,
"byteOffset": 17305600
},
{
"name": "model.layers.7.self_attn.o_proj.q_scale",
"shape": [
4096,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 843776,
"byteOffset": 24055808
},
{
"name": "model.layers.8.mlp.gate_up_proj.q_scale",
"shape": [
28672,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5906432,
"byteOffset": 24899584
}
],
"md5sum": "feae454b75af3eb5058badbc8d5046db"
},
{
"dataPath": "params_shard_35.bin",
"format": "raw-shard",
"nbytes": 23527424,
"records": [
{
"name": "model.layers.10.mlp.down_proj.q_weight",
"shape": [
4096,
1436
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 23527424,
"byteOffset": 0
}
],
"md5sum": "cb02e5ba4a97d7487cc7206a5eab9a4a"
},
{
"dataPath": "params_shard_36.bin",
"format": "raw-shard",
"nbytes": 47251456,
"records": [
{
"name": "model.layers.10.mlp.gate_up_proj.q_weight",
"shape": [
28672,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 47251456,
"byteOffset": 0
}
],
"md5sum": "045f33980fe1ec528eefdda72b5c8a30"
},
{
"dataPath": "params_shard_37.bin",
"format": "raw-shard",
"nbytes": 27848704,
"records": [
{
"name": "model.layers.8.self_attn.qkv_proj.q_weight",
"shape": [
6144,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 10125312,
"byteOffset": 0
},
{
"name": "model.layers.8.self_attn.qkv_proj.q_scale",
"shape": [
6144,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1265664,
"byteOffset": 10125312
},
{
"name": "model.layers.8.self_attn.o_proj.q_weight",
"shape": [
4096,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 6750208,
"byteOffset": 11390976
},
{
"name": "model.layers.8.self_attn.o_proj.q_scale",
"shape": [
4096,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 843776,
"byteOffset": 18141184
},
{
"name": "model.layers.10.input_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 18984960
},
{
"name": "model.layers.10.mlp.down_proj.q_scale",
"shape": [
4096,
359
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2940928,
"byteOffset": 18993152
},
{
"name": "model.layers.10.mlp.gate_up_proj.q_scale",
"shape": [
28672,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5906432,
"byteOffset": 21934080
},
{
"name": "model.layers.10.post_attention_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 27840512
}
],
"md5sum": "010ec614a1ebc001b47f9108fcb9b502"
},
{
"dataPath": "params_shard_38.bin",
"format": "raw-shard",
"nbytes": 23527424,
"records": [
{
"name": "model.layers.11.mlp.down_proj.q_weight",
"shape": [
4096,
1436
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 23527424,
"byteOffset": 0
}
],
"md5sum": "a2a3158a5d185983fc64ac0d553b8dc3"
},
{
"dataPath": "params_shard_39.bin",
"format": "raw-shard",
"nbytes": 47251456,
"records": [
{
"name": "model.layers.11.mlp.gate_up_proj.q_weight",
"shape": [
28672,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 47251456,
"byteOffset": 0
}
],
"md5sum": "dafed300ba384211949646b9de8ee2bb"
},
{
"dataPath": "params_shard_40.bin",
"format": "raw-shard",
"nbytes": 27848704,
"records": [
{
"name": "model.layers.10.self_attn.qkv_proj.q_weight",
"shape": [
6144,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 10125312,
"byteOffset": 0
},
{
"name": "model.layers.10.self_attn.qkv_proj.q_scale",
"shape": [
6144,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1265664,
"byteOffset": 10125312
},
{
"name": "model.layers.10.self_attn.o_proj.q_weight",
"shape": [
4096,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 6750208,
"byteOffset": 11390976
},
{
"name": "model.layers.10.self_attn.o_proj.q_scale",
"shape": [
4096,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 843776,
"byteOffset": 18141184
},
{
"name": "model.layers.11.input_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 18984960
},
{
"name": "model.layers.11.mlp.down_proj.q_scale",
"shape": [
4096,
359
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2940928,
"byteOffset": 18993152
},
{
"name": "model.layers.11.mlp.gate_up_proj.q_scale",
"shape": [
28672,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5906432,
"byteOffset": 21934080
},
{
"name": "model.layers.11.post_attention_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 27840512
}
],
"md5sum": "9ed8cd5c686f59b2d61b0a2aeb75d34e"
},
{
"dataPath": "params_shard_41.bin",
"format": "raw-shard",
"nbytes": 23527424,
"records": [
{
"name": "model.layers.12.mlp.down_proj.q_weight",
"shape": [
4096,
1436
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 23527424,
"byteOffset": 0
}
],
"md5sum": "a28a9f935806434258e3d54aa0bf907d"
},
{
"dataPath": "params_shard_42.bin",
"format": "raw-shard",
"nbytes": 47251456,
"records": [
{
"name": "model.layers.12.mlp.gate_up_proj.q_weight",
"shape": [
28672,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 47251456,
"byteOffset": 0
}
],
"md5sum": "bf1ac96e4a43a114ed791a66b404b2ca"
},
{
"dataPath": "params_shard_43.bin",
"format": "raw-shard",
"nbytes": 27848704,
"records": [
{
"name": "model.layers.11.self_attn.qkv_proj.q_weight",
"shape": [
6144,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 10125312,
"byteOffset": 0
},
{
"name": "model.layers.11.self_attn.qkv_proj.q_scale",
"shape": [
6144,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1265664,
"byteOffset": 10125312
},
{
"name": "model.layers.11.self_attn.o_proj.q_weight",
"shape": [
4096,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 6750208,
"byteOffset": 11390976
},
{
"name": "model.layers.11.self_attn.o_proj.q_scale",
"shape": [
4096,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 843776,
"byteOffset": 18141184
},
{
"name": "model.layers.12.input_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 18984960
},
{
"name": "model.layers.12.mlp.down_proj.q_scale",
"shape": [
4096,
359
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2940928,
"byteOffset": 18993152
},
{
"name": "model.layers.12.mlp.gate_up_proj.q_scale",
"shape": [
28672,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5906432,
"byteOffset": 21934080
},
{
"name": "model.layers.12.post_attention_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 27840512
}
],
"md5sum": "051f51a04e9a7a16a221f1c783a11e8a"
},
{
"dataPath": "params_shard_44.bin",
"format": "raw-shard",
"nbytes": 23527424,
"records": [
{
"name": "model.layers.13.mlp.down_proj.q_weight",
"shape": [
4096,
1436
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 23527424,
"byteOffset": 0
}
],
"md5sum": "58a96d963733fbc6c4c6d29787efdfa8"
},
{
"dataPath": "params_shard_45.bin",
"format": "raw-shard",
"nbytes": 47251456,
"records": [
{
"name": "model.layers.13.mlp.gate_up_proj.q_weight",
"shape": [
28672,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 47251456,
"byteOffset": 0
}
],
"md5sum": "5ae2a04162f26faf21eb9abb3bc4ef12"
},
{
"dataPath": "params_shard_46.bin",
"format": "raw-shard",
"nbytes": 27848704,
"records": [
{
"name": "model.layers.12.self_attn.qkv_proj.q_weight",
"shape": [
6144,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 10125312,
"byteOffset": 0
},
{
"name": "model.layers.12.self_attn.qkv_proj.q_scale",
"shape": [
6144,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1265664,
"byteOffset": 10125312
},
{
"name": "model.layers.12.self_attn.o_proj.q_weight",
"shape": [
4096,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 6750208,
"byteOffset": 11390976
},
{
"name": "model.layers.12.self_attn.o_proj.q_scale",
"shape": [
4096,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 843776,
"byteOffset": 18141184
},
{
"name": "model.layers.13.input_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 18984960
},
{
"name": "model.layers.13.mlp.down_proj.q_scale",
"shape": [
4096,
359
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2940928,
"byteOffset": 18993152
},
{
"name": "model.layers.13.mlp.gate_up_proj.q_scale",
"shape": [
28672,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5906432,
"byteOffset": 21934080
},
{
"name": "model.layers.13.post_attention_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 27840512
}
],
"md5sum": "8a89a7358088c202dd5ed29c7cb3f50e"
},
{
"dataPath": "params_shard_47.bin",
"format": "raw-shard",
"nbytes": 23527424,
"records": [
{
"name": "model.layers.14.mlp.down_proj.q_weight",
"shape": [
4096,
1436
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 23527424,
"byteOffset": 0
}
],
"md5sum": "90ffdd094e1df4155a460b3280777ce2"
},
{
"dataPath": "params_shard_48.bin",
"format": "raw-shard",
"nbytes": 47251456,
"records": [
{
"name": "model.layers.14.mlp.gate_up_proj.q_weight",
"shape": [
28672,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 47251456,
"byteOffset": 0
}
],
"md5sum": "b485cc145543d67b35ce81f950e11ba5"
},
{
"dataPath": "params_shard_49.bin",
"format": "raw-shard",
"nbytes": 27848704,
"records": [
{
"name": "model.layers.13.self_attn.qkv_proj.q_weight",
"shape": [
6144,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 10125312,
"byteOffset": 0
},
{
"name": "model.layers.13.self_attn.qkv_proj.q_scale",
"shape": [
6144,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1265664,
"byteOffset": 10125312
},
{
"name": "model.layers.13.self_attn.o_proj.q_weight",
"shape": [
4096,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 6750208,
"byteOffset": 11390976
},
{
"name": "model.layers.13.self_attn.o_proj.q_scale",
"shape": [
4096,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 843776,
"byteOffset": 18141184
},
{
"name": "model.layers.14.input_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 18984960
},
{
"name": "model.layers.14.mlp.down_proj.q_scale",
"shape": [
4096,
359
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2940928,
"byteOffset": 18993152
},
{
"name": "model.layers.14.mlp.gate_up_proj.q_scale",
"shape": [
28672,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5906432,
"byteOffset": 21934080
},
{
"name": "model.layers.14.post_attention_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 27840512
}
],
"md5sum": "ae3c3e4ade93237881c16432dfbaf334"
},
{
"dataPath": "params_shard_50.bin",
"format": "raw-shard",
"nbytes": 23527424,
"records": [
{
"name": "model.layers.15.mlp.down_proj.q_weight",
"shape": [
4096,
1436
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 23527424,
"byteOffset": 0
}
],
"md5sum": "a205bc432c81d7721163e13aa421e705"
},
{
"dataPath": "params_shard_51.bin",
"format": "raw-shard",
"nbytes": 47251456,
"records": [
{
"name": "model.layers.15.mlp.gate_up_proj.q_weight",
"shape": [
28672,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 47251456,
"byteOffset": 0
}
],
"md5sum": "5aa60d8a6c37e6456ac081d77d765af8"
},
{
"dataPath": "params_shard_52.bin",
"format": "raw-shard",
"nbytes": 27848704,
"records": [
{
"name": "model.layers.14.self_attn.qkv_proj.q_weight",
"shape": [
6144,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 10125312,
"byteOffset": 0
},
{
"name": "model.layers.14.self_attn.qkv_proj.q_scale",
"shape": [
6144,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1265664,
"byteOffset": 10125312
},
{
"name": "model.layers.14.self_attn.o_proj.q_weight",
"shape": [
4096,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 6750208,
"byteOffset": 11390976
},
{
"name": "model.layers.14.self_attn.o_proj.q_scale",
"shape": [
4096,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 843776,
"byteOffset": 18141184
},
{
"name": "model.layers.15.input_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 18984960
},
{
"name": "model.layers.15.mlp.down_proj.q_scale",
"shape": [
4096,
359
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2940928,
"byteOffset": 18993152
},
{
"name": "model.layers.15.mlp.gate_up_proj.q_scale",
"shape": [
28672,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5906432,
"byteOffset": 21934080
},
{
"name": "model.layers.15.post_attention_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 27840512
}
],
"md5sum": "852055d2788d9767dafcbd1529adb118"
},
{
"dataPath": "params_shard_53.bin",
"format": "raw-shard",
"nbytes": 23527424,
"records": [
{
"name": "model.layers.16.mlp.down_proj.q_weight",
"shape": [
4096,
1436
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 23527424,
"byteOffset": 0
}
],
"md5sum": "33f98fcbc661813908f644f37b0a09cf"
},
{
"dataPath": "params_shard_54.bin",
"format": "raw-shard",
"nbytes": 47251456,
"records": [
{
"name": "model.layers.16.mlp.gate_up_proj.q_weight",
"shape": [
28672,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 47251456,
"byteOffset": 0
}
],
"md5sum": "e86d2b3e682ef25259b1e67f689b0ab3"
},
{
"dataPath": "params_shard_55.bin",
"format": "raw-shard",
"nbytes": 27848704,
"records": [
{
"name": "model.layers.15.self_attn.qkv_proj.q_weight",
"shape": [
6144,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 10125312,
"byteOffset": 0
},
{
"name": "model.layers.15.self_attn.qkv_proj.q_scale",
"shape": [
6144,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1265664,
"byteOffset": 10125312
},
{
"name": "model.layers.15.self_attn.o_proj.q_weight",
"shape": [
4096,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 6750208,
"byteOffset": 11390976
},
{
"name": "model.layers.15.self_attn.o_proj.q_scale",
"shape": [
4096,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 843776,
"byteOffset": 18141184
},
{
"name": "model.layers.16.input_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 18984960
},
{
"name": "model.layers.16.mlp.down_proj.q_scale",
"shape": [
4096,
359
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2940928,
"byteOffset": 18993152
},
{
"name": "model.layers.16.mlp.gate_up_proj.q_scale",
"shape": [
28672,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5906432,
"byteOffset": 21934080
},
{
"name": "model.layers.16.post_attention_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 27840512
}
],
"md5sum": "6ae96e729d1df8a26d9dee698b0aa6d0"
},
{
"dataPath": "params_shard_56.bin",
"format": "raw-shard",
"nbytes": 23527424,
"records": [
{
"name": "model.layers.17.mlp.down_proj.q_weight",
"shape": [
4096,
1436
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 23527424,
"byteOffset": 0
}
],
"md5sum": "c3580886ded4a0a0bf9c49ae0d7e9eab"
},
{
"dataPath": "params_shard_57.bin",
"format": "raw-shard",
"nbytes": 47251456,
"records": [
{
"name": "model.layers.17.mlp.gate_up_proj.q_weight",
"shape": [
28672,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 47251456,
"byteOffset": 0
}
],
"md5sum": "fbefac3ac300b0ed381cdb8243702c9d"
},
{
"dataPath": "params_shard_58.bin",
"format": "raw-shard",
"nbytes": 27848704,
"records": [
{
"name": "model.layers.16.self_attn.qkv_proj.q_weight",
"shape": [
6144,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 10125312,
"byteOffset": 0
},
{
"name": "model.layers.16.self_attn.qkv_proj.q_scale",
"shape": [
6144,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1265664,
"byteOffset": 10125312
},
{
"name": "model.layers.16.self_attn.o_proj.q_weight",
"shape": [
4096,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 6750208,
"byteOffset": 11390976
},
{
"name": "model.layers.16.self_attn.o_proj.q_scale",
"shape": [
4096,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 843776,
"byteOffset": 18141184
},
{
"name": "model.layers.17.input_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 18984960
},
{
"name": "model.layers.17.mlp.down_proj.q_scale",
"shape": [
4096,
359
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2940928,
"byteOffset": 18993152
},
{
"name": "model.layers.17.mlp.gate_up_proj.q_scale",
"shape": [
28672,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5906432,
"byteOffset": 21934080
},
{
"name": "model.layers.17.post_attention_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 27840512
}
],
"md5sum": "ab0f3fb95dc44e581470dcb8463e9947"
},
{
"dataPath": "params_shard_59.bin",
"format": "raw-shard",
"nbytes": 23527424,
"records": [
{
"name": "model.layers.18.mlp.down_proj.q_weight",
"shape": [
4096,
1436
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 23527424,
"byteOffset": 0
}
],
"md5sum": "602b0a7abeca898f370474c658ae9e44"
},
{
"dataPath": "params_shard_60.bin",
"format": "raw-shard",
"nbytes": 47251456,
"records": [
{
"name": "model.layers.18.mlp.gate_up_proj.q_weight",
"shape": [
28672,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 47251456,
"byteOffset": 0
}
],
"md5sum": "7b0af9788ab5049af50f69ae5be43845"
},
{
"dataPath": "params_shard_61.bin",
"format": "raw-shard",
"nbytes": 27848704,
"records": [
{
"name": "model.layers.17.self_attn.qkv_proj.q_weight",
"shape": [
6144,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 10125312,
"byteOffset": 0
},
{
"name": "model.layers.17.self_attn.qkv_proj.q_scale",
"shape": [
6144,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1265664,
"byteOffset": 10125312
},
{
"name": "model.layers.17.self_attn.o_proj.q_weight",
"shape": [
4096,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 6750208,
"byteOffset": 11390976
},
{
"name": "model.layers.17.self_attn.o_proj.q_scale",
"shape": [
4096,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 843776,
"byteOffset": 18141184
},
{
"name": "model.layers.18.input_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 18984960
},
{
"name": "model.layers.18.mlp.down_proj.q_scale",
"shape": [
4096,
359
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2940928,
"byteOffset": 18993152
},
{
"name": "model.layers.18.mlp.gate_up_proj.q_scale",
"shape": [
28672,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5906432,
"byteOffset": 21934080
},
{
"name": "model.layers.18.post_attention_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 27840512
}
],
"md5sum": "61313498dbb9e34c58686447e4009182"
},
{
"dataPath": "params_shard_62.bin",
"format": "raw-shard",
"nbytes": 30375936,
"records": [
{
"name": "model.layers.18.self_attn.qkv_proj.q_weight",
"shape": [
6144,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 10125312,
"byteOffset": 0
},
{
"name": "model.layers.18.self_attn.qkv_proj.q_scale",
"shape": [
6144,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1265664,
"byteOffset": 10125312
},
{
"name": "model.layers.18.self_attn.o_proj.q_weight",
"shape": [
4096,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 6750208,
"byteOffset": 11390976
},
{
"name": "model.layers.18.self_attn.o_proj.q_scale",
"shape": [
4096,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 843776,
"byteOffset": 18141184
},
{
"name": "model.layers.19.self_attn.qkv_proj.q_weight",
"shape": [
6144,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 10125312,
"byteOffset": 18984960
},
{
"name": "model.layers.19.self_attn.qkv_proj.q_scale",
"shape": [
6144,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1265664,
"byteOffset": 29110272
}
],
"md5sum": "648e8e9130280c8b563b6c4aae8804d1"
},
{
"dataPath": "params_shard_63.bin",
"format": "raw-shard",
"nbytes": 31129600,
"records": [
{
"name": "model.layers.19.self_attn.o_proj.q_weight",
"shape": [
4096,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 6750208,
"byteOffset": 0
},
{
"name": "model.layers.19.self_attn.o_proj.q_scale",
"shape": [
4096,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 843776,
"byteOffset": 6750208
},
{
"name": "model.layers.8.input_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 7593984
},
{
"name": "model.layers.8.mlp.down_proj.q_weight",
"shape": [
4096,
1436
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 23527424,
"byteOffset": 7602176
}
],
"md5sum": "da5a826b8b94bf79dfe27e058959255f"
},
{
"dataPath": "params_shard_64.bin",
"format": "raw-shard",
"nbytes": 47251456,
"records": [
{
"name": "model.layers.9.mlp.gate_up_proj.q_weight",
"shape": [
28672,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 47251456,
"byteOffset": 0
}
],
"md5sum": "7d40270878e79abc091b6d4af1aec723"
},
{
"dataPath": "params_shard_65.bin",
"format": "raw-shard",
"nbytes": 29425664,
"records": [
{
"name": "model.layers.8.mlp.down_proj.q_scale",
"shape": [
4096,
359
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2940928,
"byteOffset": 0
},
{
"name": "model.layers.8.post_attention_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 2940928
},
{
"name": "model.layers.9.input_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 2949120
},
{
"name": "model.layers.9.mlp.down_proj.q_weight",
"shape": [
4096,
1436
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 23527424,
"byteOffset": 2957312
},
{
"name": "model.layers.9.mlp.down_proj.q_scale",
"shape": [
4096,
359
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2940928,
"byteOffset": 26484736
}
],
"md5sum": "91175068c39f0a668d8cbb4df1057c07"
},
{
"dataPath": "params_shard_66.bin",
"format": "raw-shard",
"nbytes": 23527424,
"records": [
{
"name": "model.layers.19.mlp.down_proj.q_weight",
"shape": [
4096,
1436
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 23527424,
"byteOffset": 0
}
],
"md5sum": "32bd360438bdd49d8459611c73dd1336"
},
{
"dataPath": "params_shard_67.bin",
"format": "raw-shard",
"nbytes": 47251456,
"records": [
{
"name": "model.layers.19.mlp.gate_up_proj.q_weight",
"shape": [
28672,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 47251456,
"byteOffset": 0
}
],
"md5sum": "266037bf2909c6ac4d340c56d341ec8e"
},
{
"dataPath": "params_shard_68.bin",
"format": "raw-shard",
"nbytes": 27848704,
"records": [
{
"name": "model.layers.9.mlp.gate_up_proj.q_scale",
"shape": [
28672,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5906432,
"byteOffset": 0
},
{
"name": "model.layers.9.post_attention_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 5906432
},
{
"name": "model.layers.9.self_attn.qkv_proj.q_weight",
"shape": [
6144,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 10125312,
"byteOffset": 5914624
},
{
"name": "model.layers.9.self_attn.qkv_proj.q_scale",
"shape": [
6144,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1265664,
"byteOffset": 16039936
},
{
"name": "model.layers.9.self_attn.o_proj.q_weight",
"shape": [
4096,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 6750208,
"byteOffset": 17305600
},
{
"name": "model.layers.9.self_attn.o_proj.q_scale",
"shape": [
4096,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 843776,
"byteOffset": 24055808
},
{
"name": "model.layers.19.input_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 24899584
},
{
"name": "model.layers.19.mlp.down_proj.q_scale",
"shape": [
4096,
359
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2940928,
"byteOffset": 24907776
}
],
"md5sum": "12093f2e7d6129be315e991c1f563109"
},
{
"dataPath": "params_shard_69.bin",
"format": "raw-shard",
"nbytes": 47251456,
"records": [
{
"name": "model.layers.20.mlp.gate_up_proj.q_weight",
"shape": [
28672,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 47251456,
"byteOffset": 0
}
],
"md5sum": "0644855615fbd8f540b0d6da3071597b"
},
{
"dataPath": "params_shard_70.bin",
"format": "raw-shard",
"nbytes": 32391168,
"records": [
{
"name": "model.layers.19.mlp.gate_up_proj.q_scale",
"shape": [
28672,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5906432,
"byteOffset": 0
},
{
"name": "model.layers.19.post_attention_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 5906432
},
{
"name": "model.layers.20.input_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 5914624
},
{
"name": "model.layers.20.mlp.down_proj.q_weight",
"shape": [
4096,
1436
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 23527424,
"byteOffset": 5922816
},
{
"name": "model.layers.20.mlp.down_proj.q_scale",
"shape": [
4096,
359
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2940928,
"byteOffset": 29450240
}
],
"md5sum": "5284fd5f08177bb391cdf0201f9eae06"
},
{
"dataPath": "params_shard_71.bin",
"format": "raw-shard",
"nbytes": 23527424,
"records": [
{
"name": "model.layers.21.mlp.down_proj.q_weight",
"shape": [
4096,
1436
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 23527424,
"byteOffset": 0
}
],
"md5sum": "7b33ff91b42c5c69314cdc8c8e6df81b"
},
{
"dataPath": "params_shard_72.bin",
"format": "raw-shard",
"nbytes": 47251456,
"records": [
{
"name": "model.layers.21.mlp.gate_up_proj.q_weight",
"shape": [
28672,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 47251456,
"byteOffset": 0
}
],
"md5sum": "4d2f0f7194dda87dbc26a05df89c24e4"
},
{
"dataPath": "params_shard_73.bin",
"format": "raw-shard",
"nbytes": 27848704,
"records": [
{
"name": "model.layers.20.mlp.gate_up_proj.q_scale",
"shape": [
28672,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5906432,
"byteOffset": 0
},
{
"name": "model.layers.20.post_attention_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 5906432
},
{
"name": "model.layers.20.self_attn.qkv_proj.q_weight",
"shape": [
6144,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 10125312,
"byteOffset": 5914624
},
{
"name": "model.layers.20.self_attn.qkv_proj.q_scale",
"shape": [
6144,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1265664,
"byteOffset": 16039936
},
{
"name": "model.layers.20.self_attn.o_proj.q_weight",
"shape": [
4096,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 6750208,
"byteOffset": 17305600
},
{
"name": "model.layers.20.self_attn.o_proj.q_scale",
"shape": [
4096,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 843776,
"byteOffset": 24055808
},
{
"name": "model.layers.21.input_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 24899584
},
{
"name": "model.layers.21.mlp.down_proj.q_scale",
"shape": [
4096,
359
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2940928,
"byteOffset": 24907776
}
],
"md5sum": "5a416819c3805ab2272f2c1d09b2eeb4"
},
{
"dataPath": "params_shard_74.bin",
"format": "raw-shard",
"nbytes": 23527424,
"records": [
{
"name": "model.layers.22.mlp.down_proj.q_weight",
"shape": [
4096,
1436
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 23527424,
"byteOffset": 0
}
],
"md5sum": "92830b78d5781606a99c9c40cddad980"
},
{
"dataPath": "params_shard_75.bin",
"format": "raw-shard",
"nbytes": 47251456,
"records": [
{
"name": "model.layers.22.mlp.gate_up_proj.q_weight",
"shape": [
28672,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 47251456,
"byteOffset": 0
}
],
"md5sum": "f1b194301a9f9b664e06da230e669881"
},
{
"dataPath": "params_shard_76.bin",
"format": "raw-shard",
"nbytes": 27848704,
"records": [
{
"name": "model.layers.21.mlp.gate_up_proj.q_scale",
"shape": [
28672,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5906432,
"byteOffset": 0
},
{
"name": "model.layers.21.post_attention_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 5906432
},
{
"name": "model.layers.21.self_attn.qkv_proj.q_weight",
"shape": [
6144,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 10125312,
"byteOffset": 5914624
},
{
"name": "model.layers.21.self_attn.qkv_proj.q_scale",
"shape": [
6144,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1265664,
"byteOffset": 16039936
},
{
"name": "model.layers.21.self_attn.o_proj.q_weight",
"shape": [
4096,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 6750208,
"byteOffset": 17305600
},
{
"name": "model.layers.21.self_attn.o_proj.q_scale",
"shape": [
4096,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 843776,
"byteOffset": 24055808
},
{
"name": "model.layers.22.input_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 24899584
},
{
"name": "model.layers.22.mlp.down_proj.q_scale",
"shape": [
4096,
359
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2940928,
"byteOffset": 24907776
}
],
"md5sum": "1b41e0f6535be053b6856349d98a482d"
},
{
"dataPath": "params_shard_77.bin",
"format": "raw-shard",
"nbytes": 23527424,
"records": [
{
"name": "model.layers.23.mlp.down_proj.q_weight",
"shape": [
4096,
1436
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 23527424,
"byteOffset": 0
}
],
"md5sum": "f1f1ccb477682704d94d3d225bda5cd5"
},
{
"dataPath": "params_shard_78.bin",
"format": "raw-shard",
"nbytes": 47251456,
"records": [
{
"name": "model.layers.23.mlp.gate_up_proj.q_weight",
"shape": [
28672,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 47251456,
"byteOffset": 0
}
],
"md5sum": "a84fb1e8aa532004ad869f6607a47756"
},
{
"dataPath": "params_shard_79.bin",
"format": "raw-shard",
"nbytes": 27848704,
"records": [
{
"name": "model.layers.22.mlp.gate_up_proj.q_scale",
"shape": [
28672,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5906432,
"byteOffset": 0
},
{
"name": "model.layers.22.post_attention_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 5906432
},
{
"name": "model.layers.22.self_attn.qkv_proj.q_weight",
"shape": [
6144,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 10125312,
"byteOffset": 5914624
},
{
"name": "model.layers.22.self_attn.qkv_proj.q_scale",
"shape": [
6144,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1265664,
"byteOffset": 16039936
},
{
"name": "model.layers.22.self_attn.o_proj.q_weight",
"shape": [
4096,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 6750208,
"byteOffset": 17305600
},
{
"name": "model.layers.22.self_attn.o_proj.q_scale",
"shape": [
4096,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 843776,
"byteOffset": 24055808
},
{
"name": "model.layers.23.input_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 24899584
},
{
"name": "model.layers.23.mlp.down_proj.q_scale",
"shape": [
4096,
359
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2940928,
"byteOffset": 24907776
}
],
"md5sum": "fbcc5cded415e4ffe024b144541b73dd"
},
{
"dataPath": "params_shard_80.bin",
"format": "raw-shard",
"nbytes": 23527424,
"records": [
{
"name": "model.layers.24.mlp.down_proj.q_weight",
"shape": [
4096,
1436
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 23527424,
"byteOffset": 0
}
],
"md5sum": "ff57504860315c3a48dbde7da26eba81"
},
{
"dataPath": "params_shard_81.bin",
"format": "raw-shard",
"nbytes": 47251456,
"records": [
{
"name": "model.layers.24.mlp.gate_up_proj.q_weight",
"shape": [
28672,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 47251456,
"byteOffset": 0
}
],
"md5sum": "771e18888d1142ac0e0ede3e23444aa3"
},
{
"dataPath": "params_shard_82.bin",
"format": "raw-shard",
"nbytes": 27848704,
"records": [
{
"name": "model.layers.23.mlp.gate_up_proj.q_scale",
"shape": [
28672,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5906432,
"byteOffset": 0
},
{
"name": "model.layers.23.post_attention_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 5906432
},
{
"name": "model.layers.23.self_attn.qkv_proj.q_weight",
"shape": [
6144,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 10125312,
"byteOffset": 5914624
},
{
"name": "model.layers.23.self_attn.qkv_proj.q_scale",
"shape": [
6144,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1265664,
"byteOffset": 16039936
},
{
"name": "model.layers.23.self_attn.o_proj.q_weight",
"shape": [
4096,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 6750208,
"byteOffset": 17305600
},
{
"name": "model.layers.23.self_attn.o_proj.q_scale",
"shape": [
4096,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 843776,
"byteOffset": 24055808
},
{
"name": "model.layers.24.input_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 24899584
},
{
"name": "model.layers.24.mlp.down_proj.q_scale",
"shape": [
4096,
359
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2940928,
"byteOffset": 24907776
}
],
"md5sum": "98e887e02d60cc79eb8247da7f96f89d"
},
{
"dataPath": "params_shard_83.bin",
"format": "raw-shard",
"nbytes": 23527424,
"records": [
{
"name": "model.layers.25.mlp.down_proj.q_weight",
"shape": [
4096,
1436
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 23527424,
"byteOffset": 0
}
],
"md5sum": "adeaaba35589cce9a8509ed242ddf520"
},
{
"dataPath": "params_shard_84.bin",
"format": "raw-shard",
"nbytes": 47251456,
"records": [
{
"name": "model.layers.25.mlp.gate_up_proj.q_weight",
"shape": [
28672,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 47251456,
"byteOffset": 0
}
],
"md5sum": "bfa510e811d96bcc3eb75ced4d2cc50e"
},
{
"dataPath": "params_shard_85.bin",
"format": "raw-shard",
"nbytes": 27848704,
"records": [
{
"name": "model.layers.24.mlp.gate_up_proj.q_scale",
"shape": [
28672,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5906432,
"byteOffset": 0
},
{
"name": "model.layers.24.post_attention_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 5906432
},
{
"name": "model.layers.24.self_attn.qkv_proj.q_weight",
"shape": [
6144,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 10125312,
"byteOffset": 5914624
},
{
"name": "model.layers.24.self_attn.qkv_proj.q_scale",
"shape": [
6144,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1265664,
"byteOffset": 16039936
},
{
"name": "model.layers.24.self_attn.o_proj.q_weight",
"shape": [
4096,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 6750208,
"byteOffset": 17305600
},
{
"name": "model.layers.24.self_attn.o_proj.q_scale",
"shape": [
4096,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 843776,
"byteOffset": 24055808
},
{
"name": "model.layers.25.input_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 24899584
},
{
"name": "model.layers.25.mlp.down_proj.q_scale",
"shape": [
4096,
359
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2940928,
"byteOffset": 24907776
}
],
"md5sum": "41f616066a8ec148c0e7ff71ae7659cb"
},
{
"dataPath": "params_shard_86.bin",
"format": "raw-shard",
"nbytes": 23527424,
"records": [
{
"name": "model.layers.26.mlp.down_proj.q_weight",
"shape": [
4096,
1436
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 23527424,
"byteOffset": 0
}
],
"md5sum": "176a2262ae34f6b371dbd1ad760bb3e4"
},
{
"dataPath": "params_shard_87.bin",
"format": "raw-shard",
"nbytes": 47251456,
"records": [
{
"name": "model.layers.26.mlp.gate_up_proj.q_weight",
"shape": [
28672,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 47251456,
"byteOffset": 0
}
],
"md5sum": "f69bf8f46c8ea6d3fe0f0985a4120620"
},
{
"dataPath": "params_shard_88.bin",
"format": "raw-shard",
"nbytes": 27848704,
"records": [
{
"name": "model.layers.25.mlp.gate_up_proj.q_scale",
"shape": [
28672,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5906432,
"byteOffset": 0
},
{
"name": "model.layers.25.post_attention_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 5906432
},
{
"name": "model.layers.25.self_attn.qkv_proj.q_weight",
"shape": [
6144,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 10125312,
"byteOffset": 5914624
},
{
"name": "model.layers.25.self_attn.qkv_proj.q_scale",
"shape": [
6144,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1265664,
"byteOffset": 16039936
},
{
"name": "model.layers.25.self_attn.o_proj.q_weight",
"shape": [
4096,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 6750208,
"byteOffset": 17305600
},
{
"name": "model.layers.25.self_attn.o_proj.q_scale",
"shape": [
4096,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 843776,
"byteOffset": 24055808
},
{
"name": "model.layers.26.input_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 24899584
},
{
"name": "model.layers.26.mlp.down_proj.q_scale",
"shape": [
4096,
359
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2940928,
"byteOffset": 24907776
}
],
"md5sum": "53e22d0eadfd2188662bedd009d273de"
},
{
"dataPath": "params_shard_89.bin",
"format": "raw-shard",
"nbytes": 23527424,
"records": [
{
"name": "model.layers.27.mlp.down_proj.q_weight",
"shape": [
4096,
1436
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 23527424,
"byteOffset": 0
}
],
"md5sum": "7e3a99f94e9450a3a5c3efead825988a"
},
{
"dataPath": "params_shard_90.bin",
"format": "raw-shard",
"nbytes": 47251456,
"records": [
{
"name": "model.layers.27.mlp.gate_up_proj.q_weight",
"shape": [
28672,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 47251456,
"byteOffset": 0
}
],
"md5sum": "e112529e042a12f10d9d422bd2ca4e1d"
},
{
"dataPath": "params_shard_91.bin",
"format": "raw-shard",
"nbytes": 27848704,
"records": [
{
"name": "model.layers.26.mlp.gate_up_proj.q_scale",
"shape": [
28672,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5906432,
"byteOffset": 0
},
{
"name": "model.layers.26.post_attention_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 5906432
},
{
"name": "model.layers.26.self_attn.qkv_proj.q_weight",
"shape": [
6144,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 10125312,
"byteOffset": 5914624
},
{
"name": "model.layers.26.self_attn.qkv_proj.q_scale",
"shape": [
6144,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1265664,
"byteOffset": 16039936
},
{
"name": "model.layers.26.self_attn.o_proj.q_weight",
"shape": [
4096,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 6750208,
"byteOffset": 17305600
},
{
"name": "model.layers.26.self_attn.o_proj.q_scale",
"shape": [
4096,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 843776,
"byteOffset": 24055808
},
{
"name": "model.layers.27.input_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 24899584
},
{
"name": "model.layers.27.mlp.down_proj.q_scale",
"shape": [
4096,
359
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2940928,
"byteOffset": 24907776
}
],
"md5sum": "706b6aa64caf80d0d718e053dc8a46ac"
},
{
"dataPath": "params_shard_92.bin",
"format": "raw-shard",
"nbytes": 23527424,
"records": [
{
"name": "model.layers.28.mlp.down_proj.q_weight",
"shape": [
4096,
1436
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 23527424,
"byteOffset": 0
}
],
"md5sum": "5c2a37f803be7d1b04aa05d71825f9a6"
},
{
"dataPath": "params_shard_93.bin",
"format": "raw-shard",
"nbytes": 47251456,
"records": [
{
"name": "model.layers.28.mlp.gate_up_proj.q_weight",
"shape": [
28672,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 47251456,
"byteOffset": 0
}
],
"md5sum": "a8cbebcd9cc8174d417965ba239c87ee"
},
{
"dataPath": "params_shard_94.bin",
"format": "raw-shard",
"nbytes": 27848704,
"records": [
{
"name": "model.layers.27.mlp.gate_up_proj.q_scale",
"shape": [
28672,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5906432,
"byteOffset": 0
},
{
"name": "model.layers.27.post_attention_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 5906432
},
{
"name": "model.layers.27.self_attn.qkv_proj.q_weight",
"shape": [
6144,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 10125312,
"byteOffset": 5914624
},
{
"name": "model.layers.27.self_attn.qkv_proj.q_scale",
"shape": [
6144,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1265664,
"byteOffset": 16039936
},
{
"name": "model.layers.27.self_attn.o_proj.q_weight",
"shape": [
4096,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 6750208,
"byteOffset": 17305600
},
{
"name": "model.layers.27.self_attn.o_proj.q_scale",
"shape": [
4096,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 843776,
"byteOffset": 24055808
},
{
"name": "model.layers.28.input_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 24899584
},
{
"name": "model.layers.28.mlp.down_proj.q_scale",
"shape": [
4096,
359
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2940928,
"byteOffset": 24907776
}
],
"md5sum": "938442076e637796b31f05929bb0582b"
},
{
"dataPath": "params_shard_95.bin",
"format": "raw-shard",
"nbytes": 23527424,
"records": [
{
"name": "model.layers.29.mlp.down_proj.q_weight",
"shape": [
4096,
1436
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 23527424,
"byteOffset": 0
}
],
"md5sum": "b02399ea783cd18cbfe20d388a542aea"
},
{
"dataPath": "params_shard_96.bin",
"format": "raw-shard",
"nbytes": 47251456,
"records": [
{
"name": "model.layers.29.mlp.gate_up_proj.q_weight",
"shape": [
28672,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 47251456,
"byteOffset": 0
}
],
"md5sum": "32c8ea000e11dbad870935a293325500"
},
{
"dataPath": "params_shard_97.bin",
"format": "raw-shard",
"nbytes": 27848704,
"records": [
{
"name": "model.layers.28.mlp.gate_up_proj.q_scale",
"shape": [
28672,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5906432,
"byteOffset": 0
},
{
"name": "model.layers.28.post_attention_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 5906432
},
{
"name": "model.layers.28.self_attn.qkv_proj.q_weight",
"shape": [
6144,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 10125312,
"byteOffset": 5914624
},
{
"name": "model.layers.28.self_attn.qkv_proj.q_scale",
"shape": [
6144,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1265664,
"byteOffset": 16039936
},
{
"name": "model.layers.28.self_attn.o_proj.q_weight",
"shape": [
4096,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 6750208,
"byteOffset": 17305600
},
{
"name": "model.layers.28.self_attn.o_proj.q_scale",
"shape": [
4096,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 843776,
"byteOffset": 24055808
},
{
"name": "model.layers.29.input_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 24899584
},
{
"name": "model.layers.29.mlp.down_proj.q_scale",
"shape": [
4096,
359
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2940928,
"byteOffset": 24907776
}
],
"md5sum": "b25cee1785f23496e6c69f6d393b94da"
},
{
"dataPath": "params_shard_98.bin",
"format": "raw-shard",
"nbytes": 24899584,
"records": [
{
"name": "model.layers.29.mlp.gate_up_proj.q_scale",
"shape": [
28672,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5906432,
"byteOffset": 0
},
{
"name": "model.layers.29.post_attention_layernorm.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 5906432
},
{
"name": "model.layers.29.self_attn.qkv_proj.q_weight",
"shape": [
6144,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 10125312,
"byteOffset": 5914624
},
{
"name": "model.layers.29.self_attn.qkv_proj.q_scale",
"shape": [
6144,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1265664,
"byteOffset": 16039936
},
{
"name": "model.layers.29.self_attn.o_proj.q_weight",
"shape": [
4096,
412
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 6750208,
"byteOffset": 17305600
},
{
"name": "model.layers.29.self_attn.o_proj.q_scale",
"shape": [
4096,
103
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 843776,
"byteOffset": 24055808
}
],
"md5sum": "46fc5ca3bc301942884152d5e661a1f8"
}
]
}