Tai Truong
Add nb-llama-3.2-1B instruct model weights
48f7157
{
"metadata": {
"ParamSize": 163,
"ParamBytes": 695242752.0,
"BitsPerParam": 4.500628909972241
},
"records": [
{
"dataPath": "params_shard_0.bin",
"format": "raw-shard",
"nbytes": 131334144,
"records": [
{
"name": "model.embed_tokens.q_weight",
"shape": [
128256,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 131334144,
"byteOffset": 0
}
],
"md5sum": "8be0e3dab65f4c84b76a97120fc77fea"
},
{
"dataPath": "params_shard_1.bin",
"format": "raw-shard",
"nbytes": 16777216,
"records": [
{
"name": "model.layers.0.mlp.gate_up_proj.q_weight",
"shape": [
256,
16384
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 16777216,
"byteOffset": 0
}
],
"md5sum": "d04b33f62f903d458f357e64f177348b"
},
{
"dataPath": "params_shard_2.bin",
"format": "raw-shard",
"nbytes": 31498240,
"records": [
{
"name": "model.embed_tokens.q_scale",
"shape": [
128256,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 16416768,
"byteOffset": 0
},
{
"name": "model.layers.0.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 16416768
},
{
"name": "model.layers.0.mlp.down_proj.q_weight",
"shape": [
1024,
2048
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 16420864
},
{
"name": "model.layers.0.mlp.down_proj.q_scale",
"shape": [
256,
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 24809472
},
{
"name": "model.layers.0.mlp.gate_up_proj.q_scale",
"shape": [
64,
16384
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 25858048
},
{
"name": "model.layers.0.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 27955200
},
{
"name": "model.layers.0.self_attn.qkv_proj.q_weight",
"shape": [
256,
3072
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 27959296
},
{
"name": "model.layers.0.self_attn.qkv_proj.q_scale",
"shape": [
64,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 393216,
"byteOffset": 31105024
}
],
"md5sum": "80d8c59ef00681923488dd7ed4b0f8ba"
},
{
"dataPath": "params_shard_3.bin",
"format": "raw-shard",
"nbytes": 30679040,
"records": [
{
"name": "model.layers.0.self_attn.o_proj.q_weight",
"shape": [
256,
2048
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 0
},
{
"name": "model.layers.0.self_attn.o_proj.q_scale",
"shape": [
64,
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 262144,
"byteOffset": 2097152
},
{
"name": "model.layers.1.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 2359296
},
{
"name": "model.layers.1.mlp.down_proj.q_weight",
"shape": [
1024,
2048
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 2363392
},
{
"name": "model.layers.1.mlp.down_proj.q_scale",
"shape": [
256,
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 10752000
},
{
"name": "model.layers.1.mlp.gate_up_proj.q_weight",
"shape": [
256,
16384
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 16777216,
"byteOffset": 11800576
},
{
"name": "model.layers.1.mlp.gate_up_proj.q_scale",
"shape": [
64,
16384
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 28577792
},
{
"name": "model.layers.1.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 30674944
}
],
"md5sum": "08a7c4a79b612619b5461a55aa38f548"
},
{
"dataPath": "params_shard_4.bin",
"format": "raw-shard",
"nbytes": 32116736,
"records": [
{
"name": "model.layers.1.self_attn.qkv_proj.q_weight",
"shape": [
256,
3072
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 0
},
{
"name": "model.layers.1.self_attn.qkv_proj.q_scale",
"shape": [
64,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 393216,
"byteOffset": 3145728
},
{
"name": "model.layers.1.self_attn.o_proj.q_weight",
"shape": [
256,
2048
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 3538944
},
{
"name": "model.layers.1.self_attn.o_proj.q_scale",
"shape": [
64,
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 262144,
"byteOffset": 5636096
},
{
"name": "model.layers.10.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 5898240
},
{
"name": "model.layers.10.mlp.down_proj.q_weight",
"shape": [
1024,
2048
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 5902336
},
{
"name": "model.layers.10.mlp.down_proj.q_scale",
"shape": [
256,
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 14290944
},
{
"name": "model.layers.10.mlp.gate_up_proj.q_weight",
"shape": [
256,
16384
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 16777216,
"byteOffset": 15339520
}
],
"md5sum": "45ef2eb64a2bd87980e414bff84d5aaa"
},
{
"dataPath": "params_shard_5.bin",
"format": "raw-shard",
"nbytes": 16777216,
"records": [
{
"name": "model.layers.11.mlp.gate_up_proj.q_weight",
"shape": [
256,
16384
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 16777216,
"byteOffset": 0
}
],
"md5sum": "19dea51b570a5dd30c50ef6ed2117c12"
},
{
"dataPath": "params_shard_6.bin",
"format": "raw-shard",
"nbytes": 25444352,
"records": [
{
"name": "model.layers.10.mlp.gate_up_proj.q_scale",
"shape": [
64,
16384
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 0
},
{
"name": "model.layers.10.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 2097152
},
{
"name": "model.layers.10.self_attn.qkv_proj.q_weight",
"shape": [
256,
3072
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 2101248
},
{
"name": "model.layers.10.self_attn.qkv_proj.q_scale",
"shape": [
64,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 393216,
"byteOffset": 5246976
},
{
"name": "model.layers.10.self_attn.o_proj.q_weight",
"shape": [
256,
2048
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 5640192
},
{
"name": "model.layers.10.self_attn.o_proj.q_scale",
"shape": [
64,
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 262144,
"byteOffset": 7737344
},
{
"name": "model.layers.11.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 7999488
},
{
"name": "model.layers.11.mlp.down_proj.q_weight",
"shape": [
1024,
2048
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 8003584
},
{
"name": "model.layers.11.mlp.down_proj.q_scale",
"shape": [
256,
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 16392192
},
{
"name": "model.layers.11.mlp.gate_up_proj.q_scale",
"shape": [
64,
16384
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 17440768
},
{
"name": "model.layers.11.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 19537920
},
{
"name": "model.layers.11.self_attn.qkv_proj.q_weight",
"shape": [
256,
3072
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 19542016
},
{
"name": "model.layers.11.self_attn.qkv_proj.q_scale",
"shape": [
64,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 393216,
"byteOffset": 22687744
},
{
"name": "model.layers.11.self_attn.o_proj.q_weight",
"shape": [
256,
2048
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 23080960
},
{
"name": "model.layers.11.self_attn.o_proj.q_scale",
"shape": [
64,
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 262144,
"byteOffset": 25178112
},
{
"name": "model.layers.12.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 25440256
}
],
"md5sum": "0b0ae9116e8c35f810fa91c23ffc9825"
},
{
"dataPath": "params_shard_7.bin",
"format": "raw-shard",
"nbytes": 31854592,
"records": [
{
"name": "model.layers.12.mlp.down_proj.q_weight",
"shape": [
1024,
2048
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 0
},
{
"name": "model.layers.12.mlp.down_proj.q_scale",
"shape": [
256,
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 8388608
},
{
"name": "model.layers.12.mlp.gate_up_proj.q_weight",
"shape": [
256,
16384
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 16777216,
"byteOffset": 9437184
},
{
"name": "model.layers.12.mlp.gate_up_proj.q_scale",
"shape": [
64,
16384
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 26214400
},
{
"name": "model.layers.12.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 28311552
},
{
"name": "model.layers.12.self_attn.qkv_proj.q_weight",
"shape": [
256,
3072
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 28315648
},
{
"name": "model.layers.12.self_attn.qkv_proj.q_scale",
"shape": [
64,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 393216,
"byteOffset": 31461376
}
],
"md5sum": "35f4eba4148180af9ca07d414904433a"
},
{
"dataPath": "params_shard_8.bin",
"format": "raw-shard",
"nbytes": 30679040,
"records": [
{
"name": "model.layers.12.self_attn.o_proj.q_weight",
"shape": [
256,
2048
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 0
},
{
"name": "model.layers.12.self_attn.o_proj.q_scale",
"shape": [
64,
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 262144,
"byteOffset": 2097152
},
{
"name": "model.layers.13.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 2359296
},
{
"name": "model.layers.13.mlp.down_proj.q_weight",
"shape": [
1024,
2048
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 2363392
},
{
"name": "model.layers.13.mlp.down_proj.q_scale",
"shape": [
256,
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 10752000
},
{
"name": "model.layers.13.mlp.gate_up_proj.q_weight",
"shape": [
256,
16384
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 16777216,
"byteOffset": 11800576
},
{
"name": "model.layers.13.mlp.gate_up_proj.q_scale",
"shape": [
64,
16384
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 28577792
},
{
"name": "model.layers.13.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 30674944
}
],
"md5sum": "19074e02c753bfc72bea098a9265994a"
},
{
"dataPath": "params_shard_9.bin",
"format": "raw-shard",
"nbytes": 32116736,
"records": [
{
"name": "model.layers.13.self_attn.qkv_proj.q_weight",
"shape": [
256,
3072
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 0
},
{
"name": "model.layers.13.self_attn.qkv_proj.q_scale",
"shape": [
64,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 393216,
"byteOffset": 3145728
},
{
"name": "model.layers.13.self_attn.o_proj.q_weight",
"shape": [
256,
2048
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 3538944
},
{
"name": "model.layers.13.self_attn.o_proj.q_scale",
"shape": [
64,
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 262144,
"byteOffset": 5636096
},
{
"name": "model.layers.14.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 5898240
},
{
"name": "model.layers.14.mlp.down_proj.q_weight",
"shape": [
1024,
2048
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 5902336
},
{
"name": "model.layers.14.mlp.down_proj.q_scale",
"shape": [
256,
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 14290944
},
{
"name": "model.layers.14.mlp.gate_up_proj.q_weight",
"shape": [
256,
16384
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 16777216,
"byteOffset": 15339520
}
],
"md5sum": "3b522276f928828507884a3e984c9bc1"
},
{
"dataPath": "params_shard_10.bin",
"format": "raw-shard",
"nbytes": 16777216,
"records": [
{
"name": "model.layers.15.mlp.gate_up_proj.q_weight",
"shape": [
256,
16384
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 16777216,
"byteOffset": 0
}
],
"md5sum": "5d110543543eaac23e9719cc07614ce2"
},
{
"dataPath": "params_shard_11.bin",
"format": "raw-shard",
"nbytes": 25444352,
"records": [
{
"name": "model.layers.14.mlp.gate_up_proj.q_scale",
"shape": [
64,
16384
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 0
},
{
"name": "model.layers.14.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 2097152
},
{
"name": "model.layers.14.self_attn.qkv_proj.q_weight",
"shape": [
256,
3072
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 2101248
},
{
"name": "model.layers.14.self_attn.qkv_proj.q_scale",
"shape": [
64,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 393216,
"byteOffset": 5246976
},
{
"name": "model.layers.14.self_attn.o_proj.q_weight",
"shape": [
256,
2048
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 5640192
},
{
"name": "model.layers.14.self_attn.o_proj.q_scale",
"shape": [
64,
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 262144,
"byteOffset": 7737344
},
{
"name": "model.layers.15.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 7999488
},
{
"name": "model.layers.15.mlp.down_proj.q_weight",
"shape": [
1024,
2048
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 8003584
},
{
"name": "model.layers.15.mlp.down_proj.q_scale",
"shape": [
256,
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 16392192
},
{
"name": "model.layers.15.mlp.gate_up_proj.q_scale",
"shape": [
64,
16384
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 17440768
},
{
"name": "model.layers.15.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 19537920
},
{
"name": "model.layers.15.self_attn.qkv_proj.q_weight",
"shape": [
256,
3072
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 19542016
},
{
"name": "model.layers.15.self_attn.qkv_proj.q_scale",
"shape": [
64,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 393216,
"byteOffset": 22687744
},
{
"name": "model.layers.15.self_attn.o_proj.q_weight",
"shape": [
256,
2048
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 23080960
},
{
"name": "model.layers.15.self_attn.o_proj.q_scale",
"shape": [
64,
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 262144,
"byteOffset": 25178112
},
{
"name": "model.layers.2.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 25440256
}
],
"md5sum": "b9e0a6877ade0886f4869a278d0ba6b7"
},
{
"dataPath": "params_shard_12.bin",
"format": "raw-shard",
"nbytes": 31854592,
"records": [
{
"name": "model.layers.2.mlp.down_proj.q_weight",
"shape": [
1024,
2048
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 0
},
{
"name": "model.layers.2.mlp.down_proj.q_scale",
"shape": [
256,
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 8388608
},
{
"name": "model.layers.2.mlp.gate_up_proj.q_weight",
"shape": [
256,
16384
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 16777216,
"byteOffset": 9437184
},
{
"name": "model.layers.2.mlp.gate_up_proj.q_scale",
"shape": [
64,
16384
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 26214400
},
{
"name": "model.layers.2.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 28311552
},
{
"name": "model.layers.2.self_attn.qkv_proj.q_weight",
"shape": [
256,
3072
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 28315648
},
{
"name": "model.layers.2.self_attn.qkv_proj.q_scale",
"shape": [
64,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 393216,
"byteOffset": 31461376
}
],
"md5sum": "c775d0bd34e476762a4aeb480b7d393f"
},
{
"dataPath": "params_shard_13.bin",
"format": "raw-shard",
"nbytes": 30679040,
"records": [
{
"name": "model.layers.2.self_attn.o_proj.q_weight",
"shape": [
256,
2048
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 0
},
{
"name": "model.layers.2.self_attn.o_proj.q_scale",
"shape": [
64,
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 262144,
"byteOffset": 2097152
},
{
"name": "model.layers.3.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 2359296
},
{
"name": "model.layers.3.mlp.down_proj.q_weight",
"shape": [
1024,
2048
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 2363392
},
{
"name": "model.layers.3.mlp.down_proj.q_scale",
"shape": [
256,
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 10752000
},
{
"name": "model.layers.3.mlp.gate_up_proj.q_weight",
"shape": [
256,
16384
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 16777216,
"byteOffset": 11800576
},
{
"name": "model.layers.3.mlp.gate_up_proj.q_scale",
"shape": [
64,
16384
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 28577792
},
{
"name": "model.layers.3.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 30674944
}
],
"md5sum": "45633fa3293370ea927c2f3325bc89e4"
},
{
"dataPath": "params_shard_14.bin",
"format": "raw-shard",
"nbytes": 32116736,
"records": [
{
"name": "model.layers.3.self_attn.qkv_proj.q_weight",
"shape": [
256,
3072
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 0
},
{
"name": "model.layers.3.self_attn.qkv_proj.q_scale",
"shape": [
64,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 393216,
"byteOffset": 3145728
},
{
"name": "model.layers.3.self_attn.o_proj.q_weight",
"shape": [
256,
2048
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 3538944
},
{
"name": "model.layers.3.self_attn.o_proj.q_scale",
"shape": [
64,
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 262144,
"byteOffset": 5636096
},
{
"name": "model.layers.4.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 5898240
},
{
"name": "model.layers.4.mlp.down_proj.q_weight",
"shape": [
1024,
2048
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 5902336
},
{
"name": "model.layers.4.mlp.down_proj.q_scale",
"shape": [
256,
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 14290944
},
{
"name": "model.layers.4.mlp.gate_up_proj.q_weight",
"shape": [
256,
16384
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 16777216,
"byteOffset": 15339520
}
],
"md5sum": "a4e255255e567507c39ee69e39a43fe2"
},
{
"dataPath": "params_shard_15.bin",
"format": "raw-shard",
"nbytes": 16777216,
"records": [
{
"name": "model.layers.5.mlp.gate_up_proj.q_weight",
"shape": [
256,
16384
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 16777216,
"byteOffset": 0
}
],
"md5sum": "52aebd37c84f3e4a75120ea665942e1e"
},
{
"dataPath": "params_shard_16.bin",
"format": "raw-shard",
"nbytes": 25444352,
"records": [
{
"name": "model.layers.4.mlp.gate_up_proj.q_scale",
"shape": [
64,
16384
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 0
},
{
"name": "model.layers.4.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 2097152
},
{
"name": "model.layers.4.self_attn.qkv_proj.q_weight",
"shape": [
256,
3072
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 2101248
},
{
"name": "model.layers.4.self_attn.qkv_proj.q_scale",
"shape": [
64,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 393216,
"byteOffset": 5246976
},
{
"name": "model.layers.4.self_attn.o_proj.q_weight",
"shape": [
256,
2048
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 5640192
},
{
"name": "model.layers.4.self_attn.o_proj.q_scale",
"shape": [
64,
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 262144,
"byteOffset": 7737344
},
{
"name": "model.layers.5.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 7999488
},
{
"name": "model.layers.5.mlp.down_proj.q_weight",
"shape": [
1024,
2048
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 8003584
},
{
"name": "model.layers.5.mlp.down_proj.q_scale",
"shape": [
256,
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 16392192
},
{
"name": "model.layers.5.mlp.gate_up_proj.q_scale",
"shape": [
64,
16384
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 17440768
},
{
"name": "model.layers.5.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 19537920
},
{
"name": "model.layers.5.self_attn.qkv_proj.q_weight",
"shape": [
256,
3072
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 19542016
},
{
"name": "model.layers.5.self_attn.qkv_proj.q_scale",
"shape": [
64,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 393216,
"byteOffset": 22687744
},
{
"name": "model.layers.5.self_attn.o_proj.q_weight",
"shape": [
256,
2048
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 23080960
},
{
"name": "model.layers.5.self_attn.o_proj.q_scale",
"shape": [
64,
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 262144,
"byteOffset": 25178112
},
{
"name": "model.layers.6.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 25440256
}
],
"md5sum": "f76b31c3827cf4c3700fd8a3ca3489a3"
},
{
"dataPath": "params_shard_17.bin",
"format": "raw-shard",
"nbytes": 31854592,
"records": [
{
"name": "model.layers.6.mlp.down_proj.q_weight",
"shape": [
1024,
2048
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 0
},
{
"name": "model.layers.6.mlp.down_proj.q_scale",
"shape": [
256,
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 8388608
},
{
"name": "model.layers.6.mlp.gate_up_proj.q_weight",
"shape": [
256,
16384
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 16777216,
"byteOffset": 9437184
},
{
"name": "model.layers.6.mlp.gate_up_proj.q_scale",
"shape": [
64,
16384
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 26214400
},
{
"name": "model.layers.6.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 28311552
},
{
"name": "model.layers.6.self_attn.qkv_proj.q_weight",
"shape": [
256,
3072
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 28315648
},
{
"name": "model.layers.6.self_attn.qkv_proj.q_scale",
"shape": [
64,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 393216,
"byteOffset": 31461376
}
],
"md5sum": "96d3bfa1cf2af7b02ed7e8dc95eb9686"
},
{
"dataPath": "params_shard_18.bin",
"format": "raw-shard",
"nbytes": 30679040,
"records": [
{
"name": "model.layers.6.self_attn.o_proj.q_weight",
"shape": [
256,
2048
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 0
},
{
"name": "model.layers.6.self_attn.o_proj.q_scale",
"shape": [
64,
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 262144,
"byteOffset": 2097152
},
{
"name": "model.layers.7.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 2359296
},
{
"name": "model.layers.7.mlp.down_proj.q_weight",
"shape": [
1024,
2048
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 2363392
},
{
"name": "model.layers.7.mlp.down_proj.q_scale",
"shape": [
256,
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 10752000
},
{
"name": "model.layers.7.mlp.gate_up_proj.q_weight",
"shape": [
256,
16384
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 16777216,
"byteOffset": 11800576
},
{
"name": "model.layers.7.mlp.gate_up_proj.q_scale",
"shape": [
64,
16384
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 28577792
},
{
"name": "model.layers.7.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 30674944
}
],
"md5sum": "2d4e9196ef42530a1142a5f20a83193b"
},
{
"dataPath": "params_shard_19.bin",
"format": "raw-shard",
"nbytes": 32116736,
"records": [
{
"name": "model.layers.7.self_attn.qkv_proj.q_weight",
"shape": [
256,
3072
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 0
},
{
"name": "model.layers.7.self_attn.qkv_proj.q_scale",
"shape": [
64,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 393216,
"byteOffset": 3145728
},
{
"name": "model.layers.7.self_attn.o_proj.q_weight",
"shape": [
256,
2048
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 3538944
},
{
"name": "model.layers.7.self_attn.o_proj.q_scale",
"shape": [
64,
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 262144,
"byteOffset": 5636096
},
{
"name": "model.layers.8.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 5898240
},
{
"name": "model.layers.8.mlp.down_proj.q_weight",
"shape": [
1024,
2048
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 5902336
},
{
"name": "model.layers.8.mlp.down_proj.q_scale",
"shape": [
256,
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 14290944
},
{
"name": "model.layers.8.mlp.gate_up_proj.q_weight",
"shape": [
256,
16384
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 16777216,
"byteOffset": 15339520
}
],
"md5sum": "66411161e16cc9fdff02a7b2aa47e573"
},
{
"dataPath": "params_shard_20.bin",
"format": "raw-shard",
"nbytes": 16777216,
"records": [
{
"name": "model.layers.9.mlp.gate_up_proj.q_weight",
"shape": [
256,
16384
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 16777216,
"byteOffset": 0
}
],
"md5sum": "f321375eb8856057cb398789d3f25ec7"
},
{
"dataPath": "params_shard_21.bin",
"format": "raw-shard",
"nbytes": 25444352,
"records": [
{
"name": "model.layers.8.mlp.gate_up_proj.q_scale",
"shape": [
64,
16384
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 0
},
{
"name": "model.layers.8.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 2097152
},
{
"name": "model.layers.8.self_attn.qkv_proj.q_weight",
"shape": [
256,
3072
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 2101248
},
{
"name": "model.layers.8.self_attn.qkv_proj.q_scale",
"shape": [
64,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 393216,
"byteOffset": 5246976
},
{
"name": "model.layers.8.self_attn.o_proj.q_weight",
"shape": [
256,
2048
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 5640192
},
{
"name": "model.layers.8.self_attn.o_proj.q_scale",
"shape": [
64,
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 262144,
"byteOffset": 7737344
},
{
"name": "model.layers.9.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 7999488
},
{
"name": "model.layers.9.mlp.down_proj.q_weight",
"shape": [
1024,
2048
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 8003584
},
{
"name": "model.layers.9.mlp.down_proj.q_scale",
"shape": [
256,
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 16392192
},
{
"name": "model.layers.9.mlp.gate_up_proj.q_scale",
"shape": [
64,
16384
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 17440768
},
{
"name": "model.layers.9.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 19537920
},
{
"name": "model.layers.9.self_attn.qkv_proj.q_weight",
"shape": [
256,
3072
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 19542016
},
{
"name": "model.layers.9.self_attn.qkv_proj.q_scale",
"shape": [
64,
3072
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 393216,
"byteOffset": 22687744
},
{
"name": "model.layers.9.self_attn.o_proj.q_weight",
"shape": [
256,
2048
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 23080960
},
{
"name": "model.layers.9.self_attn.o_proj.q_scale",
"shape": [
64,
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 262144,
"byteOffset": 25178112
},
{
"name": "model.norm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 25440256
}
],
"md5sum": "814b291d98a7e3bc830fb078970fbaaf"
}
]
}