Qwen-7B-Chat-q4f16_1-MLC / ndarray-cache.json
riczhou's picture
Upload folder using huggingface_hub
f34bc4d verified
{
"metadata": {
"ParamSize": 357,
"ParamBytes": 4344193024.0,
"BitsPerParam": 4.500982181743143
},
"records": [
{
"dataPath": "params_shard_0.bin",
"format": "raw-shard",
"nbytes": 311164928,
"records": [
{
"name": "lm_head.q_weight",
"shape": [
151936,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 311164928,
"byteOffset": 0
}
],
"md5sum": "9feb0be3c4ce0f6c63520b773e2880cb"
},
{
"dataPath": "params_shard_1.bin",
"format": "raw-shard",
"nbytes": 38895616,
"records": [
{
"name": "lm_head.q_scale",
"shape": [
151936,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 38895616,
"byteOffset": 0
}
],
"md5sum": "95f4c2cf4f32017df8f69998b2f1a08d"
},
{
"dataPath": "params_shard_2.bin",
"format": "raw-shard",
"nbytes": 25165824,
"records": [
{
"name": "transformer.h.0.attn.c_attn.q_weight",
"shape": [
12288,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 25165824,
"byteOffset": 0
}
],
"md5sum": "76753573dfb78d41961123771ddc5e11"
},
{
"dataPath": "params_shard_3.bin",
"format": "raw-shard",
"nbytes": 28540928,
"records": [
{
"name": "transformer.h.31.mlp.c_proj.q_weight",
"shape": [
4096,
1376
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 22544384,
"byteOffset": 0
},
{
"name": "transformer.h.31.mlp.c_proj.q_scale",
"shape": [
4096,
344
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2818048,
"byteOffset": 22544384
},
{
"name": "transformer.ln_f.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 25362432
},
{
"name": "transformer.h.0.attn.c_attn.bias",
"shape": [
12288
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 24576,
"byteOffset": 25370624
},
{
"name": "transformer.h.0.attn.c_attn.q_scale",
"shape": [
12288,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 25395200
}
],
"md5sum": "15add630ecd7d3bc8da56bd183395105"
},
{
"dataPath": "params_shard_4.bin",
"format": "raw-shard",
"nbytes": 31997952,
"records": [
{
"name": "transformer.h.0.attn.c_proj.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 0
},
{
"name": "transformer.h.0.attn.c_proj.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 8388608
},
{
"name": "transformer.h.0.ln_1.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 9437184
},
{
"name": "transformer.h.0.ln_2.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 9445376
},
{
"name": "transformer.h.0.mlp.c_proj.q_weight",
"shape": [
4096,
1376
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 22544384,
"byteOffset": 9453568
}
],
"md5sum": "19b3579f7ccbb3dd9fbad428680709ec"
},
{
"dataPath": "params_shard_5.bin",
"format": "raw-shard",
"nbytes": 45088768,
"records": [
{
"name": "transformer.h.0.mlp.gate_up_proj.q_weight",
"shape": [
22016,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 45088768,
"byteOffset": 0
}
],
"md5sum": "180f059e9aed6e4349ef44c00f1515bf"
},
{
"dataPath": "params_shard_6.bin",
"format": "raw-shard",
"nbytes": 25165824,
"records": [
{
"name": "transformer.h.1.attn.c_attn.q_weight",
"shape": [
12288,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 25165824,
"byteOffset": 0
}
],
"md5sum": "f10cec7841d551ffe9e0c494ed0d1fb8"
},
{
"dataPath": "params_shard_7.bin",
"format": "raw-shard",
"nbytes": 45088768,
"records": [
{
"name": "transformer.h.1.mlp.gate_up_proj.q_weight",
"shape": [
22016,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 45088768,
"byteOffset": 0
}
],
"md5sum": "933cd463bb8b51e152d4e6d426185e47"
},
{
"dataPath": "params_shard_8.bin",
"format": "raw-shard",
"nbytes": 311164928,
"records": [
{
"name": "transformer.wte.q_weight",
"shape": [
151936,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 311164928,
"byteOffset": 0
}
],
"md5sum": "e52ee8be3f8762d4131a3995a10ada21"
},
{
"dataPath": "params_shard_9.bin",
"format": "raw-shard",
"nbytes": 38895616,
"records": [
{
"name": "transformer.wte.q_scale",
"shape": [
151936,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 38895616,
"byteOffset": 0
}
],
"md5sum": "95f0ebdf47e08e8f02930106a18a0ef2"
},
{
"dataPath": "params_shard_10.bin",
"format": "raw-shard",
"nbytes": 22544384,
"records": [
{
"name": "transformer.h.1.mlp.c_proj.q_weight",
"shape": [
4096,
1376
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 22544384,
"byteOffset": 0
}
],
"md5sum": "7b29781e99906d3233aea77d92d0bbd4"
},
{
"dataPath": "params_shard_11.bin",
"format": "raw-shard",
"nbytes": 25165824,
"records": [
{
"name": "transformer.h.2.attn.c_attn.q_weight",
"shape": [
12288,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 25165824,
"byteOffset": 0
}
],
"md5sum": "78fd6464c901a6f878acbf02d1f13c53"
},
{
"dataPath": "params_shard_12.bin",
"format": "raw-shard",
"nbytes": 32702464,
"records": [
{
"name": "transformer.h.0.mlp.c_proj.q_scale",
"shape": [
4096,
344
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2818048,
"byteOffset": 0
},
{
"name": "transformer.h.0.mlp.gate_up_proj.q_scale",
"shape": [
22016,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5636096,
"byteOffset": 2818048
},
{
"name": "transformer.h.1.attn.c_attn.bias",
"shape": [
12288
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 24576,
"byteOffset": 8454144
},
{
"name": "transformer.h.1.attn.c_attn.q_scale",
"shape": [
12288,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 8478720
},
{
"name": "transformer.h.1.attn.c_proj.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 11624448
},
{
"name": "transformer.h.1.attn.c_proj.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 20013056
},
{
"name": "transformer.h.1.ln_1.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 21061632
},
{
"name": "transformer.h.1.ln_2.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 21069824
},
{
"name": "transformer.h.1.mlp.gate_up_proj.q_scale",
"shape": [
22016,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5636096,
"byteOffset": 21078016
},
{
"name": "transformer.h.1.mlp.c_proj.q_scale",
"shape": [
4096,
344
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2818048,
"byteOffset": 26714112
},
{
"name": "transformer.h.2.attn.c_attn.bias",
"shape": [
12288
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 24576,
"byteOffset": 29532160
},
{
"name": "transformer.h.2.attn.c_attn.q_scale",
"shape": [
12288,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 29556736
}
],
"md5sum": "4d7d84a367ec0a79d3e05fbe004b2a1a"
},
{
"dataPath": "params_shard_13.bin",
"format": "raw-shard",
"nbytes": 31997952,
"records": [
{
"name": "transformer.h.2.attn.c_proj.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 0
},
{
"name": "transformer.h.2.attn.c_proj.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 8388608
},
{
"name": "transformer.h.2.ln_1.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 9437184
},
{
"name": "transformer.h.2.ln_2.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 9445376
},
{
"name": "transformer.h.2.mlp.c_proj.q_weight",
"shape": [
4096,
1376
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 22544384,
"byteOffset": 9453568
}
],
"md5sum": "e9b0be162965c911377bd8322624174d"
},
{
"dataPath": "params_shard_14.bin",
"format": "raw-shard",
"nbytes": 45088768,
"records": [
{
"name": "transformer.h.2.mlp.gate_up_proj.q_weight",
"shape": [
22016,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 45088768,
"byteOffset": 0
}
],
"md5sum": "081d012c74b0ae170353714658c1ed02"
},
{
"dataPath": "params_shard_15.bin",
"format": "raw-shard",
"nbytes": 25165824,
"records": [
{
"name": "transformer.h.3.attn.c_attn.q_weight",
"shape": [
12288,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 25165824,
"byteOffset": 0
}
],
"md5sum": "ee49ceaffd050d8472f7d92cbbee0f3a"
},
{
"dataPath": "params_shard_16.bin",
"format": "raw-shard",
"nbytes": 22544384,
"records": [
{
"name": "transformer.h.3.mlp.c_proj.q_weight",
"shape": [
4096,
1376
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 22544384,
"byteOffset": 0
}
],
"md5sum": "2344cfcb26a81ece8ca457544df60c7f"
},
{
"dataPath": "params_shard_17.bin",
"format": "raw-shard",
"nbytes": 45088768,
"records": [
{
"name": "transformer.h.3.mlp.gate_up_proj.q_weight",
"shape": [
22016,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 45088768,
"byteOffset": 0
}
],
"md5sum": "f85a11ae1a581b106bb15ea4247fbd47"
},
{
"dataPath": "params_shard_18.bin",
"format": "raw-shard",
"nbytes": 25165824,
"records": [
{
"name": "transformer.h.4.attn.c_attn.q_weight",
"shape": [
12288,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 25165824,
"byteOffset": 0
}
],
"md5sum": "dcfbf68da7f4a53c04e951c789487fe6"
},
{
"dataPath": "params_shard_19.bin",
"format": "raw-shard",
"nbytes": 32702464,
"records": [
{
"name": "transformer.h.2.mlp.c_proj.q_scale",
"shape": [
4096,
344
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2818048,
"byteOffset": 0
},
{
"name": "transformer.h.2.mlp.gate_up_proj.q_scale",
"shape": [
22016,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5636096,
"byteOffset": 2818048
},
{
"name": "transformer.h.3.attn.c_attn.bias",
"shape": [
12288
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 24576,
"byteOffset": 8454144
},
{
"name": "transformer.h.3.attn.c_attn.q_scale",
"shape": [
12288,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 8478720
},
{
"name": "transformer.h.3.attn.c_proj.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 11624448
},
{
"name": "transformer.h.3.attn.c_proj.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 20013056
},
{
"name": "transformer.h.3.ln_1.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 21061632
},
{
"name": "transformer.h.3.ln_2.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 21069824
},
{
"name": "transformer.h.3.mlp.c_proj.q_scale",
"shape": [
4096,
344
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2818048,
"byteOffset": 21078016
},
{
"name": "transformer.h.3.mlp.gate_up_proj.q_scale",
"shape": [
22016,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5636096,
"byteOffset": 23896064
},
{
"name": "transformer.h.4.attn.c_attn.bias",
"shape": [
12288
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 24576,
"byteOffset": 29532160
},
{
"name": "transformer.h.4.attn.c_attn.q_scale",
"shape": [
12288,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 29556736
}
],
"md5sum": "4f080ab1d2312df41c52082309614d97"
},
{
"dataPath": "params_shard_20.bin",
"format": "raw-shard",
"nbytes": 31997952,
"records": [
{
"name": "transformer.h.4.attn.c_proj.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 0
},
{
"name": "transformer.h.4.attn.c_proj.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 8388608
},
{
"name": "transformer.h.4.ln_1.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 9437184
},
{
"name": "transformer.h.4.ln_2.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 9445376
},
{
"name": "transformer.h.4.mlp.c_proj.q_weight",
"shape": [
4096,
1376
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 22544384,
"byteOffset": 9453568
}
],
"md5sum": "def81502b3377a688d5cf49c2082e052"
},
{
"dataPath": "params_shard_21.bin",
"format": "raw-shard",
"nbytes": 45088768,
"records": [
{
"name": "transformer.h.4.mlp.gate_up_proj.q_weight",
"shape": [
22016,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 45088768,
"byteOffset": 0
}
],
"md5sum": "0308ee7c5fe14ae8016dba611f6c857a"
},
{
"dataPath": "params_shard_22.bin",
"format": "raw-shard",
"nbytes": 25165824,
"records": [
{
"name": "transformer.h.5.attn.c_attn.q_weight",
"shape": [
12288,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 25165824,
"byteOffset": 0
}
],
"md5sum": "e8da9954cb30ad55df9d72ea93ead922"
},
{
"dataPath": "params_shard_23.bin",
"format": "raw-shard",
"nbytes": 22544384,
"records": [
{
"name": "transformer.h.5.mlp.c_proj.q_weight",
"shape": [
4096,
1376
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 22544384,
"byteOffset": 0
}
],
"md5sum": "825c7c8970b29d7a57de88c168e3d520"
},
{
"dataPath": "params_shard_24.bin",
"format": "raw-shard",
"nbytes": 45088768,
"records": [
{
"name": "transformer.h.5.mlp.gate_up_proj.q_weight",
"shape": [
22016,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 45088768,
"byteOffset": 0
}
],
"md5sum": "7d47f442587ef3b02ce862e80b4b8779"
},
{
"dataPath": "params_shard_25.bin",
"format": "raw-shard",
"nbytes": 25165824,
"records": [
{
"name": "transformer.h.6.attn.c_attn.q_weight",
"shape": [
12288,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 25165824,
"byteOffset": 0
}
],
"md5sum": "88f9d941a873aa5ad820707c9956f7bd"
},
{
"dataPath": "params_shard_26.bin",
"format": "raw-shard",
"nbytes": 32702464,
"records": [
{
"name": "transformer.h.4.mlp.c_proj.q_scale",
"shape": [
4096,
344
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2818048,
"byteOffset": 0
},
{
"name": "transformer.h.4.mlp.gate_up_proj.q_scale",
"shape": [
22016,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5636096,
"byteOffset": 2818048
},
{
"name": "transformer.h.5.attn.c_attn.bias",
"shape": [
12288
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 24576,
"byteOffset": 8454144
},
{
"name": "transformer.h.5.attn.c_attn.q_scale",
"shape": [
12288,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 8478720
},
{
"name": "transformer.h.5.attn.c_proj.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 11624448
},
{
"name": "transformer.h.5.attn.c_proj.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 20013056
},
{
"name": "transformer.h.5.ln_1.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 21061632
},
{
"name": "transformer.h.5.ln_2.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 21069824
},
{
"name": "transformer.h.5.mlp.c_proj.q_scale",
"shape": [
4096,
344
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2818048,
"byteOffset": 21078016
},
{
"name": "transformer.h.5.mlp.gate_up_proj.q_scale",
"shape": [
22016,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5636096,
"byteOffset": 23896064
},
{
"name": "transformer.h.6.attn.c_attn.bias",
"shape": [
12288
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 24576,
"byteOffset": 29532160
},
{
"name": "transformer.h.6.attn.c_attn.q_scale",
"shape": [
12288,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 29556736
}
],
"md5sum": "7533175ce68af4a963ca00c5167f6921"
},
{
"dataPath": "params_shard_27.bin",
"format": "raw-shard",
"nbytes": 45088768,
"records": [
{
"name": "transformer.h.6.mlp.gate_up_proj.q_weight",
"shape": [
22016,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 45088768,
"byteOffset": 0
}
],
"md5sum": "6ab9dbd40321a800e28223e3ca1957c3"
},
{
"dataPath": "params_shard_28.bin",
"format": "raw-shard",
"nbytes": 25165824,
"records": [
{
"name": "transformer.h.10.attn.c_attn.q_weight",
"shape": [
12288,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 25165824,
"byteOffset": 0
}
],
"md5sum": "5c6db236a979f102f39e2514df7d721b"
},
{
"dataPath": "params_shard_29.bin",
"format": "raw-shard",
"nbytes": 22544384,
"records": [
{
"name": "transformer.h.10.mlp.c_proj.q_weight",
"shape": [
4096,
1376
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 22544384,
"byteOffset": 0
}
],
"md5sum": "a05a94ab72bada46fb2b2abf16a13e9b"
},
{
"dataPath": "params_shard_30.bin",
"format": "raw-shard",
"nbytes": 45088768,
"records": [
{
"name": "transformer.h.10.mlp.gate_up_proj.q_weight",
"shape": [
22016,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 45088768,
"byteOffset": 0
}
],
"md5sum": "f4fce380eefff61ab112c0cbe0616bec"
},
{
"dataPath": "params_shard_31.bin",
"format": "raw-shard",
"nbytes": 30531584,
"records": [
{
"name": "transformer.h.6.attn.c_proj.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 0
},
{
"name": "transformer.h.6.attn.c_proj.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 8388608
},
{
"name": "transformer.h.6.ln_1.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 9437184
},
{
"name": "transformer.h.6.ln_2.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 9445376
},
{
"name": "transformer.h.6.mlp.gate_up_proj.q_scale",
"shape": [
22016,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5636096,
"byteOffset": 9453568
},
{
"name": "transformer.h.10.attn.c_attn.bias",
"shape": [
12288
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 24576,
"byteOffset": 15089664
},
{
"name": "transformer.h.10.attn.c_attn.q_scale",
"shape": [
12288,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 15114240
},
{
"name": "transformer.h.10.attn.c_proj.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 18259968
},
{
"name": "transformer.h.10.attn.c_proj.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 26648576
},
{
"name": "transformer.h.10.ln_1.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 27697152
},
{
"name": "transformer.h.10.ln_2.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 27705344
},
{
"name": "transformer.h.10.mlp.c_proj.q_scale",
"shape": [
4096,
344
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2818048,
"byteOffset": 27713536
}
],
"md5sum": "2a1b9fc07029d2ca0d8179e29950666a"
},
{
"dataPath": "params_shard_32.bin",
"format": "raw-shard",
"nbytes": 30826496,
"records": [
{
"name": "transformer.h.10.mlp.gate_up_proj.q_scale",
"shape": [
22016,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5636096,
"byteOffset": 0
},
{
"name": "transformer.h.11.attn.c_attn.bias",
"shape": [
12288
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 24576,
"byteOffset": 5636096
},
{
"name": "transformer.h.11.attn.c_attn.q_weight",
"shape": [
12288,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 25165824,
"byteOffset": 5660672
}
],
"md5sum": "00a6de460491a3cd1f79f1acabe973fb"
},
{
"dataPath": "params_shard_33.bin",
"format": "raw-shard",
"nbytes": 45088768,
"records": [
{
"name": "transformer.h.11.mlp.gate_up_proj.q_weight",
"shape": [
22016,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 45088768,
"byteOffset": 0
}
],
"md5sum": "0a1048e5538601078888071fdef69cb2"
},
{
"dataPath": "params_shard_34.bin",
"format": "raw-shard",
"nbytes": 22544384,
"records": [
{
"name": "transformer.h.6.mlp.c_proj.q_weight",
"shape": [
4096,
1376
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 22544384,
"byteOffset": 0
}
],
"md5sum": "36469b8888ebc31d3d32ec7e7ffdca9a"
},
{
"dataPath": "params_shard_35.bin",
"format": "raw-shard",
"nbytes": 25165824,
"records": [
{
"name": "transformer.h.7.attn.c_attn.q_weight",
"shape": [
12288,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 25165824,
"byteOffset": 0
}
],
"md5sum": "a82969f7a4bd8389a2aa15432a910813"
},
{
"dataPath": "params_shard_36.bin",
"format": "raw-shard",
"nbytes": 32612352,
"records": [
{
"name": "transformer.h.11.attn.c_attn.q_scale",
"shape": [
12288,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 0
},
{
"name": "transformer.h.11.attn.c_proj.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 3145728
},
{
"name": "transformer.h.11.attn.c_proj.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 11534336
},
{
"name": "transformer.h.11.ln_1.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 12582912
},
{
"name": "transformer.h.11.ln_2.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 12591104
},
{
"name": "transformer.h.11.mlp.gate_up_proj.q_scale",
"shape": [
22016,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5636096,
"byteOffset": 12599296
},
{
"name": "transformer.h.6.mlp.c_proj.q_scale",
"shape": [
4096,
344
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2818048,
"byteOffset": 18235392
},
{
"name": "transformer.h.7.attn.c_attn.bias",
"shape": [
12288
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 24576,
"byteOffset": 21053440
},
{
"name": "transformer.h.7.attn.c_attn.q_scale",
"shape": [
12288,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 21078016
},
{
"name": "transformer.h.7.attn.c_proj.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 24223744
}
],
"md5sum": "f84374ca9bd8fea5703f31e1b193bffe"
},
{
"dataPath": "params_shard_37.bin",
"format": "raw-shard",
"nbytes": 45088768,
"records": [
{
"name": "transformer.h.7.mlp.gate_up_proj.q_weight",
"shape": [
22016,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 45088768,
"byteOffset": 0
}
],
"md5sum": "806131693cfbb42e355fe906138eea2b"
},
{
"dataPath": "params_shard_38.bin",
"format": "raw-shard",
"nbytes": 25165824,
"records": [
{
"name": "transformer.h.8.attn.c_attn.q_weight",
"shape": [
12288,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 25165824,
"byteOffset": 0
}
],
"md5sum": "78f57993d32290cdea75768142d48a32"
},
{
"dataPath": "params_shard_39.bin",
"format": "raw-shard",
"nbytes": 32088064,
"records": [
{
"name": "transformer.h.7.attn.c_proj.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 0
},
{
"name": "transformer.h.7.ln_1.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 1048576
},
{
"name": "transformer.h.7.ln_2.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 1056768
},
{
"name": "transformer.h.7.mlp.c_proj.q_weight",
"shape": [
4096,
1376
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 22544384,
"byteOffset": 1064960
},
{
"name": "transformer.h.7.mlp.c_proj.q_scale",
"shape": [
4096,
344
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2818048,
"byteOffset": 23609344
},
{
"name": "transformer.h.7.mlp.gate_up_proj.q_scale",
"shape": [
22016,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5636096,
"byteOffset": 26427392
},
{
"name": "transformer.h.8.attn.c_attn.bias",
"shape": [
12288
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 24576,
"byteOffset": 32063488
}
],
"md5sum": "c99eb07b9d6a5c55d26ab0ead6fb4e36"
},
{
"dataPath": "params_shard_40.bin",
"format": "raw-shard",
"nbytes": 22544384,
"records": [
{
"name": "transformer.h.8.mlp.c_proj.q_weight",
"shape": [
4096,
1376
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 22544384,
"byteOffset": 0
}
],
"md5sum": "a1d6115e862f671069aee6b04e2698f7"
},
{
"dataPath": "params_shard_41.bin",
"format": "raw-shard",
"nbytes": 45088768,
"records": [
{
"name": "transformer.h.8.mlp.gate_up_proj.q_weight",
"shape": [
22016,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 45088768,
"byteOffset": 0
}
],
"md5sum": "74d589b08f43670add8522fd4332f8bf"
},
{
"dataPath": "params_shard_42.bin",
"format": "raw-shard",
"nbytes": 25165824,
"records": [
{
"name": "transformer.h.9.attn.c_attn.q_weight",
"shape": [
12288,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 25165824,
"byteOffset": 0
}
],
"md5sum": "14c24a1d99940f8195c137326c6b91da"
},
{
"dataPath": "params_shard_43.bin",
"format": "raw-shard",
"nbytes": 32612352,
"records": [
{
"name": "transformer.h.8.attn.c_attn.q_scale",
"shape": [
12288,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 0
},
{
"name": "transformer.h.8.attn.c_proj.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 3145728
},
{
"name": "transformer.h.8.attn.c_proj.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 11534336
},
{
"name": "transformer.h.8.ln_1.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 12582912
},
{
"name": "transformer.h.8.ln_2.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 12591104
},
{
"name": "transformer.h.8.mlp.c_proj.q_scale",
"shape": [
4096,
344
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2818048,
"byteOffset": 12599296
},
{
"name": "transformer.h.8.mlp.gate_up_proj.q_scale",
"shape": [
22016,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5636096,
"byteOffset": 15417344
},
{
"name": "transformer.h.9.attn.c_attn.bias",
"shape": [
12288
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 24576,
"byteOffset": 21053440
},
{
"name": "transformer.h.9.attn.c_attn.q_scale",
"shape": [
12288,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 21078016
},
{
"name": "transformer.h.9.attn.c_proj.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 24223744
}
],
"md5sum": "0d5d8e4fc651a87a676df71b14873cb9"
},
{
"dataPath": "params_shard_44.bin",
"format": "raw-shard",
"nbytes": 45088768,
"records": [
{
"name": "transformer.h.9.mlp.gate_up_proj.q_weight",
"shape": [
22016,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 45088768,
"byteOffset": 0
}
],
"md5sum": "1e96c1cd93fb7770dea6a2e75c00623f"
},
{
"dataPath": "params_shard_45.bin",
"format": "raw-shard",
"nbytes": 22544384,
"records": [
{
"name": "transformer.h.11.mlp.c_proj.q_weight",
"shape": [
4096,
1376
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 22544384,
"byteOffset": 0
}
],
"md5sum": "7661a76dface8d4ec1b4e6831b569b9d"
},
{
"dataPath": "params_shard_46.bin",
"format": "raw-shard",
"nbytes": 32063488,
"records": [
{
"name": "transformer.h.9.attn.c_proj.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 0
},
{
"name": "transformer.h.9.ln_1.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 1048576
},
{
"name": "transformer.h.9.ln_2.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 1056768
},
{
"name": "transformer.h.9.mlp.c_proj.q_weight",
"shape": [
4096,
1376
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 22544384,
"byteOffset": 1064960
},
{
"name": "transformer.h.9.mlp.c_proj.q_scale",
"shape": [
4096,
344
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2818048,
"byteOffset": 23609344
},
{
"name": "transformer.h.9.mlp.gate_up_proj.q_scale",
"shape": [
22016,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5636096,
"byteOffset": 26427392
}
],
"md5sum": "74dbe7586364cb3347249758a0b11de8"
},
{
"dataPath": "params_shard_47.bin",
"format": "raw-shard",
"nbytes": 31154176,
"records": [
{
"name": "transformer.h.11.mlp.c_proj.q_scale",
"shape": [
4096,
344
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2818048,
"byteOffset": 0
},
{
"name": "transformer.h.12.attn.c_attn.bias",
"shape": [
12288
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 24576,
"byteOffset": 2818048
},
{
"name": "transformer.h.12.attn.c_attn.q_weight",
"shape": [
12288,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 25165824,
"byteOffset": 2842624
},
{
"name": "transformer.h.12.attn.c_attn.q_scale",
"shape": [
12288,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 28008448
}
],
"md5sum": "a2f69731aac9b1a4067a81b72e5e123e"
},
{
"dataPath": "params_shard_48.bin",
"format": "raw-shard",
"nbytes": 31997952,
"records": [
{
"name": "transformer.h.12.attn.c_proj.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 0
},
{
"name": "transformer.h.12.attn.c_proj.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 8388608
},
{
"name": "transformer.h.12.ln_1.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 9437184
},
{
"name": "transformer.h.12.ln_2.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 9445376
},
{
"name": "transformer.h.12.mlp.c_proj.q_weight",
"shape": [
4096,
1376
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 22544384,
"byteOffset": 9453568
}
],
"md5sum": "8ddc10ced476bd273f5dd62d9bbc037d"
},
{
"dataPath": "params_shard_49.bin",
"format": "raw-shard",
"nbytes": 45088768,
"records": [
{
"name": "transformer.h.12.mlp.gate_up_proj.q_weight",
"shape": [
22016,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 45088768,
"byteOffset": 0
}
],
"md5sum": "b813b120b947cb45746c9c304d45c270"
},
{
"dataPath": "params_shard_50.bin",
"format": "raw-shard",
"nbytes": 25165824,
"records": [
{
"name": "transformer.h.13.attn.c_attn.q_weight",
"shape": [
12288,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 25165824,
"byteOffset": 0
}
],
"md5sum": "ff392702c2124f1215359ede9780edfd"
},
{
"dataPath": "params_shard_51.bin",
"format": "raw-shard",
"nbytes": 22544384,
"records": [
{
"name": "transformer.h.13.mlp.c_proj.q_weight",
"shape": [
4096,
1376
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 22544384,
"byteOffset": 0
}
],
"md5sum": "8a56b844216a274a2a19865b3312a7b8"
},
{
"dataPath": "params_shard_52.bin",
"format": "raw-shard",
"nbytes": 45088768,
"records": [
{
"name": "transformer.h.13.mlp.gate_up_proj.q_weight",
"shape": [
22016,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 45088768,
"byteOffset": 0
}
],
"md5sum": "125fdf3a1baa0215779c8cb6e91f1aaf"
},
{
"dataPath": "params_shard_53.bin",
"format": "raw-shard",
"nbytes": 25165824,
"records": [
{
"name": "transformer.h.14.attn.c_attn.q_weight",
"shape": [
12288,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 25165824,
"byteOffset": 0
}
],
"md5sum": "8ba2ab7462405bad8e06b2e41e744779"
},
{
"dataPath": "params_shard_54.bin",
"format": "raw-shard",
"nbytes": 32702464,
"records": [
{
"name": "transformer.h.12.mlp.c_proj.q_scale",
"shape": [
4096,
344
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2818048,
"byteOffset": 0
},
{
"name": "transformer.h.12.mlp.gate_up_proj.q_scale",
"shape": [
22016,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5636096,
"byteOffset": 2818048
},
{
"name": "transformer.h.13.attn.c_attn.bias",
"shape": [
12288
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 24576,
"byteOffset": 8454144
},
{
"name": "transformer.h.13.attn.c_attn.q_scale",
"shape": [
12288,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 8478720
},
{
"name": "transformer.h.13.attn.c_proj.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 11624448
},
{
"name": "transformer.h.13.attn.c_proj.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 20013056
},
{
"name": "transformer.h.13.ln_1.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 21061632
},
{
"name": "transformer.h.13.ln_2.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 21069824
},
{
"name": "transformer.h.13.mlp.c_proj.q_scale",
"shape": [
4096,
344
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2818048,
"byteOffset": 21078016
},
{
"name": "transformer.h.13.mlp.gate_up_proj.q_scale",
"shape": [
22016,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5636096,
"byteOffset": 23896064
},
{
"name": "transformer.h.14.attn.c_attn.bias",
"shape": [
12288
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 24576,
"byteOffset": 29532160
},
{
"name": "transformer.h.14.attn.c_attn.q_scale",
"shape": [
12288,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 29556736
}
],
"md5sum": "f4b4d81de805dcf75cdfb577d1726418"
},
{
"dataPath": "params_shard_55.bin",
"format": "raw-shard",
"nbytes": 31997952,
"records": [
{
"name": "transformer.h.14.attn.c_proj.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 0
},
{
"name": "transformer.h.14.attn.c_proj.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 8388608
},
{
"name": "transformer.h.14.ln_1.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 9437184
},
{
"name": "transformer.h.14.ln_2.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 9445376
},
{
"name": "transformer.h.14.mlp.c_proj.q_weight",
"shape": [
4096,
1376
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 22544384,
"byteOffset": 9453568
}
],
"md5sum": "0484a49b41a167bdc1a3d7931585bd71"
},
{
"dataPath": "params_shard_56.bin",
"format": "raw-shard",
"nbytes": 45088768,
"records": [
{
"name": "transformer.h.14.mlp.gate_up_proj.q_weight",
"shape": [
22016,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 45088768,
"byteOffset": 0
}
],
"md5sum": "1747999530dc556c8442839f7b58d31b"
},
{
"dataPath": "params_shard_57.bin",
"format": "raw-shard",
"nbytes": 25165824,
"records": [
{
"name": "transformer.h.15.attn.c_attn.q_weight",
"shape": [
12288,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 25165824,
"byteOffset": 0
}
],
"md5sum": "bde2abc11a57fa7ac427eedab52377b7"
},
{
"dataPath": "params_shard_58.bin",
"format": "raw-shard",
"nbytes": 22544384,
"records": [
{
"name": "transformer.h.15.mlp.c_proj.q_weight",
"shape": [
4096,
1376
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 22544384,
"byteOffset": 0
}
],
"md5sum": "d4a1f192cb644db8f1f70e978b41c818"
},
{
"dataPath": "params_shard_59.bin",
"format": "raw-shard",
"nbytes": 45088768,
"records": [
{
"name": "transformer.h.15.mlp.gate_up_proj.q_weight",
"shape": [
22016,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 45088768,
"byteOffset": 0
}
],
"md5sum": "b6d59fa2a3fa8cd8ce06e9ebe76fb4b1"
},
{
"dataPath": "params_shard_60.bin",
"format": "raw-shard",
"nbytes": 25165824,
"records": [
{
"name": "transformer.h.16.attn.c_attn.q_weight",
"shape": [
12288,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 25165824,
"byteOffset": 0
}
],
"md5sum": "8cd9afc1f6cabff45b6591f57a9ac379"
},
{
"dataPath": "params_shard_61.bin",
"format": "raw-shard",
"nbytes": 32702464,
"records": [
{
"name": "transformer.h.14.mlp.c_proj.q_scale",
"shape": [
4096,
344
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2818048,
"byteOffset": 0
},
{
"name": "transformer.h.14.mlp.gate_up_proj.q_scale",
"shape": [
22016,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5636096,
"byteOffset": 2818048
},
{
"name": "transformer.h.15.attn.c_attn.bias",
"shape": [
12288
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 24576,
"byteOffset": 8454144
},
{
"name": "transformer.h.15.attn.c_attn.q_scale",
"shape": [
12288,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 8478720
},
{
"name": "transformer.h.15.attn.c_proj.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 11624448
},
{
"name": "transformer.h.15.attn.c_proj.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 20013056
},
{
"name": "transformer.h.15.ln_1.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 21061632
},
{
"name": "transformer.h.15.ln_2.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 21069824
},
{
"name": "transformer.h.15.mlp.c_proj.q_scale",
"shape": [
4096,
344
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2818048,
"byteOffset": 21078016
},
{
"name": "transformer.h.15.mlp.gate_up_proj.q_scale",
"shape": [
22016,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5636096,
"byteOffset": 23896064
},
{
"name": "transformer.h.16.attn.c_attn.bias",
"shape": [
12288
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 24576,
"byteOffset": 29532160
},
{
"name": "transformer.h.16.attn.c_attn.q_scale",
"shape": [
12288,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 29556736
}
],
"md5sum": "ebbb74adc6dfa6025c1cd1a5dd43a974"
},
{
"dataPath": "params_shard_62.bin",
"format": "raw-shard",
"nbytes": 45088768,
"records": [
{
"name": "transformer.h.16.mlp.gate_up_proj.q_weight",
"shape": [
22016,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 45088768,
"byteOffset": 0
}
],
"md5sum": "1410d459722d942c750ad61b97cf69c2"
},
{
"dataPath": "params_shard_63.bin",
"format": "raw-shard",
"nbytes": 22544384,
"records": [
{
"name": "transformer.h.16.mlp.c_proj.q_weight",
"shape": [
4096,
1376
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 22544384,
"byteOffset": 0
}
],
"md5sum": "463076745e658864b439f20491e97d41"
},
{
"dataPath": "params_shard_64.bin",
"format": "raw-shard",
"nbytes": 25165824,
"records": [
{
"name": "transformer.h.17.attn.c_attn.q_weight",
"shape": [
12288,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 25165824,
"byteOffset": 0
}
],
"md5sum": "96d6f19a989f2855908be7f804a7b4aa"
},
{
"dataPath": "params_shard_65.bin",
"format": "raw-shard",
"nbytes": 22544384,
"records": [
{
"name": "transformer.h.17.mlp.c_proj.q_weight",
"shape": [
4096,
1376
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 22544384,
"byteOffset": 0
}
],
"md5sum": "aa4ed6ca5e5ff1546c8fb104405e3b51"
},
{
"dataPath": "params_shard_66.bin",
"format": "raw-shard",
"nbytes": 45088768,
"records": [
{
"name": "transformer.h.17.mlp.gate_up_proj.q_weight",
"shape": [
22016,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 45088768,
"byteOffset": 0
}
],
"md5sum": "1225f102a160522b682c3038ed2689b1"
},
{
"dataPath": "params_shard_67.bin",
"format": "raw-shard",
"nbytes": 33349632,
"records": [
{
"name": "transformer.h.16.attn.c_proj.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 0
},
{
"name": "transformer.h.16.attn.c_proj.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 8388608
},
{
"name": "transformer.h.16.ln_1.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 9437184
},
{
"name": "transformer.h.16.ln_2.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 9445376
},
{
"name": "transformer.h.16.mlp.gate_up_proj.q_scale",
"shape": [
22016,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5636096,
"byteOffset": 9453568
},
{
"name": "transformer.h.16.mlp.c_proj.q_scale",
"shape": [
4096,
344
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2818048,
"byteOffset": 15089664
},
{
"name": "transformer.h.17.attn.c_attn.bias",
"shape": [
12288
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 24576,
"byteOffset": 17907712
},
{
"name": "transformer.h.17.attn.c_attn.q_scale",
"shape": [
12288,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 17932288
},
{
"name": "transformer.h.17.attn.c_proj.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 21078016
},
{
"name": "transformer.h.17.attn.c_proj.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 29466624
},
{
"name": "transformer.h.17.ln_1.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 30515200
},
{
"name": "transformer.h.17.ln_2.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 30523392
},
{
"name": "transformer.h.17.mlp.c_proj.q_scale",
"shape": [
4096,
344
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2818048,
"byteOffset": 30531584
}
],
"md5sum": "38951fb890de56064c4902a2715f3cb1"
},
{
"dataPath": "params_shard_68.bin",
"format": "raw-shard",
"nbytes": 30826496,
"records": [
{
"name": "transformer.h.17.mlp.gate_up_proj.q_scale",
"shape": [
22016,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5636096,
"byteOffset": 0
},
{
"name": "transformer.h.18.attn.c_attn.bias",
"shape": [
12288
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 24576,
"byteOffset": 5636096
},
{
"name": "transformer.h.18.attn.c_attn.q_weight",
"shape": [
12288,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 25165824,
"byteOffset": 5660672
}
],
"md5sum": "5bea56413882a31277e3d7b9af241f01"
},
{
"dataPath": "params_shard_69.bin",
"format": "raw-shard",
"nbytes": 22544384,
"records": [
{
"name": "transformer.h.18.mlp.c_proj.q_weight",
"shape": [
4096,
1376
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 22544384,
"byteOffset": 0
}
],
"md5sum": "51d95de743aee252fbe1be01177e2617"
},
{
"dataPath": "params_shard_70.bin",
"format": "raw-shard",
"nbytes": 45088768,
"records": [
{
"name": "transformer.h.18.mlp.gate_up_proj.q_weight",
"shape": [
22016,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 45088768,
"byteOffset": 0
}
],
"md5sum": "995d5949a0be50b38d92726885a0a577"
},
{
"dataPath": "params_shard_71.bin",
"format": "raw-shard",
"nbytes": 25165824,
"records": [
{
"name": "transformer.h.19.attn.c_attn.q_weight",
"shape": [
12288,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 25165824,
"byteOffset": 0
}
],
"md5sum": "19330d85524e0eb22c6d12cb0caf9f01"
},
{
"dataPath": "params_shard_72.bin",
"format": "raw-shard",
"nbytes": 32612352,
"records": [
{
"name": "transformer.h.18.attn.c_attn.q_scale",
"shape": [
12288,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 0
},
{
"name": "transformer.h.18.attn.c_proj.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 3145728
},
{
"name": "transformer.h.18.attn.c_proj.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 11534336
},
{
"name": "transformer.h.18.ln_1.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 12582912
},
{
"name": "transformer.h.18.ln_2.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 12591104
},
{
"name": "transformer.h.18.mlp.c_proj.q_scale",
"shape": [
4096,
344
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2818048,
"byteOffset": 12599296
},
{
"name": "transformer.h.18.mlp.gate_up_proj.q_scale",
"shape": [
22016,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5636096,
"byteOffset": 15417344
},
{
"name": "transformer.h.19.attn.c_attn.bias",
"shape": [
12288
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 24576,
"byteOffset": 21053440
},
{
"name": "transformer.h.19.attn.c_attn.q_scale",
"shape": [
12288,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 21078016
},
{
"name": "transformer.h.19.attn.c_proj.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 24223744
}
],
"md5sum": "b9b71a44a601fd37bfc8acac769db86e"
},
{
"dataPath": "params_shard_73.bin",
"format": "raw-shard",
"nbytes": 45088768,
"records": [
{
"name": "transformer.h.19.mlp.gate_up_proj.q_weight",
"shape": [
22016,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 45088768,
"byteOffset": 0
}
],
"md5sum": "b740c7b8b6cd422031d0cb7b294578e5"
},
{
"dataPath": "params_shard_74.bin",
"format": "raw-shard",
"nbytes": 25165824,
"records": [
{
"name": "transformer.h.20.attn.c_attn.q_weight",
"shape": [
12288,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 25165824,
"byteOffset": 0
}
],
"md5sum": "78ec04ae45e77245d12c996118a72ea2"
},
{
"dataPath": "params_shard_75.bin",
"format": "raw-shard",
"nbytes": 32088064,
"records": [
{
"name": "transformer.h.19.attn.c_proj.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 0
},
{
"name": "transformer.h.19.ln_1.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 1048576
},
{
"name": "transformer.h.19.ln_2.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 1056768
},
{
"name": "transformer.h.19.mlp.c_proj.q_weight",
"shape": [
4096,
1376
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 22544384,
"byteOffset": 1064960
},
{
"name": "transformer.h.19.mlp.c_proj.q_scale",
"shape": [
4096,
344
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2818048,
"byteOffset": 23609344
},
{
"name": "transformer.h.19.mlp.gate_up_proj.q_scale",
"shape": [
22016,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5636096,
"byteOffset": 26427392
},
{
"name": "transformer.h.20.attn.c_attn.bias",
"shape": [
12288
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 24576,
"byteOffset": 32063488
}
],
"md5sum": "e562557b2559b8deb6427a1d3e28878c"
},
{
"dataPath": "params_shard_76.bin",
"format": "raw-shard",
"nbytes": 22544384,
"records": [
{
"name": "transformer.h.20.mlp.c_proj.q_weight",
"shape": [
4096,
1376
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 22544384,
"byteOffset": 0
}
],
"md5sum": "e1e253e60c07986cb98009b36020b608"
},
{
"dataPath": "params_shard_77.bin",
"format": "raw-shard",
"nbytes": 45088768,
"records": [
{
"name": "transformer.h.20.mlp.gate_up_proj.q_weight",
"shape": [
22016,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 45088768,
"byteOffset": 0
}
],
"md5sum": "3524bc0d0ff6fc3de3c11859958e7202"
},
{
"dataPath": "params_shard_78.bin",
"format": "raw-shard",
"nbytes": 25165824,
"records": [
{
"name": "transformer.h.21.attn.c_attn.q_weight",
"shape": [
12288,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 25165824,
"byteOffset": 0
}
],
"md5sum": "8143a6d7e277847e543737bb7dba3ad2"
},
{
"dataPath": "params_shard_79.bin",
"format": "raw-shard",
"nbytes": 32612352,
"records": [
{
"name": "transformer.h.20.attn.c_attn.q_scale",
"shape": [
12288,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 0
},
{
"name": "transformer.h.20.attn.c_proj.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 3145728
},
{
"name": "transformer.h.20.attn.c_proj.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 11534336
},
{
"name": "transformer.h.20.ln_1.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 12582912
},
{
"name": "transformer.h.20.ln_2.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 12591104
},
{
"name": "transformer.h.20.mlp.c_proj.q_scale",
"shape": [
4096,
344
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2818048,
"byteOffset": 12599296
},
{
"name": "transformer.h.20.mlp.gate_up_proj.q_scale",
"shape": [
22016,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5636096,
"byteOffset": 15417344
},
{
"name": "transformer.h.21.attn.c_attn.bias",
"shape": [
12288
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 24576,
"byteOffset": 21053440
},
{
"name": "transformer.h.21.attn.c_attn.q_scale",
"shape": [
12288,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 21078016
},
{
"name": "transformer.h.21.attn.c_proj.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 24223744
}
],
"md5sum": "72f774cba46349cbb248abb86dd160cc"
},
{
"dataPath": "params_shard_80.bin",
"format": "raw-shard",
"nbytes": 45088768,
"records": [
{
"name": "transformer.h.21.mlp.gate_up_proj.q_weight",
"shape": [
22016,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 45088768,
"byteOffset": 0
}
],
"md5sum": "97ce95eb596c347fa2f318aa95c3805a"
},
{
"dataPath": "params_shard_81.bin",
"format": "raw-shard",
"nbytes": 25165824,
"records": [
{
"name": "transformer.h.22.attn.c_attn.q_weight",
"shape": [
12288,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 25165824,
"byteOffset": 0
}
],
"md5sum": "89766e5f5c9983bc27aa23eb84f52c78"
},
{
"dataPath": "params_shard_82.bin",
"format": "raw-shard",
"nbytes": 32088064,
"records": [
{
"name": "transformer.h.21.attn.c_proj.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 0
},
{
"name": "transformer.h.21.ln_1.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 1048576
},
{
"name": "transformer.h.21.ln_2.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 1056768
},
{
"name": "transformer.h.21.mlp.gate_up_proj.q_scale",
"shape": [
22016,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5636096,
"byteOffset": 1064960
},
{
"name": "transformer.h.21.mlp.c_proj.q_weight",
"shape": [
4096,
1376
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 22544384,
"byteOffset": 6701056
},
{
"name": "transformer.h.21.mlp.c_proj.q_scale",
"shape": [
4096,
344
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2818048,
"byteOffset": 29245440
},
{
"name": "transformer.h.22.attn.c_attn.bias",
"shape": [
12288
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 24576,
"byteOffset": 32063488
}
],
"md5sum": "5b3c5cc6887d162639481e96c606984b"
},
{
"dataPath": "params_shard_83.bin",
"format": "raw-shard",
"nbytes": 22544384,
"records": [
{
"name": "transformer.h.22.mlp.c_proj.q_weight",
"shape": [
4096,
1376
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 22544384,
"byteOffset": 0
}
],
"md5sum": "25018072f571a0f8e5dcc7be28f3b961"
},
{
"dataPath": "params_shard_84.bin",
"format": "raw-shard",
"nbytes": 45088768,
"records": [
{
"name": "transformer.h.22.mlp.gate_up_proj.q_weight",
"shape": [
22016,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 45088768,
"byteOffset": 0
}
],
"md5sum": "735ebf0a67960ec7abfcc32279b68e09"
},
{
"dataPath": "params_shard_85.bin",
"format": "raw-shard",
"nbytes": 25165824,
"records": [
{
"name": "transformer.h.23.attn.c_attn.q_weight",
"shape": [
12288,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 25165824,
"byteOffset": 0
}
],
"md5sum": "ce80e144e180f5fa1221f488748594a5"
},
{
"dataPath": "params_shard_86.bin",
"format": "raw-shard",
"nbytes": 32612352,
"records": [
{
"name": "transformer.h.22.attn.c_attn.q_scale",
"shape": [
12288,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 0
},
{
"name": "transformer.h.22.attn.c_proj.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 3145728
},
{
"name": "transformer.h.22.attn.c_proj.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 11534336
},
{
"name": "transformer.h.22.ln_1.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 12582912
},
{
"name": "transformer.h.22.ln_2.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 12591104
},
{
"name": "transformer.h.22.mlp.c_proj.q_scale",
"shape": [
4096,
344
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2818048,
"byteOffset": 12599296
},
{
"name": "transformer.h.22.mlp.gate_up_proj.q_scale",
"shape": [
22016,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5636096,
"byteOffset": 15417344
},
{
"name": "transformer.h.23.attn.c_attn.bias",
"shape": [
12288
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 24576,
"byteOffset": 21053440
},
{
"name": "transformer.h.23.attn.c_attn.q_scale",
"shape": [
12288,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 21078016
},
{
"name": "transformer.h.23.attn.c_proj.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 24223744
}
],
"md5sum": "4e61b94b5baaf7856e09c59add4f4573"
},
{
"dataPath": "params_shard_87.bin",
"format": "raw-shard",
"nbytes": 45088768,
"records": [
{
"name": "transformer.h.23.mlp.gate_up_proj.q_weight",
"shape": [
22016,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 45088768,
"byteOffset": 0
}
],
"md5sum": "bca6d6a5af15600e3e32d02cee295281"
},
{
"dataPath": "params_shard_88.bin",
"format": "raw-shard",
"nbytes": 25165824,
"records": [
{
"name": "transformer.h.24.attn.c_attn.q_weight",
"shape": [
12288,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 25165824,
"byteOffset": 0
}
],
"md5sum": "6c3ff2c5d8169adb0337a2f1a35204f0"
},
{
"dataPath": "params_shard_89.bin",
"format": "raw-shard",
"nbytes": 32088064,
"records": [
{
"name": "transformer.h.23.attn.c_proj.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 0
},
{
"name": "transformer.h.23.ln_1.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 1048576
},
{
"name": "transformer.h.23.ln_2.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 1056768
},
{
"name": "transformer.h.23.mlp.c_proj.q_weight",
"shape": [
4096,
1376
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 22544384,
"byteOffset": 1064960
},
{
"name": "transformer.h.23.mlp.c_proj.q_scale",
"shape": [
4096,
344
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2818048,
"byteOffset": 23609344
},
{
"name": "transformer.h.23.mlp.gate_up_proj.q_scale",
"shape": [
22016,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5636096,
"byteOffset": 26427392
},
{
"name": "transformer.h.24.attn.c_attn.bias",
"shape": [
12288
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 24576,
"byteOffset": 32063488
}
],
"md5sum": "657b5b472b6246094f4ba692aeb75bdd"
},
{
"dataPath": "params_shard_90.bin",
"format": "raw-shard",
"nbytes": 22544384,
"records": [
{
"name": "transformer.h.24.mlp.c_proj.q_weight",
"shape": [
4096,
1376
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 22544384,
"byteOffset": 0
}
],
"md5sum": "08d8072ec14de0e45e95848012567a5e"
},
{
"dataPath": "params_shard_91.bin",
"format": "raw-shard",
"nbytes": 45088768,
"records": [
{
"name": "transformer.h.24.mlp.gate_up_proj.q_weight",
"shape": [
22016,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 45088768,
"byteOffset": 0
}
],
"md5sum": "7859dd43f8e74876792a114a5e635cc9"
},
{
"dataPath": "params_shard_92.bin",
"format": "raw-shard",
"nbytes": 25165824,
"records": [
{
"name": "transformer.h.25.attn.c_attn.q_weight",
"shape": [
12288,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 25165824,
"byteOffset": 0
}
],
"md5sum": "ee7c3d4a78cbb5ce52d2248b98365b57"
},
{
"dataPath": "params_shard_93.bin",
"format": "raw-shard",
"nbytes": 32612352,
"records": [
{
"name": "transformer.h.24.attn.c_attn.q_scale",
"shape": [
12288,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 0
},
{
"name": "transformer.h.24.attn.c_proj.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 3145728
},
{
"name": "transformer.h.24.attn.c_proj.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 11534336
},
{
"name": "transformer.h.24.ln_1.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 12582912
},
{
"name": "transformer.h.24.ln_2.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 12591104
},
{
"name": "transformer.h.24.mlp.c_proj.q_scale",
"shape": [
4096,
344
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2818048,
"byteOffset": 12599296
},
{
"name": "transformer.h.24.mlp.gate_up_proj.q_scale",
"shape": [
22016,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5636096,
"byteOffset": 15417344
},
{
"name": "transformer.h.25.attn.c_attn.bias",
"shape": [
12288
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 24576,
"byteOffset": 21053440
},
{
"name": "transformer.h.25.attn.c_attn.q_scale",
"shape": [
12288,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 21078016
},
{
"name": "transformer.h.25.attn.c_proj.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 24223744
}
],
"md5sum": "f7072b1e6e56a14a57ad0fc64782bbcd"
},
{
"dataPath": "params_shard_94.bin",
"format": "raw-shard",
"nbytes": 45088768,
"records": [
{
"name": "transformer.h.25.mlp.gate_up_proj.q_weight",
"shape": [
22016,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 45088768,
"byteOffset": 0
}
],
"md5sum": "43fce639f8f255bbe79700ce28ae9f88"
},
{
"dataPath": "params_shard_95.bin",
"format": "raw-shard",
"nbytes": 25165824,
"records": [
{
"name": "transformer.h.26.attn.c_attn.q_weight",
"shape": [
12288,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 25165824,
"byteOffset": 0
}
],
"md5sum": "14f32968096a9b035e4fb87c5b3715e1"
},
{
"dataPath": "params_shard_96.bin",
"format": "raw-shard",
"nbytes": 32088064,
"records": [
{
"name": "transformer.h.25.attn.c_proj.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 0
},
{
"name": "transformer.h.25.ln_1.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 1048576
},
{
"name": "transformer.h.25.ln_2.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 1056768
},
{
"name": "transformer.h.25.mlp.c_proj.q_weight",
"shape": [
4096,
1376
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 22544384,
"byteOffset": 1064960
},
{
"name": "transformer.h.25.mlp.c_proj.q_scale",
"shape": [
4096,
344
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2818048,
"byteOffset": 23609344
},
{
"name": "transformer.h.25.mlp.gate_up_proj.q_scale",
"shape": [
22016,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5636096,
"byteOffset": 26427392
},
{
"name": "transformer.h.26.attn.c_attn.bias",
"shape": [
12288
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 24576,
"byteOffset": 32063488
}
],
"md5sum": "746c34579217dc65f3e080b45734f3de"
},
{
"dataPath": "params_shard_97.bin",
"format": "raw-shard",
"nbytes": 45088768,
"records": [
{
"name": "transformer.h.26.mlp.gate_up_proj.q_weight",
"shape": [
22016,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 45088768,
"byteOffset": 0
}
],
"md5sum": "05196ff177b0cd15cfb57ba74b1c9beb"
},
{
"dataPath": "params_shard_98.bin",
"format": "raw-shard",
"nbytes": 22544384,
"records": [
{
"name": "transformer.h.26.mlp.c_proj.q_weight",
"shape": [
4096,
1376
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 22544384,
"byteOffset": 0
}
],
"md5sum": "d59ec9da80759bff7e8870079b153c3f"
},
{
"dataPath": "params_shard_99.bin",
"format": "raw-shard",
"nbytes": 25165824,
"records": [
{
"name": "transformer.h.27.attn.c_attn.q_weight",
"shape": [
12288,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 25165824,
"byteOffset": 0
}
],
"md5sum": "b94ca4841b9bf28a25e455074f0a5607"
},
{
"dataPath": "params_shard_100.bin",
"format": "raw-shard",
"nbytes": 32612352,
"records": [
{
"name": "transformer.h.26.attn.c_attn.q_scale",
"shape": [
12288,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 0
},
{
"name": "transformer.h.26.attn.c_proj.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 3145728
},
{
"name": "transformer.h.26.attn.c_proj.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 11534336
},
{
"name": "transformer.h.26.ln_1.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 12582912
},
{
"name": "transformer.h.26.ln_2.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 12591104
},
{
"name": "transformer.h.26.mlp.gate_up_proj.q_scale",
"shape": [
22016,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5636096,
"byteOffset": 12599296
},
{
"name": "transformer.h.26.mlp.c_proj.q_scale",
"shape": [
4096,
344
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2818048,
"byteOffset": 18235392
},
{
"name": "transformer.h.27.attn.c_attn.bias",
"shape": [
12288
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 24576,
"byteOffset": 21053440
},
{
"name": "transformer.h.27.attn.c_attn.q_scale",
"shape": [
12288,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 21078016
},
{
"name": "transformer.h.27.attn.c_proj.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 24223744
}
],
"md5sum": "f7edbb01eb395562e9ea9baaeb6c9b5d"
},
{
"dataPath": "params_shard_101.bin",
"format": "raw-shard",
"nbytes": 45088768,
"records": [
{
"name": "transformer.h.27.mlp.gate_up_proj.q_weight",
"shape": [
22016,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 45088768,
"byteOffset": 0
}
],
"md5sum": "c4620a88c59345940642f87ed1ae9bd2"
},
{
"dataPath": "params_shard_102.bin",
"format": "raw-shard",
"nbytes": 25165824,
"records": [
{
"name": "transformer.h.28.attn.c_attn.q_weight",
"shape": [
12288,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 25165824,
"byteOffset": 0
}
],
"md5sum": "225121e1d46cfb3732fd37caf41a658b"
},
{
"dataPath": "params_shard_103.bin",
"format": "raw-shard",
"nbytes": 32088064,
"records": [
{
"name": "transformer.h.27.attn.c_proj.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 0
},
{
"name": "transformer.h.27.ln_1.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 1048576
},
{
"name": "transformer.h.27.ln_2.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 1056768
},
{
"name": "transformer.h.27.mlp.c_proj.q_weight",
"shape": [
4096,
1376
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 22544384,
"byteOffset": 1064960
},
{
"name": "transformer.h.27.mlp.c_proj.q_scale",
"shape": [
4096,
344
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2818048,
"byteOffset": 23609344
},
{
"name": "transformer.h.27.mlp.gate_up_proj.q_scale",
"shape": [
22016,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5636096,
"byteOffset": 26427392
},
{
"name": "transformer.h.28.attn.c_attn.bias",
"shape": [
12288
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 24576,
"byteOffset": 32063488
}
],
"md5sum": "00fdb64e54f957e284b81494f57ebe84"
},
{
"dataPath": "params_shard_104.bin",
"format": "raw-shard",
"nbytes": 22544384,
"records": [
{
"name": "transformer.h.28.mlp.c_proj.q_weight",
"shape": [
4096,
1376
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 22544384,
"byteOffset": 0
}
],
"md5sum": "87c749f7c128c4f20f6458f777ba74d6"
},
{
"dataPath": "params_shard_105.bin",
"format": "raw-shard",
"nbytes": 45088768,
"records": [
{
"name": "transformer.h.28.mlp.gate_up_proj.q_weight",
"shape": [
22016,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 45088768,
"byteOffset": 0
}
],
"md5sum": "3327b611657e05125cfcee48093bb579"
},
{
"dataPath": "params_shard_106.bin",
"format": "raw-shard",
"nbytes": 25165824,
"records": [
{
"name": "transformer.h.29.attn.c_attn.q_weight",
"shape": [
12288,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 25165824,
"byteOffset": 0
}
],
"md5sum": "cd8ddf55ae4c51bf0e098c989c8c8fe6"
},
{
"dataPath": "params_shard_107.bin",
"format": "raw-shard",
"nbytes": 32612352,
"records": [
{
"name": "transformer.h.28.attn.c_attn.q_scale",
"shape": [
12288,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 0
},
{
"name": "transformer.h.28.attn.c_proj.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 3145728
},
{
"name": "transformer.h.28.attn.c_proj.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 11534336
},
{
"name": "transformer.h.28.ln_1.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 12582912
},
{
"name": "transformer.h.28.ln_2.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 12591104
},
{
"name": "transformer.h.28.mlp.c_proj.q_scale",
"shape": [
4096,
344
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2818048,
"byteOffset": 12599296
},
{
"name": "transformer.h.28.mlp.gate_up_proj.q_scale",
"shape": [
22016,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5636096,
"byteOffset": 15417344
},
{
"name": "transformer.h.29.attn.c_attn.bias",
"shape": [
12288
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 24576,
"byteOffset": 21053440
},
{
"name": "transformer.h.29.attn.c_attn.q_scale",
"shape": [
12288,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 21078016
},
{
"name": "transformer.h.29.attn.c_proj.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 24223744
}
],
"md5sum": "e5fca527a72a9da266d46aee6178f48f"
},
{
"dataPath": "params_shard_108.bin",
"format": "raw-shard",
"nbytes": 45088768,
"records": [
{
"name": "transformer.h.29.mlp.gate_up_proj.q_weight",
"shape": [
22016,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 45088768,
"byteOffset": 0
}
],
"md5sum": "c14b144c1c785aaf2e00b9ffbaca1d01"
},
{
"dataPath": "params_shard_109.bin",
"format": "raw-shard",
"nbytes": 25165824,
"records": [
{
"name": "transformer.h.30.attn.c_attn.q_weight",
"shape": [
12288,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 25165824,
"byteOffset": 0
}
],
"md5sum": "9cf3d700ca1aa2bdffbfc8c174e98ca0"
},
{
"dataPath": "params_shard_110.bin",
"format": "raw-shard",
"nbytes": 32088064,
"records": [
{
"name": "transformer.h.29.attn.c_proj.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 0
},
{
"name": "transformer.h.29.ln_1.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 1048576
},
{
"name": "transformer.h.29.ln_2.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 1056768
},
{
"name": "transformer.h.29.mlp.c_proj.q_weight",
"shape": [
4096,
1376
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 22544384,
"byteOffset": 1064960
},
{
"name": "transformer.h.29.mlp.c_proj.q_scale",
"shape": [
4096,
344
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2818048,
"byteOffset": 23609344
},
{
"name": "transformer.h.29.mlp.gate_up_proj.q_scale",
"shape": [
22016,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5636096,
"byteOffset": 26427392
},
{
"name": "transformer.h.30.attn.c_attn.bias",
"shape": [
12288
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 24576,
"byteOffset": 32063488
}
],
"md5sum": "9949eb63e10acc7950a1c6d611973e27"
},
{
"dataPath": "params_shard_111.bin",
"format": "raw-shard",
"nbytes": 22544384,
"records": [
{
"name": "transformer.h.30.mlp.c_proj.q_weight",
"shape": [
4096,
1376
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 22544384,
"byteOffset": 0
}
],
"md5sum": "8cbff8dd317047e1dd4ae8102096eb85"
},
{
"dataPath": "params_shard_112.bin",
"format": "raw-shard",
"nbytes": 45088768,
"records": [
{
"name": "transformer.h.30.mlp.gate_up_proj.q_weight",
"shape": [
22016,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 45088768,
"byteOffset": 0
}
],
"md5sum": "db5897611295e59190df9d170dc777f9"
},
{
"dataPath": "params_shard_113.bin",
"format": "raw-shard",
"nbytes": 25165824,
"records": [
{
"name": "transformer.h.31.attn.c_attn.q_weight",
"shape": [
12288,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 25165824,
"byteOffset": 0
}
],
"md5sum": "f933686d39e286fcac60129ad204e9ac"
},
{
"dataPath": "params_shard_114.bin",
"format": "raw-shard",
"nbytes": 32612352,
"records": [
{
"name": "transformer.h.30.attn.c_attn.q_scale",
"shape": [
12288,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 0
},
{
"name": "transformer.h.30.attn.c_proj.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 3145728
},
{
"name": "transformer.h.30.attn.c_proj.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 11534336
},
{
"name": "transformer.h.30.ln_1.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 12582912
},
{
"name": "transformer.h.30.ln_2.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 12591104
},
{
"name": "transformer.h.30.mlp.c_proj.q_scale",
"shape": [
4096,
344
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2818048,
"byteOffset": 12599296
},
{
"name": "transformer.h.30.mlp.gate_up_proj.q_scale",
"shape": [
22016,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5636096,
"byteOffset": 15417344
},
{
"name": "transformer.h.31.attn.c_attn.bias",
"shape": [
12288
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 24576,
"byteOffset": 21053440
},
{
"name": "transformer.h.31.attn.c_attn.q_scale",
"shape": [
12288,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 21078016
},
{
"name": "transformer.h.31.attn.c_proj.q_weight",
"shape": [
4096,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 24223744
}
],
"md5sum": "637dbac345090c2dd1a2cf87f011a894"
},
{
"dataPath": "params_shard_115.bin",
"format": "raw-shard",
"nbytes": 45088768,
"records": [
{
"name": "transformer.h.31.mlp.gate_up_proj.q_weight",
"shape": [
22016,
512
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 45088768,
"byteOffset": 0
}
],
"md5sum": "3bc2f1c2fa8029f39c3652c13344994c"
},
{
"dataPath": "params_shard_116.bin",
"format": "raw-shard",
"nbytes": 6701056,
"records": [
{
"name": "transformer.h.31.attn.c_proj.q_scale",
"shape": [
4096,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 0
},
{
"name": "transformer.h.31.ln_1.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 1048576
},
{
"name": "transformer.h.31.ln_2.weight",
"shape": [
4096
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8192,
"byteOffset": 1056768
},
{
"name": "transformer.h.31.mlp.gate_up_proj.q_scale",
"shape": [
22016,
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5636096,
"byteOffset": 1064960
}
],
"md5sum": "cd51c96985777f6b2d7a49fde3e4f9c4"
}
]
}